Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
219 changes: 160 additions & 59 deletions electron/native/wgc-capture/src/audio_sample_utils.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -478,6 +478,18 @@ void mixAudioInPlace(
}
}

namespace {

// How far behind real time AudioMixer writes; see mixLoop. Wider than the capture
// threads' poll jitter with room for a stall, and harmless beyond that: the
// output timeline does not move with it.
// ponytail: absorbs jitter, not clock drift. A device slower than the steady clock
// spends the cushion over a long take, then leaves one gap as it re-anchors;
// resampling each source to the mixer clock is the upgrade if that is ever heard.
constexpr uint32_t MixerCushionMs = 100;

} // namespace

AudioMixer::AudioMixer(
const AudioInputFormat& format,
const AudioInputFormat& systemFormat,
Expand Down Expand Up @@ -515,29 +527,53 @@ bool AudioMixer::start() {
return true;
}

void AudioMixer::resetSources() {
systemQueue_.clear();
microphoneQueue_.clear();
systemDecimator_.reset();
microphoneDecimator_.reset();
systemStarved_ = false;
microphoneStarved_ = false;
}

void AudioMixer::beginTimeline() {
{
std::scoped_lock lock(mutex_);
systemQueue_.clear();
microphoneQueue_.clear();
systemDecimator_.reset();
microphoneDecimator_.reset();
resetSources();
emittedFrames_ = 0;
mixedFrames_ = 0;
timelineStarted_ = true;
// Here, not when mixLoop next wakes: the video's T0 is taken right after
// this call, and every millisecond the mixer took to wake would have
// placed the whole track that much early -- tens of them on a busy machine.
clockStart_ = std::chrono::steady_clock::now();
clockAnchored_ = true;
}
cv_.notify_all();
}

void AudioMixer::setPaused(bool paused) {
{
std::scoped_lock lock(mutex_);
paused_ = paused;
if (paused_) {
systemQueue_.clear();
microphoneQueue_.clear();
systemDecimator_.reset();
microphoneDecimator_.reset();
std::unique_lock lock(mutex_);
if (paused && !paused_) {
// The queues still hold what the cushion kept back; mixLoop writes it
// up to this instant, and the resume clears whatever is left.
pausedAt_ = std::chrono::steady_clock::now();
pauseFlushed_ = !timelineStarted_;
} else if (!paused && paused_) {
// A resume right behind the pause would otherwise clear the cushion
// before mixLoop has written it: the last 100 ms of voice before the
// pause, gone (measured, 80 to 90 ms).
cv_.wait(lock, [this] { return pauseFlushed_ || stopRequested_.load(); });
resetSources();
// Resumed at this instant, where the pause left off: the flush has
// run, so `emittedFrames_` is the pause point and mixLoop is idle.
clockStart_ = std::chrono::steady_clock::now() -
std::chrono::duration_cast<std::chrono::steady_clock::duration>(std::chrono::duration<double>(
static_cast<double>(emittedFrames_) / format_.sampleRate));
clockAnchored_ = true;
}
paused_ = paused;
Comment thread
coderabbitai[bot] marked this conversation as resolved.
}
cv_.notify_all();
}
Expand All @@ -560,7 +596,7 @@ void AudioMixer::pushSystem(const BYTE* data, DWORD byteCount) {
if (paused_) {
return;
}
append(systemQueue_, data, byteCount, systemFormat_, 1.0, systemDecimator_);
append(systemQueue_, systemStarved_, data, byteCount, systemFormat_, 1.0, systemDecimator_);
}
cv_.notify_all();
}
Expand All @@ -581,13 +617,16 @@ void AudioMixer::pushMicrophone(const BYTE* data, DWORD byteCount) {
// in: the queue would hold a flat-topped signal and the mix would add
// further distortion on top of it, with whatever headroom the opposite
// polarity of the system stream offered already destroyed.
append(microphoneQueue_, data, byteCount, microphoneFormat_, 1.0, microphoneDecimator_);
append(
microphoneQueue_, microphoneStarved_, data, byteCount, microphoneFormat_, 1.0,
microphoneDecimator_);
}
cv_.notify_all();
}

void AudioMixer::append(
std::vector<BYTE>& queue,
bool& starved,
const BYTE* data,
DWORD byteCount,
const AudioInputFormat& sourceFormat,
Expand All @@ -598,16 +637,35 @@ void AudioMixer::append(
}

convertAudioWithGain(data, byteCount, sourceFormat, format_, gain, gainBuffer_, decimator);
// A source that ran dry is coming back: loopback after a silence, or a device
// that stalled for longer than the cushion. Its packet belongs at now, which
// is as far ahead of what mixLoop has taken as real time is -- the cushion
// plus however late mixLoop is running, which on a loaded machine is tens of
// milliseconds. Queued at the front, it would land that much early.
if (starved) {
uint64_t lagFrames = 0;
if (clockAnchored_) {
const double elapsed =
std::chrono::duration<double>(std::chrono::steady_clock::now() - clockStart_).count();
const auto nowFrames = static_cast<uint64_t>(std::max(0.0, elapsed) * format_.sampleRate);
lagFrames = nowFrames > mixedFrames_ ? nowFrames - mixedFrames_ : 0;
}
queue.assign(static_cast<size_t>(lagFrames) * format_.blockAlign, 0);
starved = false;
}
queue.insert(queue.end(), gainBuffer_.begin(), gainBuffer_.end());
}

bool AudioMixer::pop(std::vector<BYTE>& queue, std::vector<BYTE>& chunk, size_t byteCount) {
bool AudioMixer::pop(
std::vector<BYTE>& queue, bool& starved, std::vector<BYTE>& chunk, size_t byteCount) {
chunk.assign(byteCount, 0);
if (queue.size() < byteCount) {
starved = true;
}
if (queue.empty()) {
chunk.assign(byteCount, 0);
return false;
}

chunk.assign(byteCount, 0);
const size_t copiedBytes = std::min(byteCount, queue.size());
std::memcpy(chunk.data(), queue.data(), copiedBytes);
queue.erase(queue.begin(), queue.begin() + static_cast<std::ptrdiff_t>(copiedBytes));
Expand All @@ -631,72 +689,60 @@ bool AudioMixer::pop(std::vector<BYTE>& queue, std::vector<BYTE>& chunk, size_t
* to cause a system-audio desync it merely stopped concealing
* (getopenscreen/openscreen#406).
*
* `audioClockStart` is anchored so that `emittedFrames_` always describes the
* A cushion behind real time (getopenscreen/openscreen#911). The capture threads
* poll WASAPI and push whatever has piled up -- every 15.6 ms at the default
* timer resolution, and later than that on a busy machine -- so a packet can
* reach its queue after the chunk it belongs to is due. Mixing at real time
* zero-filled that chunk and the packet played one chunk late: a hole of up to
* 10 ms in the middle of a continuous voice, heard as crackle. Measured on real
* takes: ten or more such holes in 25 s. Writing `MixerCushionMs` behind real time
* gives a late packet that long to arrive. The timestamps do not move -- they
* still come from `emittedFrames_` -- so the cushion only delays the writing,
* and a pause or a stop writes what it still holds up to that instant.
*
* `clockStart_` is anchored so that `emittedFrames_` always describes the
* time elapsed since the timeline began; re-deriving it on resume is what lets a
* pause interrupt the clock without shifting everything recorded after it.
*/
void AudioMixer::mixLoop() {
const uint32_t chunkFrames = std::max<uint32_t>(1, format_.sampleRate / 100);
const size_t chunkBytes = static_cast<size_t>(chunkFrames) * format_.blockAlign;
const uint64_t cushionFrames = static_cast<uint64_t>(format_.sampleRate) * MixerCushionMs / 1000;
std::vector<BYTE> mixedChunk;
std::vector<BYTE> sourceChunk;
std::chrono::steady_clock::time_point audioClockStart;
bool audioClockAnchored = false;
// This loop's copy of `clockStart_`, taken under the lock at the top of each
// pass: beginTimeline and a resume move it from other threads.
std::chrono::steady_clock::time_point clockStart;

const auto framesToDuration = [&](uint64_t frames) {
return std::chrono::duration_cast<std::chrono::steady_clock::duration>(
std::chrono::duration<double>(static_cast<double>(frames) / format_.sampleRate));
};
const auto framesAt = [&](std::chrono::steady_clock::time_point time) {
const double elapsed = std::chrono::duration<double>(time - clockStart).count();
return static_cast<uint64_t>(std::max(0.0, elapsed) * format_.sampleRate);
};

while (true) {
{
std::unique_lock lock(mutex_);
cv_.wait_for(lock, std::chrono::milliseconds(20), [&] {
return stopRequested_.load() || (timelineStarted_ && !paused_);
});

if (stopRequested_) {
break;
}
if (!timelineStarted_ || paused_) {
// A pause stops the clock rather than resetting it: the anchor is
// re-derived from `emittedFrames_` on resume, so what follows keeps
// the position it would have had.
audioClockAnchored = false;
continue;
}
}

const auto now = std::chrono::steady_clock::now();
if (!audioClockAnchored) {
audioClockStart = now - framesToDuration(emittedFrames_);
audioClockAnchored = true;
}

// How much of the timeline real time has covered. Emitting up to here --
// from the queues where they have data, from silence where they do not --
// is what keeps the audio clock pinned to the take rather than to whether
// anything happened to be playing.
const auto elapsed = std::chrono::duration<double>(now - audioClockStart).count();
const uint64_t targetFrames = static_cast<uint64_t>(elapsed * format_.sampleRate);

// Writes every chunk that ends by `targetFrames` -- from the queues where they
// have data, from silence where they do not, which is what keeps the audio
// clock pinned to the take rather than to whether anything happened to be
// playing. False once the output refuses a chunk.
const auto emitUntil = [&](uint64_t targetFrames) {
while (emittedFrames_ + chunkFrames <= targetFrames) {
{
std::scoped_lock lock(mutex_);
if (stopRequested_ || !timelineStarted_ || paused_) {
break;
}
mixedChunk.assign(chunkBytes, 0);
if (includeSystem_) {
pop(systemQueue_, sourceChunk, chunkBytes);
pop(systemQueue_, systemStarved_, sourceChunk, chunkBytes);
mixAudioInPlace(mixedChunk, sourceChunk.data(), static_cast<DWORD>(sourceChunk.size()), format_);
}
if (includeMicrophone_) {
pop(microphoneQueue_, sourceChunk, chunkBytes);
pop(microphoneQueue_, microphoneStarved_, sourceChunk, chunkBytes);
mixAudioInPlace(
mixedChunk, sourceChunk.data(), static_cast<DWORD>(sourceChunk.size()), format_,
microphoneGain_);
}
mixedFrames_ = emittedFrames_ + chunkFrames;
}

const int64_t timestampHns =
Expand All @@ -705,15 +751,70 @@ void AudioMixer::mixLoop() {
static_cast<int64_t>((static_cast<uint64_t>(chunkFrames) * HnsPerSecond) / format_.sampleRate);
if (!output_(mixedChunk.data(), static_cast<DWORD>(mixedChunk.size()), timestampHns, durationHns)) {
stopRequested_ = true;
break;
return false;
}
emittedFrames_ += chunkFrames;
}
return true;
};

if (stopRequested_) {
while (true) {
bool stopping = false;
bool running = false;
bool flush = false;
std::chrono::steady_clock::time_point pausedAt;
{
std::unique_lock lock(mutex_);
cv_.wait_for(lock, std::chrono::milliseconds(20), [&] {
return stopRequested_.load() || (timelineStarted_ && !paused_);
});
stopping = stopRequested_;
running = timelineStarted_ && !paused_;
pausedAt = pausedAt_;
clockStart = clockStart_;
if (stopping || !running) {
flush = clockAnchored_;
clockAnchored_ = false;
}
}

if (stopping || !running) {
// A pause stops the clock rather than resetting it: the anchor is
// re-derived from `emittedFrames_` on resume, so what follows keeps
// the position it would have had -- once the cushion is written up to
// the pause, which is where that position is. A stop writes it up to
// now, unless it lands on a pause this loop has not handled yet: the
// paused span is out of the video, so it stays out of the audio.
const bool written =
!flush || emitUntil(framesAt(running ? std::chrono::steady_clock::now() : pausedAt));
{
std::scoped_lock lock(mutex_);
pauseFlushed_ = true;
}
cv_.notify_all();
if (!written || stopping) {
break;
}
continue;
}

const uint64_t realFrames = framesAt(std::chrono::steady_clock::now());
if (!emitUntil(realFrames > cushionFrames ? realFrames - cushionFrames : 0)) {
break;
}

std::this_thread::sleep_until(audioClockStart + framesToDuration(emittedFrames_ + chunkFrames));
// Woken early by a pause or a stop, so that one writes the cushion out at once.
std::unique_lock lock(mutex_);
cv_.wait_until(
lock, clockStart + framesToDuration(emittedFrames_ + chunkFrames + cushionFrames),
[&] { return stopRequested_.load() || paused_; });
Comment thread
coderabbitai[bot] marked this conversation as resolved.
}
// Whatever ended the loop, a resume waiting on the pause flush must not wait
// for a loop that is gone. Set under the lock, so the wakeup cannot slip
// between the waiter's check and its sleep.
{
std::scoped_lock lock(mutex_);
pauseFlushed_ = true;
}
cv_.notify_all();
}
18 changes: 17 additions & 1 deletion electron/native/wgc-capture/src/audio_sample_utils.h
Original file line number Diff line number Diff line change
Expand Up @@ -5,6 +5,7 @@
#include <Windows.h>

#include <atomic>
#include <chrono>
#include <condition_variable>
#include <cstdint>
#include <functional>
Expand Down Expand Up @@ -103,12 +104,14 @@ class AudioMixer {
private:
void append(
std::vector<BYTE>& queue,
bool& starved,
const BYTE* data,
DWORD byteCount,
const AudioInputFormat& sourceFormat,
double gain,
AudioDecimatorState& decimator);
bool pop(std::vector<BYTE>& queue, std::vector<BYTE>& chunk, size_t byteCount);
bool pop(std::vector<BYTE>& queue, bool& starved, std::vector<BYTE>& chunk, size_t byteCount);
void resetSources();
void mixLoop();

AudioInputFormat format_{};
Expand All @@ -124,10 +127,23 @@ class AudioMixer {
std::vector<BYTE> microphoneQueue_;
AudioDecimatorState systemDecimator_;
AudioDecimatorState microphoneDecimator_;
// Set when pop() runs a source dry, cleared by its next packet (see append).
bool systemStarved_ = false;
bool microphoneStarved_ = false;
std::vector<BYTE> gainBuffer_;
std::thread thread_;
std::atomic<bool> stopRequested_ = false;
bool timelineStarted_ = false;
bool paused_ = false;
std::chrono::steady_clock::time_point pausedAt_{};
// Set by mixLoop once it has written the cushion up to `pausedAt_`; a resume
// waits for it, or it would clear audio that belongs before the pause.
bool pauseFlushed_ = true;
// mixLoop's clock, shared so a source coming back after running dry can be
// placed at real time (see append). Written by mixLoop under `mutex_`.
std::chrono::steady_clock::time_point clockStart_{};
bool clockAnchored_ = false;
// Frames mixLoop has taken from the queues so far, under `mutex_`.
uint64_t mixedFrames_ = 0;
uint64_t emittedFrames_ = 0;
};
Loading
Loading