diff --git a/electron/native/wgc-capture/src/audio_sample_utils.cpp b/electron/native/wgc-capture/src/audio_sample_utils.cpp index 54e77e40f..cab9aab61 100644 --- a/electron/native/wgc-capture/src/audio_sample_utils.cpp +++ b/electron/native/wgc-capture/src/audio_sample_utils.cpp @@ -478,6 +478,18 @@ void mixAudioInPlace( } } +namespace { + +// How far behind real time AudioMixer writes; see mixLoop. Wider than the capture +// threads' poll jitter with room for a stall, and harmless beyond that: the +// output timeline does not move with it. +// ponytail: absorbs jitter, not clock drift. A device slower than the steady clock +// spends the cushion over a long take, then leaves one gap as it re-anchors; +// resampling each source to the mixer clock is the upgrade if that is ever heard. +constexpr uint32_t MixerCushionMs = 100; + +} // namespace + AudioMixer::AudioMixer( const AudioInputFormat& format, const AudioInputFormat& systemFormat, @@ -515,29 +527,53 @@ bool AudioMixer::start() { return true; } +void AudioMixer::resetSources() { + systemQueue_.clear(); + microphoneQueue_.clear(); + systemDecimator_.reset(); + microphoneDecimator_.reset(); + systemStarved_ = false; + microphoneStarved_ = false; +} + void AudioMixer::beginTimeline() { { std::scoped_lock lock(mutex_); - systemQueue_.clear(); - microphoneQueue_.clear(); - systemDecimator_.reset(); - microphoneDecimator_.reset(); + resetSources(); emittedFrames_ = 0; + mixedFrames_ = 0; timelineStarted_ = true; + // Here, not when mixLoop next wakes: the video's T0 is taken right after + // this call, and every millisecond the mixer took to wake would have + // placed the whole track that much early -- tens of them on a busy machine. + clockStart_ = std::chrono::steady_clock::now(); + clockAnchored_ = true; } cv_.notify_all(); } void AudioMixer::setPaused(bool paused) { { - std::scoped_lock lock(mutex_); - paused_ = paused; - if (paused_) { - systemQueue_.clear(); - microphoneQueue_.clear(); - systemDecimator_.reset(); - microphoneDecimator_.reset(); + std::unique_lock lock(mutex_); + if (paused && !paused_) { + // The queues still hold what the cushion kept back; mixLoop writes it + // up to this instant, and the resume clears whatever is left. + pausedAt_ = std::chrono::steady_clock::now(); + pauseFlushed_ = !timelineStarted_; + } else if (!paused && paused_) { + // A resume right behind the pause would otherwise clear the cushion + // before mixLoop has written it: the last 100 ms of voice before the + // pause, gone (measured, 80 to 90 ms). + cv_.wait(lock, [this] { return pauseFlushed_ || stopRequested_.load(); }); + resetSources(); + // Resumed at this instant, where the pause left off: the flush has + // run, so `emittedFrames_` is the pause point and mixLoop is idle. + clockStart_ = std::chrono::steady_clock::now() - + std::chrono::duration_cast(std::chrono::duration( + static_cast(emittedFrames_) / format_.sampleRate)); + clockAnchored_ = true; } + paused_ = paused; } cv_.notify_all(); } @@ -560,7 +596,7 @@ void AudioMixer::pushSystem(const BYTE* data, DWORD byteCount) { if (paused_) { return; } - append(systemQueue_, data, byteCount, systemFormat_, 1.0, systemDecimator_); + append(systemQueue_, systemStarved_, data, byteCount, systemFormat_, 1.0, systemDecimator_); } cv_.notify_all(); } @@ -581,13 +617,16 @@ void AudioMixer::pushMicrophone(const BYTE* data, DWORD byteCount) { // in: the queue would hold a flat-topped signal and the mix would add // further distortion on top of it, with whatever headroom the opposite // polarity of the system stream offered already destroyed. - append(microphoneQueue_, data, byteCount, microphoneFormat_, 1.0, microphoneDecimator_); + append( + microphoneQueue_, microphoneStarved_, data, byteCount, microphoneFormat_, 1.0, + microphoneDecimator_); } cv_.notify_all(); } void AudioMixer::append( std::vector& queue, + bool& starved, const BYTE* data, DWORD byteCount, const AudioInputFormat& sourceFormat, @@ -598,16 +637,35 @@ void AudioMixer::append( } convertAudioWithGain(data, byteCount, sourceFormat, format_, gain, gainBuffer_, decimator); + // A source that ran dry is coming back: loopback after a silence, or a device + // that stalled for longer than the cushion. Its packet belongs at now, which + // is as far ahead of what mixLoop has taken as real time is -- the cushion + // plus however late mixLoop is running, which on a loaded machine is tens of + // milliseconds. Queued at the front, it would land that much early. + if (starved) { + uint64_t lagFrames = 0; + if (clockAnchored_) { + const double elapsed = + std::chrono::duration(std::chrono::steady_clock::now() - clockStart_).count(); + const auto nowFrames = static_cast(std::max(0.0, elapsed) * format_.sampleRate); + lagFrames = nowFrames > mixedFrames_ ? nowFrames - mixedFrames_ : 0; + } + queue.assign(static_cast(lagFrames) * format_.blockAlign, 0); + starved = false; + } queue.insert(queue.end(), gainBuffer_.begin(), gainBuffer_.end()); } -bool AudioMixer::pop(std::vector& queue, std::vector& chunk, size_t byteCount) { +bool AudioMixer::pop( + std::vector& queue, bool& starved, std::vector& chunk, size_t byteCount) { + chunk.assign(byteCount, 0); + if (queue.size() < byteCount) { + starved = true; + } if (queue.empty()) { - chunk.assign(byteCount, 0); return false; } - chunk.assign(byteCount, 0); const size_t copiedBytes = std::min(byteCount, queue.size()); std::memcpy(chunk.data(), queue.data(), copiedBytes); queue.erase(queue.begin(), queue.begin() + static_cast(copiedBytes)); @@ -631,72 +689,60 @@ bool AudioMixer::pop(std::vector& queue, std::vector& chunk, size_t * to cause a system-audio desync it merely stopped concealing * (getopenscreen/openscreen#406). * - * `audioClockStart` is anchored so that `emittedFrames_` always describes the + * A cushion behind real time (getopenscreen/openscreen#911). The capture threads + * poll WASAPI and push whatever has piled up -- every 15.6 ms at the default + * timer resolution, and later than that on a busy machine -- so a packet can + * reach its queue after the chunk it belongs to is due. Mixing at real time + * zero-filled that chunk and the packet played one chunk late: a hole of up to + * 10 ms in the middle of a continuous voice, heard as crackle. Measured on real + * takes: ten or more such holes in 25 s. Writing `MixerCushionMs` behind real time + * gives a late packet that long to arrive. The timestamps do not move -- they + * still come from `emittedFrames_` -- so the cushion only delays the writing, + * and a pause or a stop writes what it still holds up to that instant. + * + * `clockStart_` is anchored so that `emittedFrames_` always describes the * time elapsed since the timeline began; re-deriving it on resume is what lets a * pause interrupt the clock without shifting everything recorded after it. */ void AudioMixer::mixLoop() { const uint32_t chunkFrames = std::max(1, format_.sampleRate / 100); const size_t chunkBytes = static_cast(chunkFrames) * format_.blockAlign; + const uint64_t cushionFrames = static_cast(format_.sampleRate) * MixerCushionMs / 1000; std::vector mixedChunk; std::vector sourceChunk; - std::chrono::steady_clock::time_point audioClockStart; - bool audioClockAnchored = false; + // This loop's copy of `clockStart_`, taken under the lock at the top of each + // pass: beginTimeline and a resume move it from other threads. + std::chrono::steady_clock::time_point clockStart; const auto framesToDuration = [&](uint64_t frames) { return std::chrono::duration_cast( std::chrono::duration(static_cast(frames) / format_.sampleRate)); }; + const auto framesAt = [&](std::chrono::steady_clock::time_point time) { + const double elapsed = std::chrono::duration(time - clockStart).count(); + return static_cast(std::max(0.0, elapsed) * format_.sampleRate); + }; - while (true) { - { - std::unique_lock lock(mutex_); - cv_.wait_for(lock, std::chrono::milliseconds(20), [&] { - return stopRequested_.load() || (timelineStarted_ && !paused_); - }); - - if (stopRequested_) { - break; - } - if (!timelineStarted_ || paused_) { - // A pause stops the clock rather than resetting it: the anchor is - // re-derived from `emittedFrames_` on resume, so what follows keeps - // the position it would have had. - audioClockAnchored = false; - continue; - } - } - - const auto now = std::chrono::steady_clock::now(); - if (!audioClockAnchored) { - audioClockStart = now - framesToDuration(emittedFrames_); - audioClockAnchored = true; - } - - // How much of the timeline real time has covered. Emitting up to here -- - // from the queues where they have data, from silence where they do not -- - // is what keeps the audio clock pinned to the take rather than to whether - // anything happened to be playing. - const auto elapsed = std::chrono::duration(now - audioClockStart).count(); - const uint64_t targetFrames = static_cast(elapsed * format_.sampleRate); - + // Writes every chunk that ends by `targetFrames` -- from the queues where they + // have data, from silence where they do not, which is what keeps the audio + // clock pinned to the take rather than to whether anything happened to be + // playing. False once the output refuses a chunk. + const auto emitUntil = [&](uint64_t targetFrames) { while (emittedFrames_ + chunkFrames <= targetFrames) { { std::scoped_lock lock(mutex_); - if (stopRequested_ || !timelineStarted_ || paused_) { - break; - } mixedChunk.assign(chunkBytes, 0); if (includeSystem_) { - pop(systemQueue_, sourceChunk, chunkBytes); + pop(systemQueue_, systemStarved_, sourceChunk, chunkBytes); mixAudioInPlace(mixedChunk, sourceChunk.data(), static_cast(sourceChunk.size()), format_); } if (includeMicrophone_) { - pop(microphoneQueue_, sourceChunk, chunkBytes); + pop(microphoneQueue_, microphoneStarved_, sourceChunk, chunkBytes); mixAudioInPlace( mixedChunk, sourceChunk.data(), static_cast(sourceChunk.size()), format_, microphoneGain_); } + mixedFrames_ = emittedFrames_ + chunkFrames; } const int64_t timestampHns = @@ -705,15 +751,70 @@ void AudioMixer::mixLoop() { static_cast((static_cast(chunkFrames) * HnsPerSecond) / format_.sampleRate); if (!output_(mixedChunk.data(), static_cast(mixedChunk.size()), timestampHns, durationHns)) { stopRequested_ = true; - break; + return false; } emittedFrames_ += chunkFrames; } + return true; + }; - if (stopRequested_) { + while (true) { + bool stopping = false; + bool running = false; + bool flush = false; + std::chrono::steady_clock::time_point pausedAt; + { + std::unique_lock lock(mutex_); + cv_.wait_for(lock, std::chrono::milliseconds(20), [&] { + return stopRequested_.load() || (timelineStarted_ && !paused_); + }); + stopping = stopRequested_; + running = timelineStarted_ && !paused_; + pausedAt = pausedAt_; + clockStart = clockStart_; + if (stopping || !running) { + flush = clockAnchored_; + clockAnchored_ = false; + } + } + + if (stopping || !running) { + // A pause stops the clock rather than resetting it: the anchor is + // re-derived from `emittedFrames_` on resume, so what follows keeps + // the position it would have had -- once the cushion is written up to + // the pause, which is where that position is. A stop writes it up to + // now, unless it lands on a pause this loop has not handled yet: the + // paused span is out of the video, so it stays out of the audio. + const bool written = + !flush || emitUntil(framesAt(running ? std::chrono::steady_clock::now() : pausedAt)); + { + std::scoped_lock lock(mutex_); + pauseFlushed_ = true; + } + cv_.notify_all(); + if (!written || stopping) { + break; + } + continue; + } + + const uint64_t realFrames = framesAt(std::chrono::steady_clock::now()); + if (!emitUntil(realFrames > cushionFrames ? realFrames - cushionFrames : 0)) { break; } - std::this_thread::sleep_until(audioClockStart + framesToDuration(emittedFrames_ + chunkFrames)); + // Woken early by a pause or a stop, so that one writes the cushion out at once. + std::unique_lock lock(mutex_); + cv_.wait_until( + lock, clockStart + framesToDuration(emittedFrames_ + chunkFrames + cushionFrames), + [&] { return stopRequested_.load() || paused_; }); } + // Whatever ended the loop, a resume waiting on the pause flush must not wait + // for a loop that is gone. Set under the lock, so the wakeup cannot slip + // between the waiter's check and its sleep. + { + std::scoped_lock lock(mutex_); + pauseFlushed_ = true; + } + cv_.notify_all(); } diff --git a/electron/native/wgc-capture/src/audio_sample_utils.h b/electron/native/wgc-capture/src/audio_sample_utils.h index f869d5e94..81d99d021 100644 --- a/electron/native/wgc-capture/src/audio_sample_utils.h +++ b/electron/native/wgc-capture/src/audio_sample_utils.h @@ -5,6 +5,7 @@ #include #include +#include #include #include #include @@ -103,12 +104,14 @@ class AudioMixer { private: void append( std::vector& queue, + bool& starved, const BYTE* data, DWORD byteCount, const AudioInputFormat& sourceFormat, double gain, AudioDecimatorState& decimator); - bool pop(std::vector& queue, std::vector& chunk, size_t byteCount); + bool pop(std::vector& queue, bool& starved, std::vector& chunk, size_t byteCount); + void resetSources(); void mixLoop(); AudioInputFormat format_{}; @@ -124,10 +127,23 @@ class AudioMixer { std::vector microphoneQueue_; AudioDecimatorState systemDecimator_; AudioDecimatorState microphoneDecimator_; + // Set when pop() runs a source dry, cleared by its next packet (see append). + bool systemStarved_ = false; + bool microphoneStarved_ = false; std::vector gainBuffer_; std::thread thread_; std::atomic stopRequested_ = false; bool timelineStarted_ = false; bool paused_ = false; + std::chrono::steady_clock::time_point pausedAt_{}; + // Set by mixLoop once it has written the cushion up to `pausedAt_`; a resume + // waits for it, or it would clear audio that belongs before the pause. + bool pauseFlushed_ = true; + // mixLoop's clock, shared so a source coming back after running dry can be + // placed at real time (see append). Written by mixLoop under `mutex_`. + std::chrono::steady_clock::time_point clockStart_{}; + bool clockAnchored_ = false; + // Frames mixLoop has taken from the queues so far, under `mutex_`. + uint64_t mixedFrames_ = 0; uint64_t emittedFrames_ = 0; }; diff --git a/electron/native/wgc-capture/src/audio_sample_utils_test.cpp b/electron/native/wgc-capture/src/audio_sample_utils_test.cpp index b28f738c8..ea235d1e9 100644 --- a/electron/native/wgc-capture/src/audio_sample_utils_test.cpp +++ b/electron/native/wgc-capture/src/audio_sample_utils_test.cpp @@ -1446,6 +1446,167 @@ int main() { } } + // --- Capture jitter: the mixer's cushion (getopenscreen/openscreen#911) --- + // + // The capture threads poll WASAPI every 5 ms -- 15.6 ms at the default timer + // resolution -- and push whatever packets have piled up, while the mixer emits + // on its own clock. With no margin between the two, every packet that arrived + // just after the mixer's tick was zero-filled: holes of up to 10 ms in the + // middle of the voice, heard as a crackle in the recording itself. `capture` + // below is that loop without WASAPI: a 10 ms packet becomes available every + // 10 ms from `from`, and a poller that sleeps 5 ms pushes what has piled up. + { + using std::chrono::milliseconds; + using Clock = std::chrono::steady_clock; + const AudioInputFormat f32 = makeFormat(MFAudioFormat_Float, 48000, 2, 32); + const auto dcPacket = [&](float value) { + std::vector bytes(480 * f32.blockAlign, 0); + auto* samples = reinterpret_cast(bytes.data()); + for (size_t i = 0; i < 480 * 2; i += 1) { + samples[i] = value; + } + return bytes; + }; + const auto capture = [](const auto& push, Clock::time_point from, int packets, + int stallAtPacket, int stallMs) { + int pushed = 0; + bool stalled = false; + while (pushed < packets) { + const auto now = Clock::now(); + const int available = now < from + ? 0 + : std::min(packets, static_cast((now - from) / milliseconds(10))); + for (; pushed < available; pushed += 1) { + push(); + } + if (!stalled && pushed >= stallAtPacket) { + stalled = true; + std::this_thread::sleep_for(milliseconds(stallMs)); + } + std::this_thread::sleep_for(milliseconds(5)); + } + }; + // Left channel of everything the mixer writes, from beginTimeline to stop. + const auto runTake = [&](bool includeSystem, bool includeMic, const auto& drive) { + std::mutex guard; + std::vector left; + AudioMixer mixer( + target48k, f32, f32, includeSystem, includeMic, 1.0, + [&](const BYTE* data, DWORD byteCount, int64_t, int64_t) { + std::scoped_lock lock(guard); + const auto* samples = reinterpret_cast(data); + for (size_t i = 0; i < byteCount / target48k.blockAlign; i += 1) { + left.push_back(samples[i * 2]); + } + return true; + }); + expect("jitter-mixer-start", mixer.start(), ""); + mixer.beginTimeline(); + drive(mixer, Clock::now()); + mixer.stop(); + std::scoped_lock lock(guard); + return left; + }; + const auto msAt = [](size_t frame) { return static_cast(frame) * 1000.0 / 48000.0; }; + + // (1) A microphone streaming a steady DC through ordinary poll jitter and one + // 40 ms stall: every sample between its first and its last must be voiced. + { + const auto packet = dcPacket(0.5f); + const auto left = runTake(false, true, [&](AudioMixer& mixer, Clock::time_point t0) { + capture( + [&] { mixer.pushMicrophone(packet.data(), static_cast(packet.size())); }, + t0, 120, 50, 40); + }); + const auto voiced = [](int16_t s) { return s != 0; }; + const auto first = std::find_if(left.begin(), left.end(), voiced); + const auto last = std::find_if(left.rbegin(), left.rend(), voiced).base(); + const size_t span = first < last ? static_cast(last - first) : 0; + const size_t holes = first < last ? static_cast(std::count(first, last, int16_t{0})) : 0; + char detail[128]{}; + sprintf_s(detail, "span=%zu holes=%zu of 57600 pushed", span, holes); + std::cout << "JITTER_RAW streaming " << detail << std::endl; + expect("mixer-streaming-source-has-no-holes", span >= 110 * 480 && holes == 0, detail); + } + + // (2) The cushion must not move a source that starts after silence -- loopback + // delivers nothing while nothing plays. Played from 400 ms, it lands there. + { + const auto packet = dcPacket(0.5f); + const auto left = runTake(true, false, [&](AudioMixer& mixer, Clock::time_point t0) { + capture( + [&] { mixer.pushSystem(packet.data(), static_cast(packet.size())); }, + t0 + milliseconds(400), 30, 1000, 0); + }); + const auto first = std::find_if(left.begin(), left.end(), [](int16_t s) { return s != 0; }); + const double at = msAt(static_cast(first - left.begin())); + char detail[96]{}; + sprintf_s(detail, "first sound at %.1f ms, played at 400 ms", at); + std::cout << "JITTER_RAW late-source " << detail << std::endl; + expect("mixer-late-source-lands-when-it-played", first != left.end() && std::abs(at - 400.0) <= 40.0, detail); + } + + // (3) Nor across a pause: what the cushion still holds at the pause is written + // before it, so what follows the resume starts where the pause began. + { + const auto beforePause = dcPacket(0.5f); + const auto afterPause = dcPacket(-0.5f); + double pausedAtMs = 0.0; + const auto left = runTake(false, true, [&](AudioMixer& mixer, Clock::time_point t0) { + capture( + [&] { mixer.pushMicrophone(beforePause.data(), static_cast(beforePause.size())); }, + t0, 40, 1000, 0); + pausedAtMs = std::chrono::duration(Clock::now() - t0).count(); + mixer.setPaused(true); + std::this_thread::sleep_for(milliseconds(200)); + mixer.setPaused(false); + capture( + [&] { mixer.pushMicrophone(afterPause.data(), static_cast(afterPause.size())); }, + Clock::now(), 30, 1000, 0); + }); + const auto first = std::find_if(left.begin(), left.end(), [](int16_t s) { return s < 0; }); + const double at = msAt(static_cast(first - left.begin())); + char detail[96]{}; + sprintf_s(detail, "resumed sound at %.1f ms, paused at %.1f ms", at, pausedAtMs); + std::cout << "JITTER_RAW pause " << detail << std::endl; + expect("mixer-resume-continues-at-the-pause", first != left.end() && std::abs(at - pausedAtMs) <= 40.0, detail); + } + + // (4) The same with a resume that follows the pause at once, before the + // mixer has had a chance to see the pause: the resume must wait for the + // cushion to be written, or it throws the cushion away and what follows + // lands a cushion early. + { + const auto beforePause = dcPacket(0.5f); + const auto afterPause = dcPacket(-0.5f); + double pausedAtMs = 0.0; + const auto left = runTake(false, true, [&](AudioMixer& mixer, Clock::time_point t0) { + capture( + [&] { mixer.pushMicrophone(beforePause.data(), static_cast(beforePause.size())); }, + t0, 40, 1000, 0); + pausedAtMs = std::chrono::duration(Clock::now() - t0).count(); + mixer.setPaused(true); + mixer.setPaused(false); + capture( + [&] { mixer.pushMicrophone(afterPause.data(), static_cast(afterPause.size())); }, + Clock::now(), 30, 1000, 0); + }); + const auto first = std::find_if(left.begin(), left.end(), [](int16_t s) { return s < 0; }); + const double at = msAt(static_cast(first - left.begin())); + // 40 packets were pushed before the pause: 400 ms of voice, all of it + // due before the pause, none of it to be thrown away with the cushion. + const auto lastBefore = std::find_if(left.rbegin(), left.rend(), [](int16_t s) { return s > 0; }); + const double voiceEnd = msAt(static_cast(left.rend() - lastBefore)); + char detail[128]{}; + sprintf_s( + detail, "voice before the pause ends at %.1f ms of 400, resumed at %.1f ms, paused at %.1f ms", + voiceEnd, at, pausedAtMs); + std::cout << "JITTER_RAW instant-resume " << detail << std::endl; + expect("mixer-instant-resume-keeps-the-voice-before-it", voiceEnd >= 390.0, detail); + expect("mixer-instant-resume-continues-at-the-pause", first != left.end() && std::abs(at - pausedAtMs) <= 40.0, detail); + } + } + // --- mixAudioInPlace, per output format ----------------------------------- // // The gain rides in mixAudioInPlace on every format branch, but the diff --git a/technical-documentation/architecture/recording.md b/technical-documentation/architecture/recording.md index 3201817c1..35bafa57d 100644 --- a/technical-documentation/architecture/recording.md +++ b/technical-documentation/architecture/recording.md @@ -70,7 +70,7 @@ Windows and macOS both write their screen video as a fragmented MP4 — `MFCreat A session writes a screen video and a `.session.json` manifest. Windows normally muxes the webcam into that MP4; when `webcamPath` is supplied, it writes a separate webcam video. macOS currently writes the webcam as a separate Electron sidecar (`webcamVideoPath`) because native webcam composition is not part of the helper. Linux follows the Electron recorder's separate media-path convention. Audio that the selected backend captures is encoded into its screen output. -The Windows helper mixes system loopback and microphone into one track, and timestamps it from a running count of emitted frames. That count is advanced by a clock rather than by the arrival of samples: a chunk goes out every 10 ms for as long as the recording runs, filled from whichever source has data and with silence where neither does. Advancing it only when a queue held samples is what made a take that began in silence emit nothing at all — WASAPI loopback delivers no packets while nothing is playing — so the first sound landed at timestamp zero and the track came out shorter than the take. A working microphone concealed it by streaming continuously, which is why it appeared as a system-audio desync on a machine whose microphone had failed. `npm run test:wgc-audio-timeline:win` measures where a tone played at a known instant actually lands. +The Windows helper mixes system loopback and microphone into one track, and timestamps it from a running count of emitted frames. That count is advanced by a clock rather than by the arrival of samples: a chunk goes out every 10 ms for as long as the recording runs, filled from whichever source has data and with silence where neither does. Advancing it only when a queue held samples is what made a take that began in silence emit nothing at all — WASAPI loopback delivers no packets while nothing is playing — so the first sound landed at timestamp zero and the track came out shorter than the take. A working microphone concealed it by streaming continuously, which is why it appeared as a system-audio desync on a machine whose microphone had failed. `npm run test:wgc-audio-timeline:win` measures where a tone played at a known instant actually lands. The mixer writes 100 ms behind real time, and that cushion is what keeps a continuous voice whole: the capture threads poll WASAPI and hand packets over late (15.6 ms at the default timer resolution, more on a busy machine), and mixing at real time zero-filled every chunk whose packet had not arrived yet. The result was holes of up to 10 ms inside words, heard as crackle (#911). The timestamps still come from the frame count, so the cushion delays the writing, not the audio. A pause or a stop writes what it still holds up to that instant, and a resume waits for that before clearing the queues. A source that ran dry, such as loopback after a silence, is queued behind silence as long as the mixer's real lag, so it still lands when it played. The clock is anchored when the timeline begins and at each resume, not when the mixer thread next wakes. Before the keep-alive below, a mic-only recording did nothing to the render (output) endpoint: system-audio capture was the only path that even opened it, and it reads rather than writes — which reads as idle to some wireless headsets' own firmware idle timers, and they power the set down mid-take (#724: a Corsair Void dropped reliably seven to ten minutes in, as a full disconnect the OS never observes; the USB dongle stays enumerated and the WASAPI endpoints stay ACTIVE throughout, so `IMMNotificationClient` sees nothing, and neither do the device's PnP properties). The helper therefore keeps the endpoint fed: when system audio is not being captured, it opens a second, ordinary shared-mode render stream on the default output device and writes a 19 kHz tone at 0.3% amplitude for as long as the take runs — above what the large majority of adults can hear, and real signal rather than packets flagged `AUDCLNT_BUFFERFLAGS_SILENT`, because the same hardware testing that found the timer also found that digital silence does not hold it off while an actual waveform does. The gate is not an optimization: loopback already fills the endpoint with real content when it runs, and anything this stream wrote beside it would be mixed straight into the recording's own system-audio track. It is a workaround, not a fix — the timer lives in the headset's firmware, below everything an application can observe or configure, and the per-vendor setting (iCUE for Corsair, equivalents elsewhere) is the only way to turn it off; "inaudible" is likewise per listener, since hearing range and driver harmonics vary. `OPENSCREEN_WGC_DISABLE_AUDIO_KEEPALIVE=1` turns the stream off. `OPENSCREEN_WGC_LOG_AUDIO_DEVICE_EVENTS=1` runs a diagnostic device watcher alongside the take, logging endpoint state transitions as JSON events to stderr — with the expectation, learned from #724, that a real firmware drop logs nothing at all, for exactly the reason above. diff --git a/technical-documentation/testing/manual-e2e-checklist.md b/technical-documentation/testing/manual-e2e-checklist.md index 9432d654e..3017b8152 100644 --- a/technical-documentation/testing/manual-e2e-checklist.md +++ b/technical-documentation/testing/manual-e2e-checklist.md @@ -734,6 +734,7 @@ Edits are saved as they land. The dot after the project name reads "Unsaved" (ho - [ ] **post-1.10.0** — Record with no encoder override and confirm the helper's `encoder-selection` log line reports `videoEncoderRuntime: "hardware"`. The plain sink-writer path never asked for hardware transforms before, so every ordinary recording ran the software encoder; on a slow machine that is what blew the stop-shutdown budget. - [ ] **post-1.10.0** — Confirm forcing the software encoder still reports `"software"`, so the default is a default and not a hard-wire. - [ ] **post-1.10.0** — Record with microphone and system audio and confirm the resulting MP4 carries a valid AAC track at a legal rate (48 kHz). +- [ ] Speak continuously for 30 s with the microphone on, then play the file in a neutral player (VLC, ffplay) and confirm there is no crackle. Include the moment you reach for the HUD to stop: that is where the holes of #911 clustered. In a waveform view, the defect looks like drops to digital silence of up to 10 ms in the middle of words. - [ ] **post-1.10.0** — On a device whose native rate AAC cannot take (96 kHz), confirm the recording still succeeds with the rate snapped to 48 kHz rather than failing at `SetInputMediaType`. The helper's own `audio_sample_utils_test` covers the accept/reject probes at build time; this check is the end-to-end half. - [ ] **post-1.10.0** — Confirm a long recording's audio stays in sync, so the downsample remainder is carried across packets rather than drifting. - [ ] **post-1.10.0** — On a device that can be set to 96 kHz, record system audio while a 36 kHz tone plays and confirm the recording carries **no** 12 kHz component. That fold is what an inadequate anti-alias filter produces, and neither of the two checks above would catch it: the rate-snap check only asks that the recording succeeds, and the sync check only asks that frame counts stay aligned. Probe tones must sit well inside what AAC keeps — 12 kHz is fine; a 20 kHz probe was absent from the app's recording, and a 192 kbps AAC encode alone (tested with ffmpeg) removes it too, so it cannot be measured. Needs an endpoint whose **shared-mode** format is above 48 kHz and an integer multiple of it — check the Advanced tab's format list, and check every endpoint, not just the current format of the default one; a USB DAC is one way to get such an endpoint. Exclusive-mode support is not enough, because loopback reports the shared-mode format. If every endpoint really is 48 kHz, the check cannot run at all: forcing a lower encoder target instead does not work, since every AAC rate that would divide 48 kHz is rejected by the Media Foundation encoder on the host tested.