From 8417212ed2e6d40b57cf55e7df512a6af98f1d4f Mon Sep 17 00:00:00 2001 From: baketnk Date: Thu, 24 Sep 2026 14:04:31 -0400 Subject: [PATCH] Keep capture stream running between PTT utterances --- CMakeLists.txt | 4 +++ README.md | 4 ++- docs/poc.md | 13 +++++--- src/audio.cpp | 63 ++++++++++++++++++++++++-------------- src/audio.hpp | 12 +++++--- src/runtime.cpp | 25 +++++++++------ tests/audio_test.cpp | 64 +++++++++++++++++++++++++++++++++++++++ tests/fake_sdl/SDL3/SDL.h | 22 ++++++++++++++ 8 files changed, 165 insertions(+), 42 deletions(-) create mode 100644 tests/audio_test.cpp create mode 100644 tests/fake_sdl/SDL3/SDL.h diff --git a/CMakeLists.txt b/CMakeLists.txt index 63ab9b6..5dbb5ee 100644 --- a/CMakeLists.txt +++ b/CMakeLists.txt @@ -100,6 +100,10 @@ if(BUILD_TESTING AND NOT CMAKE_CROSSCOMPILING) "-DEXPECTED_VERSION=${FRAMEYAP_VERSION}" -P "${CMAKE_CURRENT_SOURCE_DIR}/tests/cli.cmake") add_test(NAME frameyap.install_preflight COMMAND sh "${CMAKE_CURRENT_SOURCE_DIR}/tests/install-preflight.sh" "${CMAKE_CURRENT_SOURCE_DIR}/scripts/install-preflight.sh") + add_executable(frameyap_audio_test tests/audio_test.cpp src/audio.cpp) + target_include_directories(frameyap_audio_test PRIVATE src tests/fake_sdl) + target_compile_options(frameyap_audio_test PRIVATE -UNDEBUG) + add_test(NAME frameyap.audio_lifecycle COMMAND frameyap_audio_test) add_executable(frameyap_core_test tests/core_test.cpp) target_link_libraries(frameyap_core_test PRIVATE frameyap_core) add_test(NAME frameyap.core COMMAND frameyap_core_test) diff --git a/README.md b/README.md index dfb1e8c..d890644 100644 --- a/README.md +++ b/README.md @@ -36,7 +36,9 @@ See [third-party notes](docs/third-party.md). No GitHub release is published yet backs it up before repairs. See [overlay configuration](docs/overlay.md#user-theme-and-controller-configuration). - **Review-first:** focus your destination, then press Insert. No automatic insertion or submission. Maximum clip 20 seconds; accidental taps under 200 ms are discarded. -- Other apps may still hear/transmit your voice. FrameYap does not mute them. +- While the native app is Ready, it keeps the mic device open and discards idle + audio instead of opening/closing on every PTT. Quit releases the device. Other + apps may still hear/transmit your voice; FrameYap does not mute them. Physical gesture timing, global bindings during games, mic capture, target-app compatibility and headset comfort still need coordinated validation. Successful diff --git a/docs/poc.md b/docs/poc.md index 300d7a5..17c9e3e 100644 --- a/docs/poc.md +++ b/docs/poc.md @@ -18,14 +18,19 @@ initialize OpenVR, open a microphone, run ASR, download files or inject input. after first release. Activity/tracking loss cancels a held recording and requires neutral rearm. - SDL3 default recording device, mono float32 conversion at 16 kHz, 200 ms minimum, - 20 second maximum. Microphone is closed outside actual capture. Other apps may - still transmit your voice: this app does **not** mute VRChat or any other app. + 20 second maximum. In the native app the microphone stream opens after model + warm-up, stays running between utterances, and discards idle samples; PTT does + not open/pause/close the device. Quit, worker restart, device failure or capture + failure closes it. This avoids repeated capture-device transitions but does not + promise glitch-free playback on every audio stack. Other apps may still + transmit your voice: this app does **not** mute VRChat or any other app. - Persistent local Redux worker, correlated bounded pipes, private tmpfs clips, cancellation/reaping and deadlines; exact pinned model SHA-256 verification. Model imports are lazy and loading is offline. Normal repeats, request-local transcription failures and microphone failures retain the loaded model; - microphone device streams close outside capture. A cancelled in-flight request - or broken worker protocol may require reloading. No cloud/desktop fallback. + microphone device failure releases the stream for explicit retry. A cancelled + in-flight request or broken worker protocol may require reloading. No + cloud/desktop fallback. - Gamescope IME v2 generated bindings, per-action short-lived lease, unavailable handling, UTF-8/control validation and explicit separate Submit action for Enter. - Idempotent user-local release-archive installer: SHA-256, safe extraction, diff --git a/src/audio.cpp b/src/audio.cpp index 4d9e9a8..bdf43e8 100644 --- a/src/audio.cpp +++ b/src/audio.cpp @@ -9,37 +9,48 @@ namespace frameyap { namespace { void check(bool result) { if (!result) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError()); } } Audio::~Audio() { - cancel(); + close(); if (initialized_) SDL_QuitSubSystem(SDL_INIT_AUDIO); } -void Audio::start() { - if (stream_) throw std::runtime_error("Already recording"); - // Keep SDL's audio backend initialized across utterances, but own the - // recording device only while actually capturing. Reinitializing PipeWire - // after every clip can fail even though the local model remains healthy. +void Audio::prepare() { + if (stream_) return; if (!initialized_) { check(SDL_InitSubSystem(SDL_INIT_AUDIO)); initialized_ = true; } SDL_AudioSpec spec{SDL_AUDIO_F32, 1, 16000}; stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_RECORDING, &spec, nullptr, nullptr); if (!stream_) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError()); - pcm_.clear(); pcm_.reserve(320000); - started_ = last_data_ = std::chrono::steady_clock::now(); try { check(SDL_ResumeAudioStreamDevice(stream_)); } - catch (...) { cancel(); throw; } + catch (...) { close(); throw; } + last_data_ = std::chrono::steady_clock::now(); +} +void Audio::start() { + if (recording_) throw std::runtime_error("Already recording"); + if (!stream_) throw std::runtime_error("Microphone unavailable; retry capture"); + // Clear queued idle data at the PTT boundary; the device never pauses or + // reopens. Polling also discards idle samples so the queue stays small. + check(SDL_ClearAudioStream(stream_)); + std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear(); pcm_.reserve(320000); + started_ = last_data_ = std::chrono::steady_clock::now(); + recording_ = true; } void Audio::drain() { std::array chunk{}; - while (pcm_.size() < 320000) { + // During idle and after the 20s bound, consume and discard device samples. + // Never let idle audio accumulate for the next utterance. + while (true) { int available = SDL_GetAudioStreamAvailable(stream_); check(available >= 0); if (available < int(sizeof(float))) break; - const int count = static_cast(std::min({chunk.size(), std::size_t(available) / sizeof(float), 320000 - pcm_.size()})); + const int count = static_cast(std::min(chunk.size(), std::size_t(available) / sizeof(float))); int bytes = SDL_GetAudioStreamData(stream_, chunk.data(), count * sizeof(float)); check(bytes >= 0); if (bytes == 0) break; if (bytes % sizeof(float)) throw std::runtime_error("Misaligned microphone samples"); - for (int i = 0; i < bytes / int(sizeof(float)); ++i) - if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples"); - pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + bytes / sizeof(float)); + if (recording_) { + const auto retain = std::min(std::size_t(bytes) / sizeof(float), 320000 - pcm_.size()); + for (std::size_t i = 0; i < retain; ++i) + if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples"); + pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + retain); + } last_data_ = std::chrono::steady_clock::now(); } } @@ -47,32 +58,38 @@ bool Audio::poll() { if (!stream_) return false; SDL_Event event; while (SDL_PollEvent(&event)) { - if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording) + if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording && + event.adevice.which == SDL_GetAudioStreamDevice(stream_)) throw std::runtime_error("Recording device removed; clip discarded"); } drain(); if (std::chrono::steady_clock::now() - last_data_ > std::chrono::seconds(2)) throw std::runtime_error("Microphone stalled; clip discarded"); - return pcm_.size() == 320000 || seconds() >= 20; + return recording_ && (pcm_.size() == 320000 || seconds() >= 20); } std::vector Audio::finish() { - if (!stream_) return {}; + if (!recording_) return {}; try { - check(SDL_PauseAudioStreamDevice(stream_)); - check(SDL_FlushAudioStream(stream_)); drain(); + recording_ = false; auto result = std::move(pcm_); - cancel(); + pcm_.clear(); + // Discard any resampler remainder and never pause the capture device. + check(SDL_ClearAudioStream(stream_)); return result; } catch (...) { cancel(); throw; } } void Audio::cancel() { + recording_ = false; + std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear(); + if (stream_) SDL_ClearAudioStream(stream_); // best effort on cancellation +} +void Audio::close() { + cancel(); if (stream_) SDL_DestroyAudioStream(stream_); stream_ = nullptr; - // Best effort clearing of the owned buffer; no persistent recording/logging. - std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear(); } int Audio::seconds() const { - return stream_ ? int(std::chrono::duration_cast(std::chrono::steady_clock::now() - started_).count()) : 0; + return recording_ ? int(std::chrono::duration_cast(std::chrono::steady_clock::now() - started_).count()) : 0; } } diff --git a/src/audio.hpp b/src/audio.hpp index c80453b..99288e8 100644 --- a/src/audio.hpp +++ b/src/audio.hpp @@ -7,16 +7,20 @@ class Audio { public: Audio() = default; ~Audio(); - void start(); - bool poll(); // true at 20s limit; throws if capture was lost/stalled + void prepare(); // open/resume once, while app is ready; never retain idle samples + void start(); // arm the already-running stream without touching the device + bool poll(); // drain even while idle; true at 20s limit; throws on loss/stall std::vector finish(); - void cancel(); - bool recording() const { return stream_ != nullptr; } + void cancel(); // discard current clip, leave the device running + void close(); // release device on shutdown or capture failure + bool recording() const { return recording_; } + bool open() const { return stream_ != nullptr; } int seconds() const; private: void drain(); SDL_AudioStream* stream_ = nullptr; bool initialized_ = false; + bool recording_ = false; std::vector pcm_; std::chrono::steady_clock::time_point started_, last_data_; }; diff --git a/src/runtime.cpp b/src/runtime.cpp index a68337a..1331d54 100644 --- a/src/runtime.cpp +++ b/src/runtime.cpp @@ -47,7 +47,7 @@ int run(const Options& options) { std::string detail = "Review mode. Other apps may also hear your mic. Enter is separate."; bool quit = false; auto warm = [&] { - worker.stop(); session = Session{}; + audio.close(); worker.stop(); session = Session{}; worker.start(options.python, options.worker, options.model, options.threads); detail = "Loading local model; microphone closed. Record again when Ready."; }; @@ -63,6 +63,7 @@ int run(const Options& options) { auto start_record = [&] { if (session.state() == State::Review || session.state() == State::Transcribing || session.state() == State::Warming) return; if (!worker.ready()) { warm(); return; } + if (!audio.open()) audio.prepare(); // recovery only; normal PTT never opens the device if (session.state() == State::Error) session.cancel(); if (!session.record()) return; audio.start(); @@ -86,16 +87,19 @@ int run(const Options& options) { } if (worker.ready()) session.ready(); } catch (const std::exception& e) { - audio.cancel(); worker.stop(); + audio.close(); worker.stop(); if (session.state() != State::Review && session.state() != State::Queued) session.fail(); detail = e.what(); // Preserve an already-correlated preview if the worker dies. } try { - if (audio.recording() && audio.poll()) stop_record(); + // Prepare after model warm-up, not on PTT. Keep draining/discarding + // idle samples so neither a device transition nor old speech reaches + // the next clip. A capture failure still requires an explicit retry. + if (session.state() == State::Ready && !audio.open() && worker.ready()) audio.prepare(); + if (audio.open() && audio.poll()) stop_record(); } catch (const std::exception& e) { - // A microphone failure is not a model failure. Release the device, - // retain the loaded worker, and allow Record to retry capture. - audio.cancel(); session.fail(); detail = e.what(); + // A microphone failure is not a model failure. + audio.close(); session.fail(); detail = e.what(); } // Drawing precedes input polling: Enter is disabled until Ready is visible. const auto status = session.state() == State::Error && worker.ready() @@ -118,10 +122,11 @@ int run(const Options& options) { case UiAction::EndRecord: stop_record(); break; case UiAction::Cancel: { auto state = session.state(); - audio.cancel(); session.cancel(); detail = "Discarded. Microphone closed."; + audio.cancel(); session.cancel(); detail = "Discarded. Idle microphone samples are discarded."; if (state == State::Warming || state == State::Transcribing) { - worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker."; + audio.close(); worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker."; } else if (state == State::Error && !worker.ready()) { + audio.close(); session.fail(); detail = "Worker unavailable. Record to reload local worker."; } break; @@ -150,7 +155,7 @@ int run(const Options& options) { // only the microphone; submit failures stop their own child if // the IPC stream was partially written. if (session.state() == State::Recording || session.state() == State::Transcribing) { - audio.cancel(); session.fail(); + audio.close(); session.fail(); } detail = e.what(); } @@ -158,7 +163,7 @@ int run(const Options& options) { } std::this_thread::sleep_for(std::chrono::milliseconds(10)); } - audio.cancel(); session.cancel(); worker.stop(); + audio.close(); session.cancel(); worker.stop(); return 0; } } diff --git a/tests/audio_test.cpp b/tests/audio_test.cpp new file mode 100644 index 0000000..a415107 --- /dev/null +++ b/tests/audio_test.cpp @@ -0,0 +1,64 @@ +#include "audio.hpp" +#include +#include +#include +#include +#include +#include + +namespace { +SDL_AudioStream* current = nullptr; +int opens = 0, resumes = 0, closes = 0; +bool removed = false; +void feed(float value, int count) { assert(current); current->queued.insert(current->queued.end(), count, value); } +} +const char* SDL_GetError() { return "fake failure"; } +bool SDL_InitSubSystem(int) { return true; } +void SDL_QuitSubSystem(int) {} +SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*) { + ++opens; return current = new SDL_AudioStream; +} +bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*) { ++resumes; return true; } +SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*) { return 42; } +int SDL_GetAudioStreamAvailable(SDL_AudioStream* stream) { return int(stream->queued.size() * sizeof(float)); } +int SDL_GetAudioStreamData(SDL_AudioStream* stream, void* data, int size) { + int count = std::min(int(stream->queued.size()), size / int(sizeof(float))); + std::memcpy(data, stream->queued.data(), count * sizeof(float)); + stream->queued.erase(stream->queued.begin(), stream->queued.begin() + count); + return count * sizeof(float); +} +bool SDL_ClearAudioStream(SDL_AudioStream* stream) { stream->queued.clear(); return true; } +void SDL_DestroyAudioStream(SDL_AudioStream* stream) { ++closes; current = nullptr; delete stream; } +bool SDL_PollEvent(SDL_Event* event) { + if (!removed) return false; + removed = false; *event = SDL_Event{SDL_EVENT_AUDIO_DEVICE_REMOVED, {true, 42}}; return true; +} +int main() { + using frameyap::Audio; + { + Audio audio; + audio.prepare(); audio.prepare(); + assert(opens == 1 && resumes == 1 && audio.open() && !audio.recording()); + feed(0.7f, 10000); audio.poll(); + assert(current->queued.empty()); + feed(0.8f, 200); audio.start(); // queued idle speech cannot leak + feed(0.1f, 3200); audio.poll(); + feed(0.2f, 3200); + auto clip = audio.finish(); + assert(clip.size() == 6400 && clip.front() == 0.1f && clip.back() == 0.2f); + assert(audio.open() && !audio.recording() && opens == 1 && closes == 0 && resumes == 1); + feed(0.5f, 200); audio.start(); feed(0.3f, 4000); + audio.cancel(); assert(current->queued.empty()); + feed(0.9f, 100); audio.start(); feed(0.4f, 3200); + clip = audio.finish(); assert(clip.size() == 3200 && clip.front() == 0.4f); + assert(opens == 1 && closes == 0 && resumes == 1); + removed = true; + bool failed = false; + try { audio.poll(); } catch (const std::runtime_error&) { failed = true; } + assert(failed); + audio.close(); assert(closes == 1 && !audio.open()); + audio.prepare(); assert(opens == 2 && resumes == 2); + } + assert(closes == 2); + std::cout << "audio lifecycle checks passed (fake device only)\n"; +} diff --git a/tests/fake_sdl/SDL3/SDL.h b/tests/fake_sdl/SDL3/SDL.h new file mode 100644 index 0000000..682c4a3 --- /dev/null +++ b/tests/fake_sdl/SDL3/SDL.h @@ -0,0 +1,22 @@ +#pragma once +#include +#include +#define SDL_INIT_AUDIO 1 +#define SDL_AUDIO_F32 2 +#define SDL_AUDIO_DEVICE_DEFAULT_RECORDING 3 +#define SDL_EVENT_AUDIO_DEVICE_REMOVED 4 +using SDL_AudioDeviceID = unsigned; +struct SDL_AudioSpec { int format, channels, freq; }; +struct SDL_AudioStream { std::vector queued; }; +struct SDL_Event { int type; struct { bool recording; SDL_AudioDeviceID which; } adevice; }; +const char* SDL_GetError(); +bool SDL_InitSubSystem(int); +void SDL_QuitSubSystem(int); +SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*); +bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*); +SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*); +int SDL_GetAudioStreamAvailable(SDL_AudioStream*); +int SDL_GetAudioStreamData(SDL_AudioStream*, void*, int); +bool SDL_ClearAudioStream(SDL_AudioStream*); +void SDL_DestroyAudioStream(SDL_AudioStream*); +bool SDL_PollEvent(SDL_Event*);