Keep capture stream running between PTT utterances

This commit is contained in:
baketnk committed 2026-09-24 14:04:31 -04:00
1 parent a9d6762ebd
commit 8417212ed2
8 files changed
+164 -41

No files matched your search

+4
View File
@@ -100,6 +100,10 @@ if(BUILD_TESTING AND NOT CMAKE_CROSSCOMPILING)
"-DEXPECTED_VERSION=${FRAMEYAP_VERSION}" -P "${CMAKE_CURRENT_SOURCE_DIR}/tests/cli.cmake") "-DEXPECTED_VERSION=${FRAMEYAP_VERSION}" -P "${CMAKE_CURRENT_SOURCE_DIR}/tests/cli.cmake")
add_test(NAME frameyap.install_preflight COMMAND sh "${CMAKE_CURRENT_SOURCE_DIR}/tests/install-preflight.sh" add_test(NAME frameyap.install_preflight COMMAND sh "${CMAKE_CURRENT_SOURCE_DIR}/tests/install-preflight.sh"
"${CMAKE_CURRENT_SOURCE_DIR}/scripts/install-preflight.sh") "${CMAKE_CURRENT_SOURCE_DIR}/scripts/install-preflight.sh")
add_executable(frameyap_audio_test tests/audio_test.cpp src/audio.cpp)
target_include_directories(frameyap_audio_test PRIVATE src tests/fake_sdl)
target_compile_options(frameyap_audio_test PRIVATE -UNDEBUG)
add_test(NAME frameyap.audio_lifecycle COMMAND frameyap_audio_test)
add_executable(frameyap_core_test tests/core_test.cpp) add_executable(frameyap_core_test tests/core_test.cpp)
target_link_libraries(frameyap_core_test PRIVATE frameyap_core) target_link_libraries(frameyap_core_test PRIVATE frameyap_core)
add_test(NAME frameyap.core COMMAND frameyap_core_test) add_test(NAME frameyap.core COMMAND frameyap_core_test)
+3 -1
View File
@@ -36,7 +36,9 @@ See [third-party notes](docs/third-party.md). No GitHub release is published yet
backs it up before repairs. See [overlay configuration](docs/overlay.md#user-theme-and-controller-configuration). backs it up before repairs. See [overlay configuration](docs/overlay.md#user-theme-and-controller-configuration).
- **Review-first:** focus your destination, then press Insert. No automatic insertion - **Review-first:** focus your destination, then press Insert. No automatic insertion
or submission. Maximum clip 20 seconds; accidental taps under 200 ms are discarded. or submission. Maximum clip 20 seconds; accidental taps under 200 ms are discarded.
- Other apps may still hear/transmit your voice. FrameYap does not mute them. - While the native app is Ready, it keeps the mic device open and discards idle
audio instead of opening/closing on every PTT. Quit releases the device. Other
apps may still hear/transmit your voice; FrameYap does not mute them.
Physical gesture timing, global bindings during games, mic capture, target-app Physical gesture timing, global bindings during games, mic capture, target-app
compatibility and headset comfort still need coordinated validation. Successful compatibility and headset comfort still need coordinated validation. Successful
+9 -4
View File
@@ -18,14 +18,19 @@ initialize OpenVR, open a microphone, run ASR, download files or inject input.
after first release. after first release.
Activity/tracking loss cancels a held recording and requires neutral rearm. Activity/tracking loss cancels a held recording and requires neutral rearm.
- SDL3 default recording device, mono float32 conversion at 16 kHz, 200 ms minimum, - SDL3 default recording device, mono float32 conversion at 16 kHz, 200 ms minimum,
20 second maximum. Microphone is closed outside actual capture. Other apps may 20 second maximum. In the native app the microphone stream opens after model
still transmit your voice: this app does **not** mute VRChat or any other app. warm-up, stays running between utterances, and discards idle samples; PTT does
not open/pause/close the device. Quit, worker restart, device failure or capture
failure closes it. This avoids repeated capture-device transitions but does not
promise glitch-free playback on every audio stack. Other apps may still
transmit your voice: this app does **not** mute VRChat or any other app.
- Persistent local Redux worker, correlated bounded pipes, private tmpfs clips, - Persistent local Redux worker, correlated bounded pipes, private tmpfs clips,
cancellation/reaping and deadlines; exact pinned model SHA-256 verification. cancellation/reaping and deadlines; exact pinned model SHA-256 verification.
Model imports are lazy and loading is offline. Normal repeats, request-local Model imports are lazy and loading is offline. Normal repeats, request-local
transcription failures and microphone failures retain the loaded model; transcription failures and microphone failures retain the loaded model;
microphone device streams close outside capture. A cancelled in-flight request microphone device failure releases the stream for explicit retry. A cancelled
or broken worker protocol may require reloading. No cloud/desktop fallback. in-flight request or broken worker protocol may require reloading. No
cloud/desktop fallback.
- Gamescope IME v2 generated bindings, per-action short-lived lease, unavailable - Gamescope IME v2 generated bindings, per-action short-lived lease, unavailable
handling, UTF-8/control validation and explicit separate Submit action for Enter. handling, UTF-8/control validation and explicit separate Submit action for Enter.
- Idempotent user-local release-archive installer: SHA-256, safe extraction, - Idempotent user-local release-archive installer: SHA-256, safe extraction,
+39 -22
View File
@@ -9,37 +9,48 @@
namespace frameyap { namespace frameyap {
namespace { void check(bool result) { if (!result) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError()); } } namespace { void check(bool result) { if (!result) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError()); } }
Audio::~Audio() { Audio::~Audio() {
cancel(); close();
if (initialized_) SDL_QuitSubSystem(SDL_INIT_AUDIO); if (initialized_) SDL_QuitSubSystem(SDL_INIT_AUDIO);
} }
void Audio::start() { void Audio::prepare() {
if (stream_) throw std::runtime_error("Already recording"); if (stream_) return;
// Keep SDL's audio backend initialized across utterances, but own the
// recording device only while actually capturing. Reinitializing PipeWire
// after every clip can fail even though the local model remains healthy.
if (!initialized_) { check(SDL_InitSubSystem(SDL_INIT_AUDIO)); initialized_ = true; } if (!initialized_) { check(SDL_InitSubSystem(SDL_INIT_AUDIO)); initialized_ = true; }
SDL_AudioSpec spec{SDL_AUDIO_F32, 1, 16000}; SDL_AudioSpec spec{SDL_AUDIO_F32, 1, 16000};
stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_RECORDING, &spec, nullptr, nullptr); stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_RECORDING, &spec, nullptr, nullptr);
if (!stream_) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError()); if (!stream_) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError());
pcm_.clear(); pcm_.reserve(320000);
started_ = last_data_ = std::chrono::steady_clock::now();
try { check(SDL_ResumeAudioStreamDevice(stream_)); } try { check(SDL_ResumeAudioStreamDevice(stream_)); }
catch (...) { cancel(); throw; } catch (...) { close(); throw; }
last_data_ = std::chrono::steady_clock::now();
}
void Audio::start() {
if (recording_) throw std::runtime_error("Already recording");
if (!stream_) throw std::runtime_error("Microphone unavailable; retry capture");
// Clear queued idle data at the PTT boundary; the device never pauses or
// reopens. Polling also discards idle samples so the queue stays small.
check(SDL_ClearAudioStream(stream_));
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear(); pcm_.reserve(320000);
started_ = last_data_ = std::chrono::steady_clock::now();
recording_ = true;
} }
void Audio::drain() { void Audio::drain() {
std::array<float, 4096> chunk{}; std::array<float, 4096> chunk{};
while (pcm_.size() < 320000) { // During idle and after the 20s bound, consume and discard device samples.
// Never let idle audio accumulate for the next utterance.
while (true) {
int available = SDL_GetAudioStreamAvailable(stream_); int available = SDL_GetAudioStreamAvailable(stream_);
check(available >= 0); check(available >= 0);
if (available < int(sizeof(float))) break; if (available < int(sizeof(float))) break;
const int count = static_cast<int>(std::min({chunk.size(), std::size_t(available) / sizeof(float), 320000 - pcm_.size()})); const int count = static_cast<int>(std::min(chunk.size(), std::size_t(available) / sizeof(float)));
int bytes = SDL_GetAudioStreamData(stream_, chunk.data(), count * sizeof(float)); int bytes = SDL_GetAudioStreamData(stream_, chunk.data(), count * sizeof(float));
check(bytes >= 0); check(bytes >= 0);
if (bytes == 0) break; if (bytes == 0) break;
if (bytes % sizeof(float)) throw std::runtime_error("Misaligned microphone samples"); if (bytes % sizeof(float)) throw std::runtime_error("Misaligned microphone samples");
for (int i = 0; i < bytes / int(sizeof(float)); ++i) if (recording_) {
const auto retain = std::min(std::size_t(bytes) / sizeof(float), 320000 - pcm_.size());
for (std::size_t i = 0; i < retain; ++i)
if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples"); if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples");
pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + bytes / sizeof(float)); pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + retain);
}
last_data_ = std::chrono::steady_clock::now(); last_data_ = std::chrono::steady_clock::now();
} }
} }
@@ -47,32 +58,38 @@ bool Audio::poll() {
if (!stream_) return false; if (!stream_) return false;
SDL_Event event; SDL_Event event;
while (SDL_PollEvent(&event)) { while (SDL_PollEvent(&event)) {
if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording) if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording &&
event.adevice.which == SDL_GetAudioStreamDevice(stream_))
throw std::runtime_error("Recording device removed; clip discarded"); throw std::runtime_error("Recording device removed; clip discarded");
} }
drain(); drain();
if (std::chrono::steady_clock::now() - last_data_ > std::chrono::seconds(2)) if (std::chrono::steady_clock::now() - last_data_ > std::chrono::seconds(2))
throw std::runtime_error("Microphone stalled; clip discarded"); throw std::runtime_error("Microphone stalled; clip discarded");
return pcm_.size() == 320000 || seconds() >= 20; return recording_ && (pcm_.size() == 320000 || seconds() >= 20);
} }
std::vector<float> Audio::finish() { std::vector<float> Audio::finish() {
if (!stream_) return {}; if (!recording_) return {};
try { try {
check(SDL_PauseAudioStreamDevice(stream_));
check(SDL_FlushAudioStream(stream_));
drain(); drain();
recording_ = false;
auto result = std::move(pcm_); auto result = std::move(pcm_);
cancel(); pcm_.clear();
// Discard any resampler remainder and never pause the capture device.
check(SDL_ClearAudioStream(stream_));
return result; return result;
} catch (...) { cancel(); throw; } } catch (...) { cancel(); throw; }
} }
void Audio::cancel() { void Audio::cancel() {
recording_ = false;
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear();
if (stream_) SDL_ClearAudioStream(stream_); // best effort on cancellation
}
void Audio::close() {
cancel();
if (stream_) SDL_DestroyAudioStream(stream_); if (stream_) SDL_DestroyAudioStream(stream_);
stream_ = nullptr; stream_ = nullptr;
// Best effort clearing of the owned buffer; no persistent recording/logging.
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear();
} }
int Audio::seconds() const { int Audio::seconds() const {
return stream_ ? int(std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - started_).count()) : 0; return recording_ ? int(std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - started_).count()) : 0;
} }
} }
+8 -4
View File
@@ -7,16 +7,20 @@ class Audio {
public: public:
Audio() = default; Audio() = default;
~Audio(); ~Audio();
void start(); void prepare(); // open/resume once, while app is ready; never retain idle samples
bool poll(); // true at 20s limit; throws if capture was lost/stalled void start(); // arm the already-running stream without touching the device
bool poll(); // drain even while idle; true at 20s limit; throws on loss/stall
std::vector<float> finish(); std::vector<float> finish();
void cancel(); void cancel(); // discard current clip, leave the device running
bool recording() const { return stream_ != nullptr; } void close(); // release device on shutdown or capture failure
bool recording() const { return recording_; }
bool open() const { return stream_ != nullptr; }
int seconds() const; int seconds() const;
private: private:
void drain(); void drain();
SDL_AudioStream* stream_ = nullptr; SDL_AudioStream* stream_ = nullptr;
bool initialized_ = false; bool initialized_ = false;
bool recording_ = false;
std::vector<float> pcm_; std::vector<float> pcm_;
std::chrono::steady_clock::time_point started_, last_data_; std::chrono::steady_clock::time_point started_, last_data_;
}; };
+15 -10
View File
@@ -47,7 +47,7 @@ int run(const Options& options) {
std::string detail = "Review mode. Other apps may also hear your mic. Enter is separate."; std::string detail = "Review mode. Other apps may also hear your mic. Enter is separate.";
bool quit = false; bool quit = false;
auto warm = [&] { auto warm = [&] {
worker.stop(); session = Session{}; audio.close(); worker.stop(); session = Session{};
worker.start(options.python, options.worker, options.model, options.threads); worker.start(options.python, options.worker, options.model, options.threads);
detail = "Loading local model; microphone closed. Record again when Ready."; detail = "Loading local model; microphone closed. Record again when Ready.";
}; };
@@ -63,6 +63,7 @@ int run(const Options& options) {
auto start_record = [&] { auto start_record = [&] {
if (session.state() == State::Review || session.state() == State::Transcribing || session.state() == State::Warming) return; if (session.state() == State::Review || session.state() == State::Transcribing || session.state() == State::Warming) return;
if (!worker.ready()) { warm(); return; } if (!worker.ready()) { warm(); return; }
if (!audio.open()) audio.prepare(); // recovery only; normal PTT never opens the device
if (session.state() == State::Error) session.cancel(); if (session.state() == State::Error) session.cancel();
if (!session.record()) return; if (!session.record()) return;
audio.start(); audio.start();
@@ -86,16 +87,19 @@ int run(const Options& options) {
} }
if (worker.ready()) session.ready(); if (worker.ready()) session.ready();
} catch (const std::exception& e) { } catch (const std::exception& e) {
audio.cancel(); worker.stop(); audio.close(); worker.stop();
if (session.state() != State::Review && session.state() != State::Queued) session.fail(); if (session.state() != State::Review && session.state() != State::Queued) session.fail();
detail = e.what(); // Preserve an already-correlated preview if the worker dies. detail = e.what(); // Preserve an already-correlated preview if the worker dies.
} }
try { try {
if (audio.recording() && audio.poll()) stop_record(); // Prepare after model warm-up, not on PTT. Keep draining/discarding
// idle samples so neither a device transition nor old speech reaches
// the next clip. A capture failure still requires an explicit retry.
if (session.state() == State::Ready && !audio.open() && worker.ready()) audio.prepare();
if (audio.open() && audio.poll()) stop_record();
} catch (const std::exception& e) { } catch (const std::exception& e) {
// A microphone failure is not a model failure. Release the device, // A microphone failure is not a model failure.
// retain the loaded worker, and allow Record to retry capture. audio.close(); session.fail(); detail = e.what();
audio.cancel(); session.fail(); detail = e.what();
} }
// Drawing precedes input polling: Enter is disabled until Ready is visible. // Drawing precedes input polling: Enter is disabled until Ready is visible.
const auto status = session.state() == State::Error && worker.ready() const auto status = session.state() == State::Error && worker.ready()
@@ -118,10 +122,11 @@ int run(const Options& options) {
case UiAction::EndRecord: stop_record(); break; case UiAction::EndRecord: stop_record(); break;
case UiAction::Cancel: { case UiAction::Cancel: {
auto state = session.state(); auto state = session.state();
audio.cancel(); session.cancel(); detail = "Discarded. Microphone closed."; audio.cancel(); session.cancel(); detail = "Discarded. Idle microphone samples are discarded.";
if (state == State::Warming || state == State::Transcribing) { if (state == State::Warming || state == State::Transcribing) {
worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker."; audio.close(); worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker.";
} else if (state == State::Error && !worker.ready()) { } else if (state == State::Error && !worker.ready()) {
audio.close();
session.fail(); detail = "Worker unavailable. Record to reload local worker."; session.fail(); detail = "Worker unavailable. Record to reload local worker.";
} }
break; break;
@@ -150,7 +155,7 @@ int run(const Options& options) {
// only the microphone; submit failures stop their own child if // only the microphone; submit failures stop their own child if
// the IPC stream was partially written. // the IPC stream was partially written.
if (session.state() == State::Recording || session.state() == State::Transcribing) { if (session.state() == State::Recording || session.state() == State::Transcribing) {
audio.cancel(); session.fail(); audio.close(); session.fail();
} }
detail = e.what(); detail = e.what();
} }
@@ -158,7 +163,7 @@ int run(const Options& options) {
} }
std::this_thread::sleep_for(std::chrono::milliseconds(10)); std::this_thread::sleep_for(std::chrono::milliseconds(10));
} }
audio.cancel(); session.cancel(); worker.stop(); audio.close(); session.cancel(); worker.stop();
return 0; return 0;
} }
} }
+64
View File
@@ -0,0 +1,64 @@
#include "audio.hpp"
#include <SDL3/SDL.h>
#include <algorithm>
#include <cassert>
#include <cstring>
#include <iostream>
#include <stdexcept>
namespace {
SDL_AudioStream* current = nullptr;
int opens = 0, resumes = 0, closes = 0;
bool removed = false;
void feed(float value, int count) { assert(current); current->queued.insert(current->queued.end(), count, value); }
}
const char* SDL_GetError() { return "fake failure"; }
bool SDL_InitSubSystem(int) { return true; }
void SDL_QuitSubSystem(int) {}
SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*) {
++opens; return current = new SDL_AudioStream;
}
bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*) { ++resumes; return true; }
SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*) { return 42; }
int SDL_GetAudioStreamAvailable(SDL_AudioStream* stream) { return int(stream->queued.size() * sizeof(float)); }
int SDL_GetAudioStreamData(SDL_AudioStream* stream, void* data, int size) {
int count = std::min(int(stream->queued.size()), size / int(sizeof(float)));
std::memcpy(data, stream->queued.data(), count * sizeof(float));
stream->queued.erase(stream->queued.begin(), stream->queued.begin() + count);
return count * sizeof(float);
}
bool SDL_ClearAudioStream(SDL_AudioStream* stream) { stream->queued.clear(); return true; }
void SDL_DestroyAudioStream(SDL_AudioStream* stream) { ++closes; current = nullptr; delete stream; }
bool SDL_PollEvent(SDL_Event* event) {
if (!removed) return false;
removed = false; *event = SDL_Event{SDL_EVENT_AUDIO_DEVICE_REMOVED, {true, 42}}; return true;
}
int main() {
using frameyap::Audio;
{
Audio audio;
audio.prepare(); audio.prepare();
assert(opens == 1 && resumes == 1 && audio.open() && !audio.recording());
feed(0.7f, 10000); audio.poll();
assert(current->queued.empty());
feed(0.8f, 200); audio.start(); // queued idle speech cannot leak
feed(0.1f, 3200); audio.poll();
feed(0.2f, 3200);
auto clip = audio.finish();
assert(clip.size() == 6400 && clip.front() == 0.1f && clip.back() == 0.2f);
assert(audio.open() && !audio.recording() && opens == 1 && closes == 0 && resumes == 1);
feed(0.5f, 200); audio.start(); feed(0.3f, 4000);
audio.cancel(); assert(current->queued.empty());
feed(0.9f, 100); audio.start(); feed(0.4f, 3200);
clip = audio.finish(); assert(clip.size() == 3200 && clip.front() == 0.4f);
assert(opens == 1 && closes == 0 && resumes == 1);
removed = true;
bool failed = false;
try { audio.poll(); } catch (const std::runtime_error&) { failed = true; }
assert(failed);
audio.close(); assert(closes == 1 && !audio.open());
audio.prepare(); assert(opens == 2 && resumes == 2);
}
assert(closes == 2);
std::cout << "audio lifecycle checks passed (fake device only)\n";
}
+22
View File
@@ -0,0 +1,22 @@
#pragma once
#include <cstdint>
#include <vector>
#define SDL_INIT_AUDIO 1
#define SDL_AUDIO_F32 2
#define SDL_AUDIO_DEVICE_DEFAULT_RECORDING 3
#define SDL_EVENT_AUDIO_DEVICE_REMOVED 4
using SDL_AudioDeviceID = unsigned;
struct SDL_AudioSpec { int format, channels, freq; };
struct SDL_AudioStream { std::vector<float> queued; };
struct SDL_Event { int type; struct { bool recording; SDL_AudioDeviceID which; } adevice; };
const char* SDL_GetError();
bool SDL_InitSubSystem(int);
void SDL_QuitSubSystem(int);
SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*);
bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*);
SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*);
int SDL_GetAudioStreamAvailable(SDL_AudioStream*);
int SDL_GetAudioStreamData(SDL_AudioStream*, void*, int);
bool SDL_ClearAudioStream(SDL_AudioStream*);
void SDL_DestroyAudioStream(SDL_AudioStream*);
bool SDL_PollEvent(SDL_Event*);