Keep capture stream running between PTT utterances

This commit is contained in:
baketnk committed 2026-09-24 14:04:31 -04:00
1 parent a9d6762ebd
commit 8417212ed2
8 files changed
+165 -42

No files matched your search

+4
View File
@@ -100,6 +100,10 @@ if(BUILD_TESTING AND NOT CMAKE_CROSSCOMPILING)
"-DEXPECTED_VERSION=${FRAMEYAP_VERSION}" -P "${CMAKE_CURRENT_SOURCE_DIR}/tests/cli.cmake")
add_test(NAME frameyap.install_preflight COMMAND sh "${CMAKE_CURRENT_SOURCE_DIR}/tests/install-preflight.sh"
"${CMAKE_CURRENT_SOURCE_DIR}/scripts/install-preflight.sh")
add_executable(frameyap_audio_test tests/audio_test.cpp src/audio.cpp)
target_include_directories(frameyap_audio_test PRIVATE src tests/fake_sdl)
target_compile_options(frameyap_audio_test PRIVATE -UNDEBUG)
add_test(NAME frameyap.audio_lifecycle COMMAND frameyap_audio_test)
add_executable(frameyap_core_test tests/core_test.cpp)
target_link_libraries(frameyap_core_test PRIVATE frameyap_core)
add_test(NAME frameyap.core COMMAND frameyap_core_test)
+3 -1
View File
@@ -36,7 +36,9 @@ See [third-party notes](docs/third-party.md). No GitHub release is published yet
backs it up before repairs. See [overlay configuration](docs/overlay.md#user-theme-and-controller-configuration).
- **Review-first:** focus your destination, then press Insert. No automatic insertion
or submission. Maximum clip 20 seconds; accidental taps under 200 ms are discarded.
- Other apps may still hear/transmit your voice. FrameYap does not mute them.
- While the native app is Ready, it keeps the mic device open and discards idle
audio instead of opening/closing on every PTT. Quit releases the device. Other
apps may still hear/transmit your voice; FrameYap does not mute them.
Physical gesture timing, global bindings during games, mic capture, target-app
compatibility and headset comfort still need coordinated validation. Successful
+9 -4
View File
@@ -18,14 +18,19 @@ initialize OpenVR, open a microphone, run ASR, download files or inject input.
after first release.
Activity/tracking loss cancels a held recording and requires neutral rearm.
- SDL3 default recording device, mono float32 conversion at 16 kHz, 200 ms minimum,
20 second maximum. Microphone is closed outside actual capture. Other apps may
still transmit your voice: this app does **not** mute VRChat or any other app.
20 second maximum. In the native app the microphone stream opens after model
warm-up, stays running between utterances, and discards idle samples; PTT does
not open/pause/close the device. Quit, worker restart, device failure or capture
failure closes it. This avoids repeated capture-device transitions but does not
promise glitch-free playback on every audio stack. Other apps may still
transmit your voice: this app does **not** mute VRChat or any other app.
- Persistent local Redux worker, correlated bounded pipes, private tmpfs clips,
cancellation/reaping and deadlines; exact pinned model SHA-256 verification.
Model imports are lazy and loading is offline. Normal repeats, request-local
transcription failures and microphone failures retain the loaded model;
microphone device streams close outside capture. A cancelled in-flight request
or broken worker protocol may require reloading. No cloud/desktop fallback.
microphone device failure releases the stream for explicit retry. A cancelled
in-flight request or broken worker protocol may require reloading. No
cloud/desktop fallback.
- Gamescope IME v2 generated bindings, per-action short-lived lease, unavailable
handling, UTF-8/control validation and explicit separate Submit action for Enter.
- Idempotent user-local release-archive installer: SHA-256, safe extraction,
+40 -23
View File
@@ -9,37 +9,48 @@
namespace frameyap {
namespace { void check(bool result) { if (!result) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError()); } }
Audio::~Audio() {
cancel();
close();
if (initialized_) SDL_QuitSubSystem(SDL_INIT_AUDIO);
}
void Audio::start() {
if (stream_) throw std::runtime_error("Already recording");
// Keep SDL's audio backend initialized across utterances, but own the
// recording device only while actually capturing. Reinitializing PipeWire
// after every clip can fail even though the local model remains healthy.
void Audio::prepare() {
if (stream_) return;
if (!initialized_) { check(SDL_InitSubSystem(SDL_INIT_AUDIO)); initialized_ = true; }
SDL_AudioSpec spec{SDL_AUDIO_F32, 1, 16000};
stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_RECORDING, &spec, nullptr, nullptr);
if (!stream_) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError());
pcm_.clear(); pcm_.reserve(320000);
started_ = last_data_ = std::chrono::steady_clock::now();
try { check(SDL_ResumeAudioStreamDevice(stream_)); }
catch (...) { cancel(); throw; }
catch (...) { close(); throw; }
last_data_ = std::chrono::steady_clock::now();
}
void Audio::start() {
if (recording_) throw std::runtime_error("Already recording");
if (!stream_) throw std::runtime_error("Microphone unavailable; retry capture");
// Clear queued idle data at the PTT boundary; the device never pauses or
// reopens. Polling also discards idle samples so the queue stays small.
check(SDL_ClearAudioStream(stream_));
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear(); pcm_.reserve(320000);
started_ = last_data_ = std::chrono::steady_clock::now();
recording_ = true;
}
void Audio::drain() {
std::array<float, 4096> chunk{};
while (pcm_.size() < 320000) {
// During idle and after the 20s bound, consume and discard device samples.
// Never let idle audio accumulate for the next utterance.
while (true) {
int available = SDL_GetAudioStreamAvailable(stream_);
check(available >= 0);
if (available < int(sizeof(float))) break;
const int count = static_cast<int>(std::min({chunk.size(), std::size_t(available) / sizeof(float), 320000 - pcm_.size()}));
const int count = static_cast<int>(std::min(chunk.size(), std::size_t(available) / sizeof(float)));
int bytes = SDL_GetAudioStreamData(stream_, chunk.data(), count * sizeof(float));
check(bytes >= 0);
if (bytes == 0) break;
if (bytes % sizeof(float)) throw std::runtime_error("Misaligned microphone samples");
for (int i = 0; i < bytes / int(sizeof(float)); ++i)
if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples");
pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + bytes / sizeof(float));
if (recording_) {
const auto retain = std::min(std::size_t(bytes) / sizeof(float), 320000 - pcm_.size());
for (std::size_t i = 0; i < retain; ++i)
if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples");
pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + retain);
}
last_data_ = std::chrono::steady_clock::now();
}
}
@@ -47,32 +58,38 @@ bool Audio::poll() {
if (!stream_) return false;
SDL_Event event;
while (SDL_PollEvent(&event)) {
if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording)
if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording &&
event.adevice.which == SDL_GetAudioStreamDevice(stream_))
throw std::runtime_error("Recording device removed; clip discarded");
}
drain();
if (std::chrono::steady_clock::now() - last_data_ > std::chrono::seconds(2))
throw std::runtime_error("Microphone stalled; clip discarded");
return pcm_.size() == 320000 || seconds() >= 20;
return recording_ && (pcm_.size() == 320000 || seconds() >= 20);
}
std::vector<float> Audio::finish() {
if (!stream_) return {};
if (!recording_) return {};
try {
check(SDL_PauseAudioStreamDevice(stream_));
check(SDL_FlushAudioStream(stream_));
drain();
recording_ = false;
auto result = std::move(pcm_);
cancel();
pcm_.clear();
// Discard any resampler remainder and never pause the capture device.
check(SDL_ClearAudioStream(stream_));
return result;
} catch (...) { cancel(); throw; }
}
void Audio::cancel() {
recording_ = false;
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear();
if (stream_) SDL_ClearAudioStream(stream_); // best effort on cancellation
}
void Audio::close() {
cancel();
if (stream_) SDL_DestroyAudioStream(stream_);
stream_ = nullptr;
// Best effort clearing of the owned buffer; no persistent recording/logging.
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear();
}
int Audio::seconds() const {
return stream_ ? int(std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - started_).count()) : 0;
return recording_ ? int(std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - started_).count()) : 0;
}
}
+8 -4
View File
@@ -7,16 +7,20 @@ class Audio {
public:
Audio() = default;
~Audio();
void start();
bool poll(); // true at 20s limit; throws if capture was lost/stalled
void prepare(); // open/resume once, while app is ready; never retain idle samples
void start(); // arm the already-running stream without touching the device
bool poll(); // drain even while idle; true at 20s limit; throws on loss/stall
std::vector<float> finish();
void cancel();
bool recording() const { return stream_ != nullptr; }
void cancel(); // discard current clip, leave the device running
void close(); // release device on shutdown or capture failure
bool recording() const { return recording_; }
bool open() const { return stream_ != nullptr; }
int seconds() const;
private:
void drain();
SDL_AudioStream* stream_ = nullptr;
bool initialized_ = false;
bool recording_ = false;
std::vector<float> pcm_;
std::chrono::steady_clock::time_point started_, last_data_;
};
+15 -10
View File
@@ -47,7 +47,7 @@ int run(const Options& options) {
std::string detail = "Review mode. Other apps may also hear your mic. Enter is separate.";
bool quit = false;
auto warm = [&] {
worker.stop(); session = Session{};
audio.close(); worker.stop(); session = Session{};
worker.start(options.python, options.worker, options.model, options.threads);
detail = "Loading local model; microphone closed. Record again when Ready.";
};
@@ -63,6 +63,7 @@ int run(const Options& options) {
auto start_record = [&] {
if (session.state() == State::Review || session.state() == State::Transcribing || session.state() == State::Warming) return;
if (!worker.ready()) { warm(); return; }
if (!audio.open()) audio.prepare(); // recovery only; normal PTT never opens the device
if (session.state() == State::Error) session.cancel();
if (!session.record()) return;
audio.start();
@@ -86,16 +87,19 @@ int run(const Options& options) {
}
if (worker.ready()) session.ready();
} catch (const std::exception& e) {
audio.cancel(); worker.stop();
audio.close(); worker.stop();
if (session.state() != State::Review && session.state() != State::Queued) session.fail();
detail = e.what(); // Preserve an already-correlated preview if the worker dies.
}
try {
if (audio.recording() && audio.poll()) stop_record();
// Prepare after model warm-up, not on PTT. Keep draining/discarding
// idle samples so neither a device transition nor old speech reaches
// the next clip. A capture failure still requires an explicit retry.
if (session.state() == State::Ready && !audio.open() && worker.ready()) audio.prepare();
if (audio.open() && audio.poll()) stop_record();
} catch (const std::exception& e) {
// A microphone failure is not a model failure. Release the device,
// retain the loaded worker, and allow Record to retry capture.
audio.cancel(); session.fail(); detail = e.what();
// A microphone failure is not a model failure.
audio.close(); session.fail(); detail = e.what();
}
// Drawing precedes input polling: Enter is disabled until Ready is visible.
const auto status = session.state() == State::Error && worker.ready()
@@ -118,10 +122,11 @@ int run(const Options& options) {
case UiAction::EndRecord: stop_record(); break;
case UiAction::Cancel: {
auto state = session.state();
audio.cancel(); session.cancel(); detail = "Discarded. Microphone closed.";
audio.cancel(); session.cancel(); detail = "Discarded. Idle microphone samples are discarded.";
if (state == State::Warming || state == State::Transcribing) {
worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker.";
audio.close(); worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker.";
} else if (state == State::Error && !worker.ready()) {
audio.close();
session.fail(); detail = "Worker unavailable. Record to reload local worker.";
}
break;
@@ -150,7 +155,7 @@ int run(const Options& options) {
// only the microphone; submit failures stop their own child if
// the IPC stream was partially written.
if (session.state() == State::Recording || session.state() == State::Transcribing) {
audio.cancel(); session.fail();
audio.close(); session.fail();
}
detail = e.what();
}
@@ -158,7 +163,7 @@ int run(const Options& options) {
}
std::this_thread::sleep_for(std::chrono::milliseconds(10));
}
audio.cancel(); session.cancel(); worker.stop();
audio.close(); session.cancel(); worker.stop();
return 0;
}
}
+64
View File
@@ -0,0 +1,64 @@
#include "audio.hpp"
#include <SDL3/SDL.h>
#include <algorithm>
#include <cassert>
#include <cstring>
#include <iostream>
#include <stdexcept>
namespace {
SDL_AudioStream* current = nullptr;
int opens = 0, resumes = 0, closes = 0;
bool removed = false;
void feed(float value, int count) { assert(current); current->queued.insert(current->queued.end(), count, value); }
}
const char* SDL_GetError() { return "fake failure"; }
bool SDL_InitSubSystem(int) { return true; }
void SDL_QuitSubSystem(int) {}
SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*) {
++opens; return current = new SDL_AudioStream;
}
bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*) { ++resumes; return true; }
SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*) { return 42; }
int SDL_GetAudioStreamAvailable(SDL_AudioStream* stream) { return int(stream->queued.size() * sizeof(float)); }
int SDL_GetAudioStreamData(SDL_AudioStream* stream, void* data, int size) {
int count = std::min(int(stream->queued.size()), size / int(sizeof(float)));
std::memcpy(data, stream->queued.data(), count * sizeof(float));
stream->queued.erase(stream->queued.begin(), stream->queued.begin() + count);
return count * sizeof(float);
}
bool SDL_ClearAudioStream(SDL_AudioStream* stream) { stream->queued.clear(); return true; }
void SDL_DestroyAudioStream(SDL_AudioStream* stream) { ++closes; current = nullptr; delete stream; }
bool SDL_PollEvent(SDL_Event* event) {
if (!removed) return false;
removed = false; *event = SDL_Event{SDL_EVENT_AUDIO_DEVICE_REMOVED, {true, 42}}; return true;
}
int main() {
using frameyap::Audio;
{
Audio audio;
audio.prepare(); audio.prepare();
assert(opens == 1 && resumes == 1 && audio.open() && !audio.recording());
feed(0.7f, 10000); audio.poll();
assert(current->queued.empty());
feed(0.8f, 200); audio.start(); // queued idle speech cannot leak
feed(0.1f, 3200); audio.poll();
feed(0.2f, 3200);
auto clip = audio.finish();
assert(clip.size() == 6400 && clip.front() == 0.1f && clip.back() == 0.2f);
assert(audio.open() && !audio.recording() && opens == 1 && closes == 0 && resumes == 1);
feed(0.5f, 200); audio.start(); feed(0.3f, 4000);
audio.cancel(); assert(current->queued.empty());
feed(0.9f, 100); audio.start(); feed(0.4f, 3200);
clip = audio.finish(); assert(clip.size() == 3200 && clip.front() == 0.4f);
assert(opens == 1 && closes == 0 && resumes == 1);
removed = true;
bool failed = false;
try { audio.poll(); } catch (const std::runtime_error&) { failed = true; }
assert(failed);
audio.close(); assert(closes == 1 && !audio.open());
audio.prepare(); assert(opens == 2 && resumes == 2);
}
assert(closes == 2);
std::cout << "audio lifecycle checks passed (fake device only)\n";
}
+22
View File
@@ -0,0 +1,22 @@
#pragma once
#include <cstdint>
#include <vector>
#define SDL_INIT_AUDIO 1
#define SDL_AUDIO_F32 2
#define SDL_AUDIO_DEVICE_DEFAULT_RECORDING 3
#define SDL_EVENT_AUDIO_DEVICE_REMOVED 4
using SDL_AudioDeviceID = unsigned;
struct SDL_AudioSpec { int format, channels, freq; };
struct SDL_AudioStream { std::vector<float> queued; };
struct SDL_Event { int type; struct { bool recording; SDL_AudioDeviceID which; } adevice; };
const char* SDL_GetError();
bool SDL_InitSubSystem(int);
void SDL_QuitSubSystem(int);
SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*);
bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*);
SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*);
int SDL_GetAudioStreamAvailable(SDL_AudioStream*);
int SDL_GetAudioStreamData(SDL_AudioStream*, void*, int);
bool SDL_ClearAudioStream(SDL_AudioStream*);
void SDL_DestroyAudioStream(SDL_AudioStream*);
bool SDL_PollEvent(SDL_Event*);