mirror of
https://github.com/baketnk/frame-yap.git
synced 2026-10-06 01:00:04 +02:00
Keep capture stream running between PTT utterances
This commit is contained in:
1 parent
a9d6762ebd
commit
8417212ed2
8 files changed
+165
-42
No files matched your search
@@ -100,6 +100,10 @@ if(BUILD_TESTING AND NOT CMAKE_CROSSCOMPILING)
|
||||
"-DEXPECTED_VERSION=${FRAMEYAP_VERSION}" -P "${CMAKE_CURRENT_SOURCE_DIR}/tests/cli.cmake")
|
||||
add_test(NAME frameyap.install_preflight COMMAND sh "${CMAKE_CURRENT_SOURCE_DIR}/tests/install-preflight.sh"
|
||||
"${CMAKE_CURRENT_SOURCE_DIR}/scripts/install-preflight.sh")
|
||||
add_executable(frameyap_audio_test tests/audio_test.cpp src/audio.cpp)
|
||||
target_include_directories(frameyap_audio_test PRIVATE src tests/fake_sdl)
|
||||
target_compile_options(frameyap_audio_test PRIVATE -UNDEBUG)
|
||||
add_test(NAME frameyap.audio_lifecycle COMMAND frameyap_audio_test)
|
||||
add_executable(frameyap_core_test tests/core_test.cpp)
|
||||
target_link_libraries(frameyap_core_test PRIVATE frameyap_core)
|
||||
add_test(NAME frameyap.core COMMAND frameyap_core_test)
|
||||
|
||||
@@ -36,7 +36,9 @@ See [third-party notes](docs/third-party.md). No GitHub release is published yet
|
||||
backs it up before repairs. See [overlay configuration](docs/overlay.md#user-theme-and-controller-configuration).
|
||||
- **Review-first:** focus your destination, then press Insert. No automatic insertion
|
||||
or submission. Maximum clip 20 seconds; accidental taps under 200 ms are discarded.
|
||||
- Other apps may still hear/transmit your voice. FrameYap does not mute them.
|
||||
- While the native app is Ready, it keeps the mic device open and discards idle
|
||||
audio instead of opening/closing on every PTT. Quit releases the device. Other
|
||||
apps may still hear/transmit your voice; FrameYap does not mute them.
|
||||
|
||||
Physical gesture timing, global bindings during games, mic capture, target-app
|
||||
compatibility and headset comfort still need coordinated validation. Successful
|
||||
|
||||
+9
-4
@@ -18,14 +18,19 @@ initialize OpenVR, open a microphone, run ASR, download files or inject input.
|
||||
after first release.
|
||||
Activity/tracking loss cancels a held recording and requires neutral rearm.
|
||||
- SDL3 default recording device, mono float32 conversion at 16 kHz, 200 ms minimum,
|
||||
20 second maximum. Microphone is closed outside actual capture. Other apps may
|
||||
still transmit your voice: this app does **not** mute VRChat or any other app.
|
||||
20 second maximum. In the native app the microphone stream opens after model
|
||||
warm-up, stays running between utterances, and discards idle samples; PTT does
|
||||
not open/pause/close the device. Quit, worker restart, device failure or capture
|
||||
failure closes it. This avoids repeated capture-device transitions but does not
|
||||
promise glitch-free playback on every audio stack. Other apps may still
|
||||
transmit your voice: this app does **not** mute VRChat or any other app.
|
||||
- Persistent local Redux worker, correlated bounded pipes, private tmpfs clips,
|
||||
cancellation/reaping and deadlines; exact pinned model SHA-256 verification.
|
||||
Model imports are lazy and loading is offline. Normal repeats, request-local
|
||||
transcription failures and microphone failures retain the loaded model;
|
||||
microphone device streams close outside capture. A cancelled in-flight request
|
||||
or broken worker protocol may require reloading. No cloud/desktop fallback.
|
||||
microphone device failure releases the stream for explicit retry. A cancelled
|
||||
in-flight request or broken worker protocol may require reloading. No
|
||||
cloud/desktop fallback.
|
||||
- Gamescope IME v2 generated bindings, per-action short-lived lease, unavailable
|
||||
handling, UTF-8/control validation and explicit separate Submit action for Enter.
|
||||
- Idempotent user-local release-archive installer: SHA-256, safe extraction,
|
||||
|
||||
+40
-23
@@ -9,37 +9,48 @@
|
||||
namespace frameyap {
|
||||
namespace { void check(bool result) { if (!result) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError()); } }
|
||||
Audio::~Audio() {
|
||||
cancel();
|
||||
close();
|
||||
if (initialized_) SDL_QuitSubSystem(SDL_INIT_AUDIO);
|
||||
}
|
||||
void Audio::start() {
|
||||
if (stream_) throw std::runtime_error("Already recording");
|
||||
// Keep SDL's audio backend initialized across utterances, but own the
|
||||
// recording device only while actually capturing. Reinitializing PipeWire
|
||||
// after every clip can fail even though the local model remains healthy.
|
||||
void Audio::prepare() {
|
||||
if (stream_) return;
|
||||
if (!initialized_) { check(SDL_InitSubSystem(SDL_INIT_AUDIO)); initialized_ = true; }
|
||||
SDL_AudioSpec spec{SDL_AUDIO_F32, 1, 16000};
|
||||
stream_ = SDL_OpenAudioDeviceStream(SDL_AUDIO_DEVICE_DEFAULT_RECORDING, &spec, nullptr, nullptr);
|
||||
if (!stream_) throw std::runtime_error(std::string("Microphone: ") + SDL_GetError());
|
||||
pcm_.clear(); pcm_.reserve(320000);
|
||||
started_ = last_data_ = std::chrono::steady_clock::now();
|
||||
try { check(SDL_ResumeAudioStreamDevice(stream_)); }
|
||||
catch (...) { cancel(); throw; }
|
||||
catch (...) { close(); throw; }
|
||||
last_data_ = std::chrono::steady_clock::now();
|
||||
}
|
||||
void Audio::start() {
|
||||
if (recording_) throw std::runtime_error("Already recording");
|
||||
if (!stream_) throw std::runtime_error("Microphone unavailable; retry capture");
|
||||
// Clear queued idle data at the PTT boundary; the device never pauses or
|
||||
// reopens. Polling also discards idle samples so the queue stays small.
|
||||
check(SDL_ClearAudioStream(stream_));
|
||||
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear(); pcm_.reserve(320000);
|
||||
started_ = last_data_ = std::chrono::steady_clock::now();
|
||||
recording_ = true;
|
||||
}
|
||||
void Audio::drain() {
|
||||
std::array<float, 4096> chunk{};
|
||||
while (pcm_.size() < 320000) {
|
||||
// During idle and after the 20s bound, consume and discard device samples.
|
||||
// Never let idle audio accumulate for the next utterance.
|
||||
while (true) {
|
||||
int available = SDL_GetAudioStreamAvailable(stream_);
|
||||
check(available >= 0);
|
||||
if (available < int(sizeof(float))) break;
|
||||
const int count = static_cast<int>(std::min({chunk.size(), std::size_t(available) / sizeof(float), 320000 - pcm_.size()}));
|
||||
const int count = static_cast<int>(std::min(chunk.size(), std::size_t(available) / sizeof(float)));
|
||||
int bytes = SDL_GetAudioStreamData(stream_, chunk.data(), count * sizeof(float));
|
||||
check(bytes >= 0);
|
||||
if (bytes == 0) break;
|
||||
if (bytes % sizeof(float)) throw std::runtime_error("Misaligned microphone samples");
|
||||
for (int i = 0; i < bytes / int(sizeof(float)); ++i)
|
||||
if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples");
|
||||
pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + bytes / sizeof(float));
|
||||
if (recording_) {
|
||||
const auto retain = std::min(std::size_t(bytes) / sizeof(float), 320000 - pcm_.size());
|
||||
for (std::size_t i = 0; i < retain; ++i)
|
||||
if (!std::isfinite(chunk[i])) throw std::runtime_error("Invalid microphone samples");
|
||||
pcm_.insert(pcm_.end(), chunk.begin(), chunk.begin() + retain);
|
||||
}
|
||||
last_data_ = std::chrono::steady_clock::now();
|
||||
}
|
||||
}
|
||||
@@ -47,32 +58,38 @@ bool Audio::poll() {
|
||||
if (!stream_) return false;
|
||||
SDL_Event event;
|
||||
while (SDL_PollEvent(&event)) {
|
||||
if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording)
|
||||
if (event.type == SDL_EVENT_AUDIO_DEVICE_REMOVED && event.adevice.recording &&
|
||||
event.adevice.which == SDL_GetAudioStreamDevice(stream_))
|
||||
throw std::runtime_error("Recording device removed; clip discarded");
|
||||
}
|
||||
drain();
|
||||
if (std::chrono::steady_clock::now() - last_data_ > std::chrono::seconds(2))
|
||||
throw std::runtime_error("Microphone stalled; clip discarded");
|
||||
return pcm_.size() == 320000 || seconds() >= 20;
|
||||
return recording_ && (pcm_.size() == 320000 || seconds() >= 20);
|
||||
}
|
||||
std::vector<float> Audio::finish() {
|
||||
if (!stream_) return {};
|
||||
if (!recording_) return {};
|
||||
try {
|
||||
check(SDL_PauseAudioStreamDevice(stream_));
|
||||
check(SDL_FlushAudioStream(stream_));
|
||||
drain();
|
||||
recording_ = false;
|
||||
auto result = std::move(pcm_);
|
||||
cancel();
|
||||
pcm_.clear();
|
||||
// Discard any resampler remainder and never pause the capture device.
|
||||
check(SDL_ClearAudioStream(stream_));
|
||||
return result;
|
||||
} catch (...) { cancel(); throw; }
|
||||
}
|
||||
void Audio::cancel() {
|
||||
recording_ = false;
|
||||
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear();
|
||||
if (stream_) SDL_ClearAudioStream(stream_); // best effort on cancellation
|
||||
}
|
||||
void Audio::close() {
|
||||
cancel();
|
||||
if (stream_) SDL_DestroyAudioStream(stream_);
|
||||
stream_ = nullptr;
|
||||
// Best effort clearing of the owned buffer; no persistent recording/logging.
|
||||
std::fill(pcm_.begin(), pcm_.end(), 0.0f); pcm_.clear();
|
||||
}
|
||||
int Audio::seconds() const {
|
||||
return stream_ ? int(std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - started_).count()) : 0;
|
||||
return recording_ ? int(std::chrono::duration_cast<std::chrono::seconds>(std::chrono::steady_clock::now() - started_).count()) : 0;
|
||||
}
|
||||
}
|
||||
+8
-4
@@ -7,16 +7,20 @@ class Audio {
|
||||
public:
|
||||
Audio() = default;
|
||||
~Audio();
|
||||
void start();
|
||||
bool poll(); // true at 20s limit; throws if capture was lost/stalled
|
||||
void prepare(); // open/resume once, while app is ready; never retain idle samples
|
||||
void start(); // arm the already-running stream without touching the device
|
||||
bool poll(); // drain even while idle; true at 20s limit; throws on loss/stall
|
||||
std::vector<float> finish();
|
||||
void cancel();
|
||||
bool recording() const { return stream_ != nullptr; }
|
||||
void cancel(); // discard current clip, leave the device running
|
||||
void close(); // release device on shutdown or capture failure
|
||||
bool recording() const { return recording_; }
|
||||
bool open() const { return stream_ != nullptr; }
|
||||
int seconds() const;
|
||||
private:
|
||||
void drain();
|
||||
SDL_AudioStream* stream_ = nullptr;
|
||||
bool initialized_ = false;
|
||||
bool recording_ = false;
|
||||
std::vector<float> pcm_;
|
||||
std::chrono::steady_clock::time_point started_, last_data_;
|
||||
};
|
||||
|
||||
+15
-10
@@ -47,7 +47,7 @@ int run(const Options& options) {
|
||||
std::string detail = "Review mode. Other apps may also hear your mic. Enter is separate.";
|
||||
bool quit = false;
|
||||
auto warm = [&] {
|
||||
worker.stop(); session = Session{};
|
||||
audio.close(); worker.stop(); session = Session{};
|
||||
worker.start(options.python, options.worker, options.model, options.threads);
|
||||
detail = "Loading local model; microphone closed. Record again when Ready.";
|
||||
};
|
||||
@@ -63,6 +63,7 @@ int run(const Options& options) {
|
||||
auto start_record = [&] {
|
||||
if (session.state() == State::Review || session.state() == State::Transcribing || session.state() == State::Warming) return;
|
||||
if (!worker.ready()) { warm(); return; }
|
||||
if (!audio.open()) audio.prepare(); // recovery only; normal PTT never opens the device
|
||||
if (session.state() == State::Error) session.cancel();
|
||||
if (!session.record()) return;
|
||||
audio.start();
|
||||
@@ -86,16 +87,19 @@ int run(const Options& options) {
|
||||
}
|
||||
if (worker.ready()) session.ready();
|
||||
} catch (const std::exception& e) {
|
||||
audio.cancel(); worker.stop();
|
||||
audio.close(); worker.stop();
|
||||
if (session.state() != State::Review && session.state() != State::Queued) session.fail();
|
||||
detail = e.what(); // Preserve an already-correlated preview if the worker dies.
|
||||
}
|
||||
try {
|
||||
if (audio.recording() && audio.poll()) stop_record();
|
||||
// Prepare after model warm-up, not on PTT. Keep draining/discarding
|
||||
// idle samples so neither a device transition nor old speech reaches
|
||||
// the next clip. A capture failure still requires an explicit retry.
|
||||
if (session.state() == State::Ready && !audio.open() && worker.ready()) audio.prepare();
|
||||
if (audio.open() && audio.poll()) stop_record();
|
||||
} catch (const std::exception& e) {
|
||||
// A microphone failure is not a model failure. Release the device,
|
||||
// retain the loaded worker, and allow Record to retry capture.
|
||||
audio.cancel(); session.fail(); detail = e.what();
|
||||
// A microphone failure is not a model failure.
|
||||
audio.close(); session.fail(); detail = e.what();
|
||||
}
|
||||
// Drawing precedes input polling: Enter is disabled until Ready is visible.
|
||||
const auto status = session.state() == State::Error && worker.ready()
|
||||
@@ -118,10 +122,11 @@ int run(const Options& options) {
|
||||
case UiAction::EndRecord: stop_record(); break;
|
||||
case UiAction::Cancel: {
|
||||
auto state = session.state();
|
||||
audio.cancel(); session.cancel(); detail = "Discarded. Microphone closed.";
|
||||
audio.cancel(); session.cancel(); detail = "Discarded. Idle microphone samples are discarded.";
|
||||
if (state == State::Warming || state == State::Transcribing) {
|
||||
worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker.";
|
||||
audio.close(); worker.stop(); session.fail(); detail = "Cancelled. Record to reload local worker.";
|
||||
} else if (state == State::Error && !worker.ready()) {
|
||||
audio.close();
|
||||
session.fail(); detail = "Worker unavailable. Record to reload local worker.";
|
||||
}
|
||||
break;
|
||||
@@ -150,7 +155,7 @@ int run(const Options& options) {
|
||||
// only the microphone; submit failures stop their own child if
|
||||
// the IPC stream was partially written.
|
||||
if (session.state() == State::Recording || session.state() == State::Transcribing) {
|
||||
audio.cancel(); session.fail();
|
||||
audio.close(); session.fail();
|
||||
}
|
||||
detail = e.what();
|
||||
}
|
||||
@@ -158,7 +163,7 @@ int run(const Options& options) {
|
||||
}
|
||||
std::this_thread::sleep_for(std::chrono::milliseconds(10));
|
||||
}
|
||||
audio.cancel(); session.cancel(); worker.stop();
|
||||
audio.close(); session.cancel(); worker.stop();
|
||||
return 0;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
#include "audio.hpp"
|
||||
#include <SDL3/SDL.h>
|
||||
#include <algorithm>
|
||||
#include <cassert>
|
||||
#include <cstring>
|
||||
#include <iostream>
|
||||
#include <stdexcept>
|
||||
|
||||
namespace {
|
||||
SDL_AudioStream* current = nullptr;
|
||||
int opens = 0, resumes = 0, closes = 0;
|
||||
bool removed = false;
|
||||
void feed(float value, int count) { assert(current); current->queued.insert(current->queued.end(), count, value); }
|
||||
}
|
||||
const char* SDL_GetError() { return "fake failure"; }
|
||||
bool SDL_InitSubSystem(int) { return true; }
|
||||
void SDL_QuitSubSystem(int) {}
|
||||
SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*) {
|
||||
++opens; return current = new SDL_AudioStream;
|
||||
}
|
||||
bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*) { ++resumes; return true; }
|
||||
SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*) { return 42; }
|
||||
int SDL_GetAudioStreamAvailable(SDL_AudioStream* stream) { return int(stream->queued.size() * sizeof(float)); }
|
||||
int SDL_GetAudioStreamData(SDL_AudioStream* stream, void* data, int size) {
|
||||
int count = std::min(int(stream->queued.size()), size / int(sizeof(float)));
|
||||
std::memcpy(data, stream->queued.data(), count * sizeof(float));
|
||||
stream->queued.erase(stream->queued.begin(), stream->queued.begin() + count);
|
||||
return count * sizeof(float);
|
||||
}
|
||||
bool SDL_ClearAudioStream(SDL_AudioStream* stream) { stream->queued.clear(); return true; }
|
||||
void SDL_DestroyAudioStream(SDL_AudioStream* stream) { ++closes; current = nullptr; delete stream; }
|
||||
bool SDL_PollEvent(SDL_Event* event) {
|
||||
if (!removed) return false;
|
||||
removed = false; *event = SDL_Event{SDL_EVENT_AUDIO_DEVICE_REMOVED, {true, 42}}; return true;
|
||||
}
|
||||
int main() {
|
||||
using frameyap::Audio;
|
||||
{
|
||||
Audio audio;
|
||||
audio.prepare(); audio.prepare();
|
||||
assert(opens == 1 && resumes == 1 && audio.open() && !audio.recording());
|
||||
feed(0.7f, 10000); audio.poll();
|
||||
assert(current->queued.empty());
|
||||
feed(0.8f, 200); audio.start(); // queued idle speech cannot leak
|
||||
feed(0.1f, 3200); audio.poll();
|
||||
feed(0.2f, 3200);
|
||||
auto clip = audio.finish();
|
||||
assert(clip.size() == 6400 && clip.front() == 0.1f && clip.back() == 0.2f);
|
||||
assert(audio.open() && !audio.recording() && opens == 1 && closes == 0 && resumes == 1);
|
||||
feed(0.5f, 200); audio.start(); feed(0.3f, 4000);
|
||||
audio.cancel(); assert(current->queued.empty());
|
||||
feed(0.9f, 100); audio.start(); feed(0.4f, 3200);
|
||||
clip = audio.finish(); assert(clip.size() == 3200 && clip.front() == 0.4f);
|
||||
assert(opens == 1 && closes == 0 && resumes == 1);
|
||||
removed = true;
|
||||
bool failed = false;
|
||||
try { audio.poll(); } catch (const std::runtime_error&) { failed = true; }
|
||||
assert(failed);
|
||||
audio.close(); assert(closes == 1 && !audio.open());
|
||||
audio.prepare(); assert(opens == 2 && resumes == 2);
|
||||
}
|
||||
assert(closes == 2);
|
||||
std::cout << "audio lifecycle checks passed (fake device only)\n";
|
||||
}
|
||||
@@ -0,0 +1,22 @@
|
||||
#pragma once
|
||||
#include <cstdint>
|
||||
#include <vector>
|
||||
#define SDL_INIT_AUDIO 1
|
||||
#define SDL_AUDIO_F32 2
|
||||
#define SDL_AUDIO_DEVICE_DEFAULT_RECORDING 3
|
||||
#define SDL_EVENT_AUDIO_DEVICE_REMOVED 4
|
||||
using SDL_AudioDeviceID = unsigned;
|
||||
struct SDL_AudioSpec { int format, channels, freq; };
|
||||
struct SDL_AudioStream { std::vector<float> queued; };
|
||||
struct SDL_Event { int type; struct { bool recording; SDL_AudioDeviceID which; } adevice; };
|
||||
const char* SDL_GetError();
|
||||
bool SDL_InitSubSystem(int);
|
||||
void SDL_QuitSubSystem(int);
|
||||
SDL_AudioStream* SDL_OpenAudioDeviceStream(SDL_AudioDeviceID, const SDL_AudioSpec*, void*, void*);
|
||||
bool SDL_ResumeAudioStreamDevice(SDL_AudioStream*);
|
||||
SDL_AudioDeviceID SDL_GetAudioStreamDevice(SDL_AudioStream*);
|
||||
int SDL_GetAudioStreamAvailable(SDL_AudioStream*);
|
||||
int SDL_GetAudioStreamData(SDL_AudioStream*, void*, int);
|
||||
bool SDL_ClearAudioStream(SDL_AudioStream*);
|
||||
void SDL_DestroyAudioStream(SDL_AudioStream*);
|
||||
bool SDL_PollEvent(SDL_Event*);
|
||||
Reference in new issue
Block a user