diff --git a/crates/voxctrl-config/src/lib.rs b/crates/voxctrl-config/src/lib.rs index 40e37e3..06825fb 100644 --- a/crates/voxctrl-config/src/lib.rs +++ b/crates/voxctrl-config/src/lib.rs @@ -286,6 +286,16 @@ pub struct FeaturesConfig { pub show_notification: Option, /// Map of trigger → expansion, e.g. {"addr" → "123 Main St"} pub snippets: std::collections::HashMap, + /// Transcribe the opening seconds of a recording while it is still going, + /// so a spoken voice command shows its overlay and starts loading the TTS + /// model before the user releases the hotkey. Costs some extra + /// transcription work at the start of each recording. + #[serde(default = "default_early_command_detection")] + pub early_command_detection: bool, +} + +fn default_early_command_detection() -> bool { + true } impl Default for FeaturesConfig { @@ -297,6 +307,7 @@ impl Default for FeaturesConfig { auto_format_lists: true, show_notification: None, snippets: std::collections::HashMap::new(), + early_command_detection: true, } } } @@ -1647,4 +1658,15 @@ mod tests { serde_json::from_str::(&with_leftover).expect("older configs must still load"); } + + #[test] + fn early_command_detection_defaults_on_for_older_configs() { + use super::FeaturesConfig; + let features: FeaturesConfig = serde_json::from_str( + r#"{"remove_fillers": true, "custom_vocabulary": [], "spoken_punctuation": true, + "auto_format_lists": true, "snippets": {}}"#, + ) + .unwrap(); + assert!(features.early_command_detection); + } } diff --git a/crates/voxctrl-inference/src/lib.rs b/crates/voxctrl-inference/src/lib.rs index 0a670b3..c2fe0f1 100644 --- a/crates/voxctrl-inference/src/lib.rs +++ b/crates/voxctrl-inference/src/lib.rs @@ -225,6 +225,10 @@ pub struct InferenceRequest { pub target_id: String, /// Hotkey binding ID (if triggered by a hotkey) pub binding_id: Option, + /// Whether this is an intermediate/live transcription during active speech + pub is_interim: bool, + /// Unique session identifier for the recording turn + pub session_id: u64, } /// Final output after transcription + post-processing. @@ -239,6 +243,8 @@ pub struct InferenceOutput { /// Set when transcription failed (model missing, backend error, ...). The /// UI layer surfaces this to the user; `text` is empty in that case. pub error: Option, + pub is_interim: bool, + pub session_id: u64, } // ── Engine ──────────────────────────────────────────────────────────────────── @@ -293,6 +299,9 @@ impl InferenceEngine { /// Transcribe and post-process. Returns the final text. pub fn process(&self, req: InferenceRequest) -> Result { + let is_interim = req.is_interim; + let session_id = req.session_id; + if req.audio.is_empty() { return Ok(InferenceOutput { text: String::new(), @@ -302,6 +311,8 @@ impl InferenceEngine { inference_ms: 0, language: "en".into(), error: None, + is_interim, + session_id, }); } @@ -345,6 +356,8 @@ impl InferenceEngine { inference_ms: 0, language: "en".into(), error: None, + is_interim, + session_id, }); } @@ -400,7 +413,10 @@ impl InferenceEngine { .and_then(|b| b.openai_enabled) .unwrap_or(false); - if binding_wants_openai && !processed.is_empty() { + // Interim passes only feed the voice-command trigger check and are + // thrown away; running the hotkey's (billed, slow) OpenAI rewrite on + // every one of them would cost a request per pass for nothing. + if binding_wants_openai && !is_interim && !processed.is_empty() { // Re-read the OpenAI settings from disk so changes made in the // Settings UI (model, endpoint, API key, prompts) take effect without // restarting the app. Targets and bindings above are already hot-read @@ -477,6 +493,8 @@ impl InferenceEngine { inference_ms: result.inference_ms, language: result.language, error: None, + is_interim, + session_id, }) } @@ -616,11 +634,25 @@ pub fn run_worker_with_config( loop { crossbeam_channel::select! { recv(rx) -> req_res => { - let req = match req_res { + let mut req = match req_res { Ok(r) => r, Err(_) => break, }; + // If an interim request arrived, drain any newer requests already in the queue + // to prioritize fresher audio or the final release request. + if req.is_interim { + while let Ok(newer) = rx.try_recv() { + req = newer; + if !req.is_interim { + break; + } + } + } + + let is_interim = req.is_interim; + let session_id = req.session_id; + if !loaded { match engine.load() { Ok(()) => { @@ -637,6 +669,8 @@ pub fn run_worker_with_config( inference_ms: 0, language: String::new(), error: Some(format!("{e:#}")), + is_interim, + session_id, }); continue; } @@ -657,6 +691,8 @@ pub fn run_worker_with_config( inference_ms: 0, language: "".to_string(), error: Some(format!("{e:#}")), + is_interim, + session_id, }); } } diff --git a/crates/voxctrl-routing/src/targets.rs b/crates/voxctrl-routing/src/targets.rs index 4d3f930..d79afd8 100644 --- a/crates/voxctrl-routing/src/targets.rs +++ b/crates/voxctrl-routing/src/targets.rs @@ -29,6 +29,21 @@ pub fn notify_command_trigger(command_name: &str, text_summary: &str) { } } +pub type CommandWithdrawnCallback = Arc; +static COMMAND_WITHDRAWN_CALLBACK: OnceLock = OnceLock::new(); + +pub fn set_command_withdrawn_callback(callback: CommandWithdrawnCallback) { + let _ = COMMAND_WITHDRAWN_CALLBACK.set(callback); +} + +/// Take back a command announced early (from a partial transcript) that the +/// final transcript did not confirm, so its overlay does not linger. +pub fn notify_command_withdrawn() { + if let Some(cb) = COMMAND_WITHDRAWN_CALLBACK.get() { + cb(); + } +} + // Shared HTTP client — built once, reused for connection pooling. fn http_client() -> &'static reqwest::Client { static CLIENT: std::sync::OnceLock = std::sync::OnceLock::new(); diff --git a/docs/api.md b/docs/api.md index 77d353b..d70b711 100644 --- a/docs/api.md +++ b/docs/api.md @@ -794,6 +794,7 @@ interface FeaturesConfig { spoken_punctuation: boolean; auto_format_lists: boolean; snippets: Record; + early_command_detection: boolean; } interface OpenAiConfig { diff --git a/docs/configuration.md b/docs/configuration.md index f158199..17a4dff 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -70,7 +70,8 @@ Full schema with defaults: "custom_vocabulary": [], "spoken_punctuation": true, "auto_format_lists": true, - "snippets": {} + "snippets": {}, + "early_command_detection": true }, "openai": { "enabled": false, @@ -237,6 +238,7 @@ The `.en` variants are English-only but slightly faster. `large-v3-turbo` is a d | `auto_format_lists` | bool | `true` | Detect "first/second/third" patterns and reformat as a numbered list | | `custom_vocabulary` | string[] | `[]` | Custom words; VoxCtrl uses fuzzy Levenshtein matching to correct near-matches post-transcription | | `snippets` | object | `{}` | Short code → expansion map | +| `early_command_detection` | bool | `true` | Transcribe the first 6 seconds of a recording while it is still going, so a voice command shows its overlay and starts loading the TTS model before the hotkey is released. Routing is still decided by the final transcript. Skipped for the remote backend and CPU-only medium/large Whisper models | Example with snippets: ```json diff --git a/src-tauri/src/lib.rs b/src-tauri/src/lib.rs index 22b2039..74ba4ca 100644 --- a/src-tauri/src/lib.rs +++ b/src-tauri/src/lib.rs @@ -350,6 +350,7 @@ pub fn run() { router: router.clone(), recording: Arc::new(AtomicBool::new(false)), processing: Arc::new(AtomicBool::new(false)), + interim_in_flight: Arc::new(AtomicBool::new(false)), speaking: Arc::new(AtomicBool::new(false)), overlay_enabled: Arc::new(AtomicBool::new(cfg_data.ui.show_overlay)), mcp_recording: Arc::new(AtomicBool::new(false)), diff --git a/src-tauri/src/pipeline.rs b/src-tauri/src/pipeline.rs index 7846060..83f9cd7 100644 --- a/src-tauri/src/pipeline.rs +++ b/src-tauri/src/pipeline.rs @@ -36,6 +36,8 @@ async fn process_remote_transcription( inference_ms: 0, language: String::new(), error: Some(format!("{e:#}")), + is_interim: false, + session_id: 0, }); return; } @@ -69,6 +71,8 @@ async fn process_remote_transcription( inference_ms: result.inference_ms, language: result.language, error: None, + is_interim: false, + session_id: 0, }); return; } @@ -188,9 +192,21 @@ async fn process_remote_transcription( inference_ms: result.inference_ms, language: result.language, error: None, + is_interim: false, + session_id: 0, }); } +/// Interim passes start once this much audio exists (0.5 s at 16 kHz)... +const INTERIM_MIN_SAMPLES: usize = 8_000; +/// ...and re-run after at least this much more (0.3 s). +const INTERIM_STEP_SAMPLES: usize = 4_800; +/// Only the opening of a recording is transcribed early: the trigger and the +/// target name come first ("Hey Vox, add this to my notes"), and capping the +/// window keeps each pass short, so the final transcription never waits long +/// behind one and long dictations don't keep the model busy the whole time. +const INTERIM_MAX_SAMPLES: usize = 6 * 16_000; + pub fn spawn_audio_coordinator( state_for_audio: Arc, audio_rx: crossbeam_channel::Receiver, @@ -205,6 +221,10 @@ pub fn spawn_audio_coordinator( let mut binding_id = String::new(); let mut remote_session: Option = None; let mut is_remote_backend = false; + let mut interim_enabled = false; + let mut session_id: u64 = 0; + let mut last_interim_sample_count: usize = 0; + let mut last_interim_instant = std::time::Instant::now(); while let Ok(chunk) = audio_rx.recv() { let is_recording = state_for_audio.is_recording(); @@ -215,8 +235,28 @@ pub fn spawn_audio_coordinator( target_id = state_for_audio.active_target.blocking_lock().clone(); binding_id = state_for_audio.active_binding_id.blocking_lock().clone(); was_recording = true; + session_id = session_id.wrapping_add(1); + last_interim_sample_count = 0; + last_interim_instant = std::time::Instant::now(); + state_for_audio.set_interim_in_flight(false); let cfg = state_for_audio.config.blocking_lock().data.clone(); + // Early command detection only pays off when a command + // could match (some non-router target exists) and the + // local model is fast enough to transcribe mid-recording. + let heavy_model = cfg.engine.backend == voxctrl_config::BackendChoice::WhisperCpp + && voxctrl_inference::whisper_gpu_backend().is_none() + && (cfg.engine.whisper_cpp.model_size.starts_with("medium") + || cfg.engine.whisper_cpp.model_size.starts_with("large")); + let has_command_targets = state_for_audio + .targets + .blocking_lock() + .iter() + .any(|t| t.delivery != voxctrl_routing::DeliveryType::Command); + interim_enabled = cfg.features.early_command_detection + && cfg.engine.backend != voxctrl_config::BackendChoice::RemoteOpenAi + && !heavy_model + && has_command_targets; if cfg.engine.backend == voxctrl_config::BackendChoice::RemoteOpenAi { is_remote_backend = true; let mut merged_prompt = String::from( @@ -248,6 +288,27 @@ pub fn spawn_audio_coordinator( session.send_chunk(chunk.clone()); } accumulated_audio.extend(chunk); + + // Interim transcription of the opening of the recording, so a + // voice command is recognised while the user is still talking. + let window = accumulated_audio.len().min(INTERIM_MAX_SAMPLES); + if interim_enabled + && window >= INTERIM_MIN_SAMPLES + && window.saturating_sub(last_interim_sample_count) >= INTERIM_STEP_SAMPLES + && last_interim_instant.elapsed() >= std::time::Duration::from_millis(400) + && !state_for_audio.is_interim_in_flight() + { + state_for_audio.set_interim_in_flight(true); + last_interim_sample_count = window; + last_interim_instant = std::time::Instant::now(); + let _ = inference_tx.send(voxctrl_inference::InferenceRequest { + audio: accumulated_audio[..window].to_vec(), + target_id: target_id.clone(), + binding_id: Some(binding_id.clone()), + is_interim: true, + session_id, + }); + } } else { if was_recording { if is_remote_backend { @@ -280,6 +341,8 @@ pub fn spawn_audio_coordinator( audio: std::mem::take(&mut accumulated_audio), target_id: target_id.clone(), binding_id: Some(binding_id.clone()), + is_interim: false, + session_id, }; state_for_audio.set_processing(true); let _ = inference_tx.send(req); @@ -298,9 +361,60 @@ pub fn spawn_text_delivery_worker( rt_handle: tokio::runtime::Handle, ) { std::thread::spawn(move || { + // (session, target) already announced from an interim pass, so the + // overlay and TTS preload fire once per command, not once per pass. + let mut announced: Option<(u64, String)> = None; + while let Ok(output) = text_rx.recv() { + if output.is_interim { + state.set_interim_in_flight(false); + if output.error.is_some() || output.text.trim().is_empty() { + continue; + } + // Early command detection: surface the command overlay and + // start loading the TTS model while the user is still talking. + // This is only a head start — routing is decided solely by the + // final transcript below, which must carry the trigger itself. + let dir = voxctrl_routing::config_dir(); + let targets = voxctrl_routing::load_targets(&dir).unwrap_or_default(); + if let Some(parsed) = voxctrl_routing::targets::parse_voice_command(&output.text, &targets) { + let key = (output.session_id, parsed.matched_target_id); + if announced.as_ref() != Some(&key) { + let matched_target = targets.iter().find(|t| t.id == key.1); + let matched_label = matched_target + .map(|t| if t.label.is_empty() { t.id.clone() } else { t.label.clone() }) + .unwrap_or_else(|| key.1.clone()); + tracing::info!("Early command detected during speech: '{matched_label}' (target: {})", key.1); + voxctrl_routing::targets::notify_command_trigger(&matched_label, &parsed.payload); + + let leads_to_speech = matched_target.is_some_and(|t| { + t.delivery == voxctrl_routing::DeliveryType::Speak + || t.response_pipe.as_deref().is_some_and(|p| !p.trim().is_empty()) + }); + if leads_to_speech { + let state_c = state.clone(); + rt_handle.spawn(async move { state_c.preload_tts().await }); + } + announced = Some(key); + } + } + continue; + } + + state.set_interim_in_flight(false); state.set_processing(false); + + // A command announced from an interim pass stands only if the + // final transcript confirms it; otherwise take its overlay down. + let announced_early = announced.take().is_some_and(|(session, _)| session == output.session_id); + let withdraw_early_command = || { + if announced_early { + voxctrl_routing::targets::notify_command_withdrawn(); + } + }; + if let Some(ref err) = output.error { + withdraw_early_command(); // Always surface transcription failures — without this a // fresh install with no Whisper model records audio and // then silently drops it, which reads as "hotkeys broken". @@ -309,6 +423,7 @@ pub fn spawn_text_delivery_worker( continue; } if output.text.trim().is_empty() { + withdraw_early_command(); continue; } @@ -349,6 +464,7 @@ pub fn spawn_text_delivery_worker( }; (matched_id, cleaned_payload) } else { + withdraw_early_command(); let cleaned_text = if s1_mini_enabled && !output.text.trim().is_empty() { voxctrl_inference::s1_mini::clean_dictation(&output.text, &s1_mini_styling, None) } else { diff --git a/src-tauri/src/services.rs b/src-tauri/src/services.rs index 1fe74ed..05fd694 100644 --- a/src-tauri/src/services.rs +++ b/src-tauri/src/services.rs @@ -240,6 +240,14 @@ pub fn register_speak_target(app_handle: &tauri::AppHandle) { pub fn register_command_trigger_target(app_handle: &tauri::AppHandle) { let state = app_handle.state::>().inner().clone(); let app_handle_clone = app_handle.clone(); + + let withdraw_state = state.clone(); + let withdraw_handle = app_handle.clone(); + voxctrl_routing::targets::set_command_withdrawn_callback(std::sync::Arc::new(move || { + withdraw_state.clear_command_overlay(); + let _ = withdraw_handle.emit("command-withdrawn", ()); + })); + voxctrl_routing::targets::set_command_trigger_callback(std::sync::Arc::new( move |command_name, text_summary| { let (show_overlay, duration_secs) = if let Ok(cfg) = state.config.try_lock() { diff --git a/src-tauri/src/state.rs b/src-tauri/src/state.rs index 7e8f163..a21bf70 100644 --- a/src-tauri/src/state.rs +++ b/src-tauri/src/state.rs @@ -14,6 +14,8 @@ pub struct AppState { pub recording: Arc, /// True while speech transcription/OpenAI post-processing is running pub processing: Arc, + /// True while an interim (mid-recording) transcription pass is running + pub interim_in_flight: Arc, /// True while TTS is playing back pub speaking: Arc, /// Live mirror of `ui.show_overlay` so the hot status-forwarding loops can @@ -150,6 +152,14 @@ impl AppState { self.processing.store(v, Ordering::SeqCst); } + pub fn is_interim_in_flight(&self) -> bool { + self.interim_in_flight.load(Ordering::SeqCst) + } + + pub fn set_interim_in_flight(&self, v: bool) { + self.interim_in_flight.store(v, Ordering::SeqCst); + } + pub fn is_audio_ready(&self) -> bool { self.audio_ready.load(Ordering::SeqCst) } @@ -242,6 +252,11 @@ impl AppState { *self.command_overlay_until.lock().unwrap() = Some(std::time::Instant::now() + duration); } + /// Stop showing the command-executed overlay pill now. + pub fn clear_command_overlay(&self) { + *self.command_overlay_until.lock().unwrap() = None; + } + /// Whether the command-executed overlay pill should still be showing. pub fn is_command_overlay_active(&self) -> bool { self.command_overlay_until diff --git a/src-tauri/src/tests.rs b/src-tauri/src/tests.rs index c565757..340ad21 100644 --- a/src-tauri/src/tests.rs +++ b/src-tauri/src/tests.rs @@ -66,6 +66,7 @@ fn make_test_state() -> AppState { router: Arc::new(OutputTargetRouter::new(Vec::new())), recording: Arc::new(AtomicBool::new(false)), processing: Arc::new(AtomicBool::new(false)), + interim_in_flight: Arc::new(AtomicBool::new(false)), speaking: Arc::new(AtomicBool::new(false)), overlay_enabled: Arc::new(AtomicBool::new(true)), mcp_recording: Arc::new(AtomicBool::new(false)), diff --git a/src/lib/Overlay/Overlay.svelte b/src/lib/Overlay/Overlay.svelte index 1dbaeb7..4289c0f 100644 --- a/src/lib/Overlay/Overlay.svelte +++ b/src/lib/Overlay/Overlay.svelte @@ -37,6 +37,7 @@ let commandOverlayText = $state(""); let commandTimerId: any = null; let unlistenCommandExecuted: (() => void) | null = null; + let unlistenCommandWithdrawn: (() => void) | null = null; let unlistenOverlayStyleSelected: (() => void) | null = null; // Delay unmounting the visualizer when recording/speaking/command stops to allow CSS outro animation to finish @@ -276,11 +277,20 @@ unlistenCommandExecuted = unlisten; }); + // A command recognised mid-speech that the final transcript didn't confirm. + listen("command-withdrawn", () => { + if (commandTimerId) clearTimeout(commandTimerId); + commandOverlayActive = false; + }).then((unlisten) => { + unlistenCommandWithdrawn = unlisten; + }); + return () => { document.documentElement.classList.remove("overlay-window"); document.body.classList.remove("overlay-window"); if (unlistenAudioLevel) unlistenAudioLevel(); if (unlistenCommandExecuted) unlistenCommandExecuted(); + if (unlistenCommandWithdrawn) unlistenCommandWithdrawn(); if (unlistenOverlayStyleSelected) unlistenOverlayStyleSelected(); if (commandTimerId) clearTimeout(commandTimerId); if (animationFrameId !== null) cancelAnimationFrame(animationFrameId); diff --git a/src/lib/Settings/FeaturesTab.svelte b/src/lib/Settings/FeaturesTab.svelte index 2c79a47..47384cc 100644 --- a/src/lib/Settings/FeaturesTab.svelte +++ b/src/lib/Settings/FeaturesTab.svelte @@ -259,6 +259,20 @@ +
+

Voice Commands

+ +

+ Transcribes the first few seconds of each recording early, so a command like + “Hey Vox, say …” shows its overlay and starts loading the voice before you let go + of the hotkey. Where the text goes is still decided by the full transcript. Turn + this off to save the extra transcription work. +

+
+

Custom Dictionary

Provide a comma-separated list of words (e.g. names or jargon like "Waylin, Rufer, Enola, Kenz") that are hard to spell. The transcription process will correct these in the final text.

diff --git a/src/stores/config.ts b/src/stores/config.ts index f048c32..38428da 100644 --- a/src/stores/config.ts +++ b/src/stores/config.ts @@ -78,6 +78,7 @@ export interface FeaturesConfig { spoken_punctuation: boolean; auto_format_lists: boolean; snippets: Record; + early_command_detection: boolean; } export interface OpenAiConfig { @@ -203,6 +204,7 @@ const defaultConfig: AppConfig = { spoken_punctuation: true, auto_format_lists: true, snippets: {}, + early_command_detection: true, }, openai: { enabled: false,