Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
73 changes: 50 additions & 23 deletions crates/voxctrl-tts/src/engine.rs
Original file line number Diff line number Diff line change
Expand Up @@ -123,6 +123,38 @@ impl TtsEngineHandle {
}
}

/// Whether the selected engine's `prewarm` setting asks for its model to be
/// loaded up front.
fn prewarm_enabled(config: &TtsConfig) -> bool {
match config.engine {
TtsEngine::PocketTts => config.pocket_tts.prewarm,
TtsEngine::InflectMicro => config.inflect_micro.prewarm,
TtsEngine::BreezeTts2 => config.breeze_tts_2.prewarm,
TtsEngine::VoxCpm2 => config.vox_cpm_2.prewarm,
_ => false,
}
}

/// Load the selected engine's model if it is not resident yet. `None` when
/// there was nothing to do: already loaded, or an engine that holds no model.
fn load_selected_engine(
config: &TtsConfig,
inflect_model: &mut InflectModelSlot,
audiocpp_session: &mut Option<AudioCppSession>,
) -> Option<Result<()>> {
match config.engine {
TtsEngine::InflectMicro if inflect_model.is_none() => {
Some(ensure_inflect_micro_loaded(config, inflect_model))
}
TtsEngine::PocketTts if audiocpp_session.is_none() => Some(ensure_pocket_tts_loaded(config, audiocpp_session)),
TtsEngine::BreezeTts2 if audiocpp_session.is_none() => {
Some(ensure_breeze_tts_2_loaded(config, audiocpp_session))
}
TtsEngine::VoxCpm2 if audiocpp_session.is_none() => Some(ensure_vox_cpm_2_loaded(config, audiocpp_session)),
_ => None,
}
}

// ── TTS engine worker ─────────────────────────────────────────────────────────

pub type PlaybackCallback = Arc<dyn Fn() + Send + Sync + 'static>;
Expand Down Expand Up @@ -158,13 +190,7 @@ impl TtsEngineWorker {
model_loaded: model_loaded.clone(),
};

let prewarm = match config.engine {
TtsEngine::PocketTts => config.pocket_tts.prewarm,
TtsEngine::InflectMicro => config.inflect_micro.prewarm,
TtsEngine::BreezeTts2 => config.breeze_tts_2.prewarm,
TtsEngine::VoxCpm2 => config.vox_cpm_2.prewarm,
_ => false,
};
let prewarm = prewarm_enabled(&config);
// Pre-warming loads the model at startup and keeps it there, which is
// exactly what the on-demand memory mode exists to avoid — so the two
// settings do not fight: on-demand wins and the model waits for its
Expand Down Expand Up @@ -283,10 +309,25 @@ impl TtsEngineWorker {
"TTS worker config dynamically updated (engine={:?}, memory_mode={:?})",
new_cfg.engine, new_cfg.memory_mode
);
let was_unloading = current_config.unloads_when_idle();
current_config = new_cfg;
// Switching to on-demand starts the clock now rather than
// dropping a model that may be about to be used again.
last_used = Instant::now();

// Switching to always-loaded (e.g. from the tray) with
// pre-warming on: load now, as a startup in this mode would.
if was_unloading && !current_config.unloads_when_idle() && prewarm_enabled(&current_config) {
if let Some(Err(e)) =
load_selected_engine(&current_config, &mut inflect_model, &mut audiocpp_session)
{
warn!("TTS pre-warm after switching to always-loaded failed: {e:#}");
}
self.model_loaded.store(
inflect_model.is_some() || audiocpp_session.is_some(),
Ordering::SeqCst,
);
}
}
TtsCommand::Play { mut utterance, generation } => {
// Synthesis is about to touch the model; hold off the idle
Expand Down Expand Up @@ -427,22 +468,8 @@ impl TtsEngineWorker {
// Speculative: failures are logged, not surfaced. The same
// error is reported properly (with a toast) if an utterance
// actually arrives.
let outcome = match current_config.engine {
TtsEngine::InflectMicro if inflect_model.is_none() => {
Some(ensure_inflect_micro_loaded(&current_config, &mut inflect_model))
}
TtsEngine::PocketTts if audiocpp_session.is_none() => {
Some(ensure_pocket_tts_loaded(&current_config, &mut audiocpp_session))
}
TtsEngine::BreezeTts2 if audiocpp_session.is_none() => {
Some(ensure_breeze_tts_2_loaded(&current_config, &mut audiocpp_session))
}
TtsEngine::VoxCpm2 if audiocpp_session.is_none() => {
Some(ensure_vox_cpm_2_loaded(&current_config, &mut audiocpp_session))
}
// Already resident, or an engine that holds no model.
_ => None,
};
let outcome =
load_selected_engine(&current_config, &mut inflect_model, &mut audiocpp_session);
match outcome {
Some(Err(e)) => debug!("TTS preload skipped: {e:#}"),
Some(Ok(())) => debug!("TTS model pre-loaded and primed"),
Expand Down
2 changes: 1 addition & 1 deletion docs/tts.md
Original file line number Diff line number Diff line change
Expand Up @@ -333,7 +333,7 @@ survives being idle:

| Mode | Behaviour |
|---|---|
| `"always_loaded"` (default) | The model is loaded on first use and stays resident for the life of the TTS worker. Fastest, highest memory. |
| `"always_loaded"` (default) | The model is loaded on first use (or at startup with `prewarm`) and stays resident for the life of the TTS worker. Fastest, highest memory. Recording against a speech target, or a spoken command to one, starts that first load early, as in `"on_demand"` mode. Switching into this mode with `prewarm` on loads the model right away. |
| `"on_demand"` | The model is loaded when VoxCtrl knows it is needed, kept primed while it keeps being used, and dropped after `idle_unload_secs` (default 900 = 15 minutes) of inactivity. |

In `"on_demand"` mode TTS itself stays enabled the whole time — only the weights come and go:
Expand Down
34 changes: 15 additions & 19 deletions src-tauri/src/lib.rs
Original file line number Diff line number Diff line change
Expand Up @@ -345,6 +345,19 @@ pub fn run() {

let hotkey_health = Arc::new(voxctrl_hotkeys::ListenerHealth::default());

// TTS initial worker
let initial_tts_handle = if cfg_data.tts.enabled {
Some(voxctrl_tts::TtsEngineWorker::start(
cfg_data.tts.clone(),
cfg_data.features.custom_vocabulary.clone(),
None,
None,
None,
))
} else {
None
};

let app_state = Arc::new(AppState {
config: config.clone(),
router: router.clone(),
Expand Down Expand Up @@ -373,7 +386,7 @@ pub fn run() {
audio_tx: audio_tx.clone(),
audio_wake: audio_wake_tx,
inference_config_tx: inference_cfg_tx,
tts_handle: Arc::new(Mutex::new(None)),
tts_handle: Arc::new(Mutex::new(initial_tts_handle.clone())),
active_fifos: Arc::new(Mutex::new(std::collections::HashSet::new())),
stop_key_held: Arc::new(AtomicBool::new(false)),
speaking_tx: Arc::new(std::sync::OnceLock::new()),
Expand Down Expand Up @@ -424,26 +437,9 @@ pub fn run() {
inference_cfg_rx,
);

// TTS initial worker
let _tts_handle = if cfg_data.tts.enabled {
Some(voxctrl_tts::TtsEngineWorker::start(
cfg_data.tts.clone(),
cfg_data.features.custom_vocabulary.clone(),
None,
None,
None,
))
} else {
None
};

let state_for_tts = app_state.clone();
let tts_handle_clone = _tts_handle.clone();
let tts_handle_clone = initial_tts_handle.clone();
tokio::spawn(async move {
{
let mut handle = state_for_tts.tts_handle.lock().await;
*handle = tts_handle_clone.clone();
}
if let Some(tts) = tts_handle_clone {
state_for_tts.spawn_fifo_responders(tts).await;
}
Expand Down
8 changes: 7 additions & 1 deletion src-tauri/src/services.rs
Original file line number Diff line number Diff line change
Expand Up @@ -224,8 +224,14 @@ pub fn setup_tts_and_fifos(app_handle: &tauri::AppHandle, state: Arc<AppState>)
pub fn register_speak_target(app_handle: &tauri::AppHandle) {
let state = app_handle.state::<Arc<AppState>>().inner().clone();
voxctrl_routing::targets::set_speak_callback(std::sync::Arc::new(move |text| {
let state = state.clone();
let text_str = text.to_string();
if let Ok(handle) = state.tts_handle.try_lock() {
if let Some(ref tts) = *handle {
tts.speak(text_str);
return;
}
}
let state = state.clone();
tauri::async_runtime::spawn(async move {
let handle = state.tts_handle.lock().await;
if let Some(ref tts) = *handle {
Expand Down
17 changes: 7 additions & 10 deletions src-tauri/src/state.rs
Original file line number Diff line number Diff line change
Expand Up @@ -285,17 +285,14 @@ impl AppState {

/// Start loading the TTS model now, before anything asks it to speak.
///
/// Only meaningful in the on-demand memory mode, where the model is not
/// resident between uses: the load takes seconds, so kicking it off the
/// moment we know speech is coming (the user has just started dictating)
/// hides most of that behind the time they spend talking. In always-loaded
/// mode the model is already there and this is a no-op.
/// The load takes seconds, so kicking it off the moment we know speech is
/// coming (the user has just started dictating, or a spoken command was
/// recognised mid-speech) hides most of that behind the time they spend
/// talking. That holds in both memory modes: on-demand drops the model
/// when idle, and always-loaded without pre-warming loads it lazily. When
/// the model is already resident the worker treats this as a no-op.
pub async fn preload_tts(&self) {
let unloads_when_idle = {
let cfg = self.config.lock().await;
cfg.data.tts.enabled && cfg.data.tts.unloads_when_idle()
};
if !unloads_when_idle {
if !self.config.lock().await.data.tts.enabled {
return;
}
if let Some(tts) = self.tts_handle.lock().await.as_ref() {
Expand Down
Loading