Repository navigation
api
These are the commands the Svelte frontend (or any Tauri WebView) can call via invoke().
Source: src-tauri/src/commands.rs
import { invoke } from '@tauri-apps/api/core';Returns the current application state.
const status = await invoke<StatusPayload>('get_status');interface StatusPayload {
recording: boolean;
processing: boolean;
speaking: boolean;
mcp_recording: boolean;
audio_ready: boolean;
word_count: number;
active_target_id: string;
active_target_label: string;
}Sets the recording flag to true. The audio pipeline will start capturing.
await invoke('start_recording');Sets the recording flag to false, signaling the audio pipeline to stop.
await invoke('stop_recording');Toggles recording state. Returns the new recording state.
const nowRecording = await invoke<boolean>('toggle_recording');Returns the full application configuration.
const config = await invoke<AppConfig>('get_config');Persists configuration to ~/.config/voxctrl/config.json and emits a config-changed event to all windows.
await invoke('save_config', { newConfig: myConfig });Note the parameter name is newConfig (camelCase), not config.
Returns all output commands from targets.toml. Named targets throughout the API; "Output Commands" is the UI label for the same thing.
const targets = await invoke<OutputTarget[]>('get_targets');Writes updated targets to targets.toml, updates the in-memory cache, hot-reloads the router, and spawns any new FIFO response pipe listeners.
await invoke('save_targets', { targets: myTargets });Returns all hotkey bindings from bindings.toml.
const bindings = await invoke<HotkeyBinding[]>('get_bindings');Writes updated bindings to bindings.toml and sends a hot-reload signal to the hotkey listener thread.
await invoke('save_bindings', { bindings: myBindings });Forgets a chat target's stored conversation so the next dictation starts a new thread.
Returns how many messages were discarded. Unknown target ids return 0.
const dropped = await invoke<number>('reset_chat_conversation', { targetId: 'hermes' });Probes a chat target's GET /v1/models endpoint. Resolves with a description of the
reachable endpoint, or rejects with the failure reason. Accepts an unsaved target so the
settings UI can test edits before they are persisted.
try {
const detail = await invoke<string>('test_chat_target', { target: editingTarget });
} catch (e) {
console.error('Chat endpoint unreachable:', e);
}Opens the first-run wizard, building its window if it has been closed. A re-opened wizard starts at step one.
await invoke('open_setup_wizard');Marks setup complete (ui.setup_completed = true), persists the config, and
closes the wizard window. Pass true to open Settings afterwards.
await invoke('finish_setup_wizard', { openSettings: false });Everything first-run setup depends on, in one call: how global shortcuts are being delivered, whether text can be typed into other windows, and whether a speech model is on disk. The wizard's final screen uses it to report problems it did not itself cause.
interface SetupStatusPayload {
hotkeys: HotkeyStatusPayload;
hotkeys_active: boolean;
model_ready: boolean;
model_size: string;
model_auto_downloads: boolean; // small models fetch themselves in the background
missing_injection_tool: string | null;
pkexec_available: boolean;
manual_package_commands: string;
is_complete: boolean;
}VoxCtrl never checks on its own; every command below only runs because the user triggered it from Settings → General or the update window.
Asks GitHub for the latest published release and compares it with the running
version, and gathers the release notes of every release in between into notes. Also resolves which release file matches this installation, so
install_update does not have to fetch anything twice.
interface UpdateInfo {
version: string; // "0.4.0"
tag: string; // "v0.4.0"
current_version: string; // the version running now
notes: string; // notes for each release newer than yours, up to the latest, newest first
release_url: string;
asset_name: string | null; // the file that would be installed
download_size: number; // bytes
can_self_update: boolean; // false for .deb / source / unwritable installs
unsupported_reason: string | null;
}
interface UpdateCheckPayload {
current_version: string;
update: UpdateInfo | null; // null when this is the latest release
}
const result = await invoke<UpdateCheckPayload>('check_for_update');What the last check found, without contacting GitHub. Returns update: null
if no check has run yet.
Downloads the pending update, verifies it against the SHA-256 digest GitHub
published, replaces the running application file and restarts into it. Emits
update-progress while downloading and update-installed just before the app
exits; on any failure it emits update-failed and leaves the running version
untouched. Rejects if no update is pending, if this installation cannot update
itself, or if an install is already running.
await invoke('install_update');Opens the update window (building it if needed), and closes it.
Queues text for TTS playback.
await invoke('speak_text', { text: 'Hello world', voice: 'en-us-lessac-medium' });voice is optional — omit to use the configured default.
Returns whether a Piper voice pack is available locally.
const downloaded = await invoke<boolean>('check_voice_downloaded', {
voiceName: 'en-us-lessac-medium'
});Downloads a Piper voice pack from GitHub.
await invoke('download_voice', { voiceName: 'en-us-ryan-high' });Returns the built-in Pocket-TTS voice catalogue merged with any custom .wav clips found in voiceDir ("" = default directory). A custom clip named after a built-in voice id overrides that entry's label/source instead of adding a duplicate.
const voices = await invoke<{ id: string; label: string }[]>('list_pocket_tts_voices', {
voiceDir: '',
});Returns whether the model weights, tokenizer, and the selected voice's reference clip are all present locally (no network access).
const ready = await invoke<boolean>('check_pocket_tts_ready', {
voice: 'alba',
voiceDir: '',
});Downloads the gated model weights, tokenizer, and the selected voice's reference clip. For a custom voice resolved from voiceDir, the clip is already on disk so only the model weights/tokenizer are fetched.
await invoke('download_pocket_tts', {
voice: 'alba',
voiceDir: '',
hfToken: '<your HuggingFace token>',
});Whether this build was compiled with the inflect-micro cargo feature. When false the engine can be selected and its model downloaded, but synthesis is unavailable — the Settings panel uses this to explain why Test TTS is disabled.
const available = await invoke<boolean>('inflect_micro_available');Whether both ONNX graphs and a usable phoneme table are present in modelDir ("" = default directory). The table is detected by parsing rather than by filename, so this agrees with what synthesis will actually accept.
const ready = await invoke<boolean>('check_inflect_micro_downloaded', { modelDir: '' });Downloads duration.onnx, decode.onnx, their accompanying files, and the ordered symbol list. The export's layout is discovered by listing the Hugging Face API, and the symbol list is fetched separately because it is published in a different repository from the graphs. Independent of the inflect-micro feature — the model downloads in any build.
await invoke('download_inflect_micro', { modelDir: '' });Reports what the downloaded graphs actually declare: every input and output with its element type and shape, plus the phoneme table's filename and size. Skips the contract check, so it still answers for a model whose signature does not match. Requires the inflect-micro feature.
const signature = await invoke<unknown>('inflect_micro_inspect', { modelDir: '' });Returns whether Breeze-TTS-2 model weights and tokenizer exist locally.
Downloads gated Breeze-TTS-2 model weights from HuggingFace using the provided token.
Returns whether VoxCPM2 model assets (config.json, generation_config.json, weights) exist locally.
Downloads VoxCPM2 model assets from HuggingFace.
Returns any HF_TOKEN exported in the application environment (takes precedence over UI config).
Returns whether a Whisper model GGUF file is present locally.
const downloaded = await invoke<boolean>('check_model_downloaded', { modelSize: 'base' });Downloads a Whisper GGUF model.
await invoke('download_model', { modelSize: 'small' });Valid sizes: "tiny", "tiny.en", "base", "base.en", "small", "small.en", "medium", "medium.en", "large-v2", "large-v3", "large-v3-turbo"
Downloads the Parakeet TDT ONNX model files.
Checks whether the S1-mini GGUF model and tokenizer are downloaded locally.
Downloads the S1-mini GGUF model (~480 MB) and tokenizer assets for on-device dictation cleanup.
test_remote_stt(endpoint: string, apiKey: string | null, model: string, timeoutSecs: number) → RemoteSttTestResult
Tests connection to a Remote Speech Engine (OpenAI-compatible /v1/audio/transcriptions API) with sample audio, and queries available models.
interface RemoteSttTestResult {
success: boolean;
message: string;
latency_ms: number;
models: string[];
}Enables the monitoring flag so audio-level events are emitted for the VU meter.
await invoke('start_monitoring_audio');Disables monitoring and stops audio-level event streaming.
await invoke('stop_monitoring_audio');Returns all available input devices.
const devices = await invoke<AudioDeviceInfo[]>('list_audio_devices');interface AudioDeviceInfo {
index: number;
name: string;
}Pings an OpenAI-compatible API server (GET {endpoint}/v1/models) and lists
available models. apiKey is sent as a Bearer token when present; pass null
for servers that don't require authentication (e.g. a local server).
This command was previously named
test_ollama; the client speaks the OpenAI API and works with any compatible server.
const result = await invoke<OpenAiTestResult>('test_openai', {
endpoint: 'http://localhost:11434',
apiKey: null,
timeoutSecs: 5
});interface OpenAiTestResult {
success: boolean;
message: string;
models: string[];
}The commands behind Settings → Bug Report. What a report may contain, and why,
is in Bug Reports; the enforcement is in the
voxctrl-bugreport crate. None of these run on their own — every one is a
button press.
interface UserStatement {
summary: string; // one line; becomes the issue title
description: string; // what happened, what was expected
area: string; // from a fixed list in the UI
frequency: "always" | "sometimes" | "once";
}What the page needs to describe itself: whether this build can submit a report itself, where the log it quotes lives, the limits, and how many reports have gone out.
interface BugReportContext {
relay_configured: boolean; // false → the page hides Send and offers the rest
issues_new_url: string;
support_email: string;
log_path: string;
install_id: string;
limits: {
cooldown_seconds: number;
per_day: number;
per_month: number;
min_description_chars: number;
max_description_chars: number;
};
submissions_last_day: number;
submissions_last_month: number;
}Builds the report and returns it. markdown is the literal issue body, not a
summary of it — the page shows exactly this and nothing fuller is ever sent.
Sends nothing.
interface BugReportPreview {
markdown: string;
title: string;
fingerprint: string;
blocked_reason: string | null; // why the limits will not allow a send
can_submit: boolean;
github_url: string; // GitHub's new-issue form, prefilled
mailto_url: string;
}The machine facts are gathered once per run and cached: collecting them shells out to a system probe (PowerShell on Windows), and the page previews on every pause in typing.
Posts the report to the relay compiled into this build. Re-checks the limits first — the page may have been open since before the last submission — and records a submission only when one actually goes out, so a failed send does not spend the reporter's allowance.
interface BugReportOutcome {
ok: boolean;
issue_url: string | null;
message: string; // shown verbatim; the relay may write it
}Writes the report as Markdown to path. Never rate-limited, never needs a
network. Returns the path written.
A timestamped filename for the save dialog.
Throws away the installation identifier and the local submission history, and returns the new identifier.
The overlay window has no explicit show/hide command — it is created and
destroyed entirely from the Rust backend (tray::spawn_status_ticker, which
polls the same recording/speaking/MCP-recording state and ui.show_overlay /
tts.response_overlay / mcp.visual_feedback config that decides everything
else that reacts to them), not requested by the frontend. Once the window
exists, its content is driven by the status-tick / audio-level events
the same way every other consumer of those already reacts to them. See the
Overlay UI Guide.
Lists every user-authored custom overlay folder under the local overlays
directory (see get_custom_overlays_dir), alongside the built-in styles in
the ui.overlay_style picker.
interface CustomOverlayInfo {
name: string;
html: string;
css: string;
}
const overlays = await invoke<CustomOverlayInfo[]>('get_custom_overlays');Re-reads a single custom overlay folder by its display name — used to reload just the active style rather than the whole list.
const overlay = await invoke<CustomOverlayInfo | null>('get_custom_overlay', { name: 'Custom' });The resolved, absolute path to the custom-overlays folder, for display in Settings.
const dir = await invoke<string>('get_custom_overlays_dir');Whether pasting dictation is available on this system. supported: false (Linux
Mint) means text is always typed; reason is the sentence Settings shows next to
the disabled toggle. See Pasting Dictation.
Returns comprehensive first-run readiness: shortcut health, active speech model status, missing injection tools (wtype/xdotool), and polkit privileges.
Downloads the speech model selected by the active configuration.
Spawns the first-run Setup Wizard window on demand.
Marks setup complete (ui.setup_completed = true), emits config-changed, closes the wizard, and optionally opens Settings.
Opens the Settings window and switches directly to the specified tab ("general", "engine", "hotkeys", "audio", "tts", "features", "visual", "openai", "targets", "bugreport").
Queries connected display monitors (names, dimensions, primary flag) on the main thread.
interface MonitorInfo {
name: string | null;
width: number;
height: number;
is_primary: boolean;
}Subscribe with listen() from @tauri-apps/api/event.
import { listen } from '@tauri-apps/api/event';Emitted whenever the application state changes, and otherwise as a heartbeat under a second apart. The backend compares each payload against the last and skips sending an identical one, so an idle app is quiet without the frontend's staleness fallback ever tripping.
await listen<AppStatus>('status-tick', (event) => {
console.log(event.payload.recording);
});Emitted when the config is saved (from any window or external change).
await listen<AppConfig>('config-changed', (event) => {
config.set(event.payload);
});Emitted with the current RMS energy level (0.0–1.0+) while recording or monitoring — the overlay visualisers and the Audio Input tab's VU meter respectively. Nothing is emitted when neither is watching.
The microphone produces a level for every buffer it delivers, which is far more often than anything can draw; they are coalesced to at most one event per frame (16 ms), newest value winning.
await listen<number>('audio-level', (event) => {
updateVuMeter(event.payload);
});Emitted a few times a second while recording, only when the active custom
overlay contains a data-voxctrl-live-text element (see
Live transcript text). session_id
increases with each recording; a payload from an earlier recording than the one
already shown is ignored.
await listen<{ session_id: number; text: string }>('live-transcript', (event) => {
liveText = event.payload.text;
});The text is the interim transcript after VoxCtrl's normal clean-up (fillers, spoken punctuation, snippets, lists). It is not processed by S1-mini or the hotkey's OpenAI rewrite, and a spoken voice-command trigger is still in it, so it can differ from what is finally delivered.
Emitted while an update downloads, at most once per megabyte.
await listen<{ downloaded: number; total: number }>('update-progress', (event) => {
// total is 0 when the server sent no content length
});Emitted with the new version once it is in place. The app exits shortly after and the new build starts itself.
Emitted with a message when an update could not be installed. The running version is unchanged.
These types are defined in src/stores/config.ts:
interface AppConfig {
engine: EngineConfig;
audio: AudioConfig;
ui: UiConfig;
features: FeaturesConfig;
openai: OpenAiConfig;
tts: TtsConfig;
mcp: McpConfig;
}
interface EngineConfig {
backend: "whisper-cpp" | "moonshine" | "parakeet" | "remote-openai";
whisper_cpp: WhisperCppConfig;
moonshine: MoonshineConfig;
parakeet: ParakeetConfig;
remote_openai: RemoteOpenAiConfig;
s1_mini: S1MiniConfig;
}
interface WhisperCppConfig {
model_dir: string;
model_size: string;
device: string;
threads: number;
}
interface MoonshineConfig {
model_size: string;
language: string;
}
interface ParakeetConfig {
model_size: string;
language: string;
}
interface RemoteOpenAiConfig {
endpoint: string;
api_key: string | null;
model: string;
language: string;
timeout_secs: number;
}
interface S1MiniConfig {
enabled: boolean;
styling: string;
}
interface AudioConfig {
vad_threshold: number;
input_device_index: number | null;
evdev_device: string | null;
noise_suppression: boolean;
gain: number;
dynamic_stream: boolean;
}
interface UiConfig {
show_overlay: boolean;
overlay_style: "voice_card" | "waveform" | "pulse" | "blue_wave" | "none";
overlay_position: string;
overlay_monitor: string;
auto_show_settings: boolean;
setup_completed: boolean;
show_notification: boolean;
}
interface FeaturesConfig {
remove_fillers: boolean;
custom_vocabulary: string[];
spoken_punctuation: boolean;
auto_format_lists: boolean;
snippets: Record<string, string>;
early_command_detection: boolean;
}
interface OpenAiConfig {
enabled: boolean;
model: string;
mode: "clean" | "formal" | "casual" | "bullet" | "concise" | "custom"; // GUI preset that fills system_prompt
custom_prompt: string | null; // legacy; migrated into user_prompt on load
system_prompt: string; // system message (empty = none)
user_prompt: string; // user message template; must contain "{text}"
endpoint: string; // OpenAI-compatible API base URL (a `/v1` suffix is optional)
api_key: string | null; // sent as a Bearer token when set
timeout_secs: number;
}
interface PocketTtsConfig {
voice: string;
prewarm: boolean;
voice_dir: string; // custom .wav voice clips; empty = default directory
gpu: boolean; // Vulkan, via the audio.cpp subprocess
}
interface InflectMicroConfig {
model_dir: string; // empty = default directory
seed: number; // deterministic for a fixed seed
noise_scale: number; // 0.0-1.0 variation, default 0.667
prewarm: boolean;
}
interface BreezeTts2Config {
voice_mode: "prompt" | "clone";
cloned_voice: string; // voice id from the shared clip folder
voice_dir: string; // shared with pocket_tts; empty = default directory
speaker_prompt: string; // Voice Design description
model_dir: string; // empty = default directory
prewarm: boolean;
gpu: boolean; // Vulkan, via the audio.cpp subprocess
}
interface VoxCpm2Config {
voice_mode: "prompt" | "clone";
speaker_prompt: string;
cloned_voice: string;
voice_dir: string;
ultimate_cloning: boolean;
model_dir: string;
prewarm: boolean;
gpu: boolean;
}
interface TtsConfig {
enabled: boolean;
engine: "piper" | "espeak" | "pocket_tts" | "inflect_micro" | "breeze_tts_2" | "vox_cpm_2";
voice: string;
voice_dir: string;
stop_key: string[]; // singular field name, plural value
response_overlay: boolean;
speed: number; // not used by pocket_tts
gpu: boolean; // applies to piper; Breeze and VoxCPM2 have their own flag
hf_token: string | null; // one token for every gated model download
pocket_tts: PocketTtsConfig;
inflect_micro: InflectMicroConfig;
breeze_tts_2: BreezeTts2Config;
vox_cpm_2: VoxCpm2Config;
memory_mode: "always_loaded" | "on_demand"; // "on_demand" unloads model after 15 min idle
idle_unload_secs: number; // idle seconds before unloading (default 900)
snippets: Record<string, string>; // pronunciation guide, speech only
}
interface McpConfig {
server_enabled: boolean; // not "enabled"
record_timeout: number; // default for transcribe_voice, read per call
visual_feedback: boolean;
}
interface OutputTarget {
id: string;
label: string;
delivery: "inject" | "clipboard" | "exec" | "pipe" | "socket" | "file" | "dbus" | "http" | "webhook" | "mcp" | "speak" | "chat" | "command";
// exec
command?: string;
// pipe
pipe_path?: string;
// socket (unix or TCP)
socket_unix?: string;
socket_host?: string;
socket_port?: number;
// file
file_path?: string;
file_prefix: string;
file_timestamp: boolean;
file_timestamp_format: string; // strftime, UTC; default "%Y-%m-%dT%H:%M:%SZ"
file_mode: string; // "append" or "write"
// dbus
dbus_signal?: string;
// http / webhook
http_url?: string;
http_method?: string;
webhook_url?: string;
webhook_secret?: string;
// mcp
mcp_path?: string;
mcp_tool?: string;
// chat (conversational LLM)
chat_url?: string;
chat_model?: string;
chat_system_prompt?: string;
chat_reply_mode?: "speak" | "inject" | "clipboard" | "none";
chat_max_history?: number;
chat_timeout_secs?: number;
chat_reset_phrase?: string;
chat_api_key?: string;
strip_newlines?: boolean; // default: false; inject and command targets
processing?: TargetProcessingConfig;
response_pipe?: string;
}
interface TargetProcessingConfig {
remove_fillers?: boolean;
spoken_punctuation?: boolean;
auto_format_lists?: boolean;
code_mode?: boolean;
}
interface HotkeyBinding {
id: string;
label: string;
keys: string[];
gesture: "hold" | "toggle" | "double_tap" | "double_tap_hold";
target_id: string;
target_ids: string[];
tap_ms: number; // default: 250
hold_threshold_ms: number;// default: 200
disabled: boolean;
openai_enabled?: boolean; // legacy alias: ollama_enabled
openai_model?: string; // legacy alias: ollama_model
openai_mode?: string; // legacy alias: ollama_mode
openai_prompt?: string; // user prompt template override (must contain "{text}"); legacy alias: ollama_prompt
openai_system_prompt?: string; // system prompt override (empty = inherit global default); legacy alias: ollama_system_prompt
}
interface AppStatus {
recording: boolean;
processing: boolean;
speaking: boolean;
mcp_recording: boolean;
audio_ready?: boolean;
word_count: number;
active_target_id?: string;
active_target_label?: string;
}