-
Notifications
You must be signed in to change notification settings - Fork 0
api
These are the commands the Svelte frontend (or any Tauri WebView) can call via invoke().
Source: src-tauri/src/commands.rs
import { invoke } from '@tauri-apps/api/core';Returns the current application state.
const status = await invoke<StatusPayload>('get_status');interface StatusPayload {
recording: boolean;
processing: boolean;
speaking: boolean;
mcp_recording: boolean;
audio_ready: boolean;
word_count: number;
active_target_id: string;
active_target_label: string;
}Sets the recording flag to true. The audio pipeline will start capturing.
await invoke('start_recording');Sets the recording flag to false, signaling the audio pipeline to stop.
await invoke('stop_recording');Toggles recording state. Returns the new recording state.
const nowRecording = await invoke<boolean>('toggle_recording');Returns the full application configuration.
const config = await invoke<AppConfig>('get_config');Persists configuration to ~/.config/voxctrl/config.json and emits a config-changed event to all windows.
await invoke('save_config', { newConfig: myConfig });Note the parameter name is newConfig (camelCase), not config.
Returns all output commands from targets.toml. Named targets throughout the API; "Output Commands" is the UI label for the same thing.
const targets = await invoke<OutputTarget[]>('get_targets');Writes updated targets to targets.toml, updates the in-memory cache, hot-reloads the router, and spawns any new FIFO response pipe listeners.
await invoke('save_targets', { targets: myTargets });Returns all hotkey bindings from bindings.toml.
const bindings = await invoke<HotkeyBinding[]>('get_bindings');Writes updated bindings to bindings.toml and sends a hot-reload signal to the hotkey listener thread.
await invoke('save_bindings', { bindings: myBindings });Forgets a chat target's stored conversation so the next dictation starts a new thread.
Returns how many messages were discarded. Unknown target ids return 0.
const dropped = await invoke<number>('reset_chat_conversation', { targetId: 'hermes' });Probes a chat target's GET /v1/models endpoint. Resolves with a description of the
reachable endpoint, or rejects with the failure reason. Accepts an unsaved target so the
settings UI can test edits before they are persisted.
try {
const detail = await invoke<string>('test_chat_target', { target: editingTarget });
} catch (e) {
console.error('Chat endpoint unreachable:', e);
}Opens the first-run wizard, building its window if it has been closed. A re-opened wizard starts at step one.
await invoke('open_setup_wizard');Marks setup complete (ui.setup_completed = true), persists the config, and
closes the wizard window. Pass true to open Settings afterwards.
await invoke('finish_setup_wizard', { openSettings: false });Everything first-run setup depends on, in one call: how global shortcuts are being delivered, whether text can be typed into other windows, and whether a speech model is on disk. The wizard's final screen uses it to report problems it did not itself cause.
interface SetupStatusPayload {
hotkeys: HotkeyStatusPayload;
hotkeys_active: boolean;
model_ready: boolean;
model_size: string;
model_auto_downloads: boolean; // small models fetch themselves in the background
missing_injection_tool: string | null;
pkexec_available: boolean;
manual_package_commands: string;
is_complete: boolean;
}Asks GitHub for the latest published release and compares it with the running
version. Also resolves which release file matches this installation, so
install_update does not have to fetch anything twice.
interface UpdateInfo {
version: string; // "0.4.0"
tag: string; // "v0.4.0"
current_version: string; // the version running now
notes: string; // release notes, trimmed for a dialog
release_url: string;
asset_name: string | null; // the file that would be installed
download_size: number; // bytes
can_self_update: boolean; // false for .deb / source / unwritable installs
unsupported_reason: string | null;
}
interface UpdateCheckPayload {
current_version: string;
update: UpdateInfo | null; // null when this is the latest release
skipped: boolean; // the user pressed "Skip this version" on it
}
const result = await invoke<UpdateCheckPayload>('check_for_update');What the last check found, without contacting GitHub. Returns update: null
if no check has run yet.
Downloads the pending update, verifies it against the SHA-256 digest GitHub
published, replaces the running application file and restarts into it. Emits
update-progress while downloading and update-installed just before the app
exits; on any failure it emits update-failed and leaves the running version
untouched. Rejects if no update is pending, if this installation cannot update
itself, or if an install is already running.
await invoke('install_update');Records updates.skipped_version, so this release is not raised again. A newer
one still is.
await invoke('skip_update_version', { version: '0.4.0' });Turns the launch-time check on or off and persists it (updates.auto_check).
Opens the update window (building it if needed), and closes it.
Queues text for TTS playback.
await invoke('speak_text', { text: 'Hello world', voice: 'en-us-lessac-medium' });voice is optional — omit to use the configured default.
Returns whether a Piper voice pack is available locally.
const downloaded = await invoke<boolean>('check_voice_downloaded', {
voiceName: 'en-us-lessac-medium'
});Downloads a Piper voice pack from GitHub.
await invoke('download_voice', { voiceName: 'en-us-ryan-high' });Returns the built-in Pocket-TTS voice catalogue merged with any custom .wav clips found in voiceDir ("" = default directory). A custom clip named after a built-in voice id overrides that entry's label/source instead of adding a duplicate.
const voices = await invoke<{ id: string; label: string }[]>('list_pocket_tts_voices', {
voiceDir: '',
});Returns whether the model weights, tokenizer, and the selected voice's reference clip are all present locally (no network access).
const ready = await invoke<boolean>('check_pocket_tts_ready', {
voice: 'alba',
voiceDir: '',
});Downloads the gated model weights, tokenizer, and the selected voice's reference clip. For a custom voice resolved from voiceDir, the clip is already on disk so only the model weights/tokenizer are fetched.
await invoke('download_pocket_tts', {
voice: 'alba',
voiceDir: '',
hfToken: '<your HuggingFace token>',
});Whether this build was compiled with the inflect-micro cargo feature. When false the engine can be selected and its model downloaded, but synthesis is unavailable — the Settings panel uses this to explain why Test TTS is disabled.
const available = await invoke<boolean>('inflect_micro_available');Whether both ONNX graphs and a usable phoneme table are present in modelDir ("" = default directory). The table is detected by parsing rather than by filename, so this agrees with what synthesis will actually accept.
const ready = await invoke<boolean>('check_inflect_micro_downloaded', { modelDir: '' });Downloads duration.onnx, decode.onnx, their accompanying files, and the ordered symbol list. The export's layout is discovered by listing the Hugging Face API, and the symbol list is fetched separately because it is published in a different repository from the graphs. Independent of the inflect-micro feature — the model downloads in any build.
await invoke('download_inflect_micro', { modelDir: '' });Reports what the downloaded graphs actually declare: every input and output with its element type and shape, plus the phoneme table's filename and size. Skips the contract check, so it still answers for a model whose signature does not match. Requires the inflect-micro feature.
const signature = await invoke<unknown>('inflect_micro_inspect', { modelDir: '' });Returns whether a Whisper model GGUF file is present locally.
const downloaded = await invoke<boolean>('check_model_downloaded', { modelSize: 'base' });Downloads a Whisper GGUF model.
await invoke('download_model', { modelSize: 'small' });Valid sizes: "tiny", "tiny.en", "base", "base.en", "small", "small.en", "medium", "medium.en", "large-v2", "large-v3", "large-v3-turbo"
Enables the monitoring flag so audio-level events are emitted for the VU meter.
await invoke('start_monitoring_audio');Disables monitoring and stops audio-level event streaming.
await invoke('stop_monitoring_audio');Returns all available input devices.
const devices = await invoke<AudioDeviceInfo[]>('list_audio_devices');interface AudioDeviceInfo {
index: number;
name: string;
}Pings an OpenAI-compatible API server (GET {endpoint}/v1/models) and lists
available models. apiKey is sent as a Bearer token when present; pass null
for servers that don't require authentication (e.g. a local server).
This command was previously named
test_ollama; the client speaks the OpenAI API and works with any compatible server.
const result = await invoke<OpenAiTestResult>('test_openai', {
endpoint: 'http://localhost:11434',
apiKey: null,
timeoutSecs: 5
});interface OpenAiTestResult {
success: boolean;
message: string;
models: string[];
}The commands behind Settings → Bug Report. What a report may contain, and why,
is in Bug Reports; the enforcement is in the
voxctrl-bugreport crate. None of these run on their own — every one is a
button press.
interface UserStatement {
summary: string; // one line; becomes the issue title
description: string; // what happened, what was expected
area: string; // from a fixed list in the UI
frequency: "always" | "sometimes" | "once";
}What the page needs to describe itself: whether this build can submit a report itself, where the log it quotes lives, the limits, and how many reports have gone out.
interface BugReportContext {
relay_configured: boolean; // false → the page hides Send and offers the rest
issues_new_url: string;
support_email: string;
log_path: string;
install_id: string;
limits: {
cooldown_seconds: number;
per_day: number;
per_month: number;
min_description_chars: number;
max_description_chars: number;
};
submissions_last_day: number;
submissions_last_month: number;
}Builds the report and returns it. markdown is the literal issue body, not a
summary of it — the page shows exactly this and nothing fuller is ever sent.
Sends nothing.
interface BugReportPreview {
markdown: string;
title: string;
fingerprint: string;
blocked_reason: string | null; // why the limits will not allow a send
can_submit: boolean;
github_url: string; // GitHub's new-issue form, prefilled
mailto_url: string;
}The machine facts are gathered once per run and cached: collecting them shells out to a system probe (PowerShell on Windows), and the page previews on every pause in typing.
Posts the report to the relay compiled into this build. Re-checks the limits first — the page may have been open since before the last submission — and records a submission only when one actually goes out, so a failed send does not spend the reporter's allowance.
interface BugReportOutcome {
ok: boolean;
issue_url: string | null;
message: string; // shown verbatim; the relay may write it
}Writes the report as Markdown to path. Never rate-limited, never needs a
network. Returns the path written.
A timestamped filename for the save dialog.
Throws away the installation identifier and the local submission history, and returns the new identifier.
Makes the overlay window visible and sets always-on-top.
await invoke('show_overlay');Hides the overlay window.
await invoke('hide_overlay');Subscribe with listen() from @tauri-apps/api/event.
import { listen } from '@tauri-apps/api/event';Emitted whenever the application state changes, and otherwise as a heartbeat under a second apart. The backend compares each payload against the last and skips sending an identical one, so an idle app is quiet without the frontend's staleness fallback ever tripping.
await listen<AppStatus>('status-tick', (event) => {
console.log(event.payload.recording);
});Emitted when the config is saved (from any window or external change).
await listen<AppConfig>('config-changed', (event) => {
config.set(event.payload);
});Emitted with the current RMS energy level (0.0–1.0+) while recording or monitoring — the overlay visualisers and the Audio tab's VU meter respectively. Nothing is emitted when neither is watching.
The microphone produces a level for every buffer it delivers, which is far more often than anything can draw; they are coalesced to at most one event per frame (16 ms), newest value winning.
await listen<number>('audio-level', (event) => {
updateVuMeter(event.payload);
});Emitted while an update downloads, at most once per megabyte.
await listen<{ downloaded: number; total: number }>('update-progress', (event) => {
// total is 0 when the server sent no content length
});Emitted with the new version once it is in place. The app exits shortly after and the new build starts itself.
Emitted with a message when an update could not be installed. The running version is unchanged.
These types are defined in src/stores/config.ts:
interface AppConfig {
engine: EngineConfig;
audio: AudioConfig;
ui: UiConfig;
features: FeaturesConfig;
openai: OpenAiConfig;
tts: TtsConfig;
mcp: McpConfig;
}
interface EngineConfig {
backend: "whisper-cpp" | "moonshine"; // a legacy "auto" loads as whisper-cpp
whisper_cpp: WhisperCppConfig;
moonshine: MoonshineConfig;
}
interface WhisperCppConfig {
model_dir: string;
model_size: string;
device: string;
threads: number;
}
interface MoonshineConfig {
model_size: string;
language: string;
}
interface AudioConfig {
vad_threshold: number;
input_device_index: number | null;
evdev_device: string | null;
noise_suppression: boolean;
gain: number;
dynamic_stream: boolean;
}
interface UiConfig {
show_overlay: boolean;
overlay_style: "voice_card" | "waveform" | "pulse" | "blue_wave" | "none";
overlay_position: string;
overlay_monitor: string;
auto_show_settings: boolean;
setup_completed: boolean;
show_notification: boolean;
}
interface FeaturesConfig {
remove_fillers: boolean;
custom_vocabulary: string[];
spoken_punctuation: boolean;
auto_format_lists: boolean;
snippets: Record<string, string>;
}
interface OpenAiConfig {
enabled: boolean;
model: string;
mode: "clean" | "formal" | "casual" | "bullet" | "concise" | "custom"; // GUI preset that fills system_prompt
custom_prompt: string | null; // legacy; migrated into user_prompt on load
system_prompt: string; // system message (empty = none)
user_prompt: string; // user message template; must contain "{text}"
endpoint: string; // OpenAI-compatible API base URL (a `/v1` suffix is optional)
api_key: string | null; // sent as a Bearer token when set
timeout_secs: number;
}
interface PocketTtsConfig {
voice: string;
prewarm: boolean;
voice_dir: string; // custom .wav voice clips; empty = default directory
}
interface InflectMicroConfig {
model_dir: string; // empty = default directory
seed: number; // deterministic for a fixed seed
noise_scale: number; // 0.0-1.0 variation, default 0.667
prewarm: boolean;
}
interface BreezeTts2Config {
voice_mode: "prompt" | "clone";
cloned_voice: string; // voice id from the shared clip folder
voice_dir: string; // shared with pocket_tts; empty = default directory
speaker_prompt: string; // Voice Design description
model_dir: string; // empty = default directory
prewarm: boolean;
gpu: boolean; // needs a breeze-cuda / breeze-metal build
}
interface TtsConfig {
enabled: boolean;
engine: "piper" | "espeak" | "pocket_tts" | "inflect_micro" | "breeze_tts_2";
voice: string;
voice_dir: string;
stop_key: string[]; // singular field name, plural value
response_overlay: boolean;
speed: number; // not used by pocket_tts
gpu: boolean; // only applies to piper; Breeze has its own flag
hf_token: string | null; // one token for every gated model download;
// an exported HF_TOKEN wins and is never saved here
pocket_tts: PocketTtsConfig;
inflect_micro: InflectMicroConfig; // fixed-voice, so no voice field
breeze_tts_2: BreezeTts2Config;
memory_mode: "always_loaded" | "on_demand"; // "on_demand" unloads the model when idle
idle_unload_secs: number; // idle seconds before unloading in "on_demand" mode (default 900)
snippets: Record<string, string>; // pronunciation guide, speech only
}
interface McpConfig {
server_enabled: boolean; // not "enabled"
record_timeout: number; // default for transcribe_voice, read per call
visual_feedback: boolean;
}
interface OutputTarget {
id: string;
label: string;
delivery: "inject" | "clipboard" | "exec" | "pipe" | "socket" | "file" | "dbus" | "http" | "webhook" | "mcp" | "speak";
// exec
command?: string;
// pipe
pipe_path?: string;
// socket (unix or TCP)
socket_unix?: string;
socket_host?: string;
socket_port?: number;
// file
file_path?: string;
file_prefix: string;
file_timestamp: boolean;
file_timestamp_format: string; // strftime, UTC; default "%Y-%m-%dT%H:%M:%SZ"
file_mode: string; // "append" or "write"
// dbus
dbus_signal?: string;
// http
http_url?: string;
http_method: string;
// webhook (note: webhook_url, not http_url)
webhook_url?: string;
webhook_secret?: string;
// mcp
mcp_path?: string;
mcp_tool?: string;
// chat (OpenAI-compatible /v1/chat/completions, with conversation history)
chat_url?: string;
chat_model?: string;
chat_api_key?: string;
chat_system_prompt?: string;
chat_max_history: number; // default: 20 (0 = send the whole conversation)
chat_timeout_secs: number; // default: 120
chat_reply_mode: string; // "speak" | "inject" | "clipboard" | "none"
chat_reset_phrase?: string;
strip_newlines: boolean; // default: false; inject and command targets
processing: TargetProcessingConfig;
response_pipe?: string;
}
interface TargetProcessingConfig {
remove_fillers?: boolean;
spoken_punctuation?: boolean;
auto_format_lists?: boolean;
code_mode?: boolean;
}
interface HotkeyBinding {
id: string;
label: string;
keys: string[];
gesture: "hold" | "toggle" | "double_tap" | "double_tap_hold";
target_id: string;
target_ids: string[];
tap_ms: number; // default: 250
hold_threshold_ms: number;// default: 200
disabled: boolean;
openai_enabled?: boolean; // legacy alias: ollama_enabled
openai_model?: string; // legacy alias: ollama_model
openai_mode?: string; // legacy alias: ollama_mode
openai_prompt?: string; // user prompt template override (must contain "{text}"); legacy alias: ollama_prompt
openai_system_prompt?: string; // system prompt override (empty = inherit global default); legacy alias: ollama_system_prompt
}
interface AppStatus {
recording: boolean;
processing: boolean;
speaking: boolean;
mcp_recording: boolean;
audio_ready?: boolean;
word_count: number;
active_target_id?: string;
active_target_label?: string;
}