| 1 | //! Voice input commands — `/voice`, `/voice-send`, `/voice-control`. |
| 2 | //! |
| 3 | //! Records audio from the default microphone, sends it to the configured |
| 4 | //! provider's API for transcription, and inserts the transcribed text into |
| 5 | //! the composer. The interaction model mirrors MiMo Code's voice UX: |
| 6 | //! |
| 7 | //! `/voice` — toggle voice input on/off (records when toggled on) |
| 8 | //! `/voice-send` — toggle auto-send when the transcript ends with |
| 9 | //! "send it" / "发送" |
| 10 | //! `/voice-control` — toggle AI-assisted dictation that sees the current |
| 11 | //! composer text |
| 12 | //! |
| 13 | //! The slash commands only flip state and emit [`AppAction::VoiceCapture`]; |
| 14 | //! the actual capture runs in the UI event loop where the live [`Config`] |
| 15 | //! supplies provider credentials. That keeps the handlers side-effect free |
| 16 | //! (the registry smoke tests execute every command) and avoids caching |
| 17 | //! auth material on [`App`]. |
| 18 | //! |
| 19 | //! Recording, transcription and the headless dictation cycle live in |
| 20 | //! `crate::voice`; this module keeps the slash commands and the capture loop |
| 21 | //! that updates the composer while the user speaks. |
| 22 | |
| 23 | use std::time::Duration; |
| 24 | |
| 25 | use crate::commands::CommandResult; |
| 26 | use crate::commands::traits::{CommandInfo, RegisterCommand}; |
| 27 | use crate::config::Config; |
| 28 | use crate::tui::app::{App, AppAction}; |
| 29 | use crate::voice::{ |
| 30 | is_available, record_audio, resolve_asr_choice, split_send_suffix, transcribe_selected, |
| 31 | }; |
| 32 | use codewhale_localization::{MessageId, tr}; |
| 33 | |
| 34 | pub(in crate::commands) const VOICE_INFO: CommandInfo = CommandInfo { |
| 35 | name: "voice", |
| 36 | aliases: &["yuyin", "语音"], |
| 37 | usage: "/voice", |
| 38 | description_id: MessageId::CmdVoiceDescription, |
| 39 | }; |
| 40 | |
| 41 | pub(in crate::commands) const VOICE_SEND_INFO: CommandInfo = CommandInfo { |
| 42 | name: "voicesend", |
| 43 | aliases: &["voice-send", "yuyinsend", "语音发送"], |
| 44 | usage: "/voicesend", |
| 45 | description_id: MessageId::CmdVoiceSendDescription, |
| 46 | }; |
| 47 | |
| 48 | pub(in crate::commands) const VOICE_CONTROL_INFO: CommandInfo = CommandInfo { |
| 49 | name: "voicecontrol", |
| 50 | aliases: &["voice-control", "yuyincontrol", "语音控制"], |
| 51 | usage: "/voicecontrol", |
| 52 | description_id: MessageId::CmdVoiceControlDescription, |
| 53 | }; |
| 54 | |
| 55 | pub(in crate::commands) struct VoiceCmd; |
| 56 | pub(in crate::commands) struct VoiceSendCmd; |
| 57 | pub(in crate::commands) struct VoiceControlCmd; |
| 58 | |
| 59 | impl RegisterCommand for VoiceCmd { |
| 60 | fn info() -> &'static CommandInfo { |
| 61 | &VOICE_INFO |
| 62 | } |
| 63 | |
| 64 | fn execute(app: &mut App, _arg: Option<&str>) -> CommandResult { |
| 65 | voice(app) |
| 66 | } |
| 67 | } |
| 68 | |
| 69 | impl RegisterCommand for VoiceSendCmd { |
| 70 | fn info() -> &'static CommandInfo { |
| 71 | &VOICE_SEND_INFO |
| 72 | } |
| 73 | |
| 74 | fn execute(app: &mut App, _arg: Option<&str>) -> CommandResult { |
| 75 | voice_send(app) |
| 76 | } |
| 77 | } |
| 78 | |
| 79 | impl RegisterCommand for VoiceControlCmd { |
| 80 | fn info() -> &'static CommandInfo { |
| 81 | &VOICE_CONTROL_INFO |
| 82 | } |
| 83 | |
| 84 | fn execute(app: &mut App, _arg: Option<&str>) -> CommandResult { |
| 85 | voice_control(app) |
| 86 | } |
| 87 | } |
| 88 | |
| 89 | // --- Capture orchestration (UI event loop) --------------------------------- |
| 90 | |
| 91 | /// What the UI should do with a finished capture. |
| 92 | #[derive(Debug, Clone, PartialEq, Eq)] |
| 93 | pub enum VoiceCaptureOutcome { |
| 94 | /// Insert the transcribed text into the composer at the cursor. |
| 95 | Insert(String), |
| 96 | /// Submit this text as a message (auto-send). |
| 97 | Send(String), |
| 98 | } |
| 99 | |
| 100 | /// Status line while recording: the localized recording label, the latest |
| 101 | /// interim transcript once one exists, and how to stop. The capture is awaited |
| 102 | /// on the UI loop, so no key can end it — `record_audio` stops after a second |
| 103 | /// of silence (or `MAX_RECORD_SECS`), and the cue says exactly that. |
| 104 | fn recording_status(locale: codewhale_localization::Locale, interim: Option<&str>) -> String { |
| 105 | let label = tr(locale, MessageId::VoiceRecording); |
| 106 | let stop = tr(locale, MessageId::VoiceRecordingStopHint); |
| 107 | match interim.map(str::trim).filter(|text| !text.is_empty()) { |
| 108 | Some(text) => format!("{label} \u{2014} \u{201c}{text}\u{201d} \u{00b7} {stop}"), |
| 109 | None => format!("{label} \u{00b7} {stop}"), |
| 110 | } |
| 111 | } |
| 112 | |
| 113 | /// Perform a complete record + transcribe cycle with live interim display. |
| 114 | /// |
| 115 | /// Runs in the UI event loop (see [`AppAction::VoiceCapture`]) so provider |
| 116 | /// credentials come from the live [`Config`] rather than state cached on |
| 117 | /// [`App`]. Recording happens on a blocking thread; transcription uses the |
| 118 | /// shared async HTTP client. Every failure path returns a localized message |
| 119 | /// so callers can surface it as a status line. |
| 120 | pub async fn capture_and_transcribe( |
| 121 | app: &mut App, |
| 122 | config: &Config, |
| 123 | ) -> Result<VoiceCaptureOutcome, String> { |
| 124 | let locale = app.ui_locale; |
| 125 | |
| 126 | if !is_available() { |
| 127 | return Err(tr(locale, MessageId::VoiceErrNoRecorder).to_string()); |
| 128 | } |
| 129 | // C01-10: the credential check follows the ASR backend that will run, |
| 130 | // not the chat provider. Local Whisper and Groq need no provider key, so |
| 131 | // resolve it lazily — the same contract as `crate::voice::dictate_once` |
| 132 | // — and only refuse before recording when provider ASR is the backend. |
| 133 | let provider_key = || { |
| 134 | config |
| 135 | .active_route_api_key() |
| 136 | .map_err(|_| tr(locale, MessageId::VoiceErrNoAuth).to_string()) |
| 137 | }; |
| 138 | let (asr_kind, _asr_model) = resolve_asr_choice(config); |
| 139 | if asr_kind == "provider" { |
| 140 | provider_key()?; |
| 141 | } |
| 142 | // Show the localized recording status plus the live interim in the composer. |
| 143 | let original_input = app.composer.input.clone(); |
| 144 | let original_cursor = app.composer.cursor_position; |
| 145 | app.status_message = Some(recording_status(locale, None)); |
| 146 | |
| 147 | // Streaming interim: poll every 700ms and show partial transcript like Grok Build's |
| 148 | // VoiceEvent::Interim → VoiceState::Recording{interim}. We re-transcribe the |
| 149 | // growing buffer (local-whisper is cheap; Groq is ~300ms; provider falls back). |
| 150 | let interim_enabled = true; // always show partials — feels alive like Spark |
| 151 | |
| 152 | // Spawn recorder on blocking thread with a shared buffer for interim polling. |
| 153 | let shared_buf: std::sync::Arc<parking_lot::Mutex<Vec<i16>>> = |
| 154 | std::sync::Arc::new(parking_lot::Mutex::new(Vec::new())); |
| 155 | let shared_done = std::sync::Arc::new(std::sync::atomic::AtomicBool::new(false)); |
| 156 | let shared_buf_clone = std::sync::Arc::clone(&shared_buf); |
| 157 | let shared_done_clone = std::sync::Arc::clone(&shared_done); |
| 158 | let recorder_handle = tokio::task::spawn_blocking(move || { |
| 159 | // Bridge to existing record_audio but copy into shared buffer incrementally. |
| 160 | // For now we reuse the blocking recorder and then publish; interim will |
| 161 | // poll the final buffer. A true streaming recorder (cpal/pw-record) is |
| 162 | // the next step — see grokbuild's xai-grok-voice::audio for the subprocess |
| 163 | // isolation pattern we should mirror. |
| 164 | let result = record_audio(); |
| 165 | if let Some((samples, dur)) = result { |
| 166 | *shared_buf_clone.lock() = samples.clone(); |
| 167 | shared_done_clone.store(true, std::sync::atomic::Ordering::SeqCst); |
| 168 | Some((samples, dur)) |
| 169 | } else { |
| 170 | shared_done_clone.store(true, std::sync::atomic::Ordering::SeqCst); |
| 171 | None |
| 172 | } |
| 173 | }); |
| 174 | |
| 175 | // Interim polling loop — updates composer with "original + interim ▍" so text |
| 176 | // appears as you talk, just like Spark's live transcript. |
| 177 | let mut last_interim = String::new(); |
| 178 | let mut ticks: u32 = 0; |
| 179 | loop { |
| 180 | tokio::time::sleep(Duration::from_millis(700)).await; |
| 181 | ticks += 1; |
| 182 | if shared_done.load(std::sync::atomic::Ordering::SeqCst) { |
| 183 | break; |
| 184 | } |
| 185 | if !interim_enabled || ticks < 2 { |
| 186 | continue; // let a little audio accumulate before first interim |
| 187 | } |
| 188 | let snapshot = { shared_buf.lock().clone() }; |
| 189 | if snapshot.len() < 8000 { |
| 190 | // <0.5s of audio — not enough for meaningful ASR |
| 191 | continue; |
| 192 | } |
| 193 | // Try cheapest free ASR for interim; don't fail the whole capture on interim error. |
| 194 | let interim = transcribe_selected(config, &asr_kind, &snapshot, None) |
| 195 | .await |
| 196 | .unwrap_or_default(); |
| 197 | let trimmed = interim.trim(); |
| 198 | if !trimmed.is_empty() && trimmed != last_interim { |
| 199 | last_interim = trimmed.to_string(); |
| 200 | // Show interim inline — preserve cursor at original position, append interim with a block cursor |
| 201 | let display = if original_input.trim().is_empty() { |
| 202 | format!("{trimmed} ▍") |
| 203 | } else { |
| 204 | format!("{} {} ▍", original_input.trim_end(), trimmed) |
| 205 | }; |
| 206 | app.composer.input = display; |
| 207 | app.composer.cursor_position = original_cursor; |
| 208 | app.status_message = Some(recording_status(locale, Some(trimmed))); |
| 209 | } |
| 210 | if ticks > 40 { |
| 211 | break; // safety: ~28s max interim polling |
| 212 | } |
| 213 | } |
| 214 | |
| 215 | let (samples, _duration) = recorder_handle |
| 216 | .await |
| 217 | .ok() |
| 218 | .flatten() |
| 219 | .ok_or_else(|| tr(locale, MessageId::VoiceErrTooShort).to_string())?; |
| 220 | |
| 221 | // Restore composer to original before final insert (interim was preview only) |
| 222 | app.composer.input = original_input.clone(); |
| 223 | app.composer.cursor_position = original_cursor; |
| 224 | app.status_message = Some(tr(locale, MessageId::VoiceProcessing).to_string()); |
| 225 | |
| 226 | let text = transcribe_selected( |
| 227 | config, |
| 228 | &asr_kind, |
| 229 | &samples, |
| 230 | app.voice_control_enabled.then_some(original_input.as_str()), |
| 231 | ) |
| 232 | .await |
| 233 | .map_err(|error| format!("{}: {error}", tr(locale, MessageId::VoiceErrNetwork)))?; |
| 234 | |
| 235 | let clean = text.trim(); |
| 236 | if app.voice_send_enabled { |
| 237 | let (remainder, wants_send) = split_send_suffix(clean); |
| 238 | if wants_send { |
| 239 | // A bare "send it" submits whatever is already in the composer. |
| 240 | let outgoing = if remainder.is_empty() { |
| 241 | let existing = app.composer.input.trim().to_string(); |
| 242 | if !existing.is_empty() { |
| 243 | app.clear_input(); |
| 244 | } |
| 245 | existing |
| 246 | } else { |
| 247 | remainder.to_string() |
| 248 | }; |
| 249 | if outgoing.is_empty() { |
| 250 | return Err(tr(locale, MessageId::VoiceErrEmptySend).to_string()); |
| 251 | } |
| 252 | return Ok(VoiceCaptureOutcome::Send(outgoing)); |
| 253 | } |
| 254 | } |
| 255 | if clean.is_empty() { |
| 256 | return Err(tr(locale, MessageId::VoiceErrEmptySend).to_string()); |
| 257 | } |
| 258 | Ok(VoiceCaptureOutcome::Insert(clean.to_string())) |
| 259 | } |
| 260 | |
| 261 | // --- Command handlers ------------------------------------------------------ |
| 262 | |
| 263 | /// Handle the `/voice` command: toggle voice input. Toggling on requests a |
| 264 | /// one-shot recording + transcription via [`AppAction::VoiceCapture`]. |
| 265 | pub fn voice(app: &mut App) -> CommandResult { |
| 266 | let locale = app.ui_locale; |
| 267 | |
| 268 | if app.voice_enabled { |
| 269 | app.voice_enabled = false; |
| 270 | return CommandResult::message(tr(locale, MessageId::VoiceDisabled)); |
| 271 | } |
| 272 | if !is_available() { |
| 273 | return CommandResult::error(tr(locale, MessageId::VoiceErrNoRecorder)); |
| 274 | } |
| 275 | app.voice_enabled = true; |
| 276 | CommandResult::with_message_and_action( |
| 277 | tr(locale, MessageId::VoiceEnabled), |
| 278 | AppAction::VoiceCapture, |
| 279 | ) |
| 280 | } |
| 281 | |
| 282 | /// Handle the `/voice-send` command: toggle auto-send after transcription. |
| 283 | pub fn voice_send(app: &mut App) -> CommandResult { |
| 284 | let locale = app.ui_locale; |
| 285 | app.voice_send_enabled = !app.voice_send_enabled; |
| 286 | |
| 287 | let msg = if app.voice_send_enabled { |
| 288 | tr(locale, MessageId::VoiceSendEnabled) |
| 289 | } else { |
| 290 | tr(locale, MessageId::VoiceSendDisabled) |
| 291 | }; |
| 292 | CommandResult::message(msg) |
| 293 | } |
| 294 | |
| 295 | /// Handle the `/voice-control` command: toggle AI-assisted dictation. |
| 296 | pub fn voice_control(app: &mut App) -> CommandResult { |
| 297 | let locale = app.ui_locale; |
| 298 | app.voice_control_enabled = !app.voice_control_enabled; |
| 299 | |
| 300 | let msg = if app.voice_control_enabled { |
| 301 | tr(locale, MessageId::VoiceControlEnabled) |
| 302 | } else { |
| 303 | tr(locale, MessageId::VoiceControlDisabled) |
| 304 | }; |
| 305 | CommandResult::message(msg) |
| 306 | } |
| 307 | |
| 308 | #[cfg(test)] |
| 309 | mod tests { |
| 310 | use super::*; |
| 311 | |
| 312 | #[test] |
| 313 | fn recording_status_is_localized_and_keeps_the_interim() { |
| 314 | use codewhale_localization::Locale; |
| 315 | |
| 316 | for locale in [Locale::En, Locale::De, Locale::Ja] { |
| 317 | let label = tr(locale, MessageId::VoiceRecording).to_string(); |
| 318 | let stop = tr(locale, MessageId::VoiceRecordingStopHint).to_string(); |
| 319 | let idle = format!("{label} \u{00b7} {stop}"); |
| 320 | assert_eq!(recording_status(locale, None), idle); |
| 321 | assert_eq!(recording_status(locale, Some(" ")), idle); |
| 322 | |
| 323 | let with_interim = recording_status(locale, Some(" hello there ")); |
| 324 | assert!(with_interim.starts_with(&label), "{with_interim}"); |
| 325 | assert!(with_interim.contains("\u{201c}hello there\u{201d}")); |
| 326 | assert!( |
| 327 | with_interim.ends_with(&stop), |
| 328 | "the stop cue survives the interim: {with_interim}" |
| 329 | ); |
| 330 | assert!(!with_interim.contains("\u{2325}V"), "no hardcoded key hint"); |
| 331 | if locale != Locale::En { |
| 332 | assert!(!with_interim.contains("to finish"), "no English hint"); |
| 333 | } |
| 334 | } |
| 335 | assert_ne!( |
| 336 | recording_status(Locale::En, None), |
| 337 | recording_status(Locale::De, None) |
| 338 | ); |
| 339 | } |
| 340 | } |
| 341 |