diff --git a/desktop/src-tauri/src/huddle/models.rs b/desktop/src-tauri/src/huddle/models.rs index 04521c5f0..e420d60f2 100644 --- a/desktop/src-tauri/src/huddle/models.rs +++ b/desktop/src-tauri/src/huddle/models.rs @@ -1,8 +1,9 @@ //! Model download manager for STT (Parakeet family) and TTS (Pocket TTS) models. //! //! The STT model is selectable via the `STT_MODELS` registry (issue #2478): -//! the English Parakeet TDT-CTC 110M default, or the multilingual Parakeet TDT -//! 0.6B v3, chosen by `BUZZ_STT_MODEL` / system locale in `selected_stt_model`. +//! the English Parakeet TDT-CTC 110M default, multilingual Parakeet TDT 0.6B +//! v3, or SenseVoiceSmall, chosen by `BUZZ_STT_MODEL` / system locale in +//! `selected_stt_model`. //! //! Mental model: //! app launch → start_stt_download (background) → ~/.buzz/models// @@ -115,9 +116,9 @@ const MAX_TTS_FILE_BYTES: u64 = 400 * 1024 * 1024; // // Historically the huddle STT model was hard-pinned to an English-only build, // so non-English speech transcribed as garbage (issue #2478). The registry -// makes the model selectable: the English default stays the default, and a -// multilingual model is picked automatically for non-English locales (or -// forced with the `BUZZ_STT_MODEL` env override). Adding a model is pure data +// makes the model selectable: the English default stays the default, and each +// supported system locale maps to a model that actually covers its language +// (or is forced with the `BUZZ_STT_MODEL` env override). Adding a model is data // here plus, for a new sherpa-onnx model family, one match arm in `stt.rs`. /// Attribution sidecar filename written next to every STT model's files. @@ -131,6 +132,8 @@ pub enum SttFamily { NemoCtc, /// NeMo transducer: encoder + decoder + joiner (e.g. Parakeet TDT 0.6B v3). Transducer, + /// Single-file SenseVoice model with language auto-detection and ITN. + SenseVoice, } /// One selectable speech-to-text model. `STT_MODELS` is the single source of @@ -156,12 +159,12 @@ pub struct SttModel { pub family: SttFamily, /// Manifest version — bump to force re-download of this model. pub version: &'static str, - /// CC-BY-4.0 §3(a)(1) attribution written next to the model bytes. + /// Model-specific license and attribution notice written next to the bytes. pub license_text: &'static str, /// Human-readable language coverage (About dialog / logs). pub languages: &'static str, - /// `true` when multilingual — used by the non-English locale auto-select. - pub multilingual: bool, + /// Primary locale tags that may automatically select this model. + pub auto_select_languages: &'static [&'static str], } /// Registry of selectable STT models. Index 0 is the default (English). @@ -184,17 +187,13 @@ static STT_MODELS: &[SttModel] = &[ version: "2", license_text: STT_EN_LICENSE_TEXT, languages: "English", - multilingual: false, + auto_select_languages: &["en"], }, // NVIDIA Parakeet TDT 0.6B v3 (multilingual, int8) — packaged for // sherpa-onnx by k2-fsa. Transducer family (encoder/decoder/joiner). Auto - // language-ID + punctuation across 25 European languages. This is the - // multilingual default for non-English locales (issue #2478). - // - // Coverage note: Parakeet v3 covers European languages only. CJK/Korean - // (the concrete case in #2478) is not covered by this model; the registry - // makes adding a CJK model (e.g. SenseVoice-Small) a follow-up — one data - // entry, no engine change beyond a family already handled here. + // language-ID + punctuation across 25 European languages. Locale selection + // is restricted to those languages instead of treating it as a universal + // non-English fallback (issue #2478). // // Checksum: upstream publishes no SHA-256, so this hash was computed from // the k2-fsa `asr-models` release archive @@ -222,7 +221,29 @@ static STT_MODELS: &[SttModel] = &[ German, Greek, Hungarian, Italian, Latvian, Lithuanian, \ Maltese, Polish, Portuguese, Romanian, Russian, Slovak, \ Slovenian, Spanish, Swedish, Ukrainian", - multilingual: true, + auto_select_languages: &[ + "bg", "hr", "cs", "da", "nl", "et", "fi", "fr", "de", "el", "hu", "it", "lv", "lt", + "mt", "pl", "pt", "ro", "ru", "sk", "sl", "es", "sv", "uk", + ], + }, + // SenseVoiceSmall int8 — packaged for sherpa-onnx by k2-fsa. This covers + // the Korean case reported in #2478 as well as Chinese, Cantonese, and + // Japanese. The English locale intentionally keeps the smaller Parakeet + // English default; SenseVoice remains available through BUZZ_STT_MODEL. + SttModel { + id: "sensevoice", + dir_name: "sensevoice-small", + download_url: "https://github.com/k2-fsa/sherpa-onnx/releases/download/asr-models/\ + sherpa-onnx-sense-voice-zh-en-ja-ko-yue-int8-2024-07-17.tar.bz2", + archive_subdir: "sherpa-onnx-sense-voice-zh-en-ja-ko-yue-int8-2024-07-17", + archive_sha256: Some("7d1efa2138a65b0b488df37f8b89e3d91a60676e416f515b952358d83dfd347e"), + max_download_bytes: 250 * 1024 * 1024, + model_files: &["model.int8.onnx", "tokens.txt"], + family: SttFamily::SenseVoice, + version: "1", + license_text: STT_SENSEVOICE_LICENSE_TEXT, + languages: "Chinese, Cantonese, English, Japanese, Korean (auto-detected)", + auto_select_languages: &["zh", "yue", "ja", "ko"], }, ]; @@ -237,8 +258,8 @@ fn stt_model_by_id(id: &str) -> Option<&'static SttModel> { } /// Files that must be present for a model to be "ready": its model files plus -/// the Buzz-written attribution sidecar (the upstream archives ship no LICENSE, -/// so readiness requires the local CC-BY-4.0 notice to travel with the bytes). +/// the Buzz-written model-specific license sidecar. This keeps the applicable +/// notice next to the bytes even when an upstream archive also ships LICENSE. fn stt_expected_files(model: &SttModel) -> Vec<&'static str> { let mut files = model.model_files.to_vec(); files.push(STT_LICENSE_FILE_NAME); @@ -249,7 +270,7 @@ fn stt_expected_files(model: &SttModel) -> Vec<&'static str> { /// /// Precedence (issue #2478 options 2 + 3): /// 1. `override_id` (from `BUZZ_STT_MODEL`) when it names a known model. -/// 2. a non-English locale → the multilingual model. +/// 2. a locale covered by a registered model → that model. /// 3. otherwise the English default. /// /// Pure and dependency-free so it is unit-testable without touching disk/env. @@ -278,10 +299,11 @@ pub fn select_stt_model(override_id: Option<&str>, locale: Option<&str>) -> &'st .next() .unwrap_or("") .to_ascii_lowercase(); - if !lang.is_empty() && lang != "en" { - if let Some(model) = STT_MODELS.iter().find(|m| m.multilingual) { - return model; - } + if let Some(model) = STT_MODELS + .iter() + .find(|model| model.auto_select_languages.contains(&lang.as_str())) + { + return model; } } default_stt_model() @@ -347,6 +369,20 @@ Provided \"AS IS\", without warranty of any kind, express or implied. See the license text for full warranty disclaimer. "; +/// FunASR Model Open Source License 1.1 notice for SenseVoiceSmall. +const STT_SENSEVOICE_LICENSE_TEXT: &str = "\ +SenseVoiceSmall +Copyright (C) 2023-2028 Alibaba Group. All rights reserved. + +Licensed under the FunASR Model Open Source License Agreement, Version 1.1. +Full license: https://github.com/modelscope/FunASR/blob/main/MODEL_LICENSE + +Original model: https://github.com/FunAudioLLM/SenseVoice +Converted to ONNX with int8 quantization by the sherpa-onnx project +(https://github.com/k2-fsa/sherpa-onnx); Buzz ships this conversion unmodified. +The upstream archive also includes its own LICENSE pointer. +"; + // ── Pocket TTS model ────────────────────────────────────────────────────────── /// Final directory name under `~/.buzz/models/`. @@ -943,10 +979,9 @@ impl ModelManager { )); } - // Write the CC-BY-4.0 attribution sidecar before the atomic install, - // so it lands in the final model dir as part of the same rename. The - // upstream tarball ships no LICENSE/NOTICE, so we provide it ourselves - // per §3(a)(1) (license must travel with Shared material). + // Write the model-specific license sidecar before the atomic install, + // so it lands in the final directory as part of the same rename. Some + // upstream archives also include their own LICENSE. let license_path = extracted_subdir.join(STT_LICENSE_FILE_NAME); if let Err(e) = tokio::fs::write(&license_path, model.license_text).await { let _ = tokio::fs::remove_dir_all(&temp_dir).await; @@ -1203,18 +1238,33 @@ mod tests { assert_eq!(select_stt_model(None, Some("en")).id, "parakeet-en"); // Neutral C/POSIX locales are dropped by detect_locale, but pass the raw // primary subtag here to prove "en" specifically stays on English. - assert!(!select_stt_model(None, None).multilingual); + assert_eq!(select_stt_model(None, None).auto_select_languages, &["en"]); } #[test] - fn non_english_locale_selects_multilingual_model() { + fn supported_european_locale_selects_parakeet_v3() { for locale in ["de-DE", "uk_UA", "fr", "es-ES", "pl_PL.UTF-8"] { let model = select_stt_model(None, Some(locale)); - assert!(model.multilingual, "locale {locale} should be multilingual"); assert_eq!(model.id, "parakeet-v3", "locale {locale}"); } } + #[test] + fn cjk_locales_select_sensevoice() { + for locale in ["zh-CN", "ja_JP", "ko-KR", "yue_HK.UTF-8"] { + assert_eq!( + select_stt_model(None, Some(locale)).id, + "sensevoice", + "locale {locale}" + ); + } + } + + #[test] + fn unsupported_locale_does_not_select_an_incompatible_multilingual_model() { + assert_eq!(select_stt_model(None, Some("ar-SA")).id, "parakeet-en"); + } + #[test] fn explicit_override_wins_over_locale() { // Force multilingual even on an English locale. @@ -1232,6 +1282,10 @@ mod tests { select_stt_model(Some("PARAKEET-V3"), None).id, "parakeet-v3" ); + assert_eq!( + select_stt_model(Some("SENSEVOICE"), Some("de-DE")).id, + "sensevoice" + ); } #[test] @@ -1258,8 +1312,10 @@ mod tests { default_stt_model().archive_sha256.is_some(), "English default must ship a pinned SHA-256" ); - // At least one multilingual model exists (covers #2478). - assert!(STT_MODELS.iter().any(|m| m.multilingual)); + // At least one model automatically covers multiple languages (#2478). + assert!(STT_MODELS + .iter() + .any(|model| model.auto_select_languages.len() > 1)); // Ids are unique (case-insensitive); every model declares files + a cap. for (i, model) in STT_MODELS.iter().enumerate() { assert!(!model.model_files.is_empty(), "{} has no files", model.id); @@ -1270,6 +1326,14 @@ mod tests { "duplicate model id {}", model.id ); + for language in model.auto_select_languages { + assert!( + !other.auto_select_languages.contains(language), + "locale {language} is auto-selected by both {} and {}", + model.id, + other.id + ); + } } } } @@ -1311,4 +1375,20 @@ mod tests { } assert!(slot.is_ready(temp.path())); } + + #[test] + fn sensevoice_readiness_uses_single_model_file() { + let model = stt_model_by_id("sensevoice").expect("SenseVoice registered"); + let temp = tempfile::tempdir().expect("tempdir"); + let slot = ModelSlot::new(model.dir_name, stt_expected_files(model), model.version); + let dir = temp.path().join(model.dir_name); + std::fs::create_dir_all(&dir).expect("create dir"); + std::fs::write(dir.join(MANIFEST_FILENAME), model.version).expect("manifest"); + std::fs::write(dir.join("model.int8.onnx"), b"x").expect("model"); + std::fs::write(dir.join("tokens.txt"), b"x").expect("tokens"); + assert!(!slot.is_ready(temp.path())); + + std::fs::write(dir.join(STT_LICENSE_FILE_NAME), b"x").expect("license"); + assert!(slot.is_ready(temp.path())); + } } diff --git a/desktop/src-tauri/src/huddle/stt.rs b/desktop/src-tauri/src/huddle/stt.rs index 1eb768aaf..9f214f54c 100644 --- a/desktop/src-tauri/src/huddle/stt.rs +++ b/desktop/src-tauri/src/huddle/stt.rs @@ -241,6 +241,8 @@ fn stt_worker( // Transducer — `encoder/decoder/joiner.int8.onnx`, e.g. Parakeet 0.6B v3 // (multilingual). See k2-fsa/sherpa-onnx offline-transducer // NeMo examples. + // SenseVoice — single `model.int8.onnx`, automatic language detection, + // and inverse text normalization. // // `tokens.txt` is shared by every family and lives on the parent config. use sherpa_onnx::{OfflineRecognizer, OfflineRecognizerConfig}; @@ -286,6 +288,20 @@ fn stt_worker( cfg.model_config.transducer.decoder = Some(decoder.to_string_lossy().into_owned()); cfg.model_config.transducer.joiner = Some(joiner.to_string_lossy().into_owned()); } + SttFamily::SenseVoice => { + let model_path = model_dir.join("model.int8.onnx"); + if !model_path.exists() { + eprintln!( + "buzz-desktop: STT model.int8.onnx not found at {} — STT disabled", + model_dir.display() + ); + drain_until_shutdown(audio_rx, &shutdown); + return; + } + cfg.model_config.sense_voice.model = Some(model_path.to_string_lossy().into_owned()); + cfg.model_config.sense_voice.language = Some("auto".to_string()); + cfg.model_config.sense_voice.use_itn = true; + } } cfg.model_config.tokens = Some(tokens_path.to_string_lossy().into_owned()); cfg.model_config.num_threads = STT_NUM_THREADS;