From 94847a4ae23ad5e19aa004c343731b3bf470e65e Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Wed, 30 Sep 2026 18:35:56 +0800 Subject: [PATCH 1/8] speech: Add `SpeechState`, `SpeechButton` and system recognizers for dictation Add a `speech` module to gpui-component: a session state machine that feeds captured audio to a pluggable `SpeechRecognizer`, a `SpeechButton` and a `SpeechWaveform` to render it, and two seams, `SpeechRecognizer` and `AudioInput`, so applications can bring any speech service or audio source. With the new `speech` feature, a state without its own recognizer falls back to the platform's: `SFSpeechRecognizer` on macOS (on-device only) and `Windows.Media.SpeechRecognition` on Windows. Linux has no system recognizer and works with an application recognizer. Audio is captured with cpal. Also adds the gallery story, English and Chinese docs, and the `speech` example, a dictation notepad that reports what the machine supports. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 207 ++++- Cargo.toml | 1 + crates/assets/default-icons.txt | 2 + crates/assets/tests/icons.rs | 4 +- crates/component/Cargo.toml | 40 + crates/component/locales/ui.yml | 16 + crates/component/src/lib.rs | 1 + crates/component/src/speech/button.rs | 126 +++ crates/component/src/speech/microphone.rs | 262 ++++++ crates/component/src/speech/mod.rs | 51 ++ crates/component/src/speech/recognizer.rs | 266 ++++++ crates/component/src/speech/state.rs | 689 ++++++++++++++ crates/component/src/speech/system/macos.rs | 486 ++++++++++ crates/component/src/speech/system/mod.rs | 86 ++ .../src/speech/system/unsupported.rs | 30 + crates/component/src/speech/system/winrt.rs | 395 ++++++++ crates/component/src/speech/waveform.rs | 83 ++ crates/kit/Cargo.toml | 2 + crates/story/Cargo.toml | 2 +- crates/story/src/gallery.rs | 1 + crates/story/src/stories/mod.rs | 2 + crates/story/src/stories/speech_story.rs | 281 ++++++ examples/speech/Cargo.toml | 14 + examples/speech/Info.plist | 10 + examples/speech/README.md | 62 ++ examples/speech/build.rs | 21 + examples/speech/src/demo.rs | 166 ++++ examples/speech/src/main.rs | 845 ++++++++++++++++++ script/install-linux.sh | 2 +- skills/gpui-kit/SKILL.md | 1 + website/component/index.md | 1 + website/component/speech.md | 408 +++++++++ website/docs/installation.md | 4 +- website/zh-CN/component/index.md | 1 + website/zh-CN/component/speech.md | 336 +++++++ website/zh-CN/docs/installation.md | 4 +- 36 files changed, 4896 insertions(+), 12 deletions(-) create mode 100644 crates/component/src/speech/button.rs create mode 100644 crates/component/src/speech/microphone.rs create mode 100644 crates/component/src/speech/mod.rs create mode 100644 crates/component/src/speech/recognizer.rs create mode 100644 crates/component/src/speech/state.rs create mode 100644 crates/component/src/speech/system/macos.rs create mode 100644 crates/component/src/speech/system/mod.rs create mode 100644 crates/component/src/speech/system/unsupported.rs create mode 100644 crates/component/src/speech/system/winrt.rs create mode 100644 crates/component/src/speech/waveform.rs create mode 100644 crates/story/src/stories/speech_story.rs create mode 100644 examples/speech/Cargo.toml create mode 100644 examples/speech/Info.plist create mode 100644 examples/speech/README.md create mode 100644 examples/speech/build.rs create mode 100644 examples/speech/src/demo.rs create mode 100644 examples/speech/src/main.rs create mode 100644 website/component/speech.md create mode 100644 website/zh-CN/component/speech.md diff --git a/Cargo.lock b/Cargo.lock index 109e2e967d..9877bcc1d0 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -224,6 +224,28 @@ version = "0.2.21" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" +[[package]] +name = "alsa" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed7572b7ba83a31e20d1b48970ee402d2e3e0537dcfe0a3ff4d6eb7508617d43" +dependencies = [ + "alsa-sys", + "bitflags 2.13.1", + "cfg-if", + "libc", +] + +[[package]] +name = "alsa-sys" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db8fee663d06c4e303404ef5f40488a53e062f89ba8bfed81f42325aafad1527" +dependencies = [ + "libc", + "pkg-config", +] + [[package]] name = "ambient-authority" version = "0.0.2" @@ -1731,6 +1753,26 @@ dependencies = [ "libm", ] +[[package]] +name = "coreaudio-rs" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "321077172d79c662f64f5071a03120748d5bb652f5231570141be24cfcd2bace" +dependencies = [ + "bitflags 1.3.2", + "core-foundation-sys 0.8.7", + "coreaudio-sys", +] + +[[package]] +name = "coreaudio-sys" +version = "0.2.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9b4739a805a62757a83e5654fa3faabec0442666b263bb2287d5a8185bfd953" +dependencies = [ + "bindgen", +] + [[package]] name = "cosmic-text" version = "0.19.0" @@ -1755,6 +1797,29 @@ dependencies = [ "unicode-segmentation", ] +[[package]] +name = "cpal" +version = "0.15.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "873dab07c8f743075e57f524c583985fbaf745602acbe916a01539364369a779" +dependencies = [ + "alsa", + "core-foundation-sys 0.8.7", + "coreaudio-rs", + "dasp_sample", + "jni 0.21.1", + "js-sys", + "libc", + "mach2 0.4.3", + "ndk 0.8.0", + "ndk-context", + "oboe", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", + "windows 0.54.0", +] + [[package]] name = "cpubits" version = "0.1.1" @@ -2258,6 +2323,12 @@ dependencies = [ "parking_lot_core", ] +[[package]] +name = "dasp_sample" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c87e182de0887fd5361989c677c4e8f5000cd9491d6d563161a8f3a5519fc7f" + [[package]] name = "data-encoding" version = "2.11.1" @@ -3824,8 +3895,10 @@ name = "gpui-component" version = "0.7.0" dependencies = [ "anyhow", + "block2 0.6.2", "chrono", "core-text", + "cpal", "enum-iterator", "gpui-base", "gpui-component-macros", @@ -3843,7 +3916,10 @@ dependencies = [ "num-traits", "objc2 0.6.4", "objc2-app-kit 0.3.2", + "objc2-av-foundation", + "objc2-avf-audio", "objc2-foundation 0.3.2", + "objc2-speech", "once_cell", "paste", "raw-window-handle", @@ -4254,7 +4330,7 @@ dependencies = [ "itertools 0.14.0", "libc", "log", - "mach2", + "mach2 0.5.0", "objc", "objc2 0.6.4", "objc2-app-kit 0.3.2", @@ -5661,7 +5737,7 @@ dependencies = [ "jni 0.21.1", "kuchikiki", "libc", - "ndk", + "ndk 0.9.0", "objc2 0.6.4", "objc2-app-kit 0.3.2", "objc2-core-foundation", @@ -6162,6 +6238,15 @@ dependencies = [ "libc", ] +[[package]] +name = "mach2" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d640282b302c0bb0a2a8e0233ead9035e3bed871f0b7e81fe4a1ec829765db44" +dependencies = [ + "libc", +] + [[package]] name = "mach2" version = "0.5.0" @@ -6437,6 +6522,20 @@ dependencies = [ "getrandom 0.2.17", ] +[[package]] +name = "ndk" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2076a31b7010b17a38c01907c45b945e8f11495ee4dd588309718901b1f7a5b7" +dependencies = [ + "bitflags 2.13.1", + "jni-sys 0.3.1", + "log", + "ndk-sys 0.5.0+25.2.9519653", + "num_enum", + "thiserror 1.0.69", +] + [[package]] name = "ndk" version = "0.9.0" @@ -6446,12 +6545,27 @@ dependencies = [ "bitflags 2.13.1", "jni-sys 0.3.1", "log", - "ndk-sys", + "ndk-sys 0.6.0+11769913", "num_enum", "raw-window-handle", "thiserror 1.0.69", ] +[[package]] +name = "ndk-context" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "27b02d87554356db9e9a873add8782d4ea6e3e58ea071a9adb9a2e8ddb884a8b" + +[[package]] +name = "ndk-sys" +version = "0.5.0+25.2.9519653" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8c196769dd60fd4f363e11d948139556a344e79d451aeb2fa2fd040738ef7691" +dependencies = [ + "jni-sys 0.3.1", +] + [[package]] name = "ndk-sys" version = "0.6.0+11769913" @@ -6817,6 +6931,27 @@ dependencies = [ "objc2-quartz-core 0.3.2", ] +[[package]] +name = "objc2-av-foundation" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "478ae33fcac9df0a18db8302387c666b8ef08a3e2d62b510ca4fc278a384b6c0" +dependencies = [ + "bitflags 2.13.1", + "objc2 0.6.4", + "objc2-foundation 0.3.2", +] + +[[package]] +name = "objc2-avf-audio" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13a380031deed8e99db00065c45937da434ca987c034e13b87e4441f9e4090be" +dependencies = [ + "objc2 0.6.4", + "objc2-foundation 0.3.2", +] + [[package]] name = "objc2-cloud-kit" version = "0.3.2" @@ -7094,6 +7229,18 @@ dependencies = [ "objc2-foundation 0.3.2", ] +[[package]] +name = "objc2-speech" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9eb609f9d2a25e0f8b76954f3acaa58348ce2a91285fe867643b2224ac4d263" +dependencies = [ + "block2 0.6.2", + "objc2 0.6.4", + "objc2-avf-audio", + "objc2-foundation 0.3.2", +] + [[package]] name = "objc2-ui-kit" version = "0.3.2" @@ -7172,6 +7319,29 @@ dependencies = [ "memchr", ] +[[package]] +name = "oboe" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e8b61bebd49e5d43f5f8cc7ee2891c16e0f41ec7954d36bcb6c14c5e0de867fb" +dependencies = [ + "jni 0.21.1", + "ndk 0.8.0", + "ndk-context", + "num-derive", + "num-traits", + "oboe-sys", +] + +[[package]] +name = "oboe-sys" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6c8bb09a4a2b1d668170cfe0a7d5bc103f8999fb316c98099b6a9939c9f2e79d" +dependencies = [ + "cc", +] + [[package]] name = "once_cell" version = "1.21.4" @@ -9759,6 +9929,15 @@ dependencies = [ "system-deps", ] +[[package]] +name = "speech" +version = "0.7.0" +dependencies = [ + "anyhow", + "cpal", + "gpui-kit", +] + [[package]] name = "spin" version = "0.9.9" @@ -12040,7 +12219,7 @@ dependencies = [ "libloading", "log", "naga", - "ndk-sys", + "ndk-sys 0.6.0+11769913", "objc2 0.6.4", "objc2-core-foundation", "objc2-foundation 0.3.2", @@ -12140,6 +12319,16 @@ dependencies = [ "gpui-kit", ] +[[package]] +name = "windows" +version = "0.54.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9252e5725dbed82865af151df558e754e4a3c2c30818359eb17465f1346a1b49" +dependencies = [ + "windows-core 0.54.0", + "windows-targets 0.52.6", +] + [[package]] name = "windows" version = "0.57.0" @@ -12216,6 +12405,16 @@ dependencies = [ "windows-core 0.62.2", ] +[[package]] +name = "windows-core" +version = "0.54.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12661b9c89351d684a50a8a643ce5f608e20243b9fb84687800163429f161d65" +dependencies = [ + "windows-result 0.1.2", + "windows-targets 0.52.6", +] + [[package]] name = "windows-core" version = "0.57.0" diff --git a/Cargo.toml b/Cargo.toml index 6a8e3948dd..7070488588 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -36,6 +36,7 @@ members = [ "examples/table_in_scrollable", "examples/markdown_table", "examples/ai_recipes", + "examples/speech", "crates/shell", "crates/shell/rquickjs-compat", "crates/component-shell", diff --git a/crates/assets/default-icons.txt b/crates/assets/default-icons.txt index 044531c80f..7119c5d985 100644 --- a/crates/assets/default-icons.txt +++ b/crates/assets/default-icons.txt @@ -60,6 +60,7 @@ icons/map.svg icons/maximize.svg icons/memory-stick.svg icons/menu.svg +icons/mic.svg icons/minimize.svg icons/minus.svg icons/moon.svg @@ -88,6 +89,7 @@ icons/settings.svg icons/sort-ascending.svg icons/sort-descending.svg icons/square-terminal.svg +icons/square.svg icons/star-fill.svg icons/star-off.svg icons/star.svg diff --git a/crates/assets/tests/icons.rs b/crates/assets/tests/icons.rs index d51abd4f06..352f0c2263 100644 --- a/crates/assets/tests/icons.rs +++ b/crates/assets/tests/icons.rs @@ -62,8 +62,8 @@ fn default_assets_preserve_the_component_bundle_without_all_lucide_icons() { .collect(); let actual: BTreeSet<_> = Assets.list("icons/").unwrap().into_iter().collect(); assert_eq!(actual, expected); - assert_eq!(actual.len(), 104); - assert_eq!(Assets::iter().count(), 104); + assert_eq!(actual.len(), 106); + assert_eq!(Assets::iter().count(), 106); assert!(Assets::get("icons/search.svg").is_some()); assert!(Assets::get("icons/accessibility.svg").is_none()); for path in actual { diff --git a/crates/component/Cargo.toml b/crates/component/Cargo.toml index 545485ba43..a03b97d1fd 100644 --- a/crates/component/Cargo.toml +++ b/crates/component/Cargo.toml @@ -19,6 +19,21 @@ test-support = ["gpui/test-support", "gpui-base/test-support"] decimal = ["gpui-base/decimal"] inspector = ["gpui_macros/inspector", "gpui/inspector", "gpui-base/inspector"] tree-sitter = ["dep:tree-sitter", "dep:tree-sitter-json"] +# Speech input: microphone capture and the system speech recognizer. +speech = [ + "dep:cpal", + "dep:block2", + "dep:objc2-speech", + "dep:objc2-avf-audio", + "dep:objc2-av-foundation", + "objc2-foundation/NSBundle", + "objc2-foundation/NSDictionary", + "objc2-foundation/NSError", + "objc2-foundation/NSLocale", + "windows/Foundation_Collections", + "windows/Globalization", + "windows/Media_SpeechRecognition", +] # For syntax highlighting in Markdown and CodeEditor. tree-sitter-languages = [ @@ -137,6 +152,7 @@ instant.workspace = true # Native-only dependencies (not available on WASM) [target.'cfg(not(target_family = "wasm"))'.dependencies] smol.workspace = true +cpal = { version = "0.15.3", optional = true } tree-sitter = { version = "0.26.13", optional = true } tree-sitter-astro-next = { version="0.1.1", optional = true } tree-sitter-bash = { version = "0.23.3", optional = true } @@ -188,6 +204,30 @@ objc2-app-kit = { version = "0.3", features = [ "NSImage", ] } objc2-foundation = { version = "0.3", features = ["NSData", "NSString", "NSGeometry"] } +# Speech input (SystemRecognizer) — SFSpeechRecognizer via objc2. +block2 = { version = "0.6", optional = true } +objc2-av-foundation = { version = "0.3", default-features = false, features = [ + "std", + "AVCaptureDevice", + "AVMediaFormat", +], optional = true } +objc2-avf-audio = { version = "0.3", default-features = false, features = [ + "std", + "AVAudioBuffer", + "AVAudioFormat", + "AVAudioTypes", +], optional = true } +objc2-speech = { version = "0.3", default-features = false, features = [ + "std", + "block2", + "objc2-avf-audio", + "SFSpeechRecognitionMetadata", + "SFSpeechRecognitionRequest", + "SFSpeechRecognitionResult", + "SFSpeechRecognitionTask", + "SFSpeechRecognizer", + "SFTranscription", +], optional = true } [target.'cfg(target_os = "windows")'.dependencies] # Native menu (NativeMenu) — drives Win32 popup menus. diff --git a/crates/component/locales/ui.yml b/crates/component/locales/ui.yml index 2016977c35..8f9f999e25 100644 --- a/crates/component/locales/ui.yml +++ b/crates/component/locales/ui.yml @@ -432,3 +432,19 @@ Questionnaire: zh-CN: 请选择一个答案,或跳过此题。 zh-HK: 請選擇一個答案,或跳過此題。 zh-TW: 請選擇一個答案,或跳過此題。 +Speech: + Start: + en: Dictate + zh-CN: 语音输入 + zh-HK: 語音輸入 + zh-TW: 語音輸入 + Stop: + en: Stop dictation + zh-CN: 停止语音输入 + zh-HK: 停止語音輸入 + zh-TW: 停止語音輸入 + Unavailable: + en: Dictation unavailable + zh-CN: 语音输入不可用 + zh-HK: 語音輸入不可用 + zh-TW: 語音輸入不可用 diff --git a/crates/component/src/lib.rs b/crates/component/src/lib.rs index 3b2f2b2a69..5ebfcf0570 100644 --- a/crates/component/src/lib.rs +++ b/crates/component/src/lib.rs @@ -77,6 +77,7 @@ pub mod shimmer; pub mod sidebar; pub mod skeleton; pub mod slider; +pub mod speech; pub mod spinner; pub mod status_bar; pub mod stepper; diff --git a/crates/component/src/speech/button.rs b/crates/component/src/speech/button.rs new file mode 100644 index 0000000000..be2f337978 --- /dev/null +++ b/crates/component/src/speech/button.rs @@ -0,0 +1,126 @@ +use gpui::{App, ElementId, Entity, IntoElement, RenderOnce, Window, div}; +use rust_i18n::t; + +use crate::{ + Disableable, IconName, Selectable as _, Sizable, Size, + button::{Button, ButtonVariants as _}, +}; + +use super::{SpeechState, SpeechStatus}; + +/// The button that starts and stops a [`SpeechState`]'s session. +/// +/// Shows a microphone at rest and a stop glyph, pressed, while capturing; a +/// click toggles the session. While the final result is pending it shows a +/// spinner and ignores clicks. +/// +/// Renders nothing when the state has no recognizer on this platform, unless +/// [`Self::show_when_unsupported`] is set, and renders disabled while the +/// recognizer reports itself unavailable. +#[derive(IntoElement)] +pub struct SpeechButton { + id: ElementId, + state: Entity, + size: Size, + disabled: bool, + show_when_unsupported: bool, +} + +impl SpeechButton { + /// A button for `state`. + pub fn new(state: &Entity) -> Self { + Self { + id: ("speech-button", state.entity_id()).into(), + state: state.clone(), + size: Size::default(), + disabled: false, + show_when_unsupported: false, + } + } + + /// Render a disabled button, instead of nothing, when the state has no + /// recognizer on this platform. Default `false`. + pub fn show_when_unsupported(mut self, show: bool) -> Self { + self.show_when_unsupported = show; + self + } +} + +impl Sizable for SpeechButton { + fn with_size(mut self, size: impl Into) -> Self { + self.size = size.into(); + self + } +} + +impl Disableable for SpeechButton { + fn disabled(mut self, disabled: bool) -> Self { + self.disabled = disabled; + self + } +} + +impl RenderOnce for SpeechButton { + fn render(self, _: &mut Window, cx: &mut App) -> impl IntoElement { + let state = self.state.read(cx); + let status = state.status(); + let supported = state.has_recognizer(); + if !supported && !self.show_when_unsupported { + return div().into_any_element(); + } + // A running session stays stoppable even if the recognizer turns + // unavailable mid-way. + let available = status.is_active() || state.is_available(cx); + let capturing = status.is_capturing(); + let label = if !available { + t!("Speech.Unavailable") + } else if capturing { + t!("Speech.Stop") + } else { + t!("Speech.Start") + }; + + Button::new(self.id) + .ghost() + .with_size(self.size) + .icon(if capturing { + IconName::Square + } else { + IconName::Mic + }) + .selected(capturing) + .loading(status == SpeechStatus::Stopping) + .disabled(self.disabled || !available) + .tooltip(label.clone()) + .accessibility_label(label) + .on_click({ + let state = self.state.clone(); + move |_, _, cx| state.update(cx, |state, cx| state.toggle(cx)) + }) + .into_any_element() + } +} + +#[cfg(test)] +mod tests { + use gpui::{AppContext as _, TestAppContext}; + + use super::*; + + #[gpui::test] + fn test_speech_button_builder(cx: &mut TestAppContext) { + let state = cx.update(|cx| cx.new(|cx| SpeechState::new(cx).system_fallback(false))); + let button = SpeechButton::new(&state) + .small() + .disabled(true) + .show_when_unsupported(true); + + assert_eq!(button.size, Size::Small); + assert!(button.disabled); + assert!(button.show_when_unsupported); + assert_eq!( + button.id, + ElementId::from(("speech-button", state.entity_id())) + ); + } +} diff --git a/crates/component/src/speech/microphone.rs b/crates/component/src/speech/microphone.rs new file mode 100644 index 0000000000..fc4e941f61 --- /dev/null +++ b/crates/component/src/speech/microphone.rs @@ -0,0 +1,262 @@ +use std::f32::consts::TAU; + +use anyhow::anyhow; +use cpal::{ + FromSample, Sample, SampleFormat, SizedSample, StreamConfig, + traits::{DeviceTrait as _, HostTrait as _, StreamTrait as _}, +}; +use gpui::{App, Subscription}; +use smol::channel::{Sender, unbounded}; + +use super::{AudioFormat, AudioInput, AudioSink, SpeechError}; + +/// The default audio input device, captured through the platform's audio API +/// (Core Audio, WASAPI or ALSA). +/// +/// The device's own format is mixed down to mono and resampled to the format +/// the recognizer asks for. +/// +/// On macOS the application's `Info.plist` must describe why it uses the +/// microphone (`NSMicrophoneUsageDescription`), or the system refuses access +/// without asking. +#[derive(Debug, Default, Clone)] +pub struct Microphone { + _private: (), +} + +impl AudioInput for Microphone { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + #[cfg(target_os = "macos")] + if super::system::macos::is_microphone_denied() { + return Err(SpeechError::PermissionDenied); + } + + let device = cpal::default_host() + .default_input_device() + .ok_or(SpeechError::NoInputDevice)?; + let supported = device.default_input_config().map_err(SpeechError::input)?; + let sample_format = supported.sample_format(); + let config: StreamConfig = supported.into(); + let source_rate = config.sample_rate.0; + + let (tx, rx) = unbounded(); + let stream = match sample_format { + SampleFormat::F32 => build_stream::(&device, &config, tx), + SampleFormat::I16 => build_stream::(&device, &config, tx), + SampleFormat::U16 => build_stream::(&device, &config, tx), + SampleFormat::I32 => build_stream::(&device, &config, tx), + other => { + return Err(SpeechError::input(anyhow!( + "unsupported sample format {other}" + ))); + } + }?; + stream.play().map_err(SpeechError::input)?; + + let task = cx.spawn(async move |cx| { + let mut converter = Converter::new(source_rate, format); + let mut mono = Vec::new(); + while let Ok(first) = rx.recv().await { + // The device delivers ~10 ms chunks; forward what has queued up + // as one push so the state updates once per batch. + let mut error = None; + for capture in + std::iter::once(first).chain(std::iter::from_fn(|| rx.try_recv().ok())) + { + match capture { + Capture::Samples(samples) => mono.extend_from_slice(&samples), + Capture::Error(message) => error = Some(message), + } + } + + let samples = converter.convert(&mono); + mono.clear(); + cx.update(|cx| { + if !samples.is_empty() { + sink.push(samples, cx); + } + if let Some(message) = error.take() { + sink.error(SpeechError::input(anyhow!(message)), cx); + } + }); + } + }); + + Ok(Subscription::new(move || { + drop(stream); + drop(task); + })) + } +} + +enum Capture { + Samples(Vec), + Error(String), +} + +/// Open an input stream that sends each callback's frames, mixed down to mono, +/// through `tx`. Runs on the audio thread, so it only converts and sends. +fn build_stream( + device: &cpal::Device, + config: &StreamConfig, + tx: Sender, +) -> Result +where + T: SizedSample, + f32: FromSample, +{ + let channels = config.channels.max(1) as usize; + let error_tx = tx.clone(); + device + .build_input_stream( + config, + move |data: &[T], _| { + let mono = data + .chunks(channels) + .map(|frame| { + frame + .iter() + .map(|&sample| f32::from_sample(sample)) + .sum::() + / frame.len() as f32 + }) + .collect(); + _ = tx.try_send(Capture::Samples(mono)); + }, + move |error| { + _ = error_tx.try_send(Capture::Error(error.to_string())); + }, + None, + ) + .map_err(|error| match error { + cpal::BuildStreamError::DeviceNotAvailable => SpeechError::NoInputDevice, + error => SpeechError::input(error), + }) +} + +/// Converts mono `f32` audio at the device rate to interleaved `i16` audio in +/// the recognizer's format, carrying state across chunks. +struct Converter { + /// Source samples per output sample. + step: f64, + /// Position of the next output sample, in source samples, where `0.0` is + /// the last sample of the previous chunk. + position: f64, + previous: f32, + low_pass: Option, + channels: usize, +} + +impl Converter { + fn new(source_rate: u32, format: AudioFormat) -> Self { + let target_rate = format.sample_rate().max(1); + let source_rate = source_rate.max(1); + Self { + step: source_rate as f64 / target_rate as f64, + position: 1., + previous: 0., + // Downsampling folds everything above the new Nyquist rate back + // into the band speech lives in; filter it out first. + low_pass: (source_rate > target_rate) + .then(|| LowPass::new(target_rate as f32 * 0.45, source_rate as f32)), + channels: format.channels().max(1) as usize, + } + } + + fn convert(&mut self, input: &[f32]) -> Vec { + if input.is_empty() { + return Vec::new(); + } + let filtered; + let input = match self.low_pass.as_mut() { + Some(low_pass) => { + filtered = input + .iter() + .map(|&x| low_pass.process(x)) + .collect::>(); + &filtered[..] + } + None => input, + }; + + let sample = |ix: usize| { + if ix == 0 { + self.previous + } else { + input[ix - 1] + } + }; + let last = input.len(); + let len = last as f64; + let mut output = Vec::with_capacity(((len / self.step) as usize + 1) * self.channels); + while self.position <= len { + let ix = self.position.floor() as usize; + let frac = (self.position - ix as f64) as f32; + let next = sample((ix + 1).min(last)); + let value = sample(ix) * (1. - frac) + next * frac; + let value = (value.clamp(-1., 1.) * i16::MAX as f32) as i16; + output.extend(std::iter::repeat_n(value, self.channels)); + self.position += self.step; + } + self.position -= len; + self.previous = input[input.len() - 1]; + output + } +} + +/// Two cascaded one-pole low-pass filters, 12 dB per octave. +struct LowPass { + alpha: f32, + stages: [f32; 2], +} + +impl LowPass { + fn new(cutoff: f32, sample_rate: f32) -> Self { + Self { + alpha: 1. - (-TAU * cutoff / sample_rate).exp(), + stages: [0.; 2], + } + } + + fn process(&mut self, mut x: f32) -> f32 { + for stage in &mut self.stages { + *stage += self.alpha * (x - *stage); + x = *stage; + } + x + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn converter_keeps_rate_and_duplicates_channels() { + let mut converter = Converter::new(16_000, AudioFormat::new(16_000, 2)); + let output = converter.convert(&[0., 0.5, -0.5]); + assert_eq!(output, vec![0, 0, 16383, 16383, -16383, -16383]); + } + + #[test] + fn converter_downsamples_across_chunks() { + let mut converter = Converter::new(48_000, AudioFormat::default()); + let total: usize = (0..10).map(|_| converter.convert(&[0.; 480]).len()).sum(); + // 100 ms at 48 kHz is 100 ms at 16 kHz, whatever the chunking. + assert_eq!(total, 1_600); + } + + #[test] + fn converter_upsamples() { + let mut converter = Converter::new(8_000, AudioFormat::default()); + let total: usize = (0..10).map(|_| converter.convert(&[0.; 80]).len()).sum(); + // The first output lands on the first input sample, so the half step + // before it is never produced. + assert_eq!(total, 1_599); + } +} diff --git a/crates/component/src/speech/mod.rs b/crates/component/src/speech/mod.rs new file mode 100644 index 0000000000..24f775b288 --- /dev/null +++ b/crates/component/src/speech/mod.rs @@ -0,0 +1,51 @@ +//! Speech input: capture audio, recognize it and hand the text to the caller. +//! +//! [`SpeechState`] owns a session and [`SpeechButton`] and [`SpeechWaveform`] +//! render it. Recognition and audio capture are both replaceable: +//! implement [`SpeechRecognizer`] to use any speech service and [`AudioInput`] +//! to feed audio from anywhere. +//! +//! With the `speech` feature, a state without its own recognizer falls back to +//! the `SystemRecognizer` of macOS or Windows, and captures from the +//! `Microphone`. Linux has no system recognizer, so speech input works there +//! only with an application recognizer. + +mod button; +mod recognizer; +mod state; +mod waveform; + +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +mod microphone; +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +mod system; + +use std::rc::Rc; + +pub use button::SpeechButton; +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +pub use microphone::Microphone; +pub use recognizer::{ + AudioFormat, AudioInput, AudioSink, RecognitionSession, SpeechError, SpeechRecognizer, + SpeechSink, +}; +pub use state::{SpeechEvent, SpeechState, SpeechStatus}; +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +pub use system::SystemRecognizer; +pub use waveform::SpeechWaveform; + +/// The input a [`SpeechState`] captures from unless told otherwise. +fn default_input() -> Option> { + #[cfg(all(feature = "speech", not(target_family = "wasm")))] + return Some(Rc::new(Microphone::default())); + #[allow(unreachable_code)] + None +} + +/// The recognizer a [`SpeechState`] falls back to, if this platform has one. +fn system_recognizer() -> Option> { + #[cfg(all(feature = "speech", any(target_os = "macos", target_os = "windows")))] + return Some(Rc::new(SystemRecognizer::new())); + #[allow(unreachable_code)] + None +} diff --git a/crates/component/src/speech/recognizer.rs b/crates/component/src/speech/recognizer.rs new file mode 100644 index 0000000000..f0456fd396 --- /dev/null +++ b/crates/component/src/speech/recognizer.rs @@ -0,0 +1,266 @@ +use std::{fmt, rc::Rc, sync::Arc}; + +use gpui::{App, Context, SharedString, Subscription, WeakEntity}; + +use super::{SpeechState, state::defer_session_update}; + +/// The PCM format a [`SpeechRecognizer`] consumes. +/// +/// Audio always arrives as interleaved signed 16-bit samples; an [`AudioInput`] +/// converts whatever its device produces to this rate and channel count. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct AudioFormat { + sample_rate: u32, + channels: u16, +} + +impl AudioFormat { + /// A format of `sample_rate` samples per second on each of `channels`. + pub fn new(sample_rate: u32, channels: u16) -> Self { + Self { + sample_rate, + channels, + } + } + + /// Samples per second of each channel. + pub fn sample_rate(&self) -> u32 { + self.sample_rate + } + + /// Number of interleaved channels. + pub fn channels(&self) -> u16 { + self.channels + } +} + +impl Default for AudioFormat { + /// 16 kHz mono, the format most speech services expect. + fn default() -> Self { + Self::new(16_000, 1) + } +} + +/// Why a speech session failed. +#[derive(Debug, Clone)] +pub enum SpeechError { + /// The user or the system denied access to the microphone. + PermissionDenied, + /// No audio input device is available. + NoInputDevice, + /// Speech input is not supported on this platform or build. + Unsupported, + /// The audio input failed or its device went away. + Input(Arc), + /// The recognizer failed, e.g. it could not reach its service. + Recognizer(Arc), +} + +impl SpeechError { + /// An [`SpeechError::Input`] from any error. + pub fn input(error: impl Into) -> Self { + Self::Input(Arc::new(error.into())) + } + + /// A [`SpeechError::Recognizer`] from any error. + pub fn recognizer(error: impl Into) -> Self { + Self::Recognizer(Arc::new(error.into())) + } +} + +impl fmt::Display for SpeechError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::PermissionDenied => f.write_str("microphone access was denied"), + Self::NoInputDevice => f.write_str("no audio input device is available"), + Self::Unsupported => f.write_str("speech input is not supported"), + Self::Input(error) => write!(f, "audio input failed: {error:#}"), + Self::Recognizer(error) => write!(f, "speech recognition failed: {error:#}"), + } + } +} + +impl std::error::Error for SpeechError {} + +/// Turns speech into text, e.g. by streaming audio to a cloud service. +/// +/// Implement it for the service the application uses and pass it to +/// [`SpeechState::recognizer`]. Without one, the state falls back to the +/// platform's own recognizer where there is one (see +/// `SystemRecognizer`). +/// +/// A session runs as follows: +/// +/// 1. [`start`](Self::start) opens a session. Connecting may take a while, so +/// return immediately and buffer the audio pushed in the meantime; call +/// [`SpeechSink::ready`] once the service accepts audio. +/// 2. [`RecognitionSession::push_audio`] delivers PCM in [`Self::audio_format`]. +/// Report results through [`SpeechSink::hypothesis`] and +/// [`SpeechSink::phrase`] as they arrive. +/// 3. [`RecognitionSession::finish`] means the user stopped talking: send the +/// remaining audio, wait for the last result, then call +/// [`SpeechSink::finish`]. +/// +/// Dropping the [`RecognitionSession`] cancels it: close the connection and +/// report nothing more. Every [`SpeechSink`] method may be called from any +/// point on the main thread, including from inside `start` or `push_audio`. +pub trait SpeechRecognizer: 'static { + /// The audio format this recognizer consumes, default 16 kHz mono. + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + /// Whether the recognizer can start a session now, default `true`. + /// + /// Return `false` while it cannot work, e.g. before the user signs in; the + /// [`SpeechButton`](super::SpeechButton) then renders disabled. Called on + /// every render, so keep it cheap. + fn is_available(&self, _cx: &App) -> bool { + true + } + + /// Open a session that reports its results to `sink`. + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError>; +} + +impl SpeechRecognizer for Rc { + fn audio_format(&self) -> AudioFormat { + (**self).audio_format() + } + + fn is_available(&self, cx: &App) -> bool { + (**self).is_available(cx) + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + (**self).start(sink, cx) + } +} + +/// One running recognition, opened by [`SpeechRecognizer::start`]. +/// +/// Dropping it cancels the session. +pub trait RecognitionSession: 'static { + /// Deliver interleaved PCM samples in the recognizer's [`AudioFormat`]. + fn push_audio(&mut self, samples: &[i16], cx: &mut App); + + /// No more audio will arrive. Flush what is buffered and call + /// [`SpeechSink::finish`] once the final result is in. + fn finish(&mut self, cx: &mut App); +} + +/// Where a [`SpeechRecognizer`] reports its session's progress. +/// +/// The sink is cheap to clone and may outlive its session: once the session is +/// stopped, cancelled or replaced, its calls are ignored. Calls are applied +/// after the current update, so they never re-enter the [`SpeechState`]. +#[derive(Clone)] +pub struct SpeechSink { + pub(super) state: WeakEntity, + pub(super) session: usize, +} + +impl SpeechSink { + /// The service is connected and consuming audio. + pub fn ready(&self, cx: &mut App) { + self.apply(cx, |state, cx| state.on_ready(cx)); + } + + /// Replace the hypothesis for the phrase being spoken. + pub fn hypothesis(&self, text: impl Into, cx: &mut App) { + let text = text.into(); + self.apply(cx, move |state, cx| state.on_hypothesis(text, cx)); + } + + /// Commit a recognized phrase and clear the hypothesis. + /// + /// Phrases are joined verbatim, so include any separator the language + /// needs, such as a leading space between English sentences. + pub fn phrase(&self, text: impl Into, cx: &mut App) { + let text = text.into(); + self.apply(cx, move |state, cx| state.on_phrase(text, cx)); + } + + /// The session is complete; no more results will follow. + pub fn finish(&self, cx: &mut App) { + self.apply(cx, |state, cx| state.on_finish(cx)); + } + + /// The session failed. + pub fn error(&self, error: SpeechError, cx: &mut App) { + self.apply(cx, move |state, cx| state.on_error(error, cx)); + } + + fn apply( + &self, + cx: &mut App, + f: impl FnOnce(&mut SpeechState, &mut Context) + 'static, + ) { + defer_session_update(self.state.clone(), self.session, cx, f); + } +} + +/// A source of audio for a [`SpeechState`], such as the microphone. +/// +/// Enable the `speech` feature for the built-in `Microphone`, +/// or implement this trait to feed audio from elsewhere, e.g. a file in tests. +pub trait AudioInput: 'static { + /// Start capturing `format` audio into `sink`. + /// + /// Capture runs until the returned [`Subscription`] is dropped. + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result; +} + +impl AudioInput for Rc { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + (**self).start(format, sink, cx) + } +} + +/// Where an [`AudioInput`] delivers captured audio. +/// +/// Like [`SpeechSink`], it is cheap to clone, ignored once its session ends and +/// applied after the current update. +#[derive(Clone)] +pub struct AudioSink { + pub(super) state: WeakEntity, + pub(super) session: usize, +} + +impl AudioSink { + /// Deliver interleaved PCM samples in the requested [`AudioFormat`]. + pub fn push(&self, samples: Vec, cx: &mut App) { + self.apply(cx, move |state, cx| state.on_audio(&samples, cx)); + } + + /// Capture failed; the session ends with this error. + pub fn error(&self, error: SpeechError, cx: &mut App) { + self.apply(cx, move |state, cx| state.on_error(error, cx)); + } + + fn apply( + &self, + cx: &mut App, + f: impl FnOnce(&mut SpeechState, &mut Context) + 'static, + ) { + defer_session_update(self.state.clone(), self.session, cx, f); + } +} diff --git a/crates/component/src/speech/state.rs b/crates/component/src/speech/state.rs new file mode 100644 index 0000000000..f4a79376a2 --- /dev/null +++ b/crates/component/src/speech/state.rs @@ -0,0 +1,689 @@ +use std::{cell::OnceCell, collections::VecDeque, rc::Rc, time::Duration}; + +use gpui::{App, Context, EventEmitter, SharedString, Subscription, Task, WeakEntity}; + +use super::{AudioInput, AudioSink, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// How many recent input levels a [`SpeechState`] keeps for a waveform. +pub(super) const LEVEL_HISTORY: usize = 48; + +/// Default time [`SpeechState::stop`] waits for the final result. +const DEFAULT_STOP_TIMEOUT: Duration = Duration::from_secs(3); + +/// Where a [`SpeechState`] is in its session. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum SpeechStatus { + /// No session is running. + #[default] + Idle, + /// Audio is being captured while the recognizer connects. + Connecting, + /// Audio is being captured and recognized. + Recording, + /// Capture stopped; waiting for the recognizer's final result. + Stopping, +} + +impl SpeechStatus { + /// Whether a session is running. + pub fn is_active(self) -> bool { + self != Self::Idle + } + + /// Whether the microphone is capturing. + pub fn is_capturing(self) -> bool { + matches!(self, Self::Connecting | Self::Recording) + } +} + +/// Events emitted by [`SpeechState`]. +#[derive(Debug, Clone)] +pub enum SpeechEvent { + /// A session started and audio is being captured. + Started, + /// The transcript changed: every committed phrase followed by the current + /// hypothesis. Later events supersede earlier ones. + Partial(SharedString), + /// The session ended normally with this transcript, possibly empty. + Final(SharedString), + /// The session was cancelled and its transcript discarded. + Cancelled, + /// The session failed and ended. + Error(SpeechError), +} + +struct Session { + id: usize, + /// Dropping this stops capture. + capture: Option, + recognition: Box, + _stop_timeout: Option>, +} + +/// The state of a speech input: captures audio from an [`AudioInput`], feeds it +/// to a [`SpeechRecognizer`] and tracks the transcript. +/// +/// Render it with [`SpeechButton`](super::SpeechButton) and +/// [`SpeechWaveform`](super::SpeechWaveform), and subscribe to [`SpeechEvent`] +/// to receive the text. +/// +/// The recognizer is, in order: the one passed to [`Self::recognizer`]; else +/// the platform's `SystemRecognizer`, unless +/// [`Self::system_fallback`] turned it off; else none, and the state is not +/// available. The input defaults to the `Microphone`. +/// Both defaults need the `speech` feature. +pub struct SpeechState { + recognizer: Option>, + input: Option>, + system_fallback: bool, + system_recognizer: OnceCell>>, + stop_timeout: Duration, + status: SpeechStatus, + session: Option, + next_session: usize, + committed: String, + hypothesis: SharedString, + levels: VecDeque, +} + +impl EventEmitter for SpeechState {} + +impl SpeechState { + /// Create a speech state with the default recognizer and input. + pub fn new(_: &mut Context) -> Self { + Self { + recognizer: None, + input: super::default_input(), + system_fallback: true, + system_recognizer: OnceCell::new(), + stop_timeout: DEFAULT_STOP_TIMEOUT, + status: SpeechStatus::Idle, + session: None, + next_session: 0, + committed: String::new(), + hypothesis: SharedString::default(), + levels: VecDeque::with_capacity(LEVEL_HISTORY), + } + } + + /// Recognize speech with `recognizer` instead of the system's. + pub fn recognizer(mut self, recognizer: impl SpeechRecognizer) -> Self { + self.recognizer = Some(Rc::new(recognizer)); + self + } + + /// Capture audio from `input` instead of the microphone. + pub fn input(mut self, input: impl AudioInput) -> Self { + self.input = Some(Rc::new(input)); + self + } + + /// Whether to fall back to the platform's recognizer when no + /// [`Self::recognizer`] is set, default `true`. + /// + /// On Windows the system recognizer dictates through Microsoft's online + /// service; turn this off when audio must not leave the application. + pub fn system_fallback(mut self, system_fallback: bool) -> Self { + self.system_fallback = system_fallback; + self + } + + /// Set how long [`Self::stop`] waits for the recognizer's final result + /// before ending the session with the transcript so far, default 3 seconds. + pub fn stop_timeout(mut self, timeout: Duration) -> Self { + self.stop_timeout = timeout; + self + } + + /// Whether a recognizer and an input are configured, so that speech input + /// can work on this platform at all. + pub fn has_recognizer(&self) -> bool { + self.input.is_some() && self.active_recognizer().is_some() + } + + /// Whether a session can start now: [`Self::has_recognizer`] and the + /// recognizer reports itself available. + pub fn is_available(&self, cx: &App) -> bool { + self.input.is_some() + && self + .active_recognizer() + .is_some_and(|recognizer| recognizer.is_available(cx)) + } + + /// Where the state is in its session. + pub fn status(&self) -> SpeechStatus { + self.status + } + + /// The transcript of the current or last session: every committed phrase + /// followed by the current hypothesis. + pub fn transcript(&self) -> SharedString { + if self.hypothesis.is_empty() { + self.committed.clone().into() + } else { + format!("{}{}", self.committed, self.hypothesis).into() + } + } + + /// Recent input levels in `0.0..=1.0`, oldest first. + pub fn levels(&self) -> impl ExactSizeIterator + '_ { + self.levels.iter().copied() + } + + /// Start a session. Does nothing while one is running. + /// + /// When the recognizer or the input fails to start, emits + /// [`SpeechEvent::Error`] and stays idle. + pub fn start(&mut self, cx: &mut Context) { + if self.status.is_active() { + return; + } + + let (Some(recognizer), Some(input)) = (self.active_recognizer(), self.input.clone()) else { + cx.emit(SpeechEvent::Error(SpeechError::Unsupported)); + return; + }; + + self.next_session += 1; + let id = self.next_session; + let state = cx.weak_entity(); + let sink = SpeechSink { + state: state.clone(), + session: id, + }; + let recognition = match recognizer.start(sink, cx) { + Ok(recognition) => recognition, + Err(error) => { + cx.emit(SpeechEvent::Error(error)); + return; + } + }; + + let format = recognizer.audio_format(); + let sink = AudioSink { state, session: id }; + let capture = match input.start(format, sink, cx) { + Ok(capture) => capture, + Err(error) => { + cx.emit(SpeechEvent::Error(error)); + return; + } + }; + + self.status = SpeechStatus::Connecting; + self.committed.clear(); + self.hypothesis = SharedString::default(); + self.levels.clear(); + self.session = Some(Session { + id, + capture: Some(capture), + recognition, + _stop_timeout: None, + }); + cx.emit(SpeechEvent::Started); + cx.notify(); + } + + /// Stop capturing and wait for the final result, which arrives as + /// [`SpeechEvent::Final`]. + pub fn stop(&mut self, cx: &mut Context) { + if !self.status.is_capturing() { + return; + } + let Some(session) = self.session.as_mut() else { + return; + }; + + session.capture = None; + session.recognition.finish(cx); + + let id = session.id; + let timeout = self.stop_timeout; + session._stop_timeout = Some(cx.spawn(async move |this, cx| { + cx.background_executor().timer(timeout).await; + _ = this.update(cx, |this, cx| { + if this.is_session(id) { + this.end(SpeechEvent::Final(this.transcript()), cx); + } + }); + })); + + self.status = SpeechStatus::Stopping; + cx.notify(); + } + + /// End the session at once and discard its transcript. + pub fn cancel(&mut self, cx: &mut Context) { + if !self.status.is_active() { + return; + } + self.committed.clear(); + self.hypothesis = SharedString::default(); + self.end(SpeechEvent::Cancelled, cx); + } + + /// Start a session when idle, otherwise stop the running one. + pub fn toggle(&mut self, cx: &mut Context) { + if self.status.is_active() { + self.stop(cx); + } else { + self.start(cx); + } + } + + fn active_recognizer(&self) -> Option> { + if let Some(recognizer) = &self.recognizer { + return Some(recognizer.clone()); + } + if !self.system_fallback { + return None; + } + self.system_recognizer + .get_or_init(super::system_recognizer) + .clone() + } + + pub(super) fn is_session(&self, id: usize) -> bool { + self.session + .as_ref() + .is_some_and(|session| session.id == id) + } + + pub(super) fn on_ready(&mut self, cx: &mut Context) { + if self.status == SpeechStatus::Connecting { + self.status = SpeechStatus::Recording; + cx.notify(); + } + } + + pub(super) fn on_audio(&mut self, samples: &[i16], cx: &mut Context) { + if !self.status.is_capturing() { + return; + } + if let Some(session) = self.session.as_mut() { + session.recognition.push_audio(samples, cx); + } + if self.levels.len() == LEVEL_HISTORY { + self.levels.pop_front(); + } + self.levels.push_back(level(samples)); + cx.notify(); + } + + pub(super) fn on_hypothesis(&mut self, text: SharedString, cx: &mut Context) { + self.hypothesis = text; + cx.emit(SpeechEvent::Partial(self.transcript())); + cx.notify(); + } + + pub(super) fn on_phrase(&mut self, text: SharedString, cx: &mut Context) { + self.committed.push_str(&text); + self.hypothesis = SharedString::default(); + cx.emit(SpeechEvent::Partial(self.transcript())); + cx.notify(); + } + + pub(super) fn on_finish(&mut self, cx: &mut Context) { + self.end(SpeechEvent::Final(self.transcript()), cx); + } + + pub(super) fn on_error(&mut self, error: SpeechError, cx: &mut Context) { + self.end(SpeechEvent::Error(error), cx); + } + + /// Tear the session down, capture first, and report how it ended. + fn end(&mut self, event: SpeechEvent, cx: &mut Context) { + if let Some(mut session) = self.session.take() { + session.capture = None; + } + self.status = SpeechStatus::Idle; + self.levels.clear(); + cx.emit(event); + cx.notify(); + } +} + +/// Apply `f` to `state` after the current update, if `session` is still its +/// running session. Sinks go through this so that a recognizer or an input may +/// report from anywhere, including from inside a call the state made. +pub(super) fn defer_session_update( + state: WeakEntity, + session: usize, + cx: &mut App, + f: impl FnOnce(&mut SpeechState, &mut Context) + 'static, +) { + cx.defer(move |cx| { + _ = state.update(cx, |state, cx| { + if state.is_session(session) { + f(state, cx); + } + }); + }); +} + +/// The loudness of `samples` in `0.0..=1.0`, mapping -50 dBFS..0 dBFS linearly +/// so that normal speech fills most of the range. +fn level(samples: &[i16]) -> f32 { + if samples.is_empty() { + return 0.; + } + let sum: f64 = samples + .iter() + .map(|&sample| { + let sample = sample as f64 / i16::MAX as f64; + sample * sample + }) + .sum(); + let rms = (sum / samples.len() as f64).sqrt(); + if rms <= 0. { + return 0.; + } + let db = 20. * rms.log10(); + ((db + 50.) / 50.).clamp(0., 1.) as f32 +} + +#[cfg(test)] +mod tests { + use std::{cell::RefCell, rc::Rc, time::Duration}; + + use gpui::{App, AppContext as _, Entity, Subscription, TestAppContext}; + + use super::*; + use crate::speech::{AudioFormat, AudioInput, AudioSink, RecognitionSession, SpeechSink}; + + #[derive(Default)] + struct Recorded { + sink: Option, + samples: usize, + finished: bool, + dropped: bool, + } + + #[derive(Clone, Default)] + struct FakeRecognizer { + recorded: Rc>, + fail_to_start: bool, + } + + struct FakeSession(Rc>); + + impl SpeechRecognizer for FakeRecognizer { + fn start( + &self, + sink: SpeechSink, + _: &mut App, + ) -> Result, SpeechError> { + if self.fail_to_start { + return Err(SpeechError::recognizer(anyhow::anyhow!("offline"))); + } + *self.recorded.borrow_mut() = Recorded { + sink: Some(sink), + ..Default::default() + }; + Ok(Box::new(FakeSession(self.recorded.clone()))) + } + } + + impl RecognitionSession for FakeSession { + fn push_audio(&mut self, samples: &[i16], _: &mut App) { + self.0.borrow_mut().samples += samples.len(); + } + + fn finish(&mut self, _: &mut App) { + self.0.borrow_mut().finished = true; + } + } + + impl Drop for FakeSession { + fn drop(&mut self) { + self.0.borrow_mut().dropped = true; + } + } + + #[derive(Clone, Default)] + struct FakeInput { + sink: Rc>>, + capturing: Rc>, + } + + impl AudioInput for FakeInput { + fn start( + &self, + _: AudioFormat, + sink: AudioSink, + _: &mut App, + ) -> Result { + *self.sink.borrow_mut() = Some(sink); + *self.capturing.borrow_mut() = true; + let capturing = self.capturing.clone(); + Ok(Subscription::new(move || *capturing.borrow_mut() = false)) + } + } + + struct Fixture { + state: Entity, + recognizer: FakeRecognizer, + input: FakeInput, + events: Rc>>, + _subscription: Subscription, + } + + impl Fixture { + fn new(recognizer: FakeRecognizer, cx: &mut TestAppContext) -> Self { + let input = FakeInput::default(); + let state = cx.update(|cx| { + cx.new(|cx| { + SpeechState::new(cx) + .recognizer(recognizer.clone()) + .input(input.clone()) + .stop_timeout(Duration::from_secs(1)) + }) + }); + let events = Rc::new(RefCell::new(Vec::new())); + let _subscription = cx.update(|cx| { + let events = events.clone(); + cx.subscribe(&state, move |_, event: &SpeechEvent, _| { + events.borrow_mut().push(event.clone()); + }) + }); + Self { + state, + recognizer, + input, + events, + _subscription, + } + } + + fn sink(&self) -> SpeechSink { + self.recognizer.recorded.borrow().sink.clone().unwrap() + } + + fn audio(&self) -> AudioSink { + self.input.sink.borrow().clone().unwrap() + } + + fn status(&self, cx: &mut TestAppContext) -> SpeechStatus { + cx.read(|cx| self.state.read(cx).status()) + } + + /// Events so far, as short labels. + fn take_events(&self) -> Vec { + self.events + .borrow_mut() + .drain(..) + .map(|event| match event { + SpeechEvent::Started => "started".into(), + SpeechEvent::Partial(text) => format!("partial:{text}"), + SpeechEvent::Final(text) => format!("final:{text}"), + SpeechEvent::Cancelled => "cancelled".into(), + SpeechEvent::Error(error) => format!("error:{error}"), + }) + .collect() + } + } + + #[gpui::test] + fn session_runs_from_start_to_final(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + + f.state.update(cx, |state, cx| state.start(cx)); + assert_eq!(f.status(cx), SpeechStatus::Connecting); + assert!(*f.input.capturing.borrow()); + + cx.update(|cx| { + f.sink().ready(cx); + f.audio().push(vec![i16::MAX / 2; 160], cx); + f.sink().hypothesis("hello", cx); + }); + cx.run_until_parked(); + assert_eq!(f.status(cx), SpeechStatus::Recording); + assert_eq!(f.recognizer.recorded.borrow().samples, 160); + cx.read(|cx| { + let state = f.state.read(cx); + assert_eq!(state.levels().len(), 1); + assert!(state.levels().next().unwrap() > 0.5); + }); + + cx.update(|cx| { + f.sink().phrase("Hello.", cx); + f.sink().hypothesis(" How", cx); + }); + cx.run_until_parked(); + assert_eq!( + cx.read(|cx| f.state.read(cx).transcript()), + SharedString::from("Hello. How") + ); + + f.state.update(cx, |state, cx| state.stop(cx)); + assert_eq!(f.status(cx), SpeechStatus::Stopping); + assert!(!*f.input.capturing.borrow(), "stop releases the microphone"); + assert!(f.recognizer.recorded.borrow().finished); + + cx.update(|cx| { + f.sink().phrase(" How are you?", cx); + f.sink().finish(cx); + }); + cx.run_until_parked(); + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(f.recognizer.recorded.borrow().dropped); + assert_eq!( + f.take_events(), + [ + "started", + "partial:hello", + "partial:Hello.", + "partial:Hello. How", + "partial:Hello. How are you?", + "final:Hello. How are you?", + ] + ); + } + + #[gpui::test] + fn stop_ends_with_the_transcript_so_far_after_the_timeout(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + f.state.update(cx, |state, cx| state.start(cx)); + cx.update(|cx| f.sink().hypothesis("half a sentence", cx)); + cx.run_until_parked(); + + f.state.update(cx, |state, cx| state.stop(cx)); + cx.executor().advance_clock(Duration::from_millis(900)); + assert_eq!(f.status(cx), SpeechStatus::Stopping); + cx.executor().advance_clock(Duration::from_millis(200)); + cx.run_until_parked(); + + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert_eq!(f.take_events().last().unwrap(), "final:half a sentence"); + } + + #[gpui::test] + fn cancel_discards_the_session_and_ignores_late_results(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + f.state.update(cx, |state, cx| state.start(cx)); + let stale = f.sink(); + cx.update(|cx| stale.phrase("draft", cx)); + cx.run_until_parked(); + + f.state.update(cx, |state, cx| state.cancel(cx)); + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(!*f.input.capturing.borrow()); + assert!(f.recognizer.recorded.borrow().dropped); + + // A new session must not pick up the old session's results. + f.state.update(cx, |state, cx| state.start(cx)); + cx.update(|cx| { + stale.phrase("late", cx); + stale.finish(cx); + }); + cx.run_until_parked(); + + assert_eq!(f.status(cx), SpeechStatus::Connecting); + assert_eq!( + cx.read(|cx| f.state.read(cx).transcript()), + SharedString::default() + ); + assert_eq!( + f.take_events(), + ["started", "partial:draft", "cancelled", "started"] + ); + } + + #[gpui::test] + fn input_error_ends_the_session(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + f.state.update(cx, |state, cx| state.start(cx)); + cx.update(|cx| f.audio().error(SpeechError::NoInputDevice, cx)); + cx.run_until_parked(); + + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(f.recognizer.recorded.borrow().dropped); + assert_eq!( + f.take_events(), + ["started", "error:no audio input device is available"] + ); + } + + #[gpui::test] + fn recognizer_that_fails_to_start_leaves_the_state_idle(cx: &mut TestAppContext) { + let f = Fixture::new( + FakeRecognizer { + fail_to_start: true, + ..Default::default() + }, + cx, + ); + f.state.update(cx, |state, cx| state.start(cx)); + + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(!*f.input.capturing.borrow(), "the input never starts"); + assert_eq!( + f.take_events(), + ["error:speech recognition failed: offline"] + ); + } + + #[gpui::test] + fn state_without_a_recognizer_is_unsupported(cx: &mut TestAppContext) { + let state = cx.update(|cx| { + cx.new(|cx| { + SpeechState::new(cx) + .input(FakeInput::default()) + .system_fallback(false) + }) + }); + cx.read(|cx| { + assert!(!state.read(cx).has_recognizer()); + assert!(!state.read(cx).is_available(cx)); + }); + } + + #[test] + fn level_maps_decibels_to_the_unit_range() { + assert_eq!(level(&[]), 0.); + assert_eq!(level(&[0; 16]), 0.); + assert_eq!(level(&[i16::MAX; 16]), 1.); + // -20 dBFS sits at 0.6 on the -50..0 dB scale. + let quiet = level(&[i16::MAX / 10; 16]); + assert!((quiet - 0.6).abs() < 0.01, "{quiet}"); + } +} diff --git a/crates/component/src/speech/system/macos.rs b/crates/component/src/speech/system/macos.rs new file mode 100644 index 0000000000..95ad16a2ac --- /dev/null +++ b/crates/component/src/speech/system/macos.rs @@ -0,0 +1,486 @@ +//! `SFSpeechRecognizer`, recognizing on the device only. + +use std::{cell::RefCell, rc::Rc}; + +use anyhow::anyhow; +use block2::RcBlock; +use gpui::{App, SharedString, Task}; +use objc2::{AnyThread as _, rc::Retained, runtime::NSObjectProtocol as _, sel}; +use objc2_av_foundation::{AVAuthorizationStatus, AVCaptureDevice, AVMediaTypeAudio}; +use objc2_avf_audio::{AVAudioCommonFormat, AVAudioFormat, AVAudioPCMBuffer}; +use objc2_foundation::{NSBundle, NSError, NSLocale, NSString}; +use objc2_speech::{ + SFSpeechAudioBufferRecognitionRequest, SFSpeechRecognitionResult, SFSpeechRecognitionTask, + SFSpeechRecognizer, SFSpeechRecognizerAuthorizationStatus, +}; +use smol::channel::{Receiver, Sender, unbounded}; + +use crate::speech::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// Without this `Info.plist` key, asking for speech recognition access +/// terminates the process. +const USAGE_DESCRIPTION_KEY: &str = "NSSpeechRecognitionUsageDescription"; + +/// The error `SFSpeechRecognizer` reports when the audio held no speech. +const NO_SPEECH_DOMAIN: &str = "kAFAssistantErrorDomain"; +const NO_SPEECH_CODE: isize = 1110; + +pub(super) struct PlatformRecognizer { + /// `None` when the locale has no recognizer or the application cannot ask + /// for access. + recognizer: Option>, +} + +impl PlatformRecognizer { + pub(super) fn new(locale: Option) -> Self { + if !has_usage_description() { + tracing::warn!( + "speech recognition is unavailable: the application's Info.plist has no \ + `{USAGE_DESCRIPTION_KEY}`, and asking for access without it terminates the process" + ); + return Self { recognizer: None }; + } + + let recognizer = match &locale { + Some(locale) => { + let locale = NSLocale::initWithLocaleIdentifier( + NSLocale::alloc(), + &NSString::from_str(locale), + ); + // SAFETY: `locale` is a valid `NSLocale`; the initializer returns + // nil for a locale without a recognizer. + unsafe { SFSpeechRecognizer::initWithLocale(SFSpeechRecognizer::alloc(), &locale) } + } + // SAFETY: plain initializer; returns nil if the system language has + // no recognizer. + None => unsafe { SFSpeechRecognizer::init(SFSpeechRecognizer::alloc()) }, + }; + if recognizer.is_none() { + tracing::warn!( + "speech recognition is unavailable: no recognizer for locale {}", + locale.as_deref().unwrap_or("of the system") + ); + } + + Self { recognizer } + } + + /// The recognizer, if it can recognize on the device. + fn on_device(&self) -> Option<&Retained> { + self.recognizer + .as_ref() + // SAFETY: plain property reads on a valid recognizer. + .filter(|recognizer| unsafe { + recognizer.isAvailable() && recognizer.supportsOnDeviceRecognition() + }) + } +} + +impl SpeechRecognizer for PlatformRecognizer { + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + fn is_available(&self, _: &App) -> bool { + self.on_device().is_some() && !is_speech_denied(speech_authorization()) + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + let recognizer = self.on_device().ok_or(SpeechError::Unsupported)?.clone(); + let authorization = speech_authorization(); + if is_speech_denied(authorization) { + return Err(SpeechError::PermissionDenied); + } + + let (events, rx) = unbounded(); + let mut recognition = Recognition::new(recognizer, sink, events.clone())?; + if authorization == SFSpeechRecognizerAuthorizationStatus::Authorized { + recognition.begin(cx); + } else { + request_authorization(events); + } + + let recognition = Rc::new(RefCell::new(recognition)); + let task = cx.spawn({ + let recognition = recognition.clone(); + async move |cx| run(recognition, rx, cx).await + }); + + Ok(Box::new(Session { + recognition, + _task: task, + })) + } +} + +/// Whether the user or a policy denied this application the microphone. +/// +/// Capturing without access yields silence rather than an error on macOS, so +/// the [`Microphone`](crate::speech::Microphone) checks this first. +pub(in crate::speech) fn is_microphone_denied() -> bool { + // SAFETY: reading an immutable framework constant. + let Some(audio) = (unsafe { AVMediaTypeAudio }) else { + return false; + }; + // SAFETY: `audio` is one of the two media types the method accepts. + let status = unsafe { AVCaptureDevice::authorizationStatusForMediaType(audio) }; + status == AVAuthorizationStatus::Denied || status == AVAuthorizationStatus::Restricted +} + +fn has_usage_description() -> bool { + NSBundle::mainBundle() + .objectForInfoDictionaryKey(&NSString::from_str(USAGE_DESCRIPTION_KEY)) + .is_some() +} + +fn speech_authorization() -> SFSpeechRecognizerAuthorizationStatus { + // SAFETY: reading the status never prompts, so it is safe without the + // usage description. + unsafe { SFSpeechRecognizer::authorizationStatus() } +} + +fn is_speech_denied(status: SFSpeechRecognizerAuthorizationStatus) -> bool { + status == SFSpeechRecognizerAuthorizationStatus::Denied + || status == SFSpeechRecognizerAuthorizationStatus::Restricted +} + +/// Ask for speech recognition access; the answer arrives as an [`Event`]. +fn request_authorization(events: Sender) { + let handler = RcBlock::new(move |status: SFSpeechRecognizerAuthorizationStatus| { + _ = events.try_send(Event::Authorized( + status == SFSpeechRecognizerAuthorizationStatus::Authorized, + )); + }); + // SAFETY: only reached with the usage description present (see + // `PlatformRecognizer::new`); the block only sends on a channel, so it may + // run on any queue. + unsafe { SFSpeechRecognizer::requestAuthorization(&handler) }; +} + +/// What the framework reports, sent from its queues to the main thread. +enum Event { + Authorized(bool), + Result { + text: String, + is_final: bool, + /// The result ends an utterance, see [`Recognition::on_result`]. + ends_utterance: bool, + }, + Error { + domain: String, + code: isize, + message: String, + }, +} + +/// Deliver framework events to the recognition until it is done. +async fn run( + recognition: Rc>, + events: Receiver, + cx: &mut gpui::AsyncApp, +) { + while let Ok(event) = events.recv().await { + let done = cx.update(|cx| recognition.borrow_mut().on_event(event, cx)); + if done { + break; + } + } +} + +struct Session { + recognition: Rc>, + /// Dropping this stops delivering results. + _task: Task<()>, +} + +impl RecognitionSession for Session { + fn push_audio(&mut self, samples: &[i16], _: &mut App) { + self.recognition.borrow_mut().push_audio(samples); + } + + fn finish(&mut self, _: &mut App) { + self.recognition.borrow_mut().finish(); + } +} + +impl Drop for Session { + fn drop(&mut self) { + if let Some(task) = &self.recognition.borrow().task { + // SAFETY: cancelling a valid task; it reports nothing we still read. + unsafe { task.cancel() }; + } + } +} + +struct Recognition { + recognizer: Retained, + request: Retained, + format: Retained, + /// `None` until access is granted. + task: Option>, + sink: SpeechSink, + events: Sender, + /// Audio pushed before the task started. + pending: Vec, + finishing: bool, + done: bool, + /// The text of the last result that ended an utterance, until the next + /// result shows whether the recognizer carries it on. + utterance: Option, + /// The current hypothesis, as the framework reported it. + hypothesis: String, + /// The last character committed as a phrase, to join the next one. + last_committed: Option, +} + +impl Recognition { + fn new( + recognizer: Retained, + sink: SpeechSink, + events: Sender, + ) -> Result { + let format = AudioFormat::default(); + // 16-bit mono is also the request's native format, so appended audio + // needs no conversion. + // SAFETY: a mono PCM format, which the initializer supports. + let format = unsafe { + AVAudioFormat::initWithCommonFormat_sampleRate_channels_interleaved( + AVAudioFormat::alloc(), + AVAudioCommonFormat::PCMFormatInt16, + format.sample_rate().into(), + format.channels().into(), + false, + ) + } + .ok_or_else(|| SpeechError::recognizer(anyhow!("cannot create the audio format")))?; + + // SAFETY: plain initializer and property setters on a fresh request. + let request = unsafe { + let request = SFSpeechAudioBufferRecognitionRequest::new(); + request.setShouldReportPartialResults(true); + // Never send audio to Apple's servers. + request.setRequiresOnDeviceRecognition(true); + // Punctuation needs macOS 13. + if request.respondsToSelector(sel!(setAddsPunctuation:)) { + request.setAddsPunctuation(true); + } + request + }; + + Ok(Self { + recognizer, + request, + format, + task: None, + sink, + events, + pending: Vec::new(), + finishing: false, + done: false, + utterance: None, + hypothesis: String::new(), + last_committed: None, + }) + } + + /// Start recognizing, once access is granted. + fn begin(&mut self, cx: &mut App) { + let events = self.events.clone(); + let handler = RcBlock::new( + move |result: *mut SFSpeechRecognitionResult, error: *mut NSError| { + // SAFETY: the framework passes nil or objects valid for the call. + let (result, error) = unsafe { (result.as_ref(), error.as_ref()) }; + if let Some(result) = result { + // SAFETY: plain property reads on a valid result. + let event = unsafe { + Event::Result { + text: result.bestTranscription().formattedString().to_string(), + is_final: result.isFinal(), + ends_utterance: result.speechRecognitionMetadata().is_some(), + } + }; + _ = events.try_send(event); + } + if let Some(error) = error { + _ = events.try_send(Event::Error { + domain: error.domain().to_string(), + code: error.code(), + message: error.localizedDescription().to_string(), + }); + } + }, + ); + // SAFETY: the request is an audio buffer request and the block only + // sends on a channel, so it may run on any queue. + let task = unsafe { + self.recognizer + .recognitionTaskWithRequest_resultHandler(&self.request, &handler) + }; + self.task = Some(task); + + let pending = std::mem::take(&mut self.pending); + self.append(&pending); + if self.finishing { + // SAFETY: ending the audio of a valid request. + unsafe { self.request.endAudio() }; + } + self.sink.ready(cx); + } + + fn push_audio(&mut self, samples: &[i16]) { + if self.done || self.finishing { + return; + } + if self.task.is_some() { + self.append(samples); + } else { + self.pending.extend_from_slice(samples); + } + } + + fn finish(&mut self) { + if self.finishing { + return; + } + self.finishing = true; + if self.task.is_some() { + // SAFETY: ending the audio of a valid request. + unsafe { self.request.endAudio() }; + } + } + + fn append(&self, samples: &[i16]) { + let Ok(frames) = u32::try_from(samples.len()) else { + return; + }; + if frames == 0 { + return; + } + // SAFETY: `format` is PCM, so the initializer only fails for sizes + // beyond `u32`. + let Some(buffer) = (unsafe { + AVAudioPCMBuffer::initWithPCMFormat_frameCapacity( + AVAudioPCMBuffer::alloc(), + &self.format, + frames, + ) + }) else { + return; + }; + // SAFETY: the format is 16-bit mono, so channel 0 holds `frames` + // writable samples, and `frames` is within the capacity. + unsafe { + let channel = (*buffer.int16ChannelData()).as_ptr(); + std::ptr::copy_nonoverlapping(samples.as_ptr(), channel, samples.len()); + buffer.setFrameLength(frames); + self.request.appendAudioPCMBuffer(&buffer); + } + } + + /// Handle one framework event; returns whether the recognition is done. + fn on_event(&mut self, event: Event, cx: &mut App) -> bool { + if self.done { + return true; + } + match event { + Event::Authorized(true) => self.begin(cx), + Event::Authorized(false) => { + self.sink.error(SpeechError::PermissionDenied, cx); + self.done = true; + } + Event::Result { + text, + is_final, + ends_utterance, + } => self.on_result(text, is_final, ends_utterance, cx), + Event::Error { + domain, + code, + message, + } => { + if self.finishing && domain == NO_SPEECH_DOMAIN && code == NO_SPEECH_CODE { + // Stopping without (more) speech is a normal end. + let hypothesis = std::mem::take(&mut self.hypothesis); + self.commit(&hypothesis, cx); + self.sink.finish(cx); + } else { + self.sink.error( + SpeechError::recognizer(anyhow!("{message} ({domain} {code})")), + cx, + ); + } + self.done = true; + } + } + self.done + } + + /// Results carry the whole text of the request so far, except that some + /// macOS versions start over after a pause: the result that ends an + /// utterance has metadata, and the next one may no longer include its + /// text. Commit that utterance as a phrase only once it is dropped. + fn on_result(&mut self, text: String, is_final: bool, ends_utterance: bool, cx: &mut App) { + if let Some(utterance) = self.utterance.take() + && !text.starts_with(&utterance) + { + self.commit(&utterance, cx); + } + + if is_final { + self.hypothesis.clear(); + self.commit(&text, cx); + self.sink.finish(cx); + self.done = true; + return; + } + + if ends_utterance { + self.utterance = Some(text.clone()); + } + self.sink.hypothesis(self.joined(&text), cx); + self.hypothesis = text; + } + + fn commit(&mut self, text: &str, cx: &mut App) { + if text.is_empty() { + return; + } + let text = self.joined(text); + self.last_committed = text.chars().next_back(); + self.sink.phrase(text, cx); + } + + /// `text` with the separator it needs after the committed phrases. + fn joined(&self, text: &str) -> String { + match (self.last_committed, text.chars().next()) { + (Some(before), Some(after)) if needs_space(before, after) => format!(" {text}"), + _ => text.to_string(), + } + } +} + +/// Whether two phrases ending and starting with these characters need a space +/// between them. +fn needs_space(before: char, after: char) -> bool { + !before.is_whitespace() + && !after.is_whitespace() + && !matches!(after, ',' | '.' | '?' | '!' | ';' | ':' | ')') + && !is_unspaced_script(before) + && !is_unspaced_script(after) +} + +/// Characters of scripts written without spaces between words: Chinese, +/// Japanese and their punctuation. +fn is_unspaced_script(c: char) -> bool { + matches!(c, + '\u{3000}'..='\u{30FF}' + | '\u{3400}'..='\u{4DBF}' + | '\u{4E00}'..='\u{9FFF}' + | '\u{F900}'..='\u{FAFF}' + | '\u{FF00}'..='\u{FFEF}' + ) +} diff --git a/crates/component/src/speech/system/mod.rs b/crates/component/src/speech/system/mod.rs new file mode 100644 index 0000000000..8325d6b3f0 --- /dev/null +++ b/crates/component/src/speech/system/mod.rs @@ -0,0 +1,86 @@ +//! The platform's own speech recognizer. + +use std::{cell::OnceCell, rc::Rc}; + +use gpui::{App, SharedString}; + +use super::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +#[cfg(target_os = "macos")] +pub(super) mod macos; +#[cfg(target_os = "macos")] +use macos as platform; + +#[cfg(target_os = "windows")] +mod winrt; +#[cfg(target_os = "windows")] +use winrt as platform; + +#[cfg(not(any(target_os = "macos", target_os = "windows")))] +mod unsupported; +#[cfg(not(any(target_os = "macos", target_os = "windows")))] +use unsupported as platform; + +/// The operating system's speech recognizer. +/// +/// - **macOS**: `SFSpeechRecognizer`, recognizing on the device only. A +/// language the device cannot recognize offline is not available, so audio +/// never leaves the machine. +/// - **Windows**: `Windows.Media.SpeechRecognition`. Dictation needs the +/// language's speech pack and the "Online speech recognition" privacy +/// setting, and runs through Microsoft's online service. +/// - **Other platforms**: never available. +/// +/// A [`SpeechState`](super::SpeechState) without its own recognizer uses this +/// one; create it directly to choose the language. +pub struct SystemRecognizer { + locale: Option, + platform: OnceCell>, +} + +impl SystemRecognizer { + /// A recognizer for the system's current language. + pub fn new() -> Self { + Self { + locale: None, + platform: OnceCell::new(), + } + } + + /// Recognize `locale`, a BCP 47 language tag such as `en-US` or `zh-CN`, + /// instead of the system's language. + pub fn locale(mut self, locale: impl Into) -> Self { + self.locale = Some(locale.into()); + self.platform = OnceCell::new(); + self + } + + fn platform(&self) -> &Rc { + self.platform + .get_or_init(|| Rc::new(platform::PlatformRecognizer::new(self.locale.clone()))) + } +} + +impl Default for SystemRecognizer { + fn default() -> Self { + Self::new() + } +} + +impl SpeechRecognizer for SystemRecognizer { + fn audio_format(&self) -> AudioFormat { + self.platform().audio_format() + } + + fn is_available(&self, cx: &App) -> bool { + self.platform().is_available(cx) + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + self.platform().start(sink, cx) + } +} diff --git a/crates/component/src/speech/system/unsupported.rs b/crates/component/src/speech/system/unsupported.rs new file mode 100644 index 0000000000..13aeb09601 --- /dev/null +++ b/crates/component/src/speech/system/unsupported.rs @@ -0,0 +1,30 @@ +use gpui::{App, SharedString}; + +use crate::speech::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// Placeholder until the platform recognizer lands. +pub(super) struct PlatformRecognizer; + +impl PlatformRecognizer { + pub(super) fn new(_locale: Option) -> Self { + Self + } +} + +impl SpeechRecognizer for PlatformRecognizer { + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + fn is_available(&self, _: &App) -> bool { + false + } + + fn start( + &self, + _: SpeechSink, + _: &mut App, + ) -> Result, SpeechError> { + Err(SpeechError::Unsupported) + } +} diff --git a/crates/component/src/speech/system/winrt.rs b/crates/component/src/speech/system/winrt.rs new file mode 100644 index 0000000000..7fc7b18e87 --- /dev/null +++ b/crates/component/src/speech/system/winrt.rs @@ -0,0 +1,395 @@ +//! The Windows recognizer: continuous dictation with +//! `Windows.Media.SpeechRecognition`. +//! +//! The WinRT recognizer captures from the default microphone itself and has no +//! way to consume audio from elsewhere, so the pushed PCM is ignored; the +//! [`Microphone`](crate::speech::Microphone) keeps capturing alongside it only +//! to drive the waveform, which WASAPI's shared mode allows. +//! +//! The speech objects are agile, so they are called from the main thread, an +//! STA that GPUI initializes with `OleInitialize`, without further apartment +//! setup. Their completions and events arrive on WinRT threads, which only +//! forward them over a channel to a foreground task that reports to the sink. + +use anyhow::anyhow; +use gpui::{App, AsyncApp, SharedString, Task}; +use smol::channel::{Receiver, Sender, bounded, unbounded}; +use windows::{ + Foundation::{ + AsyncActionCompletedHandler, AsyncOperationCompletedHandler, EventRegistrationToken, + IAsyncAction, IAsyncOperation, TypedEventHandler, + }, + Globalization::Language, + Media::SpeechRecognition::{ + SpeechContinuousRecognitionCompletedEventArgs, + SpeechContinuousRecognitionResultGeneratedEventArgs, SpeechContinuousRecognitionSession, + SpeechRecognitionConfidence, SpeechRecognitionHypothesisGeneratedEventArgs, + SpeechRecognitionResult, SpeechRecognitionResultStatus, SpeechRecognitionScenario, + SpeechRecognitionTopicConstraint, SpeechRecognizer as WinSpeechRecognizer, + }, + Win32::Foundation::E_ACCESSDENIED, + core::{HRESULT, HSTRING, RuntimeType}, +}; + +use crate::speech::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// `SPERR_SPEECH_PRIVACY_POLICY_NOT_ACCEPTED`: "Online speech recognition" is +/// turned off in the privacy settings, which dictation requires. +const PRIVACY_POLICY_NOT_ACCEPTED: HRESULT = HRESULT(0x80045509_u32 as i32); +/// `MF_E_NO_CAPTURE_DEVICES_AVAILABLE`. +const NO_CAPTURE_DEVICES: HRESULT = HRESULT(0xC00DABE0_u32 as i32); + +pub(super) struct PlatformRecognizer { + /// The dictation language, or `None` when the requested one is malformed or + /// has no speech pack installed. + language: Option, + separator: &'static str, +} + +impl PlatformRecognizer { + pub(super) fn new(locale: Option) -> Self { + let language = supported_language(locale.as_deref()) + .inspect_err(|error| log::warn!("speech: no dictation language: {error:#}")) + .ok() + .flatten(); + let separator = language + .as_ref() + .and_then(|language| language.LanguageTag().ok()) + .map_or(" ", |tag| phrase_separator(&tag.to_string_lossy())); + Self { + language, + separator, + } + } +} + +impl SpeechRecognizer for PlatformRecognizer { + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + fn is_available(&self, _: &App) -> bool { + self.language.is_some() + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + let Some(language) = &self.language else { + return Err(SpeechError::Unsupported); + }; + + let recognizer = WinSpeechRecognizer::Create(language).map_err(speech_error)?; + let continuous = recognizer + .ContinuousRecognitionSession() + .map_err(speech_error)?; + let (messages, rx) = unbounded(); + let mut session = Session { + recognizer: recognizer.clone(), + continuous: continuous.clone(), + hypothesis_token: None, + result_token: None, + completed_token: None, + messages: messages.clone(), + _task: None, + }; + + let constraint = SpeechRecognitionTopicConstraint::Create( + SpeechRecognitionScenario::Dictation, + &HSTRING::from("dictation"), + ) + .map_err(speech_error)?; + recognizer + .Constraints() + .and_then(|constraints| constraints.Append(&constraint)) + .map_err(speech_error)?; + + session.hypothesis_token = Some( + recognizer + .HypothesisGenerated(&TypedEventHandler::new({ + let messages = messages.clone(); + move |_, args: &Option| { + if let Some(args) = args { + let text = args.Hypothesis()?.Text()?; + _ = messages.try_send(Message::Hypothesis(text)); + } + Ok(()) + } + })) + .map_err(speech_error)?, + ); + session.result_token = Some( + continuous + .ResultGenerated(&TypedEventHandler::new({ + let messages = messages.clone(); + move |_, args: &Option| { + if let Some(args) = args { + _ = messages.try_send(Message::Result(args.Result()?)); + } + Ok(()) + } + })) + .map_err(speech_error)?, + ); + session.completed_token = Some( + continuous + .Completed(&TypedEventHandler::new({ + let messages = messages.clone(); + move |_, args: &Option| { + if let Some(args) = args { + _ = messages.try_send(Message::Completed(args.Status()?)); + } + Ok(()) + } + })) + .map_err(speech_error)?, + ); + + let separator = self.separator; + session._task = Some(cx.spawn(async move |cx| { + let result = dictate(&recognizer, &continuous, rx, &sink, separator, cx).await; + if let Err(error) = result { + cx.update(|cx| sink.error(error, cx)); + } + })); + Ok(Box::new(session)) + } +} + +/// What the WinRT threads and [`Session::finish`] tell the dictation task. +enum Message { + Hypothesis(HSTRING), + Result(SpeechRecognitionResult), + Completed(SpeechRecognitionResultStatus), + /// The user stopped talking. + Finish, +} + +struct Session { + recognizer: WinSpeechRecognizer, + continuous: SpeechContinuousRecognitionSession, + hypothesis_token: Option, + result_token: Option, + completed_token: Option, + messages: Sender, + _task: Option>, +} + +impl RecognitionSession for Session { + /// Ignored: the WinRT recognizer captures from the microphone itself. + fn push_audio(&mut self, _: &[i16], _: &mut App) {} + + fn finish(&mut self, _: &mut App) { + // Queued behind the start, so finishing while connecting stops the + // session as soon as it runs. + _ = self.messages.try_send(Message::Finish); + } +} + +impl Drop for Session { + fn drop(&mut self) { + if let Some(token) = self.hypothesis_token.take() { + _ = self.recognizer.RemoveHypothesisGenerated(token); + } + if let Some(token) = self.result_token.take() { + _ = self.continuous.RemoveResultGenerated(token); + } + if let Some(token) = self.completed_token.take() { + _ = self.continuous.RemoveCompleted(token); + } + + // Close the recognizer once the cancellation lands, or right away when + // there is nothing to cancel. + let recognizer = self.recognizer.clone(); + let closed = self.continuous.CancelAsync().and_then(|action| { + action.SetCompleted(&AsyncActionCompletedHandler::new(move |_, _| { + _ = recognizer.Close(); + Ok(()) + })) + }); + if closed.is_err() { + _ = self.recognizer.Close(); + } + } +} + +/// Compile the dictation constraint, start the continuous session and report +/// its results to `sink` until it completes. +async fn dictate( + recognizer: &WinSpeechRecognizer, + continuous: &SpeechContinuousRecognitionSession, + messages: Receiver, + sink: &SpeechSink, + separator: &'static str, + cx: &mut AsyncApp, +) -> Result<(), SpeechError> { + let compilation = operation(recognizer.CompileConstraintsAsync().map_err(speech_error)?) + .await + .map_err(speech_error)?; + let status = compilation.Status().map_err(speech_error)?; + if status != SpeechRecognitionResultStatus::Success { + return Err(status_error(status)); + } + action(continuous.StartAsync().map_err(speech_error)?) + .await + .map_err(speech_error)?; + cx.update(|cx| sink.ready(cx)); + + let mut separator_due = false; + while let Ok(message) = messages.recv().await { + match message { + Message::Hypothesis(text) => { + let text = joined(separator_due, separator, &text); + cx.update(|cx| sink.hypothesis(text, cx)); + } + Message::Result(result) => { + let Some(text) = phrase_text(&result) else { + continue; + }; + let text = joined(separator_due, separator, &text); + separator_due = true; + cx.update(|cx| sink.phrase(text, cx)); + } + Message::Completed(status) => { + return match status { + // Stopped, cancelled, or ended by the silence timeout. + SpeechRecognitionResultStatus::Success + | SpeechRecognitionResultStatus::UserCanceled + | SpeechRecognitionResultStatus::TimeoutExceeded => { + cx.update(|cx| sink.finish(cx)); + Ok(()) + } + status => Err(status_error(status)), + }; + } + // Stopping flushes the last phrase, then completes the session. + Message::Finish => action(continuous.StopAsync().map_err(speech_error)?) + .await + .map_err(speech_error)?, + } + } + Ok(()) +} + +/// The text of a recognized phrase, unless it was rejected or empty. +fn phrase_text(result: &SpeechRecognitionResult) -> Option { + if result.Status().ok()? != SpeechRecognitionResultStatus::Success + || result.Confidence().ok()? == SpeechRecognitionConfidence::Rejected + { + return None; + } + Some(result.Text().ok()?).filter(|text| !text.is_empty()) +} + +/// `text` preceded by `separator` once a phrase has been committed. +fn joined(separator_due: bool, separator: &str, text: &HSTRING) -> SharedString { + let text = text.to_string_lossy(); + if separator_due { + format!("{separator}{text}").into() + } else { + text.into() + } +} + +/// What goes between two phrases, which dictation returns without surrounding +/// whitespace: a space, except in languages written without spaces between +/// words (Chinese, Japanese, Thai, Lao, Khmer, Burmese). +fn phrase_separator(tag: &str) -> &'static str { + let primary = tag.split('-').next().unwrap_or_default(); + let unspaced = ["zh", "yue", "ja", "th", "lo", "km", "my"]; + if unspaced + .iter() + .any(|lang| primary.eq_ignore_ascii_case(lang)) + { + "" + } else { + " " + } +} + +/// The language to dictate `locale` (or, without one, the system's speech +/// language) in, if a speech pack supports it. +/// +/// A bare language such as `en` falls back to the first supported region. +fn supported_language(locale: Option<&str>) -> windows::core::Result> { + let requested = match locale { + Some(locale) => { + let tag = HSTRING::from(locale); + if !Language::IsWellFormed(&tag)? { + return Ok(None); + } + Language::CreateLanguage(&tag)? + } + None => WinSpeechRecognizer::SystemSpeechLanguage()?, + }; + let requested = requested.LanguageTag()?.to_string_lossy(); + let region_prefix = format!("{requested}-"); + + let mut fallback = None; + for language in WinSpeechRecognizer::SupportedTopicLanguages()? { + let tag = language.LanguageTag()?.to_string_lossy(); + if tag.eq_ignore_ascii_case(&requested) { + return Ok(Some(language)); + } + if fallback.is_none() + && tag.len() > region_prefix.len() + && tag[..region_prefix.len()].eq_ignore_ascii_case(®ion_prefix) + { + fallback = Some(language); + } + } + Ok(fallback) +} + +/// Await `operation` without blocking: its completion handler, which runs on a +/// WinRT thread, only wakes this future. +async fn operation( + operation: IAsyncOperation, +) -> windows::core::Result { + let (done, wait) = bounded(1); + operation.SetCompleted(&AsyncOperationCompletedHandler::new(move |_, _| { + _ = done.try_send(()); + Ok(()) + }))?; + _ = wait.recv().await; + operation.GetResults() +} + +/// Await `action` like [`operation`]. +async fn action(action: IAsyncAction) -> windows::core::Result<()> { + let (done, wait) = bounded(1); + action.SetCompleted(&AsyncActionCompletedHandler::new(move |_, _| { + _ = done.try_send(()); + Ok(()) + }))?; + _ = wait.recv().await; + action.GetResults() +} + +fn speech_error(error: windows::core::Error) -> SpeechError { + match error.code() { + E_ACCESSDENIED => SpeechError::PermissionDenied, + NO_CAPTURE_DEVICES => SpeechError::NoInputDevice, + PRIVACY_POLICY_NOT_ACCEPTED => SpeechError::recognizer(anyhow!( + "Online speech recognition is turned off; turn it on in Settings > \ + Privacy & security > Speech" + )), + _ => SpeechError::recognizer(error), + } +} + +fn status_error(status: SpeechRecognitionResultStatus) -> SpeechError { + match status { + SpeechRecognitionResultStatus::TopicLanguageNotSupported => SpeechError::Unsupported, + SpeechRecognitionResultStatus::MicrophoneUnavailable => SpeechError::NoInputDevice, + SpeechRecognitionResultStatus::NetworkFailure => { + SpeechError::recognizer(anyhow!("could not reach the online speech service")) + } + SpeechRecognitionResultStatus::AudioQualityFailure => { + SpeechError::recognizer(anyhow!("the audio was too poor to recognize")) + } + status => SpeechError::recognizer(anyhow!("recognition ended with status {}", status.0)), + } +} diff --git a/crates/component/src/speech/waveform.rs b/crates/component/src/speech/waveform.rs new file mode 100644 index 0000000000..874e4b8d06 --- /dev/null +++ b/crates/component/src/speech/waveform.rs @@ -0,0 +1,83 @@ +use gpui::{ + App, Entity, IntoElement, ParentElement as _, Pixels, RenderOnce, Styled as _, Window, div, px, +}; + +use crate::{ActiveTheme as _, Sizable, Size, h_flex}; + +use super::{SpeechState, state::LEVEL_HISTORY}; + +/// A live bar graph of a [`SpeechState`]'s recent input levels. +/// +/// The newest level is on the trailing end. The bars sit flat and muted while +/// no audio is captured. +#[derive(IntoElement)] +pub struct SpeechWaveform { + state: Entity, + bars: usize, + size: Size, +} + +impl SpeechWaveform { + /// A waveform for `state`. + pub fn new(state: &Entity) -> Self { + Self { + state: state.clone(), + bars: 24, + size: Size::default(), + } + } + + /// Set the number of bars, default 24, at most 48. + pub fn bars(mut self, bars: usize) -> Self { + self.bars = bars.clamp(1, LEVEL_HISTORY); + self + } + + fn height(&self) -> Pixels { + match self.size { + Size::Size(height) => height, + Size::XSmall => px(12.), + Size::Small => px(16.), + Size::Medium => px(20.), + Size::Large => px(24.), + } + } +} + +impl Sizable for SpeechWaveform { + fn with_size(mut self, size: impl Into) -> Self { + self.size = size.into(); + self + } +} + +impl RenderOnce for SpeechWaveform { + fn render(self, _: &mut Window, cx: &mut App) -> impl IntoElement { + let height = self.height(); + let bar_width = px(2.); + let state = self.state.read(cx); + let color = if state.status().is_capturing() { + cx.theme().primary + } else { + cx.theme().muted_foreground + }; + let levels = state.levels(); + // Right-align the history: pad the leading bars when fewer levels have + // arrived than there are bars. + let skip = levels.len().saturating_sub(self.bars); + let padding = self.bars.saturating_sub(levels.len()); + let levels = std::iter::repeat_n(0., padding).chain(levels.skip(skip)); + + h_flex() + .h(height) + .gap(bar_width) + .items_center() + .children(levels.map(|level| { + div() + .w(bar_width) + .h((height * level).max(bar_width)) + .rounded_full() + .bg(color) + })) + } +} diff --git a/crates/kit/Cargo.toml b/crates/kit/Cargo.toml index 9b77891044..3373991d1d 100644 --- a/crates/kit/Cargo.toml +++ b/crates/kit/Cargo.toml @@ -27,6 +27,8 @@ test-support = ["gpui/test-support", "gpui_platform/test-support", "gpui-base/te profiler = ["gpui/profiler"] inspector = ["gpui/inspector", "gpui-base/inspector", "gpui-component?/inspector"] decimal = ["component", "gpui-component/decimal"] +# Speech input: microphone capture and the system speech recognizer. +speech = ["component", "gpui-component/speech"] tree-sitter = ["component", "gpui-component/tree-sitter"] tree-sitter-languages = ["component", "gpui-component/tree-sitter-languages"] tree-sitter-astro = ["component", "gpui-component/tree-sitter-astro"] diff --git a/crates/story/Cargo.toml b/crates/story/Cargo.toml index 1fe54865d9..5e10ebc402 100644 --- a/crates/story/Cargo.toml +++ b/crates/story/Cargo.toml @@ -12,7 +12,7 @@ default = ["tree-sitter"] [dependencies] anyhow.workspace = true -gpui-kit.workspace = true +gpui-kit = { workspace = true, features = ["speech"] } gpui-fps.workspace = true async-channel = "2.3.1" diff --git a/crates/story/src/gallery.rs b/crates/story/src/gallery.rs index d2b43ec5c7..4c0b7c6bed 100644 --- a/crates/story/src/gallery.rs +++ b/crates/story/src/gallery.rs @@ -126,6 +126,7 @@ impl Gallery { StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), + StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), diff --git a/crates/story/src/stories/mod.rs b/crates/story/src/stories/mod.rs index 977ee8fec0..307e7f5fb4 100644 --- a/crates/story/src/stories/mod.rs +++ b/crates/story/src/stories/mod.rs @@ -64,6 +64,7 @@ mod shimmer_story; mod sidebar_story; mod skeleton_story; mod slider_story; +mod speech_story; mod spinner_story; mod status_bar_story; mod stepper_story; @@ -143,6 +144,7 @@ pub use shimmer_story::ShimmerStory; pub use sidebar_story::SidebarStory; pub use skeleton_story::SkeletonStory; pub use slider_story::SliderStory; +pub use speech_story::SpeechStory; pub use spinner_story::SpinnerStory; pub use status_bar_story::StatusBarStory; pub use stepper_story::StepperStory; diff --git a/crates/story/src/stories/speech_story.rs b/crates/story/src/stories/speech_story.rs new file mode 100644 index 0000000000..a3e25ed0a0 --- /dev/null +++ b/crates/story/src/stories/speech_story.rs @@ -0,0 +1,281 @@ +use gpui_kit::{ + App, AppContext, Context, Entity, FocusHandle, Focusable, IntoElement, ParentElement, Render, + Styled, Subscription, Window, div, prelude::FluentBuilder as _, +}; + +use gpui_kit::component::{ + ActiveTheme as _, Sizable as _, WindowExt as _, h_flex, + input::{Input, InputState}, + notification::Notification, + speech::{ + RecognitionSession, SpeechButton, SpeechError, SpeechEvent, SpeechRecognizer, SpeechSink, + SpeechState, SpeechWaveform, + }, + v_flex, +}; + +use crate::section; + +/// What [`DemoRecognizer`] "hears", one phrase at a time. +const SCRIPT: [&str; 2] = [ + "Speech input turns what you say into text.", + "Any recognizer plugs in through one trait.", +]; + +/// How much 16 kHz mono audio [`DemoRecognizer`] takes per word: 0.3 s. +const SAMPLES_PER_WORD: usize = 4_800; + +/// A recognizer that needs no service: it types [`SCRIPT`] one word per +/// [`SAMPLES_PER_WORD`] of audio, whatever the audio contains. +/// +/// A real recognizer has the same shape: `start` opens a connection, the +/// session streams audio to it and reports the service's results to the sink. +struct DemoRecognizer; + +impl SpeechRecognizer for DemoRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // Nothing to connect to, so audio is consumed at once. + sink.ready(cx); + Ok(Box::new(DemoSession { + sink, + phrase_ix: 0, + words: 0, + samples: 0, + })) + } +} + +struct DemoSession { + sink: SpeechSink, + phrase_ix: usize, + /// Words of the current phrase recognized so far. + words: usize, + /// Samples received since the last word. + samples: usize, +} + +impl DemoSession { + /// The recognized part of the current phrase, if any. + fn spoken(&self) -> Option { + let phrase = SCRIPT.get(self.phrase_ix)?; + let words = phrase.split(' ').take(self.words).collect::>(); + if words.is_empty() { + return None; + } + // Phrases are joined verbatim, so separate sentences here. + let separator = if self.phrase_ix > 0 { " " } else { "" }; + Some(format!("{separator}{}", words.join(" "))) + } + + fn next_word(&mut self, cx: &mut App) { + let Some(phrase) = SCRIPT.get(self.phrase_ix) else { + return; + }; + self.words += 1; + if self.words < phrase.split(' ').count() { + if let Some(spoken) = self.spoken() { + self.sink.hypothesis(spoken, cx); + } + } else { + self.commit(cx); + } + } + + fn commit(&mut self, cx: &mut App) { + if let Some(spoken) = self.spoken() { + self.sink.phrase(spoken, cx); + } + self.phrase_ix += 1; + self.words = 0; + } +} + +impl RecognitionSession for DemoSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + self.samples += samples.len(); + while self.samples >= SAMPLES_PER_WORD { + self.samples -= SAMPLES_PER_WORD; + self.next_word(cx); + } + } + + fn finish(&mut self, cx: &mut App) { + self.commit(cx); + self.sink.finish(cx); + } +} + +/// Stands in for the microphone on the web, which the story cannot capture: +/// pushes a tone whose loudness rises and falls like speech. +#[cfg(target_family = "wasm")] +struct GeneratedInput; + +#[cfg(target_family = "wasm")] +impl gpui_kit::component::speech::AudioInput for GeneratedInput { + fn start( + &self, + format: gpui_kit::component::speech::AudioFormat, + sink: gpui_kit::component::speech::AudioSink, + cx: &mut App, + ) -> Result { + const CHUNK: std::time::Duration = std::time::Duration::from_millis(100); + let len = (format.sample_rate() / 10) as usize * format.channels() as usize; + let task = cx.spawn(async move |cx| { + let mut tick = 0u8; + loop { + cx.background_executor().timer(CHUNK).await; + tick = tick.wrapping_add(1); + let loudness = 0.05 + 0.25 * (tick as f32 * 0.9).sin().abs(); + let samples = (0..len) + .map(|ix| ((ix as f32 * 0.07).sin() * loudness * i16::MAX as f32) as i16) + .collect(); + cx.update(|cx| sink.push(samples, cx)); + } + }); + Ok(Subscription::new(move || drop(task))) + } +} + +/// A text field the user can dictate into. +struct Dictation { + speech: Entity, + input: Entity, +} + +impl Dictation { + fn new( + speech: Entity, + window: &mut Window, + cx: &mut Context, + ) -> (Self, Subscription) { + let input = cx.new(|cx| InputState::new(window, cx).placeholder("Type or dictate")); + let subscription = cx.subscribe_in(&speech, window, { + let input = input.clone(); + move |_, _, event, window, cx| match event { + SpeechEvent::Final(text) if !text.is_empty() => { + input.update(cx, |input, cx| input.insert(text.clone(), window, cx)); + } + SpeechEvent::Error(error) => { + window.push_notification(Notification::error(error.to_string()), cx); + } + _ => {} + } + }); + (Self { speech, input }, subscription) + } + + fn render(&self, show_when_unsupported: bool, cx: &App) -> impl IntoElement { + let speech = self.speech.read(cx); + let status = speech.status(); + + v_flex() + .w_full() + .gap_2() + .child( + Input::new(&self.input).suffix( + h_flex() + .gap_2() + .when(status.is_capturing(), |this| { + this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + }) + .child( + SpeechButton::new(&self.speech) + .xsmall() + .show_when_unsupported(show_when_unsupported), + ), + ), + ) + .when(status.is_active(), |this| { + this.child( + div() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child(speech.transcript()), + ) + }) + } +} + +pub struct SpeechStory { + focus_handle: FocusHandle, + custom: Dictation, + system: Dictation, + _subscriptions: Vec, +} + +impl super::Story for SpeechStory { + fn title() -> &'static str { + "Speech" + } + + fn description() -> &'static str { + "Dictate text through the system recognizer or your own." + } + + fn new_view(window: &mut Window, cx: &mut App) -> Entity { + Self::view(window, cx) + } +} + +impl SpeechStory { + pub fn view(window: &mut Window, cx: &mut App) -> Entity { + cx.new(|cx| Self::new(window, cx)) + } + + fn new(window: &mut Window, cx: &mut Context) -> Self { + let custom = cx.new(|cx| { + let state = SpeechState::new(cx).recognizer(DemoRecognizer); + #[cfg(target_family = "wasm")] + let state = state.input(GeneratedInput); + state + }); + let system = cx.new(SpeechState::new); + + let (custom, custom_subscription) = Dictation::new(custom, window, cx); + let (system, system_subscription) = Dictation::new(system, window, cx); + + Self { + focus_handle: cx.focus_handle(), + custom, + system, + _subscriptions: vec![custom_subscription, system_subscription], + } + } +} + +impl Focusable for SpeechStory { + fn focus_handle(&self, _: &App) -> FocusHandle { + self.focus_handle.clone() + } +} + +impl Render for SpeechStory { + fn render(&mut self, _: &mut Window, cx: &mut Context) -> impl IntoElement { + v_flex() + .size_full() + .justify_start() + .gap_3() + .child( + section("With a custom recognizer") + .description( + "A recognizer defined in this story types a scripted sentence \ + while it receives audio.", + ) + .w_128() + .child(self.custom.render(false, cx)), + ) + .child( + section("System recognizer") + .description( + "Recognizes speech on macOS and Windows. Elsewhere, and in an app \ + without the required usage descriptions, the button is disabled.", + ) + .w_128() + .child(self.system.render(true, cx)), + ) + } +} diff --git a/examples/speech/Cargo.toml b/examples/speech/Cargo.toml new file mode 100644 index 0000000000..b03aca0292 --- /dev/null +++ b/examples/speech/Cargo.toml @@ -0,0 +1,14 @@ +[package] +name = "speech" +description = "A dictation notepad for trying out and testing GPUI Component speech input." +version = "0.7.0" +publish = false +edition.workspace = true + +[dependencies] +anyhow.workspace = true +cpal = "0.15.3" +gpui-kit = { workspace = true, features = ["speech"] } + +[lints] +workspace = true diff --git a/examples/speech/Info.plist b/examples/speech/Info.plist new file mode 100644 index 0000000000..6bd6d5e02b --- /dev/null +++ b/examples/speech/Info.plist @@ -0,0 +1,10 @@ + + + + + NSMicrophoneUsageDescription + Dictation listens to the microphone to turn what you say into text. + NSSpeechRecognitionUsageDescription + Dictation recognizes speech on this Mac to type what you say. + + diff --git a/examples/speech/README.md b/examples/speech/README.md new file mode 100644 index 0000000000..eea9bbd187 --- /dev/null +++ b/examples/speech/README.md @@ -0,0 +1,62 @@ +# Speech + +Dictation, a notepad you can talk into, built on GPUI Component's speech input. It is also +the test bench for the platform recognizers: the sidebar shows what this machine +supports, and the session log records every `SpeechEvent`. + +```sh +cargo run -p speech # open the app +cargo run -p speech -- --check # print the checks and exit +``` + +`--check` prints the platform, the default input device, and whether the system +recognizer is available for each language in the picker: + +```text +Platform macOS +Input device MacBook Pro Microphone +System recognizer, by language: + en-US available English (US) + zh-CN available 简体中文 + ja-JP unavailable 日本語 +``` + +## Trying it + +- **Demo** types a scripted passage while it hears audio. It needs no service, + network, or speech permission, so it works on every platform, Linux included. + Use it to check the capture, waveform, and event flow. +- **System** uses the operating system's recognizer in the language you pick. +- Click the microphone or press ⇧⌘D (Ctrl+Shift+D on + Windows and Linux) to start, and again to stop. The text goes into the note + at the cursor. **Discard** stops without inserting anything. + +## Platform notes + +### macOS + +`build.rs` links `Info.plist` into the executable, so `cargo run` can ask for +microphone and speech recognition access without an app bundle. The first +session asks for both. Recognition stays on the Mac, so a language the Mac can't +recognize offline shows **Unavailable**. + +When you run from a terminal, macOS attributes the microphone to the terminal +app. To ask again after denying access: + +```sh +tccutil reset Microphone +tccutil reset SpeechRecognition +``` + +### Windows + +Install the language's speech pack (Settings › Time & language › Speech) and +turn on **Online speech recognition** (Settings › Privacy & security › Speech). +Dictation runs through Microsoft's online service. The recognizer records from +the default input device itself, and the waveform follows the same device. + +### Linux + +There is no system recognizer, so **System** shows **Not supported**. Use +**Demo**, or plug in your own `SpeechRecognizer`. Building needs +`libasound2-dev`. diff --git a/examples/speech/build.rs b/examples/speech/build.rs new file mode 100644 index 0000000000..81adfb3106 --- /dev/null +++ b/examples/speech/build.rs @@ -0,0 +1,21 @@ +//! Embed `Info.plist` into the macOS executable. +//! +//! macOS asks for microphone and speech recognition access only when the +//! application describes why it needs them. An unbundled `cargo run` binary has +//! no bundle to carry that description, but the system also reads a property +//! list linked into the executable's `__TEXT,__info_plist` section. +//! +//! The list holds only the two usage descriptions. A bundle identifier would +//! make frameworks treat the binary as an application bundle, and GPUI's system +//! notifications then fail to start without one. + +fn main() { + println!("cargo:rerun-if-changed=Info.plist"); + if std::env::var("CARGO_CFG_TARGET_OS").as_deref() == Ok("macos") { + let plist = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("Info.plist"); + println!( + "cargo:rustc-link-arg-bins=-Wl,-sectcreate,__TEXT,__info_plist,{}", + plist.display() + ); + } +} diff --git a/examples/speech/src/demo.rs b/examples/speech/src/demo.rs new file mode 100644 index 0000000000..4ed1a58e79 --- /dev/null +++ b/examples/speech/src/demo.rs @@ -0,0 +1,166 @@ +//! A recognizer that needs no service, permission or network: it types a +//! scripted passage while audio arrives, so the whole flow can be tried +//! anywhere, including on Linux. + +use gpui_kit::App; +use gpui_kit::component::speech::{RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// Audio per recognized token: 0.25 s at 16 kHz. +const SAMPLES_PER_TOKEN: usize = 4_000; + +pub struct DemoRecognizer { + script: &'static Script, +} + +impl DemoRecognizer { + /// A recognizer typing the passage for `language`, a BCP 47 tag. + pub fn new(language: &str) -> Self { + let script = SCRIPTS + .iter() + .find(|script| language.starts_with(script.language)) + .unwrap_or(&SCRIPTS[0]); + Self { script } + } +} + +impl SpeechRecognizer for DemoRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // There is nothing to connect to, so audio is consumed at once. + sink.ready(cx); + Ok(Box::new(DemoSession { + sink, + script: self.script, + sentence_ix: 0, + tokens: 0, + samples: 0, + })) + } +} + +struct Script { + language: &'static str, + /// Put between tokens and between sentences. + separator: &'static str, + sentences: &'static [&'static [&'static str]], +} + +const SCRIPTS: &[Script] = &[ + Script { + language: "en", + separator: " ", + sentences: &[ + &[ + "Speech", "input", "turns", "what", "you", "say", "into", "text.", + ], + &[ + "Stop", "whenever", "you", "like,", "and", "the", "words", "land", "in", "the", + "note.", + ], + &[ + "Any", "speech", "service", "plugs", "in", "through", "one", "trait.", + ], + ], + }, + Script { + language: "zh", + separator: "", + sentences: &[ + &["语音", "输入", "会把", "你说", "的话", "转成", "文字。"], + &["随时", "停止,", "文字", "就会", "写进", "笔记。"], + &[ + "任何", + "识别", + "服务", + "都能", + "通过", + "一个", + "接口", + "接进来。", + ], + ], + }, + Script { + language: "ja", + separator: "", + sentences: &[ + &["話した", "言葉が", "そのまま", "文字に", "なります。"], + &["止めると", "ノートに", "書き込まれます。"], + ], + }, +]; + +struct DemoSession { + sink: SpeechSink, + script: &'static Script, + sentence_ix: usize, + /// Tokens of the current sentence heard so far. + tokens: usize, + /// Samples received since the last token. + samples: usize, +} + +impl DemoSession { + /// The heard part of the current sentence, with the separator that joins it + /// to the sentences before. + fn heard(&self) -> Option { + let sentence = self.sentence()?; + if self.tokens == 0 { + return None; + } + let leading = if self.sentence_ix > 0 { + self.script.separator + } else { + "" + }; + Some(format!( + "{leading}{}", + sentence[..self.tokens].join(self.script.separator) + )) + } + + fn sentence(&self) -> Option<&'static [&'static str]> { + let sentences = self.script.sentences; + sentences.get(self.sentence_ix % sentences.len()).copied() + } + + fn next_token(&mut self, cx: &mut App) { + let Some(sentence) = self.sentence() else { + return; + }; + self.tokens += 1; + if self.tokens < sentence.len() { + if let Some(heard) = self.heard() { + self.sink.hypothesis(heard, cx); + } + } else { + self.commit(cx); + } + } + + fn commit(&mut self, cx: &mut App) { + if let Some(heard) = self.heard() { + self.sink.phrase(heard, cx); + } + self.sentence_ix += 1; + self.tokens = 0; + } +} + +impl RecognitionSession for DemoSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + self.samples += samples.len(); + while self.samples >= SAMPLES_PER_TOKEN { + self.samples -= SAMPLES_PER_TOKEN; + self.next_token(cx); + } + } + + fn finish(&mut self, cx: &mut App) { + self.commit(cx); + self.sink.finish(cx); + } +} diff --git a/examples/speech/src/main.rs b/examples/speech/src/main.rs new file mode 100644 index 0000000000..746f69ec55 --- /dev/null +++ b/examples/speech/src/main.rs @@ -0,0 +1,845 @@ +//! Dictation: a notepad you can talk into, built on GPUI Component's speech +//! input. It doubles as a test bench for the platform recognizers: the sidebar +//! reports what this machine supports and the session log records every event. +//! +//! `cargo run -p speech` opens the app; `cargo run -p speech -- --check` +//! prints the same checks to the terminal and exits. + +mod demo; + +use std::{ + borrow::Cow, + time::{Duration, Instant}, +}; + +use cpal::traits::{DeviceTrait as _, HostTrait as _}; +use gpui_kit::assets::{Assets, IconName as ExtraIcon, icon_assets}; +use gpui_kit::component::{ + ActiveTheme as _, Disableable as _, Icon, IndexPath, Selectable as _, Sizable as _, + StyledExt as _, TitleBar, WindowExt as _, + button::{Button, ButtonGroup, ButtonVariants as _}, + h_flex, + input::{Textarea, TextareaState}, + kbd::Kbd, + notification::Notification, + scroll::ScrollableElement as _, + select::{SearchableVec, Select, SelectEvent, SelectItem, SelectState}, + speech::{ + SpeechButton, SpeechEvent, SpeechState, SpeechStatus, SpeechWaveform, SystemRecognizer, + }, + tag::Tag, + v_flex, +}; +use gpui_kit::prelude::FluentBuilder as _; +use gpui_kit::*; + +use demo::DemoRecognizer; + +icon_assets!(ExtraIcons, [AudioLines, ScrollText, Eraser]); + +/// The default component icons plus the few extras this app uses. +struct AppAssets; + +impl AssetSource for AppAssets { + fn load(&self, path: &str) -> Result>> { + if let Some(bytes) = ExtraIcons.load(path)? { + return Ok(Some(bytes)); + } + Assets.load(path) + } + + fn list(&self, path: &str) -> Result> { + let mut paths = Assets.list(path)?; + paths.extend(ExtraIcons.list(path)?); + Ok(paths) + } +} + +actions!(dictation, [ToggleDictation]); + +const TOGGLE_KEYS: &str = "secondary-shift-d"; +/// Most log entries kept; older ones scroll away. +const LOG_LIMIT: usize = 200; + +/// Where the text comes from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Engine { + /// The operating system's recognizer. + System, + /// [`DemoRecognizer`]: a scripted passage, no service needed. + Demo, +} + +#[derive(Clone)] +struct Language { + tag: SharedString, + name: SharedString, +} + +impl SelectItem for Language { + type Value = SharedString; + + fn title(&self) -> SharedString { + self.name.clone() + } + + fn value(&self) -> &Self::Value { + &self.tag + } +} + +fn languages() -> Vec { + [ + ("en-US", "English (US)"), + ("en-GB", "English (UK)"), + ("zh-CN", "简体中文"), + ("zh-HK", "中文(香港)"), + ("ja-JP", "日本語"), + ] + .into_iter() + .map(|(tag, name)| Language { + tag: tag.into(), + name: name.into(), + }) + .collect() +} + +/// One line of the session log. +struct LogEntry { + at: Duration, + tone: Tone, + label: &'static str, + text: SharedString, +} + +#[derive(Clone, Copy, PartialEq, Eq)] +enum Tone { + Neutral, + Progress, + Success, + Danger, +} + +fn new_speech(engine: Engine, language: &str, cx: &mut App) -> Entity { + let language = language.to_string(); + cx.new(|cx| { + let state = SpeechState::new(cx); + match engine { + Engine::System => state.recognizer(SystemRecognizer::new().locale(language)), + Engine::Demo => state.recognizer(DemoRecognizer::new(&language)), + } + }) +} + +/// The name of the default audio input device, if there is one. +fn input_device_name() -> Option { + let device = cpal::default_host().default_input_device()?; + Some( + device + .name() + .unwrap_or_else(|_| "Unnamed device".into()) + .into(), + ) +} + +fn platform_name() -> &'static str { + if cfg!(target_os = "macos") { + "macOS" + } else if cfg!(target_os = "windows") { + "Windows" + } else if cfg!(target_os = "linux") { + "Linux" + } else { + "Other" + } +} + +struct DictationApp { + focus_handle: FocusHandle, + engine: Engine, + language: SharedString, + speech: Entity, + notes: Entity, + language_select: Entity>>, + input_device: Option, + /// When the current or last session started. + session_started: Option, + /// When the app opened; log times count from here. + opened: Instant, + log: Vec, + _speech_subscription: Subscription, + _subscriptions: Vec, +} + +impl DictationApp { + fn new(window: &mut Window, cx: &mut Context) -> Self { + let engine = Engine::System; + let language = SharedString::from("en-US"); + let speech = new_speech(engine, &language, cx); + let _speech_subscription = Self::subscribe_speech(&speech, window, cx); + + let notes = cx.new(|cx| { + TextareaState::new(window, cx) + .placeholder("Start typing, or dictate with the microphone below.") + }); + let language_select = cx.new(|cx| { + SelectState::new( + SearchableVec::new(languages()), + Some(IndexPath::default()), + window, + cx, + ) + }); + let _subscriptions = vec![cx.subscribe_in( + &language_select, + window, + |this, _, event: &SelectEvent>, window, cx| { + if let SelectEvent::Confirm(Some(tag)) = event { + this.language = tag.clone(); + this.rebuild_speech(window, cx); + } + }, + )]; + notes.update(cx, |notes, cx| notes.focus(window, cx)); + + Self { + focus_handle: cx.focus_handle(), + engine, + language, + speech, + notes, + language_select, + input_device: input_device_name(), + session_started: None, + opened: Instant::now(), + log: Vec::new(), + _speech_subscription, + _subscriptions, + } + } + + fn subscribe_speech( + speech: &Entity, + window: &mut Window, + cx: &mut Context, + ) -> Subscription { + cx.subscribe_in( + speech, + window, + |this, _, event: &SpeechEvent, window, cx| this.on_speech_event(event, window, cx), + ) + } + + /// Recreate the speech state for the chosen engine and language. + fn rebuild_speech(&mut self, window: &mut Window, cx: &mut Context) { + self.speech.update(cx, |speech, cx| speech.cancel(cx)); + self.speech = new_speech(self.engine, &self.language, cx); + self._speech_subscription = Self::subscribe_speech(&self.speech, window, cx); + self.input_device = input_device_name(); + cx.notify(); + } + + fn set_engine(&mut self, engine: Engine, window: &mut Window, cx: &mut Context) { + if self.engine != engine { + self.engine = engine; + self.rebuild_speech(window, cx); + } + } + + fn toggle_dictation(&mut self, _: &ToggleDictation, _: &mut Window, cx: &mut Context) { + self.speech.update(cx, |speech, cx| speech.toggle(cx)); + } + + fn on_speech_event( + &mut self, + event: &SpeechEvent, + window: &mut Window, + cx: &mut Context, + ) { + match event { + SpeechEvent::Started => { + self.session_started = Some(Instant::now()); + self.push_log(Tone::Progress, "Started", "Listening".into()); + } + SpeechEvent::Partial(text) => { + // Partial results arrive many times a second; keep one line per + // stretch of them. + if let Some(last) = self.log.last_mut() + && last.label == "Partial" + { + last.text = text.clone(); + last.at = self.opened.elapsed(); + } else { + self.push_log(Tone::Neutral, "Partial", text.clone()); + } + } + SpeechEvent::Final(text) => { + if text.is_empty() { + self.push_log(Tone::Neutral, "Final", "No speech recognized".into()); + } else { + self.push_log(Tone::Success, "Final", text.clone()); + self.insert_into_notes(text, window, cx); + } + } + SpeechEvent::Cancelled => { + self.push_log(Tone::Neutral, "Cancelled", "Transcript discarded".into()); + } + SpeechEvent::Error(error) => { + self.push_log(Tone::Danger, "Error", error.to_string().into()); + window.push_notification( + Notification::error(format!("Couldn’t dictate. {error}.")), + cx, + ); + } + } + cx.notify(); + } + + fn insert_into_notes( + &mut self, + text: &SharedString, + window: &mut Window, + cx: &mut Context, + ) { + self.notes.update(cx, |notes, cx| { + // Keep dictated passages apart from what is already there. + let value = notes.value(); + let needs_space = value + .chars() + .next_back() + .is_some_and(|c| !c.is_whitespace() && c.is_ascii()); + let text = if needs_space { + format!(" {text}") + } else { + text.to_string() + }; + notes.insert(text, window, cx); + notes.focus(window, cx); + }); + } + + fn push_log(&mut self, tone: Tone, label: &'static str, text: SharedString) { + if self.log.len() == LOG_LIMIT { + self.log.remove(0); + } + self.log.push(LogEntry { + at: self.opened.elapsed(), + tone, + label, + text, + }); + } + + fn render_sidebar(&self, cx: &mut Context) -> impl IntoElement { + let speech = self.speech.read(cx); + let active = speech.status().is_active(); + let engine_note = match self.engine { + Engine::System => { + "The operating system’s recognizer. On macOS it only recognizes on this Mac; \ + on Windows it uses Microsoft’s online service." + } + Engine::Demo => { + "Types a scripted passage while it hears audio. Needs no service, network or \ + speech permission." + } + }; + + v_flex() + .w(px(272.)) + .h_full() + .flex_shrink_0() + .gap_6() + .p_4() + .bg(cx.theme().sidebar) + .text_color(cx.theme().sidebar_foreground) + .border_r_1() + .border_color(cx.theme().sidebar_border) + .child( + sidebar_section("Recognizer", cx) + .child( + ButtonGroup::new("engine") + .small() + .outline() + .w_full() + .disabled(active) + .child( + Button::new("engine-system") + .flex_1() + .label("System") + .selected(self.engine == Engine::System), + ) + .child( + Button::new("engine-demo") + .flex_1() + .label("Demo") + .selected(self.engine == Engine::Demo), + ) + .on_click(cx.listener(|this, clicks: &Vec, window, cx| { + let engine = if clicks.contains(&1) { + Engine::Demo + } else { + Engine::System + }; + this.set_engine(engine, window, cx); + })), + ) + .child(caption(engine_note, cx)), + ) + .child( + sidebar_section("Language", cx) + .child(Select::new(&self.language_select).small().disabled(active)), + ) + .child(sidebar_section("Checks", cx).child(self.render_checks(cx))) + } + + fn render_checks(&self, cx: &mut Context) -> impl IntoElement { + let speech = self.speech.read(cx); + let recognizer = if !speech.has_recognizer() { + Tag::danger().outline().child("Not supported") + } else if speech.is_available(cx) { + Tag::success().outline().child("Available") + } else { + Tag::warning().outline().child("Unavailable") + }; + let session = match speech.status() { + SpeechStatus::Idle => Tag::secondary().outline().child("Idle"), + SpeechStatus::Connecting => Tag::info().outline().child("Connecting"), + SpeechStatus::Recording => Tag::info().outline().child("Recording"), + SpeechStatus::Stopping => Tag::info().outline().child("Finishing"), + }; + let input = match &self.input_device { + Some(name) => div() + .min_w_0() + .truncate() + .child(name.clone()) + .into_any_element(), + None => Tag::danger() + .small() + .outline() + .child("None") + .into_any_element(), + }; + + v_flex() + .gap_2() + .text_sm() + .child(check_row("Platform", div().child(platform_name()), cx)) + .child(check_row("Input device", input, cx)) + .child(check_row("Recognizer", recognizer.small(), cx)) + .child(check_row("Session", session.small(), cx)) + .when( + self.engine == Engine::System + && speech.has_recognizer() + && !speech.is_available(cx), + |this| this.child(caption(unavailable_hint(), cx)), + ) + } + + fn render_notes_header(&self, cx: &mut Context) -> impl IntoElement { + let characters = self.notes.read(cx).value().chars().count(); + + h_flex() + .items_center() + .justify_between() + .child( + v_flex() + .child(div().text_lg().font_semibold().child("Notes")) + .child( + div() + .text_xs() + .text_color(cx.theme().muted_foreground) + .child(match characters { + 0 => "Empty".to_string(), + 1 => "1 character".to_string(), + n => format!("{n} characters"), + }), + ), + ) + .child( + Button::new("clear-notes") + .ghost() + .small() + .icon(Icon::new(ExtraIcon::Eraser)) + .label("Clear") + .disabled(characters == 0) + .on_click(cx.listener(|this, _, window, cx| { + this.notes + .update(cx, |notes, cx| notes.set_value("", window, cx)); + cx.notify(); + })), + ) + } + + /// The bar that runs dictation: button, level, live text and controls. + fn render_dictation_bar(&self, cx: &mut Context) -> impl IntoElement { + let speech = self.speech.read(cx); + let status = speech.status(); + let transcript = speech.transcript(); + let supported = speech.has_recognizer(); + let available = speech.is_available(cx); + let elapsed = self + .session_started + .filter(|_| status.is_active()) + .map(|started| format_duration(started.elapsed())); + + let (title, detail): (SharedString, SharedString) = match status { + SpeechStatus::Idle if !supported => ( + "Not supported here".into(), + "This platform has no system recognizer. Switch to Demo to try dictation.".into(), + ), + SpeechStatus::Idle if !available => ("Not available".into(), unavailable_hint().into()), + SpeechStatus::Idle => ( + "Ready".into(), + "Click the microphone to dictate into the note at the cursor.".into(), + ), + SpeechStatus::Connecting => ( + "Connecting…".into(), + non_empty( + transcript, + "Start talking. What you say is kept while it connects.", + ), + ), + SpeechStatus::Recording => ( + format!("Listening · {}", elapsed.unwrap_or_default()).into(), + non_empty(transcript, "Start talking."), + ), + SpeechStatus::Stopping => ( + "Finishing…".into(), + non_empty(transcript, "Waiting for the last words."), + ), + }; + let capturing = status.is_capturing(); + let toggle_keys = Keystroke::parse(TOGGLE_KEYS).ok().map(Kbd::new); + + h_flex() + .gap_3() + .px_3() + .py_2p5() + .items_center() + .rounded(cx.theme().radius_lg) + .border_1() + .border_color(if capturing { + cx.theme().ring + } else { + cx.theme().border + }) + .bg(cx.theme().background) + .when(capturing, |this| this.shadow_sm()) + .child( + SpeechButton::new(&self.speech) + .show_when_unsupported(true) + .large(), + ) + .child( + v_flex() + .flex_1() + .min_w_0() + .gap_0p5() + .child( + div() + .text_sm() + .font_medium() + .text_color(if supported && available || status.is_active() { + cx.theme().foreground + } else { + cx.theme().muted_foreground + }) + .child(title), + ) + .child( + div() + .text_sm() + .truncate() + .text_color(if status.is_active() && !speech.transcript().is_empty() { + cx.theme().foreground + } else { + cx.theme().muted_foreground + }) + .child(detail), + ), + ) + .when(status.is_active(), |this| { + this.child(SpeechWaveform::new(&self.speech).bars(28).small()) + }) + .map(|this| { + if status.is_active() { + this.child( + Button::new("discard") + .ghost() + .small() + .label("Discard") + .tooltip("Stop without inserting the text") + .on_click(cx.listener(|this, _, _, cx| { + this.speech.update(cx, |speech, cx| speech.cancel(cx)); + })), + ) + } else { + this.when_some(toggle_keys.filter(|_| available), |this, kbd| { + this.child(kbd) + }) + } + }) + } + + fn render_log(&self, cx: &mut Context) -> impl IntoElement { + v_flex() + .h(px(168.)) + .flex_shrink_0() + .rounded(cx.theme().radius_lg) + .border_1() + .border_color(cx.theme().border) + .overflow_hidden() + .child( + h_flex() + .px_3() + .py_1p5() + .items_center() + .justify_between() + .border_b_1() + .border_color(cx.theme().border) + .child( + h_flex() + .gap_2() + .items_center() + .text_xs() + .font_medium() + .text_color(cx.theme().muted_foreground) + .child(Icon::new(ExtraIcon::ScrollText).xsmall()) + .child("Session log"), + ) + .child( + Button::new("clear-log") + .ghost() + .xsmall() + .label("Clear") + .disabled(self.log.is_empty()) + .on_click(cx.listener(|this, _, _, cx| { + this.log.clear(); + cx.notify(); + })), + ), + ) + .child( + div().flex_1().min_h_0().child( + v_flex() + .id("session-log") + .size_full() + .px_3() + .py_2() + .gap_1() + .overflow_y_scrollbar() + .when(self.log.is_empty(), |this| { + this.items_center().justify_center().child( + div() + .text_xs() + .text_color(cx.theme().muted_foreground) + .child("Events of each session appear here."), + ) + }) + .children(self.log.iter().rev().map(|entry| log_row(entry, cx))), + ), + ) + } +} + +impl Focusable for DictationApp { + fn focus_handle(&self, _: &App) -> FocusHandle { + self.focus_handle.clone() + } +} + +impl Render for DictationApp { + fn render(&mut self, _: &mut Window, cx: &mut Context) -> impl IntoElement { + v_flex() + .size_full() + .bg(cx.theme().background) + .text_color(cx.theme().foreground) + .on_action(cx.listener(Self::toggle_dictation)) + .child( + TitleBar::new().child( + h_flex() + .gap_2() + .items_center() + .text_sm() + .font_medium() + .child( + Icon::new(ExtraIcon::AudioLines) + .small() + .text_color(cx.theme().primary), + ) + .child("Dictation"), + ), + ) + .child( + h_flex() + .flex_1() + .min_h_0() + .child(self.render_sidebar(cx)) + .child( + v_flex() + .flex_1() + .min_w_0() + .h_full() + .gap_4() + .p_5() + .child(self.render_notes_header(cx)) + .child( + div() + .flex_1() + .min_h_0() + .child(Textarea::new(&self.notes).h_full()), + ) + .child(self.render_dictation_bar(cx)) + .child(self.render_log(cx)), + ), + ) + } +} + +fn sidebar_section(title: &'static str, cx: &App) -> Div { + v_flex().gap_2().child( + div() + .text_xs() + .font_medium() + .text_color(cx.theme().muted_foreground) + .child(title), + ) +} + +fn caption(text: &'static str, cx: &App) -> impl IntoElement { + div() + .text_xs() + .text_color(cx.theme().muted_foreground) + .child(text) +} + +fn check_row(label: &'static str, value: impl IntoElement, cx: &App) -> impl IntoElement { + h_flex() + .gap_3() + .items_center() + .justify_between() + .child( + div() + .flex_shrink_0() + .text_color(cx.theme().muted_foreground) + .child(label), + ) + .child(h_flex().flex_1().min_w_0().justify_end().child(value)) +} + +fn log_row(entry: &LogEntry, cx: &App) -> impl IntoElement { + let color = match entry.tone { + Tone::Neutral => cx.theme().muted_foreground, + Tone::Progress => cx.theme().info, + Tone::Success => cx.theme().success, + Tone::Danger => cx.theme().danger, + }; + + h_flex() + .gap_3() + .items_start() + .text_xs() + .child( + div() + .flex_shrink_0() + .font_family(cx.theme().mono_font_family.clone()) + .text_color(cx.theme().muted_foreground) + .child(format_timestamp(entry.at)), + ) + .child( + div() + .w(px(64.)) + .flex_shrink_0() + .font_medium() + .text_color(color) + .child(entry.label), + ) + .child(div().flex_1().min_w_0().child(entry.text.clone())) +} + +fn unavailable_hint() -> &'static str { + if cfg!(target_os = "macos") { + "This Mac can’t recognize the language offline, or speech recognition access is off \ + in System Settings › Privacy & Security." + } else if cfg!(target_os = "windows") { + "Install the language’s speech pack and turn on Online speech recognition in Settings › \ + Privacy & security › Speech." + } else { + "The recognizer can’t start right now." + } +} + +fn non_empty(text: SharedString, fallback: &'static str) -> SharedString { + if text.is_empty() { + fallback.into() + } else { + text + } +} + +fn format_duration(duration: Duration) -> String { + let seconds = duration.as_secs(); + format!("{}:{:02}", seconds / 60, seconds % 60) +} + +fn format_timestamp(duration: Duration) -> String { + let millis = duration.as_millis(); + format!( + "{:02}:{:02}.{:03}", + millis / 60_000, + millis / 1_000 % 60, + millis % 1_000 + ) +} + +/// Print what this machine supports and quit. +fn print_checks(cx: &mut App) { + println!("Platform {}", platform_name()); + println!( + "Input device {}", + input_device_name().as_deref().unwrap_or("none") + ); + println!("System recognizer, by language:"); + for language in languages() { + let recognizer = SystemRecognizer::new().locale(language.tag.clone()); + let available = + gpui_kit::component::speech::SpeechRecognizer::is_available(&recognizer, cx); + println!( + " {:<7} {:<12} {}", + language.tag.as_ref(), + if available { + "available" + } else { + "unavailable" + }, + language.name.as_ref(), + ); + } +} + +fn main() { + let check = std::env::args().any(|arg| arg == "--check"); + let app = gpui_kit::application().with_assets(AppAssets); + + app.run(move |cx| { + gpui_kit::init(cx); + if check { + print_checks(cx); + cx.quit(); + return; + } + + cx.bind_keys([KeyBinding::new(TOGGLE_KEYS, ToggleDictation, None)]); + cx.activate(true); + + let window_options = WindowOptions { + window_bounds: Some(WindowBounds::centered(size(px(980.), px(680.)), cx)), + window_min_size: Some(size(px(760.), px(520.))), + ..TitleBar::window_options() + }; + gpui_kit::open_window(window_options, cx, |window, cx| { + cx.new(|cx| DictationApp::new(window, cx)) + }) + .expect("Failed to open window"); + }); +} diff --git a/script/install-linux.sh b/script/install-linux.sh index e8a2cde716..594d7d3447 100755 --- a/script/install-linux.sh +++ b/script/install-linux.sh @@ -5,5 +5,5 @@ sudo apt update sudo apt install -y \ gcc g++ clang libfontconfig-dev libwayland-dev \ libwebkit2gtk-4.1-dev libxkbcommon-x11-dev libx11-xcb-dev \ - libssl-dev libzstd-dev \ + libssl-dev libzstd-dev libasound2-dev \ vulkan-validationlayers libvulkan1 diff --git a/skills/gpui-kit/SKILL.md b/skills/gpui-kit/SKILL.md index 6832cc2ccd..a67265bd57 100644 --- a/skills/gpui-kit/SKILL.md +++ b/skills/gpui-kit/SKILL.md @@ -133,6 +133,7 @@ fetch the component's `.md` doc. | `Editor` | `input::{Editor, EditorState}` | Stateful. Code editor, `tree-sitter` feature | | `NumberInput` | `input::{NumberInput, NumberInputEvent}` | Stateful. Numeric with step | | `OtpInput` | `input::OtpInput` | Stateful. One-time password | +| `SpeechButton` | `speech::{SpeechButton, SpeechState}` | Stateful. Dictation, `speech` feature | | `Select` | `select::{Select, SelectState}` | Stateful. Dropdown picker | | `Combobox` | `combobox::{Combobox, ComboboxState}` | Stateful. Searchable select | | `Checkbox` | `checkbox::Checkbox` | Stateless. `on_click` receives `&bool` | diff --git a/website/component/index.md b/website/component/index.md index 8ea400c11e..4bb6e390b5 100644 --- a/website/component/index.md +++ b/website/component/index.md @@ -52,6 +52,7 @@ collapsed: false - [DatePicker](date-picker) - Date selection with calendar - [TimeField](time-field) - Segmented time-of-day input - [OtpInput](otp-input) - One-time password input +- [Speech](speech) - Dictation through the system recognizer or your own - [ColorPicker](color-picker) - Color selection interface - [Form](form) - Form container and layout diff --git a/website/component/speech.md b/website/component/speech.md new file mode 100644 index 0000000000..102a83cfff --- /dev/null +++ b/website/component/speech.md @@ -0,0 +1,408 @@ +--- +title: Speech +description: Dictate text from the microphone through the system speech recognizer or any recognizer the application provides. +maturity: [experimental, platform-dependent] +--- + +# Speech + +The speech module turns what the user says into text. `SpeechState` owns a +dictation session: it captures audio from an `AudioInput`, feeds it to a +`SpeechRecognizer`, keeps the transcript, and emits `SpeechEvent`s. +`SpeechButton` starts and stops the session and `SpeechWaveform` shows the +input level while it captures. + +The application owns where the text goes. The state never edits an input on its +own; subscribe to its events and insert the final transcript where it belongs. + +Recognition and capture are both replaceable. Without further setup, the state +uses the recognizer built into macOS or Windows and the default microphone. +Implement `SpeechRecognizer` to use a cloud service or a local model, and +`AudioInput` to feed audio from elsewhere. + +## Enable the feature + +The microphone and the system recognizer are behind the `speech` feature: + +```toml +[dependencies] +gpui-kit = { version = "{{gpui_kit_version}}", features = ["speech"] } +``` + +The feature adds [cpal](https://crates.io/crates/cpal) for audio capture. On +Linux, building it needs the ALSA development package (`libasound2-dev` on +Debian and Ubuntu). + +Without the feature, and on the web, the types are still available but there is +no default input and no system recognizer. A state then works only with both an +application recognizer and an application input. + +## Import + +```rust +use gpui_kit::component::speech::{ + SpeechButton, SpeechEvent, SpeechState, SpeechStatus, SpeechWaveform, +}; +``` + +## Usage + +### Dictate into an input + +Create the state beside the input it fills, subscribe to it, and put the button +and waveform in the input's suffix: + +```rust +use gpui_kit::component::{ + WindowExt as _, h_flex, + input::{Input, InputState}, + notification::Notification, + speech::{SpeechButton, SpeechEvent, SpeechState, SpeechWaveform}, +}; + +let input = cx.new(|cx| InputState::new(window, cx)); +let speech = cx.new(SpeechState::new); + +let subscription = cx.subscribe_in(&speech, window, { + let input = input.clone(); + move |_, _, event, window, cx| match event { + SpeechEvent::Final(text) if !text.is_empty() => { + input.update(cx, |input, cx| input.insert(text.clone(), window, cx)); + } + SpeechEvent::Error(error) => { + window.push_notification(Notification::error(error.to_string()), cx); + } + _ => {} + } +}); +``` + +```rust +let capturing = self.speech.read(cx).status().is_capturing(); + +Input::new(&self.input).suffix( + h_flex() + .gap_2() + .when(capturing, |this| { + this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + }) + .child(SpeechButton::new(&self.speech).xsmall()), +) +``` + +Keep the subscription on the view that owns the input, for example in its +`_subscriptions` list. + +### Events + +| Event | When | +| --- | --- | +| `Started` | A session started and audio is being captured. | +| `Partial(text)` | The transcript changed while the user speaks. | +| `Final(text)` | The session ended normally. The text may be empty. | +| `Cancelled` | The session was cancelled and its transcript discarded. | +| `Error(error)` | The session failed and ended. | + +`Partial` and `Final` carry the **whole** transcript of the session so far: +every committed phrase followed by the current hypothesis. A later event +replaces an earlier one, so show the latest `Partial` text as a preview and +commit only the `Final` text. `SpeechState::transcript()` returns the same text +during render: + +```rust +let speech = self.speech.read(cx); + +v_flex() + .gap_2() + .child(Input::new(&self.input)) + .when(speech.status().is_active(), |this| { + this.child( + div() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child(speech.transcript()), + ) + }) +``` + +### Session control + +`SpeechButton` toggles the session. To drive it from an action, a key binding, +or another control, call the state directly: + +```rust +speech.update(cx, |speech, cx| speech.start(cx)); // Does nothing while running. +speech.update(cx, |speech, cx| speech.stop(cx)); // Emits `Final` once the result is in. +speech.update(cx, |speech, cx| speech.cancel(cx)); // Emits `Cancelled` at once. +speech.update(cx, |speech, cx| speech.toggle(cx)); // `start` when idle, otherwise `stop`. +``` + +`status()` reports where the session is: + +| `SpeechStatus` | Meaning | +| --- | --- | +| `Idle` | No session is running. | +| `Connecting` | Audio is captured while the recognizer connects. | +| `Recording` | Audio is captured and recognized. | +| `Stopping` | Capture stopped; waiting for the final result. | + +`is_active()` is true for every status but `Idle`, and `is_capturing()` for +`Connecting` and `Recording`. While `Stopping`, the button shows a spinner and +ignores clicks. If the recognizer does not deliver its final result within the +stop timeout, the session ends with the transcript so far. The timeout is +3 seconds by default: + +```rust +let speech = cx.new(|cx| SpeechState::new(cx).stop_timeout(Duration::from_secs(5))); +``` + +### Button states + +`SpeechButton` is a ghost icon `Button` with a localized tooltip and accessible +name: a microphone at rest, and a pressed stop glyph while capturing. + +- When the state has no recognizer or no input on this platform + (`has_recognizer()` is `false`), the button renders **nothing**, so an + application can place it unconditionally. Use `.show_when_unsupported(true)` + to render it disabled instead. +- When the recognizer reports itself unavailable (`is_available(cx)` is + `false`), for example because the language is not installed, the button is + disabled with an “unavailable” tooltip. +- `.disabled(true)` disables it for application reasons, such as a readonly + field. + +```rust +SpeechButton::new(&speech) + .small() + .show_when_unsupported(true) + .disabled(readonly) +``` + +## Choose a recognizer + +The state picks its recognizer in this order: + +1. the one passed to `.recognizer(...)`; +2. otherwise the platform's `SystemRecognizer`, unless `.system_fallback(false)` + turned it off; +3. otherwise none, and `has_recognizer()` is `false`. + +```rust +use gpui_kit::component::speech::SystemRecognizer; + +// Dictate Chinese regardless of the system language. +let speech = cx.new(|cx| { + SpeechState::new(cx).recognizer(SystemRecognizer::new().locale("zh-CN")) +}); + +// Use the application's own recognizer instead of the system's. +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(CloudRecognizer::new(client))); + +// Never fall back to the system recognizer. Without a recognizer of its own, +// `has_recognizer()` is false and the button renders nothing. +let speech = cx.new(|cx| SpeechState::new(cx).system_fallback(false)); +``` + +`CloudRecognizer` stands for an application type that implements +`SpeechRecognizer`; see [Implement a recognizer](#implement-a-recognizer). + +`SystemRecognizer::new()` recognizes the system's current language; +`.locale(...)` takes a BCP 47 tag such as `en-US` or `zh-CN`. A language the +platform cannot recognize makes the recognizer unavailable. + +## Implement a recognizer + +Implement `SpeechRecognizer` to use any speech service. `start` opens a +`RecognitionSession` and reports results through the `SpeechSink` it receives: + +```rust +use gpui_kit::App; +use gpui_kit::component::speech::{ + RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink, +}; + +/// Reports how much audio it has heard, as a stand-in for a real service. +struct DurationRecognizer; + +impl SpeechRecognizer for DurationRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // A service would connect here and call `ready` once connected. + sink.ready(cx); + Ok(Box::new(DurationSession { sink, samples: 0 })) + } +} + +struct DurationSession { + sink: SpeechSink, + samples: usize, +} + +impl RecognitionSession for DurationSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + // 16 kHz mono, the default `audio_format`. + self.samples += samples.len(); + let seconds = self.samples / 16_000; + self.sink.hypothesis(format!("{seconds} s of audio"), cx); + } + + fn finish(&mut self, cx: &mut App) { + let seconds = self.samples / 16_000; + self.sink.phrase(format!("{seconds} s of audio"), cx); + self.sink.finish(cx); + } +} +``` + +A session runs as follows: + +1. **`start`** opens the session. Connecting may take a while, so return at + once, buffer the audio pushed in the meantime, and call `sink.ready(cx)` + when the service accepts audio. The status moves from `Connecting` to + `Recording`. +2. **`push_audio`** delivers interleaved 16-bit PCM in the recognizer's + `audio_format()`, 16 kHz mono by default. Report the phrase being spoken + with `sink.hypothesis(...)`, which replaces the previous hypothesis, and a + recognized phrase with `sink.phrase(...)`, which commits it and clears the + hypothesis. +3. **`finish`** means the user stopped talking: send the remaining audio, wait + for the last result, then call `sink.finish(cx)`. The state then emits + `Final`. +4. **Dropping** the session cancels it. Close the connection there and report + nothing more. + +Report a failure with `sink.error(SpeechError::recognizer(error), cx)`; the +session ends with `SpeechEvent::Error`. + +A few rules keep a recognizer simple: + +- The sink is cheap to clone, so a task that reads the service's responses can + own a copy. Once its session is stopped, cancelled, or replaced, its calls are + ignored, so a late response cannot leak into the next session. +- Sink calls are applied after the current update. They may be made from + anywhere on the main thread, including from inside `start` or `push_audio`. +- Phrases are joined verbatim. Include any separator the language needs, such + as a leading space between English sentences and none between Chinese ones. +- Override `is_available` to return `false` while the recognizer cannot work, + for example before the user signs in. It is called on every render, so keep + it cheap. + +## Provide audio + +`AudioInput` is the audio seam. The built-in `Microphone` captures the default +input device and converts it to the recognizer's format. Replace it to feed +audio from elsewhere, such as a file in tests: + +```rust +use std::time::Duration; + +use gpui_kit::{App, Subscription}; +use gpui_kit::component::speech::{AudioFormat, AudioInput, AudioSink, SpeechError}; + +/// Pushes 100 ms of silence at a time. +struct Silence; + +impl AudioInput for Silence { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + let len = format.sample_rate() as usize / 10 * format.channels() as usize; + let task = cx.spawn(async move |cx| { + loop { + cx.background_executor().timer(Duration::from_millis(100)).await; + cx.update(|cx| sink.push(vec![0; len], cx)); + } + }); + // Capture runs until the state drops this subscription. + Ok(Subscription::new(move || drop(task))) + } +} + +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(recognizer).input(Silence)); +``` + +`levels()` exposes the recent input levels that `SpeechWaveform` draws, in +`0.0..=1.0` with the oldest first, for an application that renders its own +meter. + +## Platform support + +| Platform | `SystemRecognizer` | `Microphone` | +| --- | --- | --- | +| macOS | `SFSpeechRecognizer`, on the device only | Core Audio | +| Windows | `Windows.Media.SpeechRecognition`, through Microsoft's online service | WASAPI | +| Linux | None; provide a recognizer | ALSA | +| Web | None | None; provide an input | + +### macOS + +The application's `Info.plist` must describe why it uses the microphone and +speech recognition: + +```xml +NSMicrophoneUsageDescription +Dictate text into messages. +NSSpeechRecognitionUsageDescription +Turn your speech into text. +``` + +Without `NSMicrophoneUsageDescription`, the system refuses microphone access +without asking. Without `NSSpeechRecognitionUsageDescription`, the system +recognizer reports itself unavailable instead of asking, because asking without +it would terminate the process. A binary run outside an app bundle, such as +from `cargo run`, has neither key. + +Recognition runs on the device only, so audio never leaves the machine. A +language the Mac cannot recognize offline is unavailable. + +### Windows + +Dictation needs the language's speech pack and the **Online speech +recognition** setting in **Settings > Privacy & security > Speech**. Audio is +sent to Microsoft's online speech service. An application that must keep audio +on the device should use `.system_fallback(false)` with its own recognizer. + +The Windows recognizer listens to the default microphone itself. The state +still captures through its input, but only to drive the waveform, so a custom +`AudioInput` does not change what the system recognizer hears. + +### Linux + +Linux has no system recognizer. Speech input works there only with an +application recognizer; without one, `SpeechButton` renders nothing. The +`Microphone` captures through ALSA, and building it needs `libasound2-dev`. + +## API Reference + +- [SpeechState] — the session: `recognizer`, `input`, `system_fallback`, + `stop_timeout`, `start`, `stop`, `cancel`, `toggle`, `status`, + `has_recognizer`, `is_available`, `transcript`, `levels` +- [SpeechEvent] and [SpeechStatus] +- [SpeechButton] — `show_when_unsupported`, plus `Sizable` and `Disableable` +- [SpeechWaveform] — `bars` (default 24, at most 48), plus `Sizable` +- [SpeechRecognizer], [RecognitionSession] and [SpeechSink] — the recognition + seam +- [AudioInput], [AudioSink] and [AudioFormat] — the audio seam +- [SpeechError] +- [SystemRecognizer] and [Microphone] — the `speech` feature's defaults + +[SpeechState]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechState.html +[SpeechEvent]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechEvent.html +[SpeechStatus]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechStatus.html +[SpeechButton]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechButton.html +[SpeechWaveform]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechWaveform.html +[SpeechRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.SpeechRecognizer.html +[RecognitionSession]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.RecognitionSession.html +[SpeechSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechSink.html +[AudioInput]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.AudioInput.html +[AudioSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioSink.html +[AudioFormat]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioFormat.html +[SpeechError]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechError.html +[SystemRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SystemRecognizer.html +[Microphone]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.Microphone.html diff --git a/website/docs/installation.md b/website/docs/installation.md index 09aada789c..477fa2f819 100644 --- a/website/docs/installation.md +++ b/website/docs/installation.md @@ -31,8 +31,8 @@ Install the native toolchain for your operating system, then add the `gpui-kit`
sudo apt update
 sudo apt install -y gcc g++ clang libfontconfig-dev libwayland-dev \
   libwebkit2gtk-4.1-dev libxkbcommon-x11-dev libx11-xcb-dev \
-  libssl-dev libzstd-dev vulkan-validationlayers libvulkan1
-

This matches the repository's script/install-linux.sh for Ubuntu 24.04. Other distributions need equivalent development packages. To display a window, run in a graphical Wayland or X11 session with a working Vulkan driver; installing libvulkan1 alone does not install a GPU driver.

+ libssl-dev libzstd-dev libasound2-dev vulkan-validationlayers libvulkan1 +

This matches the repository's script/install-linux.sh for Ubuntu 24.04. Other distributions need equivalent development packages. libasound2-dev (ALSA) is needed only by the speech feature. To display a window, run in a graphical Wayland or X11 session with a working Vulkan driver; installing libvulkan1 alone does not install a GPU driver.

diff --git a/website/zh-CN/component/index.md b/website/zh-CN/component/index.md index 70fd697422..2ace1f5a94 100644 --- a/website/zh-CN/component/index.md +++ b/website/zh-CN/component/index.md @@ -36,6 +36,7 @@ collapsed: false - [DatePicker](date-picker) - 日期选择器 - [TimeField](time-field) - 分段时间输入 - [OtpInput](otp-input) - 一次性验证码输入 +- [Speech](speech) - 通过系统或自定义识别器进行语音输入 - [ColorPicker](color-picker) - 颜色选择器 - [Questionnaire](questionnaire) - 可组合的多步骤问卷与答案 - [Form](form) - 表单容器与布局 diff --git a/website/zh-CN/component/speech.md b/website/zh-CN/component/speech.md new file mode 100644 index 0000000000..f6b3936c7b --- /dev/null +++ b/website/zh-CN/component/speech.md @@ -0,0 +1,336 @@ +--- +title: Speech +description: 通过系统语音识别器或应用自己的识别器,把麦克风中的语音转成文本。 +maturity: [experimental, platform-dependent] +--- + +# Speech + +Speech 模块把用户说的话转成文本。`SpeechState` 管理一次语音输入会话:它从 `AudioInput` 采集音频,交给 `SpeechRecognizer` 识别,保存转写文本并发出 `SpeechEvent`。`SpeechButton` 负责开始和停止会话,`SpeechWaveform` 在采集期间显示输入音量。 + +文本写到哪里由应用决定。状态本身不会修改任何输入框;应用订阅它的事件,再把最终文本插入到合适的位置。 + +识别和采集都可以替换。不做额外配置时,状态使用 macOS 或 Windows 自带的识别器和默认麦克风。实现 `SpeechRecognizer` 可以接入云端服务或本地模型,实现 `AudioInput` 可以从其他来源提供音频。 + +## 启用 feature + +麦克风和系统识别器位于 `speech` feature 之后: + +```toml +[dependencies] +gpui-kit = { version = "{{gpui_kit_version}}", features = ["speech"] } +``` + +该 feature 会引入 [cpal](https://crates.io/crates/cpal) 采集音频。在 Linux 上构建需要 ALSA 开发包(Debian 和 Ubuntu 上为 `libasound2-dev`)。 + +未启用该 feature 时,以及在 Web 上,这些类型依然可用,但没有默认输入,也没有系统识别器。此时状态只有在应用同时提供识别器和输入时才能工作。 + +## 导入 + +```rust +use gpui_kit::component::speech::{ + SpeechButton, SpeechEvent, SpeechState, SpeechStatus, SpeechWaveform, +}; +``` + +## 用法 + +### 语音输入到 Input + +在要填充的输入框旁边创建状态并订阅它,再把按钮和波形放到输入框的 suffix 中: + +```rust +use gpui_kit::component::{ + WindowExt as _, h_flex, + input::{Input, InputState}, + notification::Notification, + speech::{SpeechButton, SpeechEvent, SpeechState, SpeechWaveform}, +}; + +let input = cx.new(|cx| InputState::new(window, cx)); +let speech = cx.new(SpeechState::new); + +let subscription = cx.subscribe_in(&speech, window, { + let input = input.clone(); + move |_, _, event, window, cx| match event { + SpeechEvent::Final(text) if !text.is_empty() => { + input.update(cx, |input, cx| input.insert(text.clone(), window, cx)); + } + SpeechEvent::Error(error) => { + window.push_notification(Notification::error(error.to_string()), cx); + } + _ => {} + } +}); +``` + +```rust +let capturing = self.speech.read(cx).status().is_capturing(); + +Input::new(&self.input).suffix( + h_flex() + .gap_2() + .when(capturing, |this| { + this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + }) + .child(SpeechButton::new(&self.speech).xsmall()), +) +``` + +把订阅保存在拥有该输入框的视图上,例如放进它的 `_subscriptions` 列表。 + +### 事件 + +| 事件 | 触发时机 | +| --- | --- | +| `Started` | 会话已开始,正在采集音频。 | +| `Partial(text)` | 用户说话期间转写文本发生变化。 | +| `Final(text)` | 会话正常结束。文本可能为空。 | +| `Cancelled` | 会话被取消,转写文本被丢弃。 | +| `Error(error)` | 会话失败并结束。 | + +`Partial` 和 `Final` 携带的是本次会话到目前为止的**完整**转写文本:所有已确认的短语,加上当前的识别假设。后一个事件会取代前一个,因此把最新的 `Partial` 文本作为预览显示,只提交 `Final` 文本。在 render 中,`SpeechState::transcript()` 返回同样的文本: + +```rust +let speech = self.speech.read(cx); + +v_flex() + .gap_2() + .child(Input::new(&self.input)) + .when(speech.status().is_active(), |this| { + this.child( + div() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child(speech.transcript()), + ) + }) +``` + +### 控制会话 + +`SpeechButton` 会切换会话。如果要从 Action、快捷键或其他控件触发,直接调用状态的方法: + +```rust +speech.update(cx, |speech, cx| speech.start(cx)); // 会话进行中时不做任何事。 +speech.update(cx, |speech, cx| speech.stop(cx)); // 结果到达后发出 `Final`。 +speech.update(cx, |speech, cx| speech.cancel(cx)); // 立即发出 `Cancelled`。 +speech.update(cx, |speech, cx| speech.toggle(cx)); // 空闲时 `start`,否则 `stop`。 +``` + +`status()` 表示会话所处的阶段: + +| `SpeechStatus` | 含义 | +| --- | --- | +| `Idle` | 没有进行中的会话。 | +| `Connecting` | 正在采集音频,识别器仍在连接。 | +| `Recording` | 正在采集并识别音频。 | +| `Stopping` | 已停止采集,等待最终结果。 | + +除 `Idle` 外,`is_active()` 都为 true;`is_capturing()` 只在 `Connecting` 和 `Recording` 时为 true。处于 `Stopping` 时,按钮显示 spinner 并忽略点击。如果识别器在停止超时内没有给出最终结果,会话会以当前已有的文本结束。超时默认为 3 秒: + +```rust +let speech = cx.new(|cx| SpeechState::new(cx).stop_timeout(Duration::from_secs(5))); +``` + +### 按钮状态 + +`SpeechButton` 是一个 ghost 样式的图标 `Button`,带有本地化的 tooltip 和无障碍名称:空闲时显示麦克风,采集时显示按下状态的停止图标。 + +- 当状态在当前平台上没有识别器或没有输入(`has_recognizer()` 为 `false`)时,按钮**不渲染任何内容**,应用可以无条件放置它。使用 `.show_when_unsupported(true)` 可以改为渲染禁用的按钮。 +- 当识别器报告自己不可用(`is_available(cx)` 为 `false`),例如语言未安装时,按钮会禁用,tooltip 提示不可用。 +- `.disabled(true)` 用于应用自身的禁用原因,例如 readonly 的输入框。 + +```rust +SpeechButton::new(&speech) + .small() + .show_when_unsupported(true) + .disabled(readonly) +``` + +## 选择识别器 + +状态按以下顺序选择识别器: + +1. 通过 `.recognizer(...)` 传入的识别器; +2. 否则使用平台的 `SystemRecognizer`,除非 `.system_fallback(false)` 关闭了回退; +3. 否则没有识别器,`has_recognizer()` 为 `false`。 + +```rust +use gpui_kit::component::speech::SystemRecognizer; + +// 无论系统语言是什么,都识别中文。 +let speech = cx.new(|cx| { + SpeechState::new(cx).recognizer(SystemRecognizer::new().locale("zh-CN")) +}); + +// 使用应用自己的识别器代替系统识别器。 +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(CloudRecognizer::new(client))); + +// 不回退到系统识别器。没有自己的识别器时, +// `has_recognizer()` 为 false,按钮不渲染任何内容。 +let speech = cx.new(|cx| SpeechState::new(cx).system_fallback(false)); +``` + +`CloudRecognizer` 代表应用中实现了 `SpeechRecognizer` 的类型,参见[实现识别器](#实现识别器)。 + +`SystemRecognizer::new()` 识别系统当前语言;`.locale(...)` 接受 BCP 47 语言标签,例如 `en-US` 或 `zh-CN`。平台无法识别的语言会使识别器不可用。 + +## 实现识别器 + +实现 `SpeechRecognizer` 即可接入任意语音服务。`start` 打开一个 `RecognitionSession`,并通过收到的 `SpeechSink` 报告结果: + +```rust +use gpui_kit::App; +use gpui_kit::component::speech::{ + RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink, +}; + +/// 报告已收到多少音频,用来代替真实服务。 +struct DurationRecognizer; + +impl SpeechRecognizer for DurationRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // 真实服务在这里建立连接,连接成功后再调用 `ready`。 + sink.ready(cx); + Ok(Box::new(DurationSession { sink, samples: 0 })) + } +} + +struct DurationSession { + sink: SpeechSink, + samples: usize, +} + +impl RecognitionSession for DurationSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + // 16 kHz 单声道,即默认的 `audio_format`。 + self.samples += samples.len(); + let seconds = self.samples / 16_000; + self.sink.hypothesis(format!("{seconds} s of audio"), cx); + } + + fn finish(&mut self, cx: &mut App) { + let seconds = self.samples / 16_000; + self.sink.phrase(format!("{seconds} s of audio"), cx); + self.sink.finish(cx); + } +} +``` + +一次会话的流程如下: + +1. **`start`** 打开会话。连接可能需要一段时间,因此应立即返回,缓存这期间推送的音频,并在服务开始接收音频时调用 `sink.ready(cx)`。状态随之从 `Connecting` 变为 `Recording`。 +2. **`push_audio`** 按识别器的 `audio_format()`(默认 16 kHz 单声道)传入交错的 16 位 PCM。用 `sink.hypothesis(...)` 报告正在说的短语,它会替换上一个识别假设;用 `sink.phrase(...)` 报告识别完成的短语,它会确认该短语并清空识别假设。 +3. **`finish`** 表示用户已停止说话:发送剩余音频,等待最后的结果,然后调用 `sink.finish(cx)`。状态随后发出 `Final`。 +4. **drop** 会话即取消会话。在 drop 时关闭连接,之后不再报告任何内容。 + +出错时调用 `sink.error(SpeechError::recognizer(error), cx)`,会话以 `SpeechEvent::Error` 结束。 + +以下几点能让识别器保持简单: + +- sink 可以低成本 clone,读取服务响应的任务可以持有一份。会话停止、取消或被替换后,它的调用会被忽略,迟到的响应不会混入下一次会话。 +- sink 的调用在当前 update 结束后生效。可以在主线程的任何位置调用,包括在 `start` 或 `push_audio` 内部。 +- 短语按原样拼接。请带上语言需要的分隔符,例如英文句子之间的前导空格;中文句子之间则不需要。 +- 识别器暂时无法工作时(例如用户登录之前),重写 `is_available` 返回 `false`。它在每次 render 时都会调用,应保持轻量。 + +## 提供音频 + +`AudioInput` 是音频的扩展点。内置的 `Microphone` 从默认输入设备采集,并转换为识别器需要的格式。如需从其他来源提供音频(例如测试中的文件),替换它即可: + +```rust +use std::time::Duration; + +use gpui_kit::{App, Subscription}; +use gpui_kit::component::speech::{AudioFormat, AudioInput, AudioSink, SpeechError}; + +/// 每次推送 100 ms 的静音。 +struct Silence; + +impl AudioInput for Silence { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + let len = format.sample_rate() as usize / 10 * format.channels() as usize; + let task = cx.spawn(async move |cx| { + loop { + cx.background_executor().timer(Duration::from_millis(100)).await; + cx.update(|cx| sink.push(vec![0; len], cx)); + } + }); + // 采集一直持续到状态 drop 这个 subscription。 + Ok(Subscription::new(move || drop(task))) + } +} + +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(recognizer).input(Silence)); +``` + +`levels()` 返回 `SpeechWaveform` 绘制所用的近期输入音量,取值 `0.0..=1.0`,按时间从旧到新排列,适合需要自绘音量表的应用。 + +## 平台支持 + +| 平台 | `SystemRecognizer` | `Microphone` | +| --- | --- | --- | +| macOS | `SFSpeechRecognizer`,仅在本机识别 | Core Audio | +| Windows | `Windows.Media.SpeechRecognition`,经由 Microsoft 在线服务 | WASAPI | +| Linux | 无,需由应用提供识别器 | ALSA | +| Web | 无 | 无,需由应用提供输入 | + +### macOS + +应用的 `Info.plist` 必须说明使用麦克风和语音识别的原因: + +```xml +NSMicrophoneUsageDescription +用于在消息中语音输入文字。 +NSSpeechRecognitionUsageDescription +用于把你的语音转成文字。 +``` + +缺少 `NSMicrophoneUsageDescription` 时,系统会直接拒绝麦克风访问,不会询问用户。缺少 `NSSpeechRecognitionUsageDescription` 时,系统识别器会报告不可用,而不是发起询问,因为缺少该键时发起询问会导致进程终止。在 app bundle 之外运行的二进制(例如通过 `cargo run` 启动)不包含这两个键。 + +识别只在本机进行,音频不会离开设备。Mac 无法离线识别的语言不可用。 + +### Windows + +语音输入需要安装对应语言的语音包,并在**设置 > 隐私和安全性 > 语音**中打开**联机语音识别**。音频会发送到 Microsoft 的在线语音服务。音频必须保留在本机的应用应使用 `.system_fallback(false)`,并提供自己的识别器。 + +Windows 识别器会自行监听默认麦克风。状态仍通过其输入采集音频,但只用于驱动波形,因此自定义 `AudioInput` 不会改变系统识别器听到的内容。 + +### Linux + +Linux 没有系统识别器,只有在应用提供识别器时才能使用语音输入;否则 `SpeechButton` 不渲染任何内容。`Microphone` 通过 ALSA 采集,构建时需要 `libasound2-dev`。 + +## API 参考 + +- [SpeechState]:会话本身,包括 `recognizer`、`input`、`system_fallback`、`stop_timeout`、`start`、`stop`、`cancel`、`toggle`、`status`、`has_recognizer`、`is_available`、`transcript`、`levels` +- [SpeechEvent] 与 [SpeechStatus] +- [SpeechButton]:`show_when_unsupported`,以及 `Sizable` 和 `Disableable` +- [SpeechWaveform]:`bars`(默认 24,最多 48),以及 `Sizable` +- [SpeechRecognizer]、[RecognitionSession] 与 [SpeechSink]:识别的扩展点 +- [AudioInput]、[AudioSink] 与 [AudioFormat]:音频的扩展点 +- [SpeechError] +- [SystemRecognizer] 与 [Microphone]:`speech` feature 提供的默认实现 + +[SpeechState]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechState.html +[SpeechEvent]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechEvent.html +[SpeechStatus]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechStatus.html +[SpeechButton]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechButton.html +[SpeechWaveform]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechWaveform.html +[SpeechRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.SpeechRecognizer.html +[RecognitionSession]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.RecognitionSession.html +[SpeechSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechSink.html +[AudioInput]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.AudioInput.html +[AudioSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioSink.html +[AudioFormat]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioFormat.html +[SpeechError]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechError.html +[SystemRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SystemRecognizer.html +[Microphone]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.Microphone.html diff --git a/website/zh-CN/docs/installation.md b/website/zh-CN/docs/installation.md index ddaba31b1e..4c6d0e3b9d 100644 --- a/website/zh-CN/docs/installation.md +++ b/website/zh-CN/docs/installation.md @@ -31,8 +31,8 @@ order: -1
sudo apt update
 sudo apt install -y gcc g++ clang libfontconfig-dev libwayland-dev \
   libwebkit2gtk-4.1-dev libxkbcommon-x11-dev libx11-xcb-dev \
-  libssl-dev libzstd-dev vulkan-validationlayers libvulkan1
-

此清单与仓库的 script/install-linux.sh 一致,适用于 Ubuntu 24.04;其他发行版需要安装对应的开发包。显示窗口还需要可用的 Wayland 或 X11 图形会话及 Vulkan 驱动;单独安装 libvulkan1 并不会安装 GPU 驱动。

+ libssl-dev libzstd-dev libasound2-dev vulkan-validationlayers libvulkan1 +

此清单与仓库的 script/install-linux.sh 一致,适用于 Ubuntu 24.04;其他发行版需要安装对应的开发包。libasound2-dev(ALSA)仅在启用 speech feature 时需要。显示窗口还需要可用的 Wayland 或 X11 图形会话及 Vulkan 驱动;单独安装 libvulkan1 并不会安装 GPU 驱动。

From 704da7a24659aa423ffe30f65f23e4e4ea1efed5 Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Fri, 2 Oct 2026 09:51:44 +0800 Subject: [PATCH 2/8] component-shell: Inventory the `speech` module and story The inventory test requires every public gpui-component module and story to be accounted for. Speech input needs a Rust `SpeechRecognizer` or the `speech` feature's platform services, which the frozen script catalog cannot construct, so both entries are classified as infrastructure with that reason. Co-Authored-By: Claude Opus 5.5 --- crates/component-shell/component-inventory.json | 12 ++++++++++++ 1 file changed, 12 insertions(+) diff --git a/crates/component-shell/component-inventory.json b/crates/component-shell/component-inventory.json index 78558c7281..a0ccf93b02 100644 --- a/crates/component-shell/component-inventory.json +++ b/crates/component-shell/component-inventory.json @@ -952,6 +952,12 @@ ] } }, + { + "source": "ui", + "name": "speech", + "classification": "infrastructure", + "explanation": "Speech input depends on an application-supplied SpeechRecognizer or the `speech` feature's microphone and system recognizer, which the frozen catalog cannot construct from script; it is not registered as a component yet." + }, { "source": "ui", "name": "spinner", @@ -2272,6 +2278,12 @@ ] } }, + { + "source": "story", + "name": "speech", + "classification": "infrastructure", + "explanation": "The Speech story drives a Rust SpeechRecognizer and the `speech` feature's platform services rather than a catalog constructor." + }, { "source": "story", "name": "spinner", From 1815f26bfe0fc974b4b699f7a85f897e4a8a4dae Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Fri, 2 Oct 2026 10:13:27 +0800 Subject: [PATCH 3/8] js_story: Add an infrastructure route for the Speech story The coverage audit mirrors every inventoried Rust Story in the JavaScript gallery. Speech is inventoried as infrastructure, so its route declares infrastructure availability and covers no registered surface. Co-Authored-By: Claude Opus 5.5 --- examples/js_story/catalog.js | 1 + examples/js_story/stories/coverage.js | 1 + examples/js_story/stories/inputs.js | 10 ++++++++++ 3 files changed, 12 insertions(+) diff --git a/examples/js_story/catalog.js b/examples/js_story/catalog.js index 9b1e75f921..c69ad128e5 100644 --- a/examples/js_story/catalog.js +++ b/examples/js_story/catalog.js @@ -90,6 +90,7 @@ const RUST_STORY_ORDER = [ "SidebarStory", "SkeletonStory", "SliderStory", + "SpeechStory", "SpinnerStory", "StatusBarStory", "StepperStory", diff --git a/examples/js_story/stories/coverage.js b/examples/js_story/stories/coverage.js index b1eb228e39..fd04ec3234 100644 --- a/examples/js_story/stories/coverage.js +++ b/examples/js_story/stories/coverage.js @@ -62,6 +62,7 @@ export const coveredBy = [ { route: "sidebar", registrations: ["Sidebar"] }, { route: "skeleton", registrations: ["Skeleton"] }, { route: "slider", registrations: ["Slider"] }, + { route: "speech", registrations: [] }, { route: "spinner", registrations: ["Spinner"] }, { route: "status-bar", registrations: ["StatusBar"] }, { route: "stepper", registrations: ["Stepper"] }, diff --git a/examples/js_story/stories/inputs.js b/examples/js_story/stories/inputs.js index cdd58c6061..6bc17f1885 100644 --- a/examples/js_story/stories/inputs.js +++ b/examples/js_story/stories/inputs.js @@ -111,6 +111,16 @@ export const stories = [ availability: "pending", api: "Slider", }), + pendingStory({ + id: "speech", + title: "Speech", + group: "Inputs", + rustStory: "SpeechStory", + description: "Dictation through a Rust speech recognizer.", + states: ["idle", "recording", "unsupported"], + availability: "infrastructure", + api: "SpeechButton", + }), pendingStory({ id: "color-picker", title: "ColorPicker", From c944f3c8fe1da03373bb06e98ae77459a004a299 Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Fri, 2 Oct 2026 11:19:31 +0800 Subject: [PATCH 4/8] speech: Name the example app after the crate `DictationApp`, the `dictation` action namespace and the window title become `SpeechApp`, `speech` and "Speech", matching the `speech` crate and the component names. Interface copy keeps the verb "dictate". Co-Authored-By: Claude Opus 5.5 --- examples/speech/Cargo.toml | 2 +- examples/speech/Info.plist | 4 ++-- examples/speech/README.md | 2 +- examples/speech/src/main.rs | 28 ++++++++++++++-------------- 4 files changed, 18 insertions(+), 18 deletions(-) diff --git a/examples/speech/Cargo.toml b/examples/speech/Cargo.toml index b03aca0292..8f936732ab 100644 --- a/examples/speech/Cargo.toml +++ b/examples/speech/Cargo.toml @@ -1,6 +1,6 @@ [package] name = "speech" -description = "A dictation notepad for trying out and testing GPUI Component speech input." +description = "A notepad you can talk into, for trying out and testing GPUI Component speech input." version = "0.7.0" publish = false edition.workspace = true diff --git a/examples/speech/Info.plist b/examples/speech/Info.plist index 6bd6d5e02b..1f7d7d9230 100644 --- a/examples/speech/Info.plist +++ b/examples/speech/Info.plist @@ -3,8 +3,8 @@ NSMicrophoneUsageDescription - Dictation listens to the microphone to turn what you say into text. + Speech listens to the microphone to turn what you say into text. NSSpeechRecognitionUsageDescription - Dictation recognizes speech on this Mac to type what you say. + Speech recognizes what you say on this Mac to type what you say. diff --git a/examples/speech/README.md b/examples/speech/README.md index eea9bbd187..de0886ff4e 100644 --- a/examples/speech/README.md +++ b/examples/speech/README.md @@ -1,6 +1,6 @@ # Speech -Dictation, a notepad you can talk into, built on GPUI Component's speech input. It is also +A notepad you can talk into, built on GPUI Component's speech input. It is also the test bench for the platform recognizers: the sidebar shows what this machine supports, and the session log records every `SpeechEvent`. diff --git a/examples/speech/src/main.rs b/examples/speech/src/main.rs index 746f69ec55..f2ec95a2ce 100644 --- a/examples/speech/src/main.rs +++ b/examples/speech/src/main.rs @@ -1,4 +1,4 @@ -//! Dictation: a notepad you can talk into, built on GPUI Component's speech +//! Speech: a notepad you can talk into, built on GPUI Component's speech //! input. It doubles as a test bench for the platform recognizers: the sidebar //! reports what this machine supports and the session log records every event. //! @@ -55,7 +55,7 @@ impl AssetSource for AppAssets { } } -actions!(dictation, [ToggleDictation]); +actions!(speech, [ToggleSpeech]); const TOGGLE_KEYS: &str = "secondary-shift-d"; /// Most log entries kept; older ones scroll away. @@ -154,7 +154,7 @@ fn platform_name() -> &'static str { } } -struct DictationApp { +struct SpeechApp { focus_handle: FocusHandle, engine: Engine, language: SharedString, @@ -171,7 +171,7 @@ struct DictationApp { _subscriptions: Vec, } -impl DictationApp { +impl SpeechApp { fn new(window: &mut Window, cx: &mut Context) -> Self { let engine = Engine::System; let language = SharedString::from("en-US"); @@ -246,7 +246,7 @@ impl DictationApp { } } - fn toggle_dictation(&mut self, _: &ToggleDictation, _: &mut Window, cx: &mut Context) { + fn toggle_speech(&mut self, _: &ToggleSpeech, _: &mut Window, cx: &mut Context) { self.speech.update(cx, |speech, cx| speech.toggle(cx)); } @@ -470,8 +470,8 @@ impl DictationApp { ) } - /// The bar that runs dictation: button, level, live text and controls. - fn render_dictation_bar(&self, cx: &mut Context) -> impl IntoElement { + /// The bar that runs a speech session: button, level, live text and controls. + fn render_speech_bar(&self, cx: &mut Context) -> impl IntoElement { let speech = self.speech.read(cx); let status = speech.status(); let transcript = speech.transcript(); @@ -642,19 +642,19 @@ impl DictationApp { } } -impl Focusable for DictationApp { +impl Focusable for SpeechApp { fn focus_handle(&self, _: &App) -> FocusHandle { self.focus_handle.clone() } } -impl Render for DictationApp { +impl Render for SpeechApp { fn render(&mut self, _: &mut Window, cx: &mut Context) -> impl IntoElement { v_flex() .size_full() .bg(cx.theme().background) .text_color(cx.theme().foreground) - .on_action(cx.listener(Self::toggle_dictation)) + .on_action(cx.listener(Self::toggle_speech)) .child( TitleBar::new().child( h_flex() @@ -667,7 +667,7 @@ impl Render for DictationApp { .small() .text_color(cx.theme().primary), ) - .child("Dictation"), + .child("Speech"), ), ) .child( @@ -689,7 +689,7 @@ impl Render for DictationApp { .min_h_0() .child(Textarea::new(&self.notes).h_full()), ) - .child(self.render_dictation_bar(cx)) + .child(self.render_speech_bar(cx)) .child(self.render_log(cx)), ), ) @@ -829,7 +829,7 @@ fn main() { return; } - cx.bind_keys([KeyBinding::new(TOGGLE_KEYS, ToggleDictation, None)]); + cx.bind_keys([KeyBinding::new(TOGGLE_KEYS, ToggleSpeech, None)]); cx.activate(true); let window_options = WindowOptions { @@ -838,7 +838,7 @@ fn main() { ..TitleBar::window_options() }; gpui_kit::open_window(window_options, cx, |window, cx| { - cx.new(|cx| DictationApp::new(window, cx)) + cx.new(|cx| SpeechApp::new(window, cx)) }) .expect("Failed to open window"); }); From a6fbb9a09f68433e8b86f186fee1ee57c4b66454 Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Fri, 2 Oct 2026 11:30:37 +0800 Subject: [PATCH 5/8] speech: Keep revised utterances and late stops from breaking a session On macOS, a result that carries an utterance on may revise its words or punctuation, so an exact prefix check took a revision for a restart and committed the old text twice. Treat a result as a restart only when it shares less than half of the utterance. On Windows, stopping a session that has already ended by itself, e.g. after the silence timeout, fails; that failure ended the session with an error and lost the transcript. Ignore it and let the queued `Completed` end the session. Co-Authored-By: Claude Opus 5.5 --- crates/component/src/speech/system/macos.rs | 44 ++++++++++++++++++++- crates/component/src/speech/system/winrt.rs | 15 +++++-- 2 files changed, 55 insertions(+), 4 deletions(-) diff --git a/crates/component/src/speech/system/macos.rs b/crates/component/src/speech/system/macos.rs index 95ad16a2ac..b36e2594f2 100644 --- a/crates/component/src/speech/system/macos.rs +++ b/crates/component/src/speech/system/macos.rs @@ -425,7 +425,7 @@ impl Recognition { /// text. Commit that utterance as a phrase only once it is dropped. fn on_result(&mut self, text: String, is_final: bool, ends_utterance: bool, cx: &mut App) { if let Some(utterance) = self.utterance.take() - && !text.starts_with(&utterance) + && starts_over(&utterance, &text) { self.commit(&utterance, cx); } @@ -463,6 +463,22 @@ impl Recognition { } } +/// Whether `text`, the result after the one that ended `utterance`, starts a new +/// utterance instead of carrying `utterance` on. +/// +/// A result that carries it on may still revise its words, punctuation or case +/// ("Hello word" becomes "Hello world, how"), so an exact prefix is too strict; +/// a result that starts over shares almost nothing with it. Less than half of +/// `utterance` in common means it started over. +fn starts_over(utterance: &str, text: &str) -> bool { + let common = utterance + .chars() + .zip(text.chars()) + .take_while(|(a, b)| a == b) + .count(); + common * 2 < utterance.chars().count() +} + /// Whether two phrases ending and starting with these characters need a space /// between them. fn needs_space(before: char, after: char) -> bool { @@ -484,3 +500,29 @@ fn is_unspaced_script(c: char) -> bool { | '\u{FF00}'..='\u{FFEF}' ) } + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn a_revised_utterance_carries_on() { + assert!(!starts_over("Hello word", "Hello world, how")); + assert!(!starts_over("Hello world", "Hello world. How are you")); + assert!(!starts_over("今天天气", "今天天气很好")); + } + + #[test] + fn a_reset_result_starts_over() { + assert!(starts_over("Hello world.", "How")); + assert!(starts_over("今天天气很好。", "明天")); + } + + #[test] + fn phrases_are_spaced_by_script() { + assert!(needs_space('.', 'H')); + assert!(!needs_space('d', ',')); + assert!(!needs_space('。', '明')); + assert!(!needs_space('好', 'O')); + } +} diff --git a/crates/component/src/speech/system/winrt.rs b/crates/component/src/speech/system/winrt.rs index 7fc7b18e87..e16d92acb8 100644 --- a/crates/component/src/speech/system/winrt.rs +++ b/crates/component/src/speech/system/winrt.rs @@ -265,9 +265,18 @@ async fn dictate( }; } // Stopping flushes the last phrase, then completes the session. - Message::Finish => action(continuous.StopAsync().map_err(speech_error)?) - .await - .map_err(speech_error)?, + Message::Finish => { + let stopped = match continuous.StopAsync() { + Ok(stop) => action(stop).await, + Err(error) => Err(error), + }; + // The session may have ended by itself, e.g. after the silence + // timeout, with its `Completed` still queued behind this; stopping + // it then fails, and that queued event ends the session instead. + if let Err(error) = stopped { + log::debug!("speech: stopping dictation failed: {error}"); + } + } } } Ok(()) From 19313e828e32f27cd6cb28011f4a086e460f048e Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Fri, 2 Oct 2026 14:12:25 +0800 Subject: [PATCH 6/8] speech: Let `SpeechButton` take style refinements A host places the button among its own controls, which often share a size and corner radius (e.g. round icon buttons in a chat composer). `SpeechButton` now implements `Styled` and refines its `Button` with that style. Co-Authored-By: Claude Opus 5.5 --- crates/component/src/speech/button.rs | 15 +++++++++++++-- 1 file changed, 13 insertions(+), 2 deletions(-) diff --git a/crates/component/src/speech/button.rs b/crates/component/src/speech/button.rs index be2f337978..e5525b30c4 100644 --- a/crates/component/src/speech/button.rs +++ b/crates/component/src/speech/button.rs @@ -1,8 +1,8 @@ -use gpui::{App, ElementId, Entity, IntoElement, RenderOnce, Window, div}; +use gpui::{App, ElementId, Entity, IntoElement, RenderOnce, StyleRefinement, Styled, Window, div}; use rust_i18n::t; use crate::{ - Disableable, IconName, Selectable as _, Sizable, Size, + Disableable, IconName, Selectable as _, Sizable, Size, StyledExt as _, button::{Button, ButtonVariants as _}, }; @@ -24,6 +24,7 @@ pub struct SpeechButton { size: Size, disabled: bool, show_when_unsupported: bool, + style: StyleRefinement, } impl SpeechButton { @@ -35,6 +36,7 @@ impl SpeechButton { size: Size::default(), disabled: false, show_when_unsupported: false, + style: StyleRefinement::default(), } } @@ -53,6 +55,14 @@ impl Sizable for SpeechButton { } } +impl Styled for SpeechButton { + /// Refines the button, e.g. its size or corner radius, to match the + /// controls around it. + fn style(&mut self) -> &mut StyleRefinement { + &mut self.style + } +} + impl Disableable for SpeechButton { fn disabled(mut self, disabled: bool) -> Self { self.disabled = disabled; @@ -93,6 +103,7 @@ impl RenderOnce for SpeechButton { .disabled(self.disabled || !available) .tooltip(label.clone()) .accessibility_label(label) + .refine_style(&self.style) .on_click({ let state = self.state.clone(); move |_, _, cx| state.update(cx, |state, cx| state.toggle(cx)) From 76e374c1baf73788ce7b8237a16e967d72d4891b Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Fri, 2 Oct 2026 14:55:21 +0800 Subject: [PATCH 7/8] speech: Make `SpeechWaveform` a live waveform that fills its width The waveform followed one level per audio push, which made it lag and jump with however the input batched its audio. Levels now come from `LevelMeter`: one per 25 ms of audio, carried across pushes, with fast attack and slow release and a noise floor that keeps silence calm. `SpeechWaveform` paints as many bars as its width holds, centered on the midline, the newest at the trailing edge. While capturing it redraws every frame and scrolls between levels; with reduced motion it only steps. It is `Styled` for its width, which replaces the fixed `bars` count. Co-Authored-By: Claude Opus 5.5 --- crates/component/src/speech/level.rs | 178 +++++++++++++++++++++++ crates/component/src/speech/mod.rs | 1 + crates/component/src/speech/state.rs | 74 ++++------ crates/component/src/speech/waveform.rs | 170 +++++++++++++++++----- crates/story/src/stories/speech_story.rs | 4 +- examples/speech/src/main.rs | 2 +- website/component/speech.md | 9 +- website/zh-CN/component/speech.md | 6 +- 8 files changed, 351 insertions(+), 93 deletions(-) create mode 100644 crates/component/src/speech/level.rs diff --git a/crates/component/src/speech/level.rs b/crates/component/src/speech/level.rs new file mode 100644 index 0000000000..2e6ba6d561 --- /dev/null +++ b/crates/component/src/speech/level.rs @@ -0,0 +1,178 @@ +use std::{collections::VecDeque, time::Duration}; + +use instant::Instant; + +/// How much audio one level covers: short enough for the bars to follow +/// syllables. +pub(super) const LEVEL_INTERVAL: Duration = Duration::from_millis(25); + +/// How many recent levels are kept: enough to fill a wide waveform (about six +/// seconds of audio). +pub(super) const LEVEL_HISTORY: usize = 256; + +/// Raw levels below this are background noise and read as silence, so a quiet +/// room shows a calm baseline instead of flickering bars. +const NOISE_FLOOR: f32 = 0.06; + +/// How far a level moves toward a louder reading per step: almost at once, so +/// peaks pop. +const ATTACK: f32 = 0.85; + +/// How far a level moves toward a quieter reading per step: slowly, so bars +/// fall back smoothly. +const RELEASE: f32 = 0.3; + +/// Turns a stream of PCM into smoothed input levels, one per +/// [`LEVEL_INTERVAL`] of audio, carrying partial intervals across pushes. +pub(super) struct LevelMeter { + /// Samples per level: [`LEVEL_INTERVAL`] at the stream's rate and channels. + window: usize, + sum: f64, + count: usize, + smoothed: f32, + levels: VecDeque, + last_level_at: Option, +} + +impl LevelMeter { + pub(super) fn new() -> Self { + Self { + window: window_for(16_000, 1), + sum: 0., + count: 0, + smoothed: 0., + levels: VecDeque::with_capacity(LEVEL_HISTORY), + last_level_at: None, + } + } + + /// Clear the history and measure a stream of `sample_rate` × `channels`. + pub(super) fn reset(&mut self, sample_rate: u32, channels: u16) { + *self = Self { + window: window_for(sample_rate, channels), + ..Self::new() + }; + } + + /// Measure `samples`; returns whether a new level was recorded. + pub(super) fn push(&mut self, samples: &[i16]) -> bool { + let mut recorded = false; + for &sample in samples { + let sample = sample as f64 / i16::MAX as f64; + self.sum += sample * sample; + self.count += 1; + if self.count == self.window { + let raw = gate(level_of_rms((self.sum / self.count as f64).sqrt())); + self.smoothed = smooth(self.smoothed, raw); + if self.levels.len() == LEVEL_HISTORY { + self.levels.pop_front(); + } + self.levels.push_back(self.smoothed); + self.sum = 0.; + self.count = 0; + recorded = true; + } + } + if recorded { + self.last_level_at = Some(Instant::now()); + } + recorded + } + + pub(super) fn levels(&self) -> impl ExactSizeIterator + '_ { + self.levels.iter().copied() + } + + /// When the newest level was recorded, to scroll smoothly between levels. + pub(super) fn last_level_at(&self) -> Option { + self.last_level_at + } +} + +fn window_for(sample_rate: u32, channels: u16) -> usize { + let per_second = sample_rate as u128 * channels.max(1) as u128; + ((per_second * LEVEL_INTERVAL.as_millis() / 1_000) as usize).max(1) +} + +/// The loudness of an RMS amplitude in `0.0..=1.0`, mapping -50 dBFS..0 dBFS +/// linearly so that normal speech fills most of the range. +pub(super) fn level_of_rms(rms: f64) -> f32 { + if rms <= 0. { + return 0.; + } + let db = 20. * rms.log10(); + ((db + 50.) / 50.).clamp(0., 1.) as f32 +} + +fn gate(level: f32) -> f32 { + if level < NOISE_FLOOR { 0. } else { level } +} + +/// One step of fast-attack, slow-release smoothing from `previous` toward `raw`. +pub(super) fn smooth(previous: f32, raw: f32) -> f32 { + let rate = if raw > previous { ATTACK } else { RELEASE }; + previous + (raw - previous) * rate +} + +#[cfg(test)] +mod tests { + use super::*; + + fn tone(amplitude: i16, len: usize) -> Vec { + (0..len) + .map(|ix| if ix % 2 == 0 { amplitude } else { -amplitude }) + .collect() + } + + #[test] + fn one_level_per_interval_across_pushes() { + let mut meter = LevelMeter::new(); + meter.reset(16_000, 1); + // 25 ms at 16 kHz is 400 samples; feed 1 000 samples in uneven pushes. + assert!(!meter.push(&tone(1_000, 150))); + assert!(meter.push(&tone(1_000, 300))); + assert!(meter.push(&tone(1_000, 550))); + assert_eq!(meter.levels().len(), 2); + assert!(meter.last_level_at().is_some()); + } + + #[test] + fn the_window_follows_the_stream_format() { + assert_eq!(window_for(16_000, 1), 400); + assert_eq!(window_for(48_000, 2), 2_400); + } + + #[test] + fn peaks_rise_fast_and_fall_slowly() { + let up = smooth(0., 1.); + assert!(up >= 0.85); + let down = smooth(1., 0.); + assert!(down >= 0.65, "{down}"); + assert!(smooth(down, 0.) < down); + } + + #[test] + fn background_noise_reads_as_silence() { + let mut meter = LevelMeter::new(); + meter.reset(16_000, 1); + // -60 dBFS: below the -50 dB floor of the scale. + meter.push(&tone(32, 400)); + assert_eq!(meter.levels().next(), Some(0.)); + } + + #[test] + fn level_maps_decibels_to_the_unit_range() { + assert_eq!(level_of_rms(0.), 0.); + assert_eq!(level_of_rms(1.), 1.); + // -20 dBFS sits at 0.6 on the -50..0 dB scale. + assert!((level_of_rms(0.1) - 0.6).abs() < 0.01); + } + + #[test] + fn history_is_bounded() { + let mut meter = LevelMeter::new(); + meter.reset(16_000, 1); + meter.push(&tone(8_000, 400 * (LEVEL_HISTORY + 10))); + assert_eq!(meter.levels().len(), LEVEL_HISTORY); + } +} diff --git a/crates/component/src/speech/mod.rs b/crates/component/src/speech/mod.rs index 24f775b288..68d363c0fb 100644 --- a/crates/component/src/speech/mod.rs +++ b/crates/component/src/speech/mod.rs @@ -11,6 +11,7 @@ //! only with an application recognizer. mod button; +mod level; mod recognizer; mod state; mod waveform; diff --git a/crates/component/src/speech/state.rs b/crates/component/src/speech/state.rs index f4a79376a2..6a869e35ce 100644 --- a/crates/component/src/speech/state.rs +++ b/crates/component/src/speech/state.rs @@ -1,11 +1,11 @@ -use std::{cell::OnceCell, collections::VecDeque, rc::Rc, time::Duration}; +use std::{cell::OnceCell, rc::Rc, time::Duration}; use gpui::{App, Context, EventEmitter, SharedString, Subscription, Task, WeakEntity}; -use super::{AudioInput, AudioSink, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; - -/// How many recent input levels a [`SpeechState`] keeps for a waveform. -pub(super) const LEVEL_HISTORY: usize = 48; +use super::{ + AudioInput, AudioSink, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink, + level::LevelMeter, +}; /// Default time [`SpeechState::stop`] waits for the final result. const DEFAULT_STOP_TIMEOUT: Duration = Duration::from_secs(3); @@ -83,7 +83,7 @@ pub struct SpeechState { next_session: usize, committed: String, hypothesis: SharedString, - levels: VecDeque, + meter: LevelMeter, } impl EventEmitter for SpeechState {} @@ -102,7 +102,7 @@ impl SpeechState { next_session: 0, committed: String::new(), hypothesis: SharedString::default(), - levels: VecDeque::with_capacity(LEVEL_HISTORY), + meter: LevelMeter::new(), } } @@ -165,9 +165,15 @@ impl SpeechState { } } - /// Recent input levels in `0.0..=1.0`, oldest first. + /// Recent input levels in `0.0..=1.0`, oldest first, one per 25 ms of + /// audio. Peaks rise at once and fall back smoothly; background noise reads + /// as `0.0`. pub fn levels(&self) -> impl ExactSizeIterator + '_ { - self.levels.iter().copied() + self.meter.levels() + } + + pub(super) fn last_level_at(&self) -> Option { + self.meter.last_level_at() } /// Start a session. Does nothing while one is running. @@ -212,7 +218,7 @@ impl SpeechState { self.status = SpeechStatus::Connecting; self.committed.clear(); self.hypothesis = SharedString::default(); - self.levels.clear(); + self.meter.reset(format.sample_rate(), format.channels()); self.session = Some(Session { id, capture: Some(capture), @@ -302,11 +308,11 @@ impl SpeechState { if let Some(session) = self.session.as_mut() { session.recognition.push_audio(samples, cx); } - if self.levels.len() == LEVEL_HISTORY { - self.levels.pop_front(); + // Redraw once per new level, not per push: levels are what the + // waveform shows. + if self.meter.push(samples) { + cx.notify(); } - self.levels.push_back(level(samples)); - cx.notify(); } pub(super) fn on_hypothesis(&mut self, text: SharedString, cx: &mut Context) { @@ -336,7 +342,7 @@ impl SpeechState { session.capture = None; } self.status = SpeechStatus::Idle; - self.levels.clear(); + self.meter = LevelMeter::new(); cx.emit(event); cx.notify(); } @@ -360,27 +366,6 @@ pub(super) fn defer_session_update( }); } -/// The loudness of `samples` in `0.0..=1.0`, mapping -50 dBFS..0 dBFS linearly -/// so that normal speech fills most of the range. -fn level(samples: &[i16]) -> f32 { - if samples.is_empty() { - return 0.; - } - let sum: f64 = samples - .iter() - .map(|&sample| { - let sample = sample as f64 / i16::MAX as f64; - sample * sample - }) - .sum(); - let rms = (sum / samples.len() as f64).sqrt(); - if rms <= 0. { - return 0.; - } - let db = 20. * rms.log10(); - ((db + 50.) / 50.).clamp(0., 1.) as f32 -} - #[cfg(test)] mod tests { use std::{cell::RefCell, rc::Rc, time::Duration}; @@ -532,15 +517,16 @@ mod tests { cx.update(|cx| { f.sink().ready(cx); - f.audio().push(vec![i16::MAX / 2; 160], cx); + // Two levels' worth: 25 ms is 400 samples at 16 kHz. + f.audio().push(vec![i16::MAX / 2; 800], cx); f.sink().hypothesis("hello", cx); }); cx.run_until_parked(); assert_eq!(f.status(cx), SpeechStatus::Recording); - assert_eq!(f.recognizer.recorded.borrow().samples, 160); + assert_eq!(f.recognizer.recorded.borrow().samples, 800); cx.read(|cx| { let state = f.state.read(cx); - assert_eq!(state.levels().len(), 1); + assert_eq!(state.levels().len(), 2); assert!(state.levels().next().unwrap() > 0.5); }); @@ -676,14 +662,4 @@ mod tests { assert!(!state.read(cx).is_available(cx)); }); } - - #[test] - fn level_maps_decibels_to_the_unit_range() { - assert_eq!(level(&[]), 0.); - assert_eq!(level(&[0; 16]), 0.); - assert_eq!(level(&[i16::MAX; 16]), 1.); - // -20 dBFS sits at 0.6 on the -50..0 dB scale. - let quiet = level(&[i16::MAX / 10; 16]); - assert!((quiet - 0.6).abs() < 0.01, "{quiet}"); - } } diff --git a/crates/component/src/speech/waveform.rs b/crates/component/src/speech/waveform.rs index 874e4b8d06..e0bf01143b 100644 --- a/crates/component/src/speech/waveform.rs +++ b/crates/component/src/speech/waveform.rs @@ -1,20 +1,31 @@ use gpui::{ - App, Entity, IntoElement, ParentElement as _, Pixels, RenderOnce, Styled as _, Window, div, px, + App, Bounds, Entity, IntoElement, ParentElement as _, Pixels, RenderOnce, StyleRefinement, + Styled, Window, canvas, div, fill, point, px, size, }; -use crate::{ActiveTheme as _, Sizable, Size, h_flex}; +use crate::{ActiveTheme as _, Sizable, Size, StyledExt as _}; -use super::{SpeechState, state::LEVEL_HISTORY}; +use super::{SpeechState, level::LEVEL_INTERVAL}; -/// A live bar graph of a [`SpeechState`]'s recent input levels. +/// The width of a waveform the caller does not size: 24 bars at the smaller +/// sizes. +const DEFAULT_WIDTH: Pixels = px(96.); + +/// A live waveform of a [`SpeechState`]'s input levels. +/// +/// Bars fill the waveform's width and grow up and down from its midline. The +/// newest level enters at the trailing edge and older ones scroll toward the +/// leading edge, redrawn every frame while audio is captured. Without capture +/// the bars rest as a muted baseline. With reduced motion the bars still show +/// each level but do not scroll between them. /// -/// The newest level is on the trailing end. The bars sit flat and muted while -/// no audio is captured. +/// Size the waveform like any element, e.g. `.w_full()` to span a row; it is +/// 96 px wide by default, and its height follows [`Sizable`]. #[derive(IntoElement)] pub struct SpeechWaveform { state: Entity, - bars: usize, size: Size, + style: StyleRefinement, } impl SpeechWaveform { @@ -22,17 +33,11 @@ impl SpeechWaveform { pub fn new(state: &Entity) -> Self { Self { state: state.clone(), - bars: 24, size: Size::default(), + style: StyleRefinement::default(), } } - /// Set the number of bars, default 24, at most 48. - pub fn bars(mut self, bars: usize) -> Self { - self.bars = bars.clamp(1, LEVEL_HISTORY); - self - } - fn height(&self) -> Pixels { match self.size { Size::Size(height) => height, @@ -51,33 +56,130 @@ impl Sizable for SpeechWaveform { } } +impl Styled for SpeechWaveform { + fn style(&mut self) -> &mut StyleRefinement { + &mut self.style + } +} + impl RenderOnce for SpeechWaveform { - fn render(self, _: &mut Window, cx: &mut App) -> impl IntoElement { + fn render(self, window: &mut Window, cx: &mut App) -> impl IntoElement { let height = self.height(); - let bar_width = px(2.); + // Bars thicken with the waveform so a tall one does not read as hairlines. + let bar = (height * 0.125).round().clamp(px(2.), px(4.)); let state = self.state.read(cx); - let color = if state.status().is_capturing() { + let capturing = state.status().is_capturing(); + let levels: Vec = state.levels().collect(); + let animate = capturing && !cx.reduce_motion(); + // How far the bars have scrolled toward the next level, so the motion + // stays smooth when levels arrive slower than the display refreshes. + let scroll = state + .last_level_at() + .filter(|_| animate) + .map(|at| (at.elapsed().as_secs_f32() / LEVEL_INTERVAL.as_secs_f32()).min(1.)) + .unwrap_or(0.); + if animate { + window.request_animation_frame(); + } + let color = if capturing { cx.theme().primary } else { cx.theme().muted_foreground }; - let levels = state.levels(); - // Right-align the history: pad the leading bars when fewer levels have - // arrived than there are bars. - let skip = levels.len().saturating_sub(self.bars); - let padding = self.bars.saturating_sub(levels.len()); - let levels = std::iter::repeat_n(0., padding).chain(levels.skip(skip)); - - h_flex() + + div() .h(height) - .gap(bar_width) - .items_center() - .children(levels.map(|level| { - div() - .w(bar_width) - .h((height * level).max(bar_width)) - .rounded_full() - .bg(color) - })) + .w(DEFAULT_WIDTH) + .flex_shrink_0() + .refine_style(&self.style) + .child( + canvas( + |_, _, _| {}, + move |bounds, _, window, _| { + for rect in bar_rects(bounds, &levels, scroll, bar, bar) { + window.paint_quad(fill(rect, color).corner_radii(bar / 2.)); + } + }, + ) + .size_full(), + ) + } +} + +/// The bars for `levels` (oldest first) in `bounds`: the newest at the trailing +/// edge, `scroll` of a step further toward the leading edge, each centered on +/// the midline and at least as tall as it is wide. +fn bar_rects( + bounds: Bounds, + levels: &[f32], + scroll: f32, + bar: Pixels, + gap: Pixels, +) -> Vec> { + let pitch = bar + gap; + let height = bounds.size.height; + let middle = bounds.origin.y + height / 2.; + // One more bar than fits, to scroll in from the trailing edge. + let count = ((bounds.size.width + gap) / pitch).floor().max(0.) as usize + 1; + (0..count) + .filter_map(|from_end| { + let right = bounds.right() - pitch * (from_end as f32 + scroll); + let left = right - bar; + if left < bounds.left() - px(0.5) { + return None; + } + let level = levels + .len() + .checked_sub(from_end + 1) + .map_or(0., |ix| levels[ix]); + let bar_height = (height * level.clamp(0., 1.)).max(bar); + Some(Bounds::new( + point(left, middle - bar_height / 2.), + size(bar, bar_height), + )) + }) + .collect() +} + +#[cfg(test)] +mod tests { + use super::*; + + fn bounds(width: f32, height: f32) -> Bounds { + Bounds::new(point(px(10.), px(20.)), size(px(width), px(height))) + } + + #[test] + fn bars_fill_the_width_with_the_newest_at_the_trailing_edge() { + let bounds = bounds(98., 20.); + let rects = bar_rects(bounds, &[0.2, 0.9], 0., px(2.), px(2.)); + // 98 px at a 4 px pitch fits 25 bars. + assert_eq!(rects.len(), 25); + assert_eq!(rects[0].right(), bounds.right()); + assert_eq!(rects[0].size.height, px(18.)); + assert_eq!(rects[1].size.height, px(4.)); + } + + #[test] + fn bars_grow_from_the_midline_and_rest_as_a_baseline() { + let bounds = bounds(40., 20.); + for rect in bar_rects(bounds, &[0.5], 0., px(2.), px(2.)) { + let middle = rect.origin.y + rect.size.height / 2.; + assert_eq!(middle, px(30.)); + assert!(rect.size.height >= px(2.)); + } + } + + #[test] + fn scrolling_moves_bars_toward_the_leading_edge() { + let bounds = bounds(40., 20.); + let still = bar_rects(bounds, &[1.], 0., px(2.), px(2.)); + let moving = bar_rects(bounds, &[1.], 0.5, px(2.), px(2.)); + assert_eq!(still[0].left() - moving[0].left(), px(2.)); + assert!( + moving + .iter() + .all(|rect| rect.left() >= bounds.left() - px(0.5)) + ); } } diff --git a/crates/story/src/stories/speech_story.rs b/crates/story/src/stories/speech_story.rs index a3e25ed0a0..19fad53ff2 100644 --- a/crates/story/src/stories/speech_story.rs +++ b/crates/story/src/stories/speech_story.rs @@ -1,6 +1,6 @@ use gpui_kit::{ App, AppContext, Context, Entity, FocusHandle, Focusable, IntoElement, ParentElement, Render, - Styled, Subscription, Window, div, prelude::FluentBuilder as _, + Styled, Subscription, Window, div, prelude::FluentBuilder as _, px, }; use gpui_kit::component::{ @@ -180,7 +180,7 @@ impl Dictation { h_flex() .gap_2() .when(status.is_capturing(), |this| { - this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + this.child(SpeechWaveform::new(&self.speech).w(px(48.)).xsmall()) }) .child( SpeechButton::new(&self.speech) diff --git a/examples/speech/src/main.rs b/examples/speech/src/main.rs index f2ec95a2ce..2688de6c0f 100644 --- a/examples/speech/src/main.rs +++ b/examples/speech/src/main.rs @@ -559,7 +559,7 @@ impl SpeechApp { ), ) .when(status.is_active(), |this| { - this.child(SpeechWaveform::new(&self.speech).bars(28).small()) + this.child(SpeechWaveform::new(&self.speech).w(px(112.)).small()) }) .map(|this| { if status.is_active() { diff --git a/website/component/speech.md b/website/component/speech.md index 102a83cfff..ec571187b9 100644 --- a/website/component/speech.md +++ b/website/component/speech.md @@ -84,7 +84,7 @@ Input::new(&self.input).suffix( h_flex() .gap_2() .when(capturing, |this| { - this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + this.child(SpeechWaveform::new(&self.speech).w(px(48.)).xsmall()) }) .child(SpeechButton::new(&self.speech).xsmall()), ) @@ -328,8 +328,9 @@ let speech = cx.new(|cx| SpeechState::new(cx).recognizer(recognizer).input(Silen ``` `levels()` exposes the recent input levels that `SpeechWaveform` draws, in -`0.0..=1.0` with the oldest first, for an application that renders its own -meter. +`0.0..=1.0` with the oldest first, one per 25 ms of audio, for an application +that renders its own meter. Peaks rise at once and fall back smoothly, and +background noise reads as `0.0`. ## Platform support @@ -385,7 +386,7 @@ application recognizer; without one, `SpeechButton` renders nothing. The `has_recognizer`, `is_available`, `transcript`, `levels` - [SpeechEvent] and [SpeechStatus] - [SpeechButton] — `show_when_unsupported`, plus `Sizable` and `Disableable` -- [SpeechWaveform] — `bars` (default 24, at most 48), plus `Sizable` +- [SpeechWaveform] — a live waveform that fills its width (96 px unless styled) and scrolls while capturing; `Sizable` sets its height, `Styled` its width - [SpeechRecognizer], [RecognitionSession] and [SpeechSink] — the recognition seam - [AudioInput], [AudioSink] and [AudioFormat] — the audio seam diff --git a/website/zh-CN/component/speech.md b/website/zh-CN/component/speech.md index f6b3936c7b..fa481fbff7 100644 --- a/website/zh-CN/component/speech.md +++ b/website/zh-CN/component/speech.md @@ -71,7 +71,7 @@ Input::new(&self.input).suffix( h_flex() .gap_2() .when(capturing, |this| { - this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + this.child(SpeechWaveform::new(&self.speech).w(px(48.)).xsmall()) }) .child(SpeechButton::new(&self.speech).xsmall()), ) @@ -273,7 +273,7 @@ impl AudioInput for Silence { let speech = cx.new(|cx| SpeechState::new(cx).recognizer(recognizer).input(Silence)); ``` -`levels()` 返回 `SpeechWaveform` 绘制所用的近期输入音量,取值 `0.0..=1.0`,按时间从旧到新排列,适合需要自绘音量表的应用。 +`levels()` 返回 `SpeechWaveform` 绘制所用的近期输入音量,取值 `0.0..=1.0`,按时间从旧到新排列,每 25 ms 音频一个值,适合需要自绘音量表的应用。音量升高时立即跟上、回落时平滑下降,背景噪音记为 `0.0`。 ## 平台支持 @@ -314,7 +314,7 @@ Linux 没有系统识别器,只有在应用提供识别器时才能使用语 - [SpeechState]:会话本身,包括 `recognizer`、`input`、`system_fallback`、`stop_timeout`、`start`、`stop`、`cancel`、`toggle`、`status`、`has_recognizer`、`is_available`、`transcript`、`levels` - [SpeechEvent] 与 [SpeechStatus] - [SpeechButton]:`show_when_unsupported`,以及 `Sizable` 和 `Disableable` -- [SpeechWaveform]:`bars`(默认 24,最多 48),以及 `Sizable` +- [SpeechWaveform]:铺满自身宽度的实时波形(不设宽度时为 96 px),采集期间滚动;`Sizable` 决定高度,`Styled` 决定宽度 - [SpeechRecognizer]、[RecognitionSession] 与 [SpeechSink]:识别的扩展点 - [AudioInput]、[AudioSink] 与 [AudioFormat]:音频的扩展点 - [SpeechError] From b155cde5354604033b13c542335ad08957bd63db Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Fri, 2 Oct 2026 17:17:37 +0800 Subject: [PATCH 8/8] speech: Make the example's demo recognizer follow the voice The demo typed its script whenever audio arrived, so it kept typing in silence and ignored how the user spoke. It now types only while the input is loud enough to be speech and ends the sentence after a 0.8 s pause. Co-Authored-By: Claude Opus 5.5 --- examples/speech/README.md | 6 ++- examples/speech/src/demo.rs | 81 ++++++++++++++++++++++++++++++++----- examples/speech/src/main.rs | 4 +- 3 files changed, 76 insertions(+), 15 deletions(-) diff --git a/examples/speech/README.md b/examples/speech/README.md index de0886ff4e..4a779295c8 100644 --- a/examples/speech/README.md +++ b/examples/speech/README.md @@ -23,8 +23,10 @@ System recognizer, by language: ## Trying it -- **Demo** types a scripted passage while it hears audio. It needs no service, - network, or speech permission, so it works on every platform, Linux included. +- **Demo** types a scripted passage as you speak: tokens appear only while you + talk, and a pause ends the sentence. It recognizes no words and needs no + service, network, or speech permission, so it works on every platform, Linux + included. Use it to check the capture, waveform, and event flow. - **System** uses the operating system's recognizer in the language you pick. - Click the microphone or press ⇧⌘D (Ctrl+Shift+D on diff --git a/examples/speech/src/demo.rs b/examples/speech/src/demo.rs index 4ed1a58e79..66db2f70ff 100644 --- a/examples/speech/src/demo.rs +++ b/examples/speech/src/demo.rs @@ -1,13 +1,26 @@ //! A recognizer that needs no service, permission or network: it types a -//! scripted passage while audio arrives, so the whole flow can be tried -//! anywhere, including on Linux. +//! scripted passage as you speak, so the whole flow can be tried anywhere, +//! including on Linux. +//! +//! It does not recognize words, but it follows your voice: tokens appear only +//! while the input is loud enough to be speech, and a pause ends the sentence. use gpui_kit::App; use gpui_kit::component::speech::{RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; -/// Audio per recognized token: 0.25 s at 16 kHz. +/// Speech per recognized token: 0.25 s at 16 kHz. const SAMPLES_PER_TOKEN: usize = 4_000; +/// Audio is judged speech or silence in windows of 20 ms. +const WINDOW_SAMPLES: usize = 320; + +/// A pause this long ends the sentence: 0.8 s. +const PAUSE_SAMPLES: usize = 12_800; + +/// A window counts as speech above about -40 dBFS, well over a quiet room's +/// noise floor and well under normal speech. +const VOICE_RMS: f64 = 330.; + pub struct DemoRecognizer { script: &'static Script, } @@ -37,6 +50,7 @@ impl SpeechRecognizer for DemoRecognizer { sentence_ix: 0, tokens: 0, samples: 0, + silence: 0, })) } } @@ -99,8 +113,10 @@ struct DemoSession { sentence_ix: usize, /// Tokens of the current sentence heard so far. tokens: usize, - /// Samples received since the last token. + /// Speech received since the last token, in samples. samples: usize, + /// Silence since the last speech, in samples. + silence: usize, } impl DemoSession { @@ -141,10 +157,12 @@ impl DemoSession { } } + /// End the current sentence, if any of it was heard. fn commit(&mut self, cx: &mut App) { - if let Some(heard) = self.heard() { - self.sink.phrase(heard, cx); - } + let Some(heard) = self.heard() else { + return; + }; + self.sink.phrase(heard, cx); self.sentence_ix += 1; self.tokens = 0; } @@ -152,10 +170,21 @@ impl DemoSession { impl RecognitionSession for DemoSession { fn push_audio(&mut self, samples: &[i16], cx: &mut App) { - self.samples += samples.len(); - while self.samples >= SAMPLES_PER_TOKEN { - self.samples -= SAMPLES_PER_TOKEN; - self.next_token(cx); + for window in samples.chunks(WINDOW_SAMPLES) { + if is_speech(window) { + self.silence = 0; + self.samples += window.len(); + while self.samples >= SAMPLES_PER_TOKEN { + self.samples -= SAMPLES_PER_TOKEN; + self.next_token(cx); + } + } else { + self.silence += window.len(); + if self.silence >= PAUSE_SAMPLES && self.tokens > 0 { + self.samples = 0; + self.commit(cx); + } + } } } @@ -164,3 +193,33 @@ impl RecognitionSession for DemoSession { self.sink.finish(cx); } } + +/// Whether `window` is loud enough to be speech. +fn is_speech(window: &[i16]) -> bool { + if window.is_empty() { + return false; + } + let energy: f64 = window.iter().map(|&s| f64::from(s) * f64::from(s)).sum(); + (energy / window.len() as f64).sqrt() > VOICE_RMS +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn silence_and_room_noise_are_not_speech() { + assert!(!is_speech(&[])); + assert!(!is_speech(&[0; WINDOW_SAMPLES])); + assert!(!is_speech(&[60; WINDOW_SAMPLES])); + } + + #[test] + fn a_voice_is_speech() { + // About -20 dBFS, ordinary speech into a laptop microphone. + let voice: Vec = (0..WINDOW_SAMPLES) + .map(|ix| if ix % 2 == 0 { 3_300 } else { -3_300 }) + .collect(); + assert!(is_speech(&voice)); + } +} diff --git a/examples/speech/src/main.rs b/examples/speech/src/main.rs index 2688de6c0f..85a81f2a5e 100644 --- a/examples/speech/src/main.rs +++ b/examples/speech/src/main.rs @@ -339,8 +339,8 @@ impl SpeechApp { on Windows it uses Microsoft’s online service." } Engine::Demo => { - "Types a scripted passage while it hears audio. Needs no service, network or \ - speech permission." + "Types a scripted passage as you speak and ends a sentence when you pause. \ + Needs no service, network or speech permission." } };