From 94847a4ae23ad5e19aa004c343731b3bf470e65e Mon Sep 17 00:00:00 2001 From: Floyd Wang Date: Wed, 30 Sep 2026 18:35:56 +0800 Subject: [PATCH] speech: Add `SpeechState`, `SpeechButton` and system recognizers for dictation Add a `speech` module to gpui-component: a session state machine that feeds captured audio to a pluggable `SpeechRecognizer`, a `SpeechButton` and a `SpeechWaveform` to render it, and two seams, `SpeechRecognizer` and `AudioInput`, so applications can bring any speech service or audio source. With the new `speech` feature, a state without its own recognizer falls back to the platform's: `SFSpeechRecognizer` on macOS (on-device only) and `Windows.Media.SpeechRecognition` on Windows. Linux has no system recognizer and works with an application recognizer. Audio is captured with cpal. Also adds the gallery story, English and Chinese docs, and the `speech` example, a dictation notepad that reports what the machine supports. Co-Authored-By: Claude Opus 5.5 --- Cargo.lock | 207 ++++- Cargo.toml | 1 + crates/assets/default-icons.txt | 2 + crates/assets/tests/icons.rs | 4 +- crates/component/Cargo.toml | 40 + crates/component/locales/ui.yml | 16 + crates/component/src/lib.rs | 1 + crates/component/src/speech/button.rs | 126 +++ crates/component/src/speech/microphone.rs | 262 ++++++ crates/component/src/speech/mod.rs | 51 ++ crates/component/src/speech/recognizer.rs | 266 ++++++ crates/component/src/speech/state.rs | 689 ++++++++++++++ crates/component/src/speech/system/macos.rs | 486 ++++++++++ crates/component/src/speech/system/mod.rs | 86 ++ .../src/speech/system/unsupported.rs | 30 + crates/component/src/speech/system/winrt.rs | 395 ++++++++ crates/component/src/speech/waveform.rs | 83 ++ crates/kit/Cargo.toml | 2 + crates/story/Cargo.toml | 2 +- crates/story/src/gallery.rs | 1 + crates/story/src/stories/mod.rs | 2 + crates/story/src/stories/speech_story.rs | 281 ++++++ examples/speech/Cargo.toml | 14 + examples/speech/Info.plist | 10 + examples/speech/README.md | 62 ++ examples/speech/build.rs | 21 + examples/speech/src/demo.rs | 166 ++++ examples/speech/src/main.rs | 845 ++++++++++++++++++ script/install-linux.sh | 2 +- skills/gpui-kit/SKILL.md | 1 + website/component/index.md | 1 + website/component/speech.md | 408 +++++++++ website/docs/installation.md | 4 +- website/zh-CN/component/index.md | 1 + website/zh-CN/component/speech.md | 336 +++++++ website/zh-CN/docs/installation.md | 4 +- 36 files changed, 4896 insertions(+), 12 deletions(-) create mode 100644 crates/component/src/speech/button.rs create mode 100644 crates/component/src/speech/microphone.rs create mode 100644 crates/component/src/speech/mod.rs create mode 100644 crates/component/src/speech/recognizer.rs create mode 100644 crates/component/src/speech/state.rs create mode 100644 crates/component/src/speech/system/macos.rs create mode 100644 crates/component/src/speech/system/mod.rs create mode 100644 crates/component/src/speech/system/unsupported.rs create mode 100644 crates/component/src/speech/system/winrt.rs create mode 100644 crates/component/src/speech/waveform.rs create mode 100644 crates/story/src/stories/speech_story.rs create mode 100644 examples/speech/Cargo.toml create mode 100644 examples/speech/Info.plist create mode 100644 examples/speech/README.md create mode 100644 examples/speech/build.rs create mode 100644 examples/speech/src/demo.rs create mode 100644 examples/speech/src/main.rs create mode 100644 website/component/speech.md create mode 100644 website/zh-CN/component/speech.md diff --git a/Cargo.lock b/Cargo.lock index 109e2e967d..9877bcc1d0 100644 --- a/Cargo.lock +++ b/Cargo.lock @@ -224,6 +224,28 @@ version = "0.2.21" source = "registry+https://github.com/rust-lang/crates.io-index" checksum = "683d7910e743518b0e34f1186f92494becacb047c7b6bf616c96772180fef923" +[[package]] +name = "alsa" +version = "0.9.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "ed7572b7ba83a31e20d1b48970ee402d2e3e0537dcfe0a3ff4d6eb7508617d43" +dependencies = [ + "alsa-sys", + "bitflags 2.13.1", + "cfg-if", + "libc", +] + +[[package]] +name = "alsa-sys" +version = "0.3.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "db8fee663d06c4e303404ef5f40488a53e062f89ba8bfed81f42325aafad1527" +dependencies = [ + "libc", + "pkg-config", +] + [[package]] name = "ambient-authority" version = "0.0.2" @@ -1731,6 +1753,26 @@ dependencies = [ "libm", ] +[[package]] +name = "coreaudio-rs" +version = "0.11.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "321077172d79c662f64f5071a03120748d5bb652f5231570141be24cfcd2bace" +dependencies = [ + "bitflags 1.3.2", + "core-foundation-sys 0.8.7", + "coreaudio-sys", +] + +[[package]] +name = "coreaudio-sys" +version = "0.2.18" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "b9b4739a805a62757a83e5654fa3faabec0442666b263bb2287d5a8185bfd953" +dependencies = [ + "bindgen", +] + [[package]] name = "cosmic-text" version = "0.19.0" @@ -1755,6 +1797,29 @@ dependencies = [ "unicode-segmentation", ] +[[package]] +name = "cpal" +version = "0.15.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "873dab07c8f743075e57f524c583985fbaf745602acbe916a01539364369a779" +dependencies = [ + "alsa", + "core-foundation-sys 0.8.7", + "coreaudio-rs", + "dasp_sample", + "jni 0.21.1", + "js-sys", + "libc", + "mach2 0.4.3", + "ndk 0.8.0", + "ndk-context", + "oboe", + "wasm-bindgen", + "wasm-bindgen-futures", + "web-sys", + "windows 0.54.0", +] + [[package]] name = "cpubits" version = "0.1.1" @@ -2258,6 +2323,12 @@ dependencies = [ "parking_lot_core", ] +[[package]] +name = "dasp_sample" +version = "0.11.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "0c87e182de0887fd5361989c677c4e8f5000cd9491d6d563161a8f3a5519fc7f" + [[package]] name = "data-encoding" version = "2.11.1" @@ -3824,8 +3895,10 @@ name = "gpui-component" version = "0.7.0" dependencies = [ "anyhow", + "block2 0.6.2", "chrono", "core-text", + "cpal", "enum-iterator", "gpui-base", "gpui-component-macros", @@ -3843,7 +3916,10 @@ dependencies = [ "num-traits", "objc2 0.6.4", "objc2-app-kit 0.3.2", + "objc2-av-foundation", + "objc2-avf-audio", "objc2-foundation 0.3.2", + "objc2-speech", "once_cell", "paste", "raw-window-handle", @@ -4254,7 +4330,7 @@ dependencies = [ "itertools 0.14.0", "libc", "log", - "mach2", + "mach2 0.5.0", "objc", "objc2 0.6.4", "objc2-app-kit 0.3.2", @@ -5661,7 +5737,7 @@ dependencies = [ "jni 0.21.1", "kuchikiki", "libc", - "ndk", + "ndk 0.9.0", "objc2 0.6.4", "objc2-app-kit 0.3.2", "objc2-core-foundation", @@ -6162,6 +6238,15 @@ dependencies = [ "libc", ] +[[package]] +name = "mach2" +version = "0.4.3" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "d640282b302c0bb0a2a8e0233ead9035e3bed871f0b7e81fe4a1ec829765db44" +dependencies = [ + "libc", +] + [[package]] name = "mach2" version = "0.5.0" @@ -6437,6 +6522,20 @@ dependencies = [ "getrandom 0.2.17", ] +[[package]] +name = "ndk" +version = "0.8.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "2076a31b7010b17a38c01907c45b945e8f11495ee4dd588309718901b1f7a5b7" +dependencies = [ + "bitflags 2.13.1", + "jni-sys 0.3.1", + "log", + "ndk-sys 0.5.0+25.2.9519653", + "num_enum", + "thiserror 1.0.69", +] + [[package]] name = "ndk" version = "0.9.0" @@ -6446,12 +6545,27 @@ dependencies = [ "bitflags 2.13.1", "jni-sys 0.3.1", "log", - "ndk-sys", + "ndk-sys 0.6.0+11769913", "num_enum", "raw-window-handle", "thiserror 1.0.69", ] +[[package]] +name = "ndk-context" +version = "0.1.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "27b02d87554356db9e9a873add8782d4ea6e3e58ea071a9adb9a2e8ddb884a8b" + +[[package]] +name = "ndk-sys" +version = "0.5.0+25.2.9519653" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "8c196769dd60fd4f363e11d948139556a344e79d451aeb2fa2fd040738ef7691" +dependencies = [ + "jni-sys 0.3.1", +] + [[package]] name = "ndk-sys" version = "0.6.0+11769913" @@ -6817,6 +6931,27 @@ dependencies = [ "objc2-quartz-core 0.3.2", ] +[[package]] +name = "objc2-av-foundation" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "478ae33fcac9df0a18db8302387c666b8ef08a3e2d62b510ca4fc278a384b6c0" +dependencies = [ + "bitflags 2.13.1", + "objc2 0.6.4", + "objc2-foundation 0.3.2", +] + +[[package]] +name = "objc2-avf-audio" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "13a380031deed8e99db00065c45937da434ca987c034e13b87e4441f9e4090be" +dependencies = [ + "objc2 0.6.4", + "objc2-foundation 0.3.2", +] + [[package]] name = "objc2-cloud-kit" version = "0.3.2" @@ -7094,6 +7229,18 @@ dependencies = [ "objc2-foundation 0.3.2", ] +[[package]] +name = "objc2-speech" +version = "0.3.2" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "f9eb609f9d2a25e0f8b76954f3acaa58348ce2a91285fe867643b2224ac4d263" +dependencies = [ + "block2 0.6.2", + "objc2 0.6.4", + "objc2-avf-audio", + "objc2-foundation 0.3.2", +] + [[package]] name = "objc2-ui-kit" version = "0.3.2" @@ -7172,6 +7319,29 @@ dependencies = [ "memchr", ] +[[package]] +name = "oboe" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "e8b61bebd49e5d43f5f8cc7ee2891c16e0f41ec7954d36bcb6c14c5e0de867fb" +dependencies = [ + "jni 0.21.1", + "ndk 0.8.0", + "ndk-context", + "num-derive", + "num-traits", + "oboe-sys", +] + +[[package]] +name = "oboe-sys" +version = "0.6.1" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "6c8bb09a4a2b1d668170cfe0a7d5bc103f8999fb316c98099b6a9939c9f2e79d" +dependencies = [ + "cc", +] + [[package]] name = "once_cell" version = "1.21.4" @@ -9759,6 +9929,15 @@ dependencies = [ "system-deps", ] +[[package]] +name = "speech" +version = "0.7.0" +dependencies = [ + "anyhow", + "cpal", + "gpui-kit", +] + [[package]] name = "spin" version = "0.9.9" @@ -12040,7 +12219,7 @@ dependencies = [ "libloading", "log", "naga", - "ndk-sys", + "ndk-sys 0.6.0+11769913", "objc2 0.6.4", "objc2-core-foundation", "objc2-foundation 0.3.2", @@ -12140,6 +12319,16 @@ dependencies = [ "gpui-kit", ] +[[package]] +name = "windows" +version = "0.54.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "9252e5725dbed82865af151df558e754e4a3c2c30818359eb17465f1346a1b49" +dependencies = [ + "windows-core 0.54.0", + "windows-targets 0.52.6", +] + [[package]] name = "windows" version = "0.57.0" @@ -12216,6 +12405,16 @@ dependencies = [ "windows-core 0.62.2", ] +[[package]] +name = "windows-core" +version = "0.54.0" +source = "registry+https://github.com/rust-lang/crates.io-index" +checksum = "12661b9c89351d684a50a8a643ce5f608e20243b9fb84687800163429f161d65" +dependencies = [ + "windows-result 0.1.2", + "windows-targets 0.52.6", +] + [[package]] name = "windows-core" version = "0.57.0" diff --git a/Cargo.toml b/Cargo.toml index 6a8e3948dd..7070488588 100644 --- a/Cargo.toml +++ b/Cargo.toml @@ -36,6 +36,7 @@ members = [ "examples/table_in_scrollable", "examples/markdown_table", "examples/ai_recipes", + "examples/speech", "crates/shell", "crates/shell/rquickjs-compat", "crates/component-shell", diff --git a/crates/assets/default-icons.txt b/crates/assets/default-icons.txt index 044531c80f..7119c5d985 100644 --- a/crates/assets/default-icons.txt +++ b/crates/assets/default-icons.txt @@ -60,6 +60,7 @@ icons/map.svg icons/maximize.svg icons/memory-stick.svg icons/menu.svg +icons/mic.svg icons/minimize.svg icons/minus.svg icons/moon.svg @@ -88,6 +89,7 @@ icons/settings.svg icons/sort-ascending.svg icons/sort-descending.svg icons/square-terminal.svg +icons/square.svg icons/star-fill.svg icons/star-off.svg icons/star.svg diff --git a/crates/assets/tests/icons.rs b/crates/assets/tests/icons.rs index d51abd4f06..352f0c2263 100644 --- a/crates/assets/tests/icons.rs +++ b/crates/assets/tests/icons.rs @@ -62,8 +62,8 @@ fn default_assets_preserve_the_component_bundle_without_all_lucide_icons() { .collect(); let actual: BTreeSet<_> = Assets.list("icons/").unwrap().into_iter().collect(); assert_eq!(actual, expected); - assert_eq!(actual.len(), 104); - assert_eq!(Assets::iter().count(), 104); + assert_eq!(actual.len(), 106); + assert_eq!(Assets::iter().count(), 106); assert!(Assets::get("icons/search.svg").is_some()); assert!(Assets::get("icons/accessibility.svg").is_none()); for path in actual { diff --git a/crates/component/Cargo.toml b/crates/component/Cargo.toml index 545485ba43..a03b97d1fd 100644 --- a/crates/component/Cargo.toml +++ b/crates/component/Cargo.toml @@ -19,6 +19,21 @@ test-support = ["gpui/test-support", "gpui-base/test-support"] decimal = ["gpui-base/decimal"] inspector = ["gpui_macros/inspector", "gpui/inspector", "gpui-base/inspector"] tree-sitter = ["dep:tree-sitter", "dep:tree-sitter-json"] +# Speech input: microphone capture and the system speech recognizer. +speech = [ + "dep:cpal", + "dep:block2", + "dep:objc2-speech", + "dep:objc2-avf-audio", + "dep:objc2-av-foundation", + "objc2-foundation/NSBundle", + "objc2-foundation/NSDictionary", + "objc2-foundation/NSError", + "objc2-foundation/NSLocale", + "windows/Foundation_Collections", + "windows/Globalization", + "windows/Media_SpeechRecognition", +] # For syntax highlighting in Markdown and CodeEditor. tree-sitter-languages = [ @@ -137,6 +152,7 @@ instant.workspace = true # Native-only dependencies (not available on WASM) [target.'cfg(not(target_family = "wasm"))'.dependencies] smol.workspace = true +cpal = { version = "0.15.3", optional = true } tree-sitter = { version = "0.26.13", optional = true } tree-sitter-astro-next = { version="0.1.1", optional = true } tree-sitter-bash = { version = "0.23.3", optional = true } @@ -188,6 +204,30 @@ objc2-app-kit = { version = "0.3", features = [ "NSImage", ] } objc2-foundation = { version = "0.3", features = ["NSData", "NSString", "NSGeometry"] } +# Speech input (SystemRecognizer) — SFSpeechRecognizer via objc2. +block2 = { version = "0.6", optional = true } +objc2-av-foundation = { version = "0.3", default-features = false, features = [ + "std", + "AVCaptureDevice", + "AVMediaFormat", +], optional = true } +objc2-avf-audio = { version = "0.3", default-features = false, features = [ + "std", + "AVAudioBuffer", + "AVAudioFormat", + "AVAudioTypes", +], optional = true } +objc2-speech = { version = "0.3", default-features = false, features = [ + "std", + "block2", + "objc2-avf-audio", + "SFSpeechRecognitionMetadata", + "SFSpeechRecognitionRequest", + "SFSpeechRecognitionResult", + "SFSpeechRecognitionTask", + "SFSpeechRecognizer", + "SFTranscription", +], optional = true } [target.'cfg(target_os = "windows")'.dependencies] # Native menu (NativeMenu) — drives Win32 popup menus. diff --git a/crates/component/locales/ui.yml b/crates/component/locales/ui.yml index 2016977c35..8f9f999e25 100644 --- a/crates/component/locales/ui.yml +++ b/crates/component/locales/ui.yml @@ -432,3 +432,19 @@ Questionnaire: zh-CN: 请选择一个答案,或跳过此题。 zh-HK: 請選擇一個答案,或跳過此題。 zh-TW: 請選擇一個答案,或跳過此題。 +Speech: + Start: + en: Dictate + zh-CN: 语音输入 + zh-HK: 語音輸入 + zh-TW: 語音輸入 + Stop: + en: Stop dictation + zh-CN: 停止语音输入 + zh-HK: 停止語音輸入 + zh-TW: 停止語音輸入 + Unavailable: + en: Dictation unavailable + zh-CN: 语音输入不可用 + zh-HK: 語音輸入不可用 + zh-TW: 語音輸入不可用 diff --git a/crates/component/src/lib.rs b/crates/component/src/lib.rs index 3b2f2b2a69..5ebfcf0570 100644 --- a/crates/component/src/lib.rs +++ b/crates/component/src/lib.rs @@ -77,6 +77,7 @@ pub mod shimmer; pub mod sidebar; pub mod skeleton; pub mod slider; +pub mod speech; pub mod spinner; pub mod status_bar; pub mod stepper; diff --git a/crates/component/src/speech/button.rs b/crates/component/src/speech/button.rs new file mode 100644 index 0000000000..be2f337978 --- /dev/null +++ b/crates/component/src/speech/button.rs @@ -0,0 +1,126 @@ +use gpui::{App, ElementId, Entity, IntoElement, RenderOnce, Window, div}; +use rust_i18n::t; + +use crate::{ + Disableable, IconName, Selectable as _, Sizable, Size, + button::{Button, ButtonVariants as _}, +}; + +use super::{SpeechState, SpeechStatus}; + +/// The button that starts and stops a [`SpeechState`]'s session. +/// +/// Shows a microphone at rest and a stop glyph, pressed, while capturing; a +/// click toggles the session. While the final result is pending it shows a +/// spinner and ignores clicks. +/// +/// Renders nothing when the state has no recognizer on this platform, unless +/// [`Self::show_when_unsupported`] is set, and renders disabled while the +/// recognizer reports itself unavailable. +#[derive(IntoElement)] +pub struct SpeechButton { + id: ElementId, + state: Entity, + size: Size, + disabled: bool, + show_when_unsupported: bool, +} + +impl SpeechButton { + /// A button for `state`. + pub fn new(state: &Entity) -> Self { + Self { + id: ("speech-button", state.entity_id()).into(), + state: state.clone(), + size: Size::default(), + disabled: false, + show_when_unsupported: false, + } + } + + /// Render a disabled button, instead of nothing, when the state has no + /// recognizer on this platform. Default `false`. + pub fn show_when_unsupported(mut self, show: bool) -> Self { + self.show_when_unsupported = show; + self + } +} + +impl Sizable for SpeechButton { + fn with_size(mut self, size: impl Into) -> Self { + self.size = size.into(); + self + } +} + +impl Disableable for SpeechButton { + fn disabled(mut self, disabled: bool) -> Self { + self.disabled = disabled; + self + } +} + +impl RenderOnce for SpeechButton { + fn render(self, _: &mut Window, cx: &mut App) -> impl IntoElement { + let state = self.state.read(cx); + let status = state.status(); + let supported = state.has_recognizer(); + if !supported && !self.show_when_unsupported { + return div().into_any_element(); + } + // A running session stays stoppable even if the recognizer turns + // unavailable mid-way. + let available = status.is_active() || state.is_available(cx); + let capturing = status.is_capturing(); + let label = if !available { + t!("Speech.Unavailable") + } else if capturing { + t!("Speech.Stop") + } else { + t!("Speech.Start") + }; + + Button::new(self.id) + .ghost() + .with_size(self.size) + .icon(if capturing { + IconName::Square + } else { + IconName::Mic + }) + .selected(capturing) + .loading(status == SpeechStatus::Stopping) + .disabled(self.disabled || !available) + .tooltip(label.clone()) + .accessibility_label(label) + .on_click({ + let state = self.state.clone(); + move |_, _, cx| state.update(cx, |state, cx| state.toggle(cx)) + }) + .into_any_element() + } +} + +#[cfg(test)] +mod tests { + use gpui::{AppContext as _, TestAppContext}; + + use super::*; + + #[gpui::test] + fn test_speech_button_builder(cx: &mut TestAppContext) { + let state = cx.update(|cx| cx.new(|cx| SpeechState::new(cx).system_fallback(false))); + let button = SpeechButton::new(&state) + .small() + .disabled(true) + .show_when_unsupported(true); + + assert_eq!(button.size, Size::Small); + assert!(button.disabled); + assert!(button.show_when_unsupported); + assert_eq!( + button.id, + ElementId::from(("speech-button", state.entity_id())) + ); + } +} diff --git a/crates/component/src/speech/microphone.rs b/crates/component/src/speech/microphone.rs new file mode 100644 index 0000000000..fc4e941f61 --- /dev/null +++ b/crates/component/src/speech/microphone.rs @@ -0,0 +1,262 @@ +use std::f32::consts::TAU; + +use anyhow::anyhow; +use cpal::{ + FromSample, Sample, SampleFormat, SizedSample, StreamConfig, + traits::{DeviceTrait as _, HostTrait as _, StreamTrait as _}, +}; +use gpui::{App, Subscription}; +use smol::channel::{Sender, unbounded}; + +use super::{AudioFormat, AudioInput, AudioSink, SpeechError}; + +/// The default audio input device, captured through the platform's audio API +/// (Core Audio, WASAPI or ALSA). +/// +/// The device's own format is mixed down to mono and resampled to the format +/// the recognizer asks for. +/// +/// On macOS the application's `Info.plist` must describe why it uses the +/// microphone (`NSMicrophoneUsageDescription`), or the system refuses access +/// without asking. +#[derive(Debug, Default, Clone)] +pub struct Microphone { + _private: (), +} + +impl AudioInput for Microphone { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + #[cfg(target_os = "macos")] + if super::system::macos::is_microphone_denied() { + return Err(SpeechError::PermissionDenied); + } + + let device = cpal::default_host() + .default_input_device() + .ok_or(SpeechError::NoInputDevice)?; + let supported = device.default_input_config().map_err(SpeechError::input)?; + let sample_format = supported.sample_format(); + let config: StreamConfig = supported.into(); + let source_rate = config.sample_rate.0; + + let (tx, rx) = unbounded(); + let stream = match sample_format { + SampleFormat::F32 => build_stream::(&device, &config, tx), + SampleFormat::I16 => build_stream::(&device, &config, tx), + SampleFormat::U16 => build_stream::(&device, &config, tx), + SampleFormat::I32 => build_stream::(&device, &config, tx), + other => { + return Err(SpeechError::input(anyhow!( + "unsupported sample format {other}" + ))); + } + }?; + stream.play().map_err(SpeechError::input)?; + + let task = cx.spawn(async move |cx| { + let mut converter = Converter::new(source_rate, format); + let mut mono = Vec::new(); + while let Ok(first) = rx.recv().await { + // The device delivers ~10 ms chunks; forward what has queued up + // as one push so the state updates once per batch. + let mut error = None; + for capture in + std::iter::once(first).chain(std::iter::from_fn(|| rx.try_recv().ok())) + { + match capture { + Capture::Samples(samples) => mono.extend_from_slice(&samples), + Capture::Error(message) => error = Some(message), + } + } + + let samples = converter.convert(&mono); + mono.clear(); + cx.update(|cx| { + if !samples.is_empty() { + sink.push(samples, cx); + } + if let Some(message) = error.take() { + sink.error(SpeechError::input(anyhow!(message)), cx); + } + }); + } + }); + + Ok(Subscription::new(move || { + drop(stream); + drop(task); + })) + } +} + +enum Capture { + Samples(Vec), + Error(String), +} + +/// Open an input stream that sends each callback's frames, mixed down to mono, +/// through `tx`. Runs on the audio thread, so it only converts and sends. +fn build_stream( + device: &cpal::Device, + config: &StreamConfig, + tx: Sender, +) -> Result +where + T: SizedSample, + f32: FromSample, +{ + let channels = config.channels.max(1) as usize; + let error_tx = tx.clone(); + device + .build_input_stream( + config, + move |data: &[T], _| { + let mono = data + .chunks(channels) + .map(|frame| { + frame + .iter() + .map(|&sample| f32::from_sample(sample)) + .sum::() + / frame.len() as f32 + }) + .collect(); + _ = tx.try_send(Capture::Samples(mono)); + }, + move |error| { + _ = error_tx.try_send(Capture::Error(error.to_string())); + }, + None, + ) + .map_err(|error| match error { + cpal::BuildStreamError::DeviceNotAvailable => SpeechError::NoInputDevice, + error => SpeechError::input(error), + }) +} + +/// Converts mono `f32` audio at the device rate to interleaved `i16` audio in +/// the recognizer's format, carrying state across chunks. +struct Converter { + /// Source samples per output sample. + step: f64, + /// Position of the next output sample, in source samples, where `0.0` is + /// the last sample of the previous chunk. + position: f64, + previous: f32, + low_pass: Option, + channels: usize, +} + +impl Converter { + fn new(source_rate: u32, format: AudioFormat) -> Self { + let target_rate = format.sample_rate().max(1); + let source_rate = source_rate.max(1); + Self { + step: source_rate as f64 / target_rate as f64, + position: 1., + previous: 0., + // Downsampling folds everything above the new Nyquist rate back + // into the band speech lives in; filter it out first. + low_pass: (source_rate > target_rate) + .then(|| LowPass::new(target_rate as f32 * 0.45, source_rate as f32)), + channels: format.channels().max(1) as usize, + } + } + + fn convert(&mut self, input: &[f32]) -> Vec { + if input.is_empty() { + return Vec::new(); + } + let filtered; + let input = match self.low_pass.as_mut() { + Some(low_pass) => { + filtered = input + .iter() + .map(|&x| low_pass.process(x)) + .collect::>(); + &filtered[..] + } + None => input, + }; + + let sample = |ix: usize| { + if ix == 0 { + self.previous + } else { + input[ix - 1] + } + }; + let last = input.len(); + let len = last as f64; + let mut output = Vec::with_capacity(((len / self.step) as usize + 1) * self.channels); + while self.position <= len { + let ix = self.position.floor() as usize; + let frac = (self.position - ix as f64) as f32; + let next = sample((ix + 1).min(last)); + let value = sample(ix) * (1. - frac) + next * frac; + let value = (value.clamp(-1., 1.) * i16::MAX as f32) as i16; + output.extend(std::iter::repeat_n(value, self.channels)); + self.position += self.step; + } + self.position -= len; + self.previous = input[input.len() - 1]; + output + } +} + +/// Two cascaded one-pole low-pass filters, 12 dB per octave. +struct LowPass { + alpha: f32, + stages: [f32; 2], +} + +impl LowPass { + fn new(cutoff: f32, sample_rate: f32) -> Self { + Self { + alpha: 1. - (-TAU * cutoff / sample_rate).exp(), + stages: [0.; 2], + } + } + + fn process(&mut self, mut x: f32) -> f32 { + for stage in &mut self.stages { + *stage += self.alpha * (x - *stage); + x = *stage; + } + x + } +} + +#[cfg(test)] +mod tests { + use super::*; + + #[test] + fn converter_keeps_rate_and_duplicates_channels() { + let mut converter = Converter::new(16_000, AudioFormat::new(16_000, 2)); + let output = converter.convert(&[0., 0.5, -0.5]); + assert_eq!(output, vec![0, 0, 16383, 16383, -16383, -16383]); + } + + #[test] + fn converter_downsamples_across_chunks() { + let mut converter = Converter::new(48_000, AudioFormat::default()); + let total: usize = (0..10).map(|_| converter.convert(&[0.; 480]).len()).sum(); + // 100 ms at 48 kHz is 100 ms at 16 kHz, whatever the chunking. + assert_eq!(total, 1_600); + } + + #[test] + fn converter_upsamples() { + let mut converter = Converter::new(8_000, AudioFormat::default()); + let total: usize = (0..10).map(|_| converter.convert(&[0.; 80]).len()).sum(); + // The first output lands on the first input sample, so the half step + // before it is never produced. + assert_eq!(total, 1_599); + } +} diff --git a/crates/component/src/speech/mod.rs b/crates/component/src/speech/mod.rs new file mode 100644 index 0000000000..24f775b288 --- /dev/null +++ b/crates/component/src/speech/mod.rs @@ -0,0 +1,51 @@ +//! Speech input: capture audio, recognize it and hand the text to the caller. +//! +//! [`SpeechState`] owns a session and [`SpeechButton`] and [`SpeechWaveform`] +//! render it. Recognition and audio capture are both replaceable: +//! implement [`SpeechRecognizer`] to use any speech service and [`AudioInput`] +//! to feed audio from anywhere. +//! +//! With the `speech` feature, a state without its own recognizer falls back to +//! the `SystemRecognizer` of macOS or Windows, and captures from the +//! `Microphone`. Linux has no system recognizer, so speech input works there +//! only with an application recognizer. + +mod button; +mod recognizer; +mod state; +mod waveform; + +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +mod microphone; +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +mod system; + +use std::rc::Rc; + +pub use button::SpeechButton; +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +pub use microphone::Microphone; +pub use recognizer::{ + AudioFormat, AudioInput, AudioSink, RecognitionSession, SpeechError, SpeechRecognizer, + SpeechSink, +}; +pub use state::{SpeechEvent, SpeechState, SpeechStatus}; +#[cfg(all(feature = "speech", not(target_family = "wasm")))] +pub use system::SystemRecognizer; +pub use waveform::SpeechWaveform; + +/// The input a [`SpeechState`] captures from unless told otherwise. +fn default_input() -> Option> { + #[cfg(all(feature = "speech", not(target_family = "wasm")))] + return Some(Rc::new(Microphone::default())); + #[allow(unreachable_code)] + None +} + +/// The recognizer a [`SpeechState`] falls back to, if this platform has one. +fn system_recognizer() -> Option> { + #[cfg(all(feature = "speech", any(target_os = "macos", target_os = "windows")))] + return Some(Rc::new(SystemRecognizer::new())); + #[allow(unreachable_code)] + None +} diff --git a/crates/component/src/speech/recognizer.rs b/crates/component/src/speech/recognizer.rs new file mode 100644 index 0000000000..f0456fd396 --- /dev/null +++ b/crates/component/src/speech/recognizer.rs @@ -0,0 +1,266 @@ +use std::{fmt, rc::Rc, sync::Arc}; + +use gpui::{App, Context, SharedString, Subscription, WeakEntity}; + +use super::{SpeechState, state::defer_session_update}; + +/// The PCM format a [`SpeechRecognizer`] consumes. +/// +/// Audio always arrives as interleaved signed 16-bit samples; an [`AudioInput`] +/// converts whatever its device produces to this rate and channel count. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)] +pub struct AudioFormat { + sample_rate: u32, + channels: u16, +} + +impl AudioFormat { + /// A format of `sample_rate` samples per second on each of `channels`. + pub fn new(sample_rate: u32, channels: u16) -> Self { + Self { + sample_rate, + channels, + } + } + + /// Samples per second of each channel. + pub fn sample_rate(&self) -> u32 { + self.sample_rate + } + + /// Number of interleaved channels. + pub fn channels(&self) -> u16 { + self.channels + } +} + +impl Default for AudioFormat { + /// 16 kHz mono, the format most speech services expect. + fn default() -> Self { + Self::new(16_000, 1) + } +} + +/// Why a speech session failed. +#[derive(Debug, Clone)] +pub enum SpeechError { + /// The user or the system denied access to the microphone. + PermissionDenied, + /// No audio input device is available. + NoInputDevice, + /// Speech input is not supported on this platform or build. + Unsupported, + /// The audio input failed or its device went away. + Input(Arc), + /// The recognizer failed, e.g. it could not reach its service. + Recognizer(Arc), +} + +impl SpeechError { + /// An [`SpeechError::Input`] from any error. + pub fn input(error: impl Into) -> Self { + Self::Input(Arc::new(error.into())) + } + + /// A [`SpeechError::Recognizer`] from any error. + pub fn recognizer(error: impl Into) -> Self { + Self::Recognizer(Arc::new(error.into())) + } +} + +impl fmt::Display for SpeechError { + fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result { + match self { + Self::PermissionDenied => f.write_str("microphone access was denied"), + Self::NoInputDevice => f.write_str("no audio input device is available"), + Self::Unsupported => f.write_str("speech input is not supported"), + Self::Input(error) => write!(f, "audio input failed: {error:#}"), + Self::Recognizer(error) => write!(f, "speech recognition failed: {error:#}"), + } + } +} + +impl std::error::Error for SpeechError {} + +/// Turns speech into text, e.g. by streaming audio to a cloud service. +/// +/// Implement it for the service the application uses and pass it to +/// [`SpeechState::recognizer`]. Without one, the state falls back to the +/// platform's own recognizer where there is one (see +/// `SystemRecognizer`). +/// +/// A session runs as follows: +/// +/// 1. [`start`](Self::start) opens a session. Connecting may take a while, so +/// return immediately and buffer the audio pushed in the meantime; call +/// [`SpeechSink::ready`] once the service accepts audio. +/// 2. [`RecognitionSession::push_audio`] delivers PCM in [`Self::audio_format`]. +/// Report results through [`SpeechSink::hypothesis`] and +/// [`SpeechSink::phrase`] as they arrive. +/// 3. [`RecognitionSession::finish`] means the user stopped talking: send the +/// remaining audio, wait for the last result, then call +/// [`SpeechSink::finish`]. +/// +/// Dropping the [`RecognitionSession`] cancels it: close the connection and +/// report nothing more. Every [`SpeechSink`] method may be called from any +/// point on the main thread, including from inside `start` or `push_audio`. +pub trait SpeechRecognizer: 'static { + /// The audio format this recognizer consumes, default 16 kHz mono. + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + /// Whether the recognizer can start a session now, default `true`. + /// + /// Return `false` while it cannot work, e.g. before the user signs in; the + /// [`SpeechButton`](super::SpeechButton) then renders disabled. Called on + /// every render, so keep it cheap. + fn is_available(&self, _cx: &App) -> bool { + true + } + + /// Open a session that reports its results to `sink`. + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError>; +} + +impl SpeechRecognizer for Rc { + fn audio_format(&self) -> AudioFormat { + (**self).audio_format() + } + + fn is_available(&self, cx: &App) -> bool { + (**self).is_available(cx) + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + (**self).start(sink, cx) + } +} + +/// One running recognition, opened by [`SpeechRecognizer::start`]. +/// +/// Dropping it cancels the session. +pub trait RecognitionSession: 'static { + /// Deliver interleaved PCM samples in the recognizer's [`AudioFormat`]. + fn push_audio(&mut self, samples: &[i16], cx: &mut App); + + /// No more audio will arrive. Flush what is buffered and call + /// [`SpeechSink::finish`] once the final result is in. + fn finish(&mut self, cx: &mut App); +} + +/// Where a [`SpeechRecognizer`] reports its session's progress. +/// +/// The sink is cheap to clone and may outlive its session: once the session is +/// stopped, cancelled or replaced, its calls are ignored. Calls are applied +/// after the current update, so they never re-enter the [`SpeechState`]. +#[derive(Clone)] +pub struct SpeechSink { + pub(super) state: WeakEntity, + pub(super) session: usize, +} + +impl SpeechSink { + /// The service is connected and consuming audio. + pub fn ready(&self, cx: &mut App) { + self.apply(cx, |state, cx| state.on_ready(cx)); + } + + /// Replace the hypothesis for the phrase being spoken. + pub fn hypothesis(&self, text: impl Into, cx: &mut App) { + let text = text.into(); + self.apply(cx, move |state, cx| state.on_hypothesis(text, cx)); + } + + /// Commit a recognized phrase and clear the hypothesis. + /// + /// Phrases are joined verbatim, so include any separator the language + /// needs, such as a leading space between English sentences. + pub fn phrase(&self, text: impl Into, cx: &mut App) { + let text = text.into(); + self.apply(cx, move |state, cx| state.on_phrase(text, cx)); + } + + /// The session is complete; no more results will follow. + pub fn finish(&self, cx: &mut App) { + self.apply(cx, |state, cx| state.on_finish(cx)); + } + + /// The session failed. + pub fn error(&self, error: SpeechError, cx: &mut App) { + self.apply(cx, move |state, cx| state.on_error(error, cx)); + } + + fn apply( + &self, + cx: &mut App, + f: impl FnOnce(&mut SpeechState, &mut Context) + 'static, + ) { + defer_session_update(self.state.clone(), self.session, cx, f); + } +} + +/// A source of audio for a [`SpeechState`], such as the microphone. +/// +/// Enable the `speech` feature for the built-in `Microphone`, +/// or implement this trait to feed audio from elsewhere, e.g. a file in tests. +pub trait AudioInput: 'static { + /// Start capturing `format` audio into `sink`. + /// + /// Capture runs until the returned [`Subscription`] is dropped. + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result; +} + +impl AudioInput for Rc { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + (**self).start(format, sink, cx) + } +} + +/// Where an [`AudioInput`] delivers captured audio. +/// +/// Like [`SpeechSink`], it is cheap to clone, ignored once its session ends and +/// applied after the current update. +#[derive(Clone)] +pub struct AudioSink { + pub(super) state: WeakEntity, + pub(super) session: usize, +} + +impl AudioSink { + /// Deliver interleaved PCM samples in the requested [`AudioFormat`]. + pub fn push(&self, samples: Vec, cx: &mut App) { + self.apply(cx, move |state, cx| state.on_audio(&samples, cx)); + } + + /// Capture failed; the session ends with this error. + pub fn error(&self, error: SpeechError, cx: &mut App) { + self.apply(cx, move |state, cx| state.on_error(error, cx)); + } + + fn apply( + &self, + cx: &mut App, + f: impl FnOnce(&mut SpeechState, &mut Context) + 'static, + ) { + defer_session_update(self.state.clone(), self.session, cx, f); + } +} diff --git a/crates/component/src/speech/state.rs b/crates/component/src/speech/state.rs new file mode 100644 index 0000000000..f4a79376a2 --- /dev/null +++ b/crates/component/src/speech/state.rs @@ -0,0 +1,689 @@ +use std::{cell::OnceCell, collections::VecDeque, rc::Rc, time::Duration}; + +use gpui::{App, Context, EventEmitter, SharedString, Subscription, Task, WeakEntity}; + +use super::{AudioInput, AudioSink, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// How many recent input levels a [`SpeechState`] keeps for a waveform. +pub(super) const LEVEL_HISTORY: usize = 48; + +/// Default time [`SpeechState::stop`] waits for the final result. +const DEFAULT_STOP_TIMEOUT: Duration = Duration::from_secs(3); + +/// Where a [`SpeechState`] is in its session. +#[derive(Debug, Clone, Copy, PartialEq, Eq, Default)] +pub enum SpeechStatus { + /// No session is running. + #[default] + Idle, + /// Audio is being captured while the recognizer connects. + Connecting, + /// Audio is being captured and recognized. + Recording, + /// Capture stopped; waiting for the recognizer's final result. + Stopping, +} + +impl SpeechStatus { + /// Whether a session is running. + pub fn is_active(self) -> bool { + self != Self::Idle + } + + /// Whether the microphone is capturing. + pub fn is_capturing(self) -> bool { + matches!(self, Self::Connecting | Self::Recording) + } +} + +/// Events emitted by [`SpeechState`]. +#[derive(Debug, Clone)] +pub enum SpeechEvent { + /// A session started and audio is being captured. + Started, + /// The transcript changed: every committed phrase followed by the current + /// hypothesis. Later events supersede earlier ones. + Partial(SharedString), + /// The session ended normally with this transcript, possibly empty. + Final(SharedString), + /// The session was cancelled and its transcript discarded. + Cancelled, + /// The session failed and ended. + Error(SpeechError), +} + +struct Session { + id: usize, + /// Dropping this stops capture. + capture: Option, + recognition: Box, + _stop_timeout: Option>, +} + +/// The state of a speech input: captures audio from an [`AudioInput`], feeds it +/// to a [`SpeechRecognizer`] and tracks the transcript. +/// +/// Render it with [`SpeechButton`](super::SpeechButton) and +/// [`SpeechWaveform`](super::SpeechWaveform), and subscribe to [`SpeechEvent`] +/// to receive the text. +/// +/// The recognizer is, in order: the one passed to [`Self::recognizer`]; else +/// the platform's `SystemRecognizer`, unless +/// [`Self::system_fallback`] turned it off; else none, and the state is not +/// available. The input defaults to the `Microphone`. +/// Both defaults need the `speech` feature. +pub struct SpeechState { + recognizer: Option>, + input: Option>, + system_fallback: bool, + system_recognizer: OnceCell>>, + stop_timeout: Duration, + status: SpeechStatus, + session: Option, + next_session: usize, + committed: String, + hypothesis: SharedString, + levels: VecDeque, +} + +impl EventEmitter for SpeechState {} + +impl SpeechState { + /// Create a speech state with the default recognizer and input. + pub fn new(_: &mut Context) -> Self { + Self { + recognizer: None, + input: super::default_input(), + system_fallback: true, + system_recognizer: OnceCell::new(), + stop_timeout: DEFAULT_STOP_TIMEOUT, + status: SpeechStatus::Idle, + session: None, + next_session: 0, + committed: String::new(), + hypothesis: SharedString::default(), + levels: VecDeque::with_capacity(LEVEL_HISTORY), + } + } + + /// Recognize speech with `recognizer` instead of the system's. + pub fn recognizer(mut self, recognizer: impl SpeechRecognizer) -> Self { + self.recognizer = Some(Rc::new(recognizer)); + self + } + + /// Capture audio from `input` instead of the microphone. + pub fn input(mut self, input: impl AudioInput) -> Self { + self.input = Some(Rc::new(input)); + self + } + + /// Whether to fall back to the platform's recognizer when no + /// [`Self::recognizer`] is set, default `true`. + /// + /// On Windows the system recognizer dictates through Microsoft's online + /// service; turn this off when audio must not leave the application. + pub fn system_fallback(mut self, system_fallback: bool) -> Self { + self.system_fallback = system_fallback; + self + } + + /// Set how long [`Self::stop`] waits for the recognizer's final result + /// before ending the session with the transcript so far, default 3 seconds. + pub fn stop_timeout(mut self, timeout: Duration) -> Self { + self.stop_timeout = timeout; + self + } + + /// Whether a recognizer and an input are configured, so that speech input + /// can work on this platform at all. + pub fn has_recognizer(&self) -> bool { + self.input.is_some() && self.active_recognizer().is_some() + } + + /// Whether a session can start now: [`Self::has_recognizer`] and the + /// recognizer reports itself available. + pub fn is_available(&self, cx: &App) -> bool { + self.input.is_some() + && self + .active_recognizer() + .is_some_and(|recognizer| recognizer.is_available(cx)) + } + + /// Where the state is in its session. + pub fn status(&self) -> SpeechStatus { + self.status + } + + /// The transcript of the current or last session: every committed phrase + /// followed by the current hypothesis. + pub fn transcript(&self) -> SharedString { + if self.hypothesis.is_empty() { + self.committed.clone().into() + } else { + format!("{}{}", self.committed, self.hypothesis).into() + } + } + + /// Recent input levels in `0.0..=1.0`, oldest first. + pub fn levels(&self) -> impl ExactSizeIterator + '_ { + self.levels.iter().copied() + } + + /// Start a session. Does nothing while one is running. + /// + /// When the recognizer or the input fails to start, emits + /// [`SpeechEvent::Error`] and stays idle. + pub fn start(&mut self, cx: &mut Context) { + if self.status.is_active() { + return; + } + + let (Some(recognizer), Some(input)) = (self.active_recognizer(), self.input.clone()) else { + cx.emit(SpeechEvent::Error(SpeechError::Unsupported)); + return; + }; + + self.next_session += 1; + let id = self.next_session; + let state = cx.weak_entity(); + let sink = SpeechSink { + state: state.clone(), + session: id, + }; + let recognition = match recognizer.start(sink, cx) { + Ok(recognition) => recognition, + Err(error) => { + cx.emit(SpeechEvent::Error(error)); + return; + } + }; + + let format = recognizer.audio_format(); + let sink = AudioSink { state, session: id }; + let capture = match input.start(format, sink, cx) { + Ok(capture) => capture, + Err(error) => { + cx.emit(SpeechEvent::Error(error)); + return; + } + }; + + self.status = SpeechStatus::Connecting; + self.committed.clear(); + self.hypothesis = SharedString::default(); + self.levels.clear(); + self.session = Some(Session { + id, + capture: Some(capture), + recognition, + _stop_timeout: None, + }); + cx.emit(SpeechEvent::Started); + cx.notify(); + } + + /// Stop capturing and wait for the final result, which arrives as + /// [`SpeechEvent::Final`]. + pub fn stop(&mut self, cx: &mut Context) { + if !self.status.is_capturing() { + return; + } + let Some(session) = self.session.as_mut() else { + return; + }; + + session.capture = None; + session.recognition.finish(cx); + + let id = session.id; + let timeout = self.stop_timeout; + session._stop_timeout = Some(cx.spawn(async move |this, cx| { + cx.background_executor().timer(timeout).await; + _ = this.update(cx, |this, cx| { + if this.is_session(id) { + this.end(SpeechEvent::Final(this.transcript()), cx); + } + }); + })); + + self.status = SpeechStatus::Stopping; + cx.notify(); + } + + /// End the session at once and discard its transcript. + pub fn cancel(&mut self, cx: &mut Context) { + if !self.status.is_active() { + return; + } + self.committed.clear(); + self.hypothesis = SharedString::default(); + self.end(SpeechEvent::Cancelled, cx); + } + + /// Start a session when idle, otherwise stop the running one. + pub fn toggle(&mut self, cx: &mut Context) { + if self.status.is_active() { + self.stop(cx); + } else { + self.start(cx); + } + } + + fn active_recognizer(&self) -> Option> { + if let Some(recognizer) = &self.recognizer { + return Some(recognizer.clone()); + } + if !self.system_fallback { + return None; + } + self.system_recognizer + .get_or_init(super::system_recognizer) + .clone() + } + + pub(super) fn is_session(&self, id: usize) -> bool { + self.session + .as_ref() + .is_some_and(|session| session.id == id) + } + + pub(super) fn on_ready(&mut self, cx: &mut Context) { + if self.status == SpeechStatus::Connecting { + self.status = SpeechStatus::Recording; + cx.notify(); + } + } + + pub(super) fn on_audio(&mut self, samples: &[i16], cx: &mut Context) { + if !self.status.is_capturing() { + return; + } + if let Some(session) = self.session.as_mut() { + session.recognition.push_audio(samples, cx); + } + if self.levels.len() == LEVEL_HISTORY { + self.levels.pop_front(); + } + self.levels.push_back(level(samples)); + cx.notify(); + } + + pub(super) fn on_hypothesis(&mut self, text: SharedString, cx: &mut Context) { + self.hypothesis = text; + cx.emit(SpeechEvent::Partial(self.transcript())); + cx.notify(); + } + + pub(super) fn on_phrase(&mut self, text: SharedString, cx: &mut Context) { + self.committed.push_str(&text); + self.hypothesis = SharedString::default(); + cx.emit(SpeechEvent::Partial(self.transcript())); + cx.notify(); + } + + pub(super) fn on_finish(&mut self, cx: &mut Context) { + self.end(SpeechEvent::Final(self.transcript()), cx); + } + + pub(super) fn on_error(&mut self, error: SpeechError, cx: &mut Context) { + self.end(SpeechEvent::Error(error), cx); + } + + /// Tear the session down, capture first, and report how it ended. + fn end(&mut self, event: SpeechEvent, cx: &mut Context) { + if let Some(mut session) = self.session.take() { + session.capture = None; + } + self.status = SpeechStatus::Idle; + self.levels.clear(); + cx.emit(event); + cx.notify(); + } +} + +/// Apply `f` to `state` after the current update, if `session` is still its +/// running session. Sinks go through this so that a recognizer or an input may +/// report from anywhere, including from inside a call the state made. +pub(super) fn defer_session_update( + state: WeakEntity, + session: usize, + cx: &mut App, + f: impl FnOnce(&mut SpeechState, &mut Context) + 'static, +) { + cx.defer(move |cx| { + _ = state.update(cx, |state, cx| { + if state.is_session(session) { + f(state, cx); + } + }); + }); +} + +/// The loudness of `samples` in `0.0..=1.0`, mapping -50 dBFS..0 dBFS linearly +/// so that normal speech fills most of the range. +fn level(samples: &[i16]) -> f32 { + if samples.is_empty() { + return 0.; + } + let sum: f64 = samples + .iter() + .map(|&sample| { + let sample = sample as f64 / i16::MAX as f64; + sample * sample + }) + .sum(); + let rms = (sum / samples.len() as f64).sqrt(); + if rms <= 0. { + return 0.; + } + let db = 20. * rms.log10(); + ((db + 50.) / 50.).clamp(0., 1.) as f32 +} + +#[cfg(test)] +mod tests { + use std::{cell::RefCell, rc::Rc, time::Duration}; + + use gpui::{App, AppContext as _, Entity, Subscription, TestAppContext}; + + use super::*; + use crate::speech::{AudioFormat, AudioInput, AudioSink, RecognitionSession, SpeechSink}; + + #[derive(Default)] + struct Recorded { + sink: Option, + samples: usize, + finished: bool, + dropped: bool, + } + + #[derive(Clone, Default)] + struct FakeRecognizer { + recorded: Rc>, + fail_to_start: bool, + } + + struct FakeSession(Rc>); + + impl SpeechRecognizer for FakeRecognizer { + fn start( + &self, + sink: SpeechSink, + _: &mut App, + ) -> Result, SpeechError> { + if self.fail_to_start { + return Err(SpeechError::recognizer(anyhow::anyhow!("offline"))); + } + *self.recorded.borrow_mut() = Recorded { + sink: Some(sink), + ..Default::default() + }; + Ok(Box::new(FakeSession(self.recorded.clone()))) + } + } + + impl RecognitionSession for FakeSession { + fn push_audio(&mut self, samples: &[i16], _: &mut App) { + self.0.borrow_mut().samples += samples.len(); + } + + fn finish(&mut self, _: &mut App) { + self.0.borrow_mut().finished = true; + } + } + + impl Drop for FakeSession { + fn drop(&mut self) { + self.0.borrow_mut().dropped = true; + } + } + + #[derive(Clone, Default)] + struct FakeInput { + sink: Rc>>, + capturing: Rc>, + } + + impl AudioInput for FakeInput { + fn start( + &self, + _: AudioFormat, + sink: AudioSink, + _: &mut App, + ) -> Result { + *self.sink.borrow_mut() = Some(sink); + *self.capturing.borrow_mut() = true; + let capturing = self.capturing.clone(); + Ok(Subscription::new(move || *capturing.borrow_mut() = false)) + } + } + + struct Fixture { + state: Entity, + recognizer: FakeRecognizer, + input: FakeInput, + events: Rc>>, + _subscription: Subscription, + } + + impl Fixture { + fn new(recognizer: FakeRecognizer, cx: &mut TestAppContext) -> Self { + let input = FakeInput::default(); + let state = cx.update(|cx| { + cx.new(|cx| { + SpeechState::new(cx) + .recognizer(recognizer.clone()) + .input(input.clone()) + .stop_timeout(Duration::from_secs(1)) + }) + }); + let events = Rc::new(RefCell::new(Vec::new())); + let _subscription = cx.update(|cx| { + let events = events.clone(); + cx.subscribe(&state, move |_, event: &SpeechEvent, _| { + events.borrow_mut().push(event.clone()); + }) + }); + Self { + state, + recognizer, + input, + events, + _subscription, + } + } + + fn sink(&self) -> SpeechSink { + self.recognizer.recorded.borrow().sink.clone().unwrap() + } + + fn audio(&self) -> AudioSink { + self.input.sink.borrow().clone().unwrap() + } + + fn status(&self, cx: &mut TestAppContext) -> SpeechStatus { + cx.read(|cx| self.state.read(cx).status()) + } + + /// Events so far, as short labels. + fn take_events(&self) -> Vec { + self.events + .borrow_mut() + .drain(..) + .map(|event| match event { + SpeechEvent::Started => "started".into(), + SpeechEvent::Partial(text) => format!("partial:{text}"), + SpeechEvent::Final(text) => format!("final:{text}"), + SpeechEvent::Cancelled => "cancelled".into(), + SpeechEvent::Error(error) => format!("error:{error}"), + }) + .collect() + } + } + + #[gpui::test] + fn session_runs_from_start_to_final(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + + f.state.update(cx, |state, cx| state.start(cx)); + assert_eq!(f.status(cx), SpeechStatus::Connecting); + assert!(*f.input.capturing.borrow()); + + cx.update(|cx| { + f.sink().ready(cx); + f.audio().push(vec![i16::MAX / 2; 160], cx); + f.sink().hypothesis("hello", cx); + }); + cx.run_until_parked(); + assert_eq!(f.status(cx), SpeechStatus::Recording); + assert_eq!(f.recognizer.recorded.borrow().samples, 160); + cx.read(|cx| { + let state = f.state.read(cx); + assert_eq!(state.levels().len(), 1); + assert!(state.levels().next().unwrap() > 0.5); + }); + + cx.update(|cx| { + f.sink().phrase("Hello.", cx); + f.sink().hypothesis(" How", cx); + }); + cx.run_until_parked(); + assert_eq!( + cx.read(|cx| f.state.read(cx).transcript()), + SharedString::from("Hello. How") + ); + + f.state.update(cx, |state, cx| state.stop(cx)); + assert_eq!(f.status(cx), SpeechStatus::Stopping); + assert!(!*f.input.capturing.borrow(), "stop releases the microphone"); + assert!(f.recognizer.recorded.borrow().finished); + + cx.update(|cx| { + f.sink().phrase(" How are you?", cx); + f.sink().finish(cx); + }); + cx.run_until_parked(); + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(f.recognizer.recorded.borrow().dropped); + assert_eq!( + f.take_events(), + [ + "started", + "partial:hello", + "partial:Hello.", + "partial:Hello. How", + "partial:Hello. How are you?", + "final:Hello. How are you?", + ] + ); + } + + #[gpui::test] + fn stop_ends_with_the_transcript_so_far_after_the_timeout(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + f.state.update(cx, |state, cx| state.start(cx)); + cx.update(|cx| f.sink().hypothesis("half a sentence", cx)); + cx.run_until_parked(); + + f.state.update(cx, |state, cx| state.stop(cx)); + cx.executor().advance_clock(Duration::from_millis(900)); + assert_eq!(f.status(cx), SpeechStatus::Stopping); + cx.executor().advance_clock(Duration::from_millis(200)); + cx.run_until_parked(); + + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert_eq!(f.take_events().last().unwrap(), "final:half a sentence"); + } + + #[gpui::test] + fn cancel_discards_the_session_and_ignores_late_results(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + f.state.update(cx, |state, cx| state.start(cx)); + let stale = f.sink(); + cx.update(|cx| stale.phrase("draft", cx)); + cx.run_until_parked(); + + f.state.update(cx, |state, cx| state.cancel(cx)); + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(!*f.input.capturing.borrow()); + assert!(f.recognizer.recorded.borrow().dropped); + + // A new session must not pick up the old session's results. + f.state.update(cx, |state, cx| state.start(cx)); + cx.update(|cx| { + stale.phrase("late", cx); + stale.finish(cx); + }); + cx.run_until_parked(); + + assert_eq!(f.status(cx), SpeechStatus::Connecting); + assert_eq!( + cx.read(|cx| f.state.read(cx).transcript()), + SharedString::default() + ); + assert_eq!( + f.take_events(), + ["started", "partial:draft", "cancelled", "started"] + ); + } + + #[gpui::test] + fn input_error_ends_the_session(cx: &mut TestAppContext) { + let f = Fixture::new(FakeRecognizer::default(), cx); + f.state.update(cx, |state, cx| state.start(cx)); + cx.update(|cx| f.audio().error(SpeechError::NoInputDevice, cx)); + cx.run_until_parked(); + + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(f.recognizer.recorded.borrow().dropped); + assert_eq!( + f.take_events(), + ["started", "error:no audio input device is available"] + ); + } + + #[gpui::test] + fn recognizer_that_fails_to_start_leaves_the_state_idle(cx: &mut TestAppContext) { + let f = Fixture::new( + FakeRecognizer { + fail_to_start: true, + ..Default::default() + }, + cx, + ); + f.state.update(cx, |state, cx| state.start(cx)); + + assert_eq!(f.status(cx), SpeechStatus::Idle); + assert!(!*f.input.capturing.borrow(), "the input never starts"); + assert_eq!( + f.take_events(), + ["error:speech recognition failed: offline"] + ); + } + + #[gpui::test] + fn state_without_a_recognizer_is_unsupported(cx: &mut TestAppContext) { + let state = cx.update(|cx| { + cx.new(|cx| { + SpeechState::new(cx) + .input(FakeInput::default()) + .system_fallback(false) + }) + }); + cx.read(|cx| { + assert!(!state.read(cx).has_recognizer()); + assert!(!state.read(cx).is_available(cx)); + }); + } + + #[test] + fn level_maps_decibels_to_the_unit_range() { + assert_eq!(level(&[]), 0.); + assert_eq!(level(&[0; 16]), 0.); + assert_eq!(level(&[i16::MAX; 16]), 1.); + // -20 dBFS sits at 0.6 on the -50..0 dB scale. + let quiet = level(&[i16::MAX / 10; 16]); + assert!((quiet - 0.6).abs() < 0.01, "{quiet}"); + } +} diff --git a/crates/component/src/speech/system/macos.rs b/crates/component/src/speech/system/macos.rs new file mode 100644 index 0000000000..95ad16a2ac --- /dev/null +++ b/crates/component/src/speech/system/macos.rs @@ -0,0 +1,486 @@ +//! `SFSpeechRecognizer`, recognizing on the device only. + +use std::{cell::RefCell, rc::Rc}; + +use anyhow::anyhow; +use block2::RcBlock; +use gpui::{App, SharedString, Task}; +use objc2::{AnyThread as _, rc::Retained, runtime::NSObjectProtocol as _, sel}; +use objc2_av_foundation::{AVAuthorizationStatus, AVCaptureDevice, AVMediaTypeAudio}; +use objc2_avf_audio::{AVAudioCommonFormat, AVAudioFormat, AVAudioPCMBuffer}; +use objc2_foundation::{NSBundle, NSError, NSLocale, NSString}; +use objc2_speech::{ + SFSpeechAudioBufferRecognitionRequest, SFSpeechRecognitionResult, SFSpeechRecognitionTask, + SFSpeechRecognizer, SFSpeechRecognizerAuthorizationStatus, +}; +use smol::channel::{Receiver, Sender, unbounded}; + +use crate::speech::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// Without this `Info.plist` key, asking for speech recognition access +/// terminates the process. +const USAGE_DESCRIPTION_KEY: &str = "NSSpeechRecognitionUsageDescription"; + +/// The error `SFSpeechRecognizer` reports when the audio held no speech. +const NO_SPEECH_DOMAIN: &str = "kAFAssistantErrorDomain"; +const NO_SPEECH_CODE: isize = 1110; + +pub(super) struct PlatformRecognizer { + /// `None` when the locale has no recognizer or the application cannot ask + /// for access. + recognizer: Option>, +} + +impl PlatformRecognizer { + pub(super) fn new(locale: Option) -> Self { + if !has_usage_description() { + tracing::warn!( + "speech recognition is unavailable: the application's Info.plist has no \ + `{USAGE_DESCRIPTION_KEY}`, and asking for access without it terminates the process" + ); + return Self { recognizer: None }; + } + + let recognizer = match &locale { + Some(locale) => { + let locale = NSLocale::initWithLocaleIdentifier( + NSLocale::alloc(), + &NSString::from_str(locale), + ); + // SAFETY: `locale` is a valid `NSLocale`; the initializer returns + // nil for a locale without a recognizer. + unsafe { SFSpeechRecognizer::initWithLocale(SFSpeechRecognizer::alloc(), &locale) } + } + // SAFETY: plain initializer; returns nil if the system language has + // no recognizer. + None => unsafe { SFSpeechRecognizer::init(SFSpeechRecognizer::alloc()) }, + }; + if recognizer.is_none() { + tracing::warn!( + "speech recognition is unavailable: no recognizer for locale {}", + locale.as_deref().unwrap_or("of the system") + ); + } + + Self { recognizer } + } + + /// The recognizer, if it can recognize on the device. + fn on_device(&self) -> Option<&Retained> { + self.recognizer + .as_ref() + // SAFETY: plain property reads on a valid recognizer. + .filter(|recognizer| unsafe { + recognizer.isAvailable() && recognizer.supportsOnDeviceRecognition() + }) + } +} + +impl SpeechRecognizer for PlatformRecognizer { + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + fn is_available(&self, _: &App) -> bool { + self.on_device().is_some() && !is_speech_denied(speech_authorization()) + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + let recognizer = self.on_device().ok_or(SpeechError::Unsupported)?.clone(); + let authorization = speech_authorization(); + if is_speech_denied(authorization) { + return Err(SpeechError::PermissionDenied); + } + + let (events, rx) = unbounded(); + let mut recognition = Recognition::new(recognizer, sink, events.clone())?; + if authorization == SFSpeechRecognizerAuthorizationStatus::Authorized { + recognition.begin(cx); + } else { + request_authorization(events); + } + + let recognition = Rc::new(RefCell::new(recognition)); + let task = cx.spawn({ + let recognition = recognition.clone(); + async move |cx| run(recognition, rx, cx).await + }); + + Ok(Box::new(Session { + recognition, + _task: task, + })) + } +} + +/// Whether the user or a policy denied this application the microphone. +/// +/// Capturing without access yields silence rather than an error on macOS, so +/// the [`Microphone`](crate::speech::Microphone) checks this first. +pub(in crate::speech) fn is_microphone_denied() -> bool { + // SAFETY: reading an immutable framework constant. + let Some(audio) = (unsafe { AVMediaTypeAudio }) else { + return false; + }; + // SAFETY: `audio` is one of the two media types the method accepts. + let status = unsafe { AVCaptureDevice::authorizationStatusForMediaType(audio) }; + status == AVAuthorizationStatus::Denied || status == AVAuthorizationStatus::Restricted +} + +fn has_usage_description() -> bool { + NSBundle::mainBundle() + .objectForInfoDictionaryKey(&NSString::from_str(USAGE_DESCRIPTION_KEY)) + .is_some() +} + +fn speech_authorization() -> SFSpeechRecognizerAuthorizationStatus { + // SAFETY: reading the status never prompts, so it is safe without the + // usage description. + unsafe { SFSpeechRecognizer::authorizationStatus() } +} + +fn is_speech_denied(status: SFSpeechRecognizerAuthorizationStatus) -> bool { + status == SFSpeechRecognizerAuthorizationStatus::Denied + || status == SFSpeechRecognizerAuthorizationStatus::Restricted +} + +/// Ask for speech recognition access; the answer arrives as an [`Event`]. +fn request_authorization(events: Sender) { + let handler = RcBlock::new(move |status: SFSpeechRecognizerAuthorizationStatus| { + _ = events.try_send(Event::Authorized( + status == SFSpeechRecognizerAuthorizationStatus::Authorized, + )); + }); + // SAFETY: only reached with the usage description present (see + // `PlatformRecognizer::new`); the block only sends on a channel, so it may + // run on any queue. + unsafe { SFSpeechRecognizer::requestAuthorization(&handler) }; +} + +/// What the framework reports, sent from its queues to the main thread. +enum Event { + Authorized(bool), + Result { + text: String, + is_final: bool, + /// The result ends an utterance, see [`Recognition::on_result`]. + ends_utterance: bool, + }, + Error { + domain: String, + code: isize, + message: String, + }, +} + +/// Deliver framework events to the recognition until it is done. +async fn run( + recognition: Rc>, + events: Receiver, + cx: &mut gpui::AsyncApp, +) { + while let Ok(event) = events.recv().await { + let done = cx.update(|cx| recognition.borrow_mut().on_event(event, cx)); + if done { + break; + } + } +} + +struct Session { + recognition: Rc>, + /// Dropping this stops delivering results. + _task: Task<()>, +} + +impl RecognitionSession for Session { + fn push_audio(&mut self, samples: &[i16], _: &mut App) { + self.recognition.borrow_mut().push_audio(samples); + } + + fn finish(&mut self, _: &mut App) { + self.recognition.borrow_mut().finish(); + } +} + +impl Drop for Session { + fn drop(&mut self) { + if let Some(task) = &self.recognition.borrow().task { + // SAFETY: cancelling a valid task; it reports nothing we still read. + unsafe { task.cancel() }; + } + } +} + +struct Recognition { + recognizer: Retained, + request: Retained, + format: Retained, + /// `None` until access is granted. + task: Option>, + sink: SpeechSink, + events: Sender, + /// Audio pushed before the task started. + pending: Vec, + finishing: bool, + done: bool, + /// The text of the last result that ended an utterance, until the next + /// result shows whether the recognizer carries it on. + utterance: Option, + /// The current hypothesis, as the framework reported it. + hypothesis: String, + /// The last character committed as a phrase, to join the next one. + last_committed: Option, +} + +impl Recognition { + fn new( + recognizer: Retained, + sink: SpeechSink, + events: Sender, + ) -> Result { + let format = AudioFormat::default(); + // 16-bit mono is also the request's native format, so appended audio + // needs no conversion. + // SAFETY: a mono PCM format, which the initializer supports. + let format = unsafe { + AVAudioFormat::initWithCommonFormat_sampleRate_channels_interleaved( + AVAudioFormat::alloc(), + AVAudioCommonFormat::PCMFormatInt16, + format.sample_rate().into(), + format.channels().into(), + false, + ) + } + .ok_or_else(|| SpeechError::recognizer(anyhow!("cannot create the audio format")))?; + + // SAFETY: plain initializer and property setters on a fresh request. + let request = unsafe { + let request = SFSpeechAudioBufferRecognitionRequest::new(); + request.setShouldReportPartialResults(true); + // Never send audio to Apple's servers. + request.setRequiresOnDeviceRecognition(true); + // Punctuation needs macOS 13. + if request.respondsToSelector(sel!(setAddsPunctuation:)) { + request.setAddsPunctuation(true); + } + request + }; + + Ok(Self { + recognizer, + request, + format, + task: None, + sink, + events, + pending: Vec::new(), + finishing: false, + done: false, + utterance: None, + hypothesis: String::new(), + last_committed: None, + }) + } + + /// Start recognizing, once access is granted. + fn begin(&mut self, cx: &mut App) { + let events = self.events.clone(); + let handler = RcBlock::new( + move |result: *mut SFSpeechRecognitionResult, error: *mut NSError| { + // SAFETY: the framework passes nil or objects valid for the call. + let (result, error) = unsafe { (result.as_ref(), error.as_ref()) }; + if let Some(result) = result { + // SAFETY: plain property reads on a valid result. + let event = unsafe { + Event::Result { + text: result.bestTranscription().formattedString().to_string(), + is_final: result.isFinal(), + ends_utterance: result.speechRecognitionMetadata().is_some(), + } + }; + _ = events.try_send(event); + } + if let Some(error) = error { + _ = events.try_send(Event::Error { + domain: error.domain().to_string(), + code: error.code(), + message: error.localizedDescription().to_string(), + }); + } + }, + ); + // SAFETY: the request is an audio buffer request and the block only + // sends on a channel, so it may run on any queue. + let task = unsafe { + self.recognizer + .recognitionTaskWithRequest_resultHandler(&self.request, &handler) + }; + self.task = Some(task); + + let pending = std::mem::take(&mut self.pending); + self.append(&pending); + if self.finishing { + // SAFETY: ending the audio of a valid request. + unsafe { self.request.endAudio() }; + } + self.sink.ready(cx); + } + + fn push_audio(&mut self, samples: &[i16]) { + if self.done || self.finishing { + return; + } + if self.task.is_some() { + self.append(samples); + } else { + self.pending.extend_from_slice(samples); + } + } + + fn finish(&mut self) { + if self.finishing { + return; + } + self.finishing = true; + if self.task.is_some() { + // SAFETY: ending the audio of a valid request. + unsafe { self.request.endAudio() }; + } + } + + fn append(&self, samples: &[i16]) { + let Ok(frames) = u32::try_from(samples.len()) else { + return; + }; + if frames == 0 { + return; + } + // SAFETY: `format` is PCM, so the initializer only fails for sizes + // beyond `u32`. + let Some(buffer) = (unsafe { + AVAudioPCMBuffer::initWithPCMFormat_frameCapacity( + AVAudioPCMBuffer::alloc(), + &self.format, + frames, + ) + }) else { + return; + }; + // SAFETY: the format is 16-bit mono, so channel 0 holds `frames` + // writable samples, and `frames` is within the capacity. + unsafe { + let channel = (*buffer.int16ChannelData()).as_ptr(); + std::ptr::copy_nonoverlapping(samples.as_ptr(), channel, samples.len()); + buffer.setFrameLength(frames); + self.request.appendAudioPCMBuffer(&buffer); + } + } + + /// Handle one framework event; returns whether the recognition is done. + fn on_event(&mut self, event: Event, cx: &mut App) -> bool { + if self.done { + return true; + } + match event { + Event::Authorized(true) => self.begin(cx), + Event::Authorized(false) => { + self.sink.error(SpeechError::PermissionDenied, cx); + self.done = true; + } + Event::Result { + text, + is_final, + ends_utterance, + } => self.on_result(text, is_final, ends_utterance, cx), + Event::Error { + domain, + code, + message, + } => { + if self.finishing && domain == NO_SPEECH_DOMAIN && code == NO_SPEECH_CODE { + // Stopping without (more) speech is a normal end. + let hypothesis = std::mem::take(&mut self.hypothesis); + self.commit(&hypothesis, cx); + self.sink.finish(cx); + } else { + self.sink.error( + SpeechError::recognizer(anyhow!("{message} ({domain} {code})")), + cx, + ); + } + self.done = true; + } + } + self.done + } + + /// Results carry the whole text of the request so far, except that some + /// macOS versions start over after a pause: the result that ends an + /// utterance has metadata, and the next one may no longer include its + /// text. Commit that utterance as a phrase only once it is dropped. + fn on_result(&mut self, text: String, is_final: bool, ends_utterance: bool, cx: &mut App) { + if let Some(utterance) = self.utterance.take() + && !text.starts_with(&utterance) + { + self.commit(&utterance, cx); + } + + if is_final { + self.hypothesis.clear(); + self.commit(&text, cx); + self.sink.finish(cx); + self.done = true; + return; + } + + if ends_utterance { + self.utterance = Some(text.clone()); + } + self.sink.hypothesis(self.joined(&text), cx); + self.hypothesis = text; + } + + fn commit(&mut self, text: &str, cx: &mut App) { + if text.is_empty() { + return; + } + let text = self.joined(text); + self.last_committed = text.chars().next_back(); + self.sink.phrase(text, cx); + } + + /// `text` with the separator it needs after the committed phrases. + fn joined(&self, text: &str) -> String { + match (self.last_committed, text.chars().next()) { + (Some(before), Some(after)) if needs_space(before, after) => format!(" {text}"), + _ => text.to_string(), + } + } +} + +/// Whether two phrases ending and starting with these characters need a space +/// between them. +fn needs_space(before: char, after: char) -> bool { + !before.is_whitespace() + && !after.is_whitespace() + && !matches!(after, ',' | '.' | '?' | '!' | ';' | ':' | ')') + && !is_unspaced_script(before) + && !is_unspaced_script(after) +} + +/// Characters of scripts written without spaces between words: Chinese, +/// Japanese and their punctuation. +fn is_unspaced_script(c: char) -> bool { + matches!(c, + '\u{3000}'..='\u{30FF}' + | '\u{3400}'..='\u{4DBF}' + | '\u{4E00}'..='\u{9FFF}' + | '\u{F900}'..='\u{FAFF}' + | '\u{FF00}'..='\u{FFEF}' + ) +} diff --git a/crates/component/src/speech/system/mod.rs b/crates/component/src/speech/system/mod.rs new file mode 100644 index 0000000000..8325d6b3f0 --- /dev/null +++ b/crates/component/src/speech/system/mod.rs @@ -0,0 +1,86 @@ +//! The platform's own speech recognizer. + +use std::{cell::OnceCell, rc::Rc}; + +use gpui::{App, SharedString}; + +use super::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +#[cfg(target_os = "macos")] +pub(super) mod macos; +#[cfg(target_os = "macos")] +use macos as platform; + +#[cfg(target_os = "windows")] +mod winrt; +#[cfg(target_os = "windows")] +use winrt as platform; + +#[cfg(not(any(target_os = "macos", target_os = "windows")))] +mod unsupported; +#[cfg(not(any(target_os = "macos", target_os = "windows")))] +use unsupported as platform; + +/// The operating system's speech recognizer. +/// +/// - **macOS**: `SFSpeechRecognizer`, recognizing on the device only. A +/// language the device cannot recognize offline is not available, so audio +/// never leaves the machine. +/// - **Windows**: `Windows.Media.SpeechRecognition`. Dictation needs the +/// language's speech pack and the "Online speech recognition" privacy +/// setting, and runs through Microsoft's online service. +/// - **Other platforms**: never available. +/// +/// A [`SpeechState`](super::SpeechState) without its own recognizer uses this +/// one; create it directly to choose the language. +pub struct SystemRecognizer { + locale: Option, + platform: OnceCell>, +} + +impl SystemRecognizer { + /// A recognizer for the system's current language. + pub fn new() -> Self { + Self { + locale: None, + platform: OnceCell::new(), + } + } + + /// Recognize `locale`, a BCP 47 language tag such as `en-US` or `zh-CN`, + /// instead of the system's language. + pub fn locale(mut self, locale: impl Into) -> Self { + self.locale = Some(locale.into()); + self.platform = OnceCell::new(); + self + } + + fn platform(&self) -> &Rc { + self.platform + .get_or_init(|| Rc::new(platform::PlatformRecognizer::new(self.locale.clone()))) + } +} + +impl Default for SystemRecognizer { + fn default() -> Self { + Self::new() + } +} + +impl SpeechRecognizer for SystemRecognizer { + fn audio_format(&self) -> AudioFormat { + self.platform().audio_format() + } + + fn is_available(&self, cx: &App) -> bool { + self.platform().is_available(cx) + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + self.platform().start(sink, cx) + } +} diff --git a/crates/component/src/speech/system/unsupported.rs b/crates/component/src/speech/system/unsupported.rs new file mode 100644 index 0000000000..13aeb09601 --- /dev/null +++ b/crates/component/src/speech/system/unsupported.rs @@ -0,0 +1,30 @@ +use gpui::{App, SharedString}; + +use crate::speech::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// Placeholder until the platform recognizer lands. +pub(super) struct PlatformRecognizer; + +impl PlatformRecognizer { + pub(super) fn new(_locale: Option) -> Self { + Self + } +} + +impl SpeechRecognizer for PlatformRecognizer { + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + fn is_available(&self, _: &App) -> bool { + false + } + + fn start( + &self, + _: SpeechSink, + _: &mut App, + ) -> Result, SpeechError> { + Err(SpeechError::Unsupported) + } +} diff --git a/crates/component/src/speech/system/winrt.rs b/crates/component/src/speech/system/winrt.rs new file mode 100644 index 0000000000..7fc7b18e87 --- /dev/null +++ b/crates/component/src/speech/system/winrt.rs @@ -0,0 +1,395 @@ +//! The Windows recognizer: continuous dictation with +//! `Windows.Media.SpeechRecognition`. +//! +//! The WinRT recognizer captures from the default microphone itself and has no +//! way to consume audio from elsewhere, so the pushed PCM is ignored; the +//! [`Microphone`](crate::speech::Microphone) keeps capturing alongside it only +//! to drive the waveform, which WASAPI's shared mode allows. +//! +//! The speech objects are agile, so they are called from the main thread, an +//! STA that GPUI initializes with `OleInitialize`, without further apartment +//! setup. Their completions and events arrive on WinRT threads, which only +//! forward them over a channel to a foreground task that reports to the sink. + +use anyhow::anyhow; +use gpui::{App, AsyncApp, SharedString, Task}; +use smol::channel::{Receiver, Sender, bounded, unbounded}; +use windows::{ + Foundation::{ + AsyncActionCompletedHandler, AsyncOperationCompletedHandler, EventRegistrationToken, + IAsyncAction, IAsyncOperation, TypedEventHandler, + }, + Globalization::Language, + Media::SpeechRecognition::{ + SpeechContinuousRecognitionCompletedEventArgs, + SpeechContinuousRecognitionResultGeneratedEventArgs, SpeechContinuousRecognitionSession, + SpeechRecognitionConfidence, SpeechRecognitionHypothesisGeneratedEventArgs, + SpeechRecognitionResult, SpeechRecognitionResultStatus, SpeechRecognitionScenario, + SpeechRecognitionTopicConstraint, SpeechRecognizer as WinSpeechRecognizer, + }, + Win32::Foundation::E_ACCESSDENIED, + core::{HRESULT, HSTRING, RuntimeType}, +}; + +use crate::speech::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// `SPERR_SPEECH_PRIVACY_POLICY_NOT_ACCEPTED`: "Online speech recognition" is +/// turned off in the privacy settings, which dictation requires. +const PRIVACY_POLICY_NOT_ACCEPTED: HRESULT = HRESULT(0x80045509_u32 as i32); +/// `MF_E_NO_CAPTURE_DEVICES_AVAILABLE`. +const NO_CAPTURE_DEVICES: HRESULT = HRESULT(0xC00DABE0_u32 as i32); + +pub(super) struct PlatformRecognizer { + /// The dictation language, or `None` when the requested one is malformed or + /// has no speech pack installed. + language: Option, + separator: &'static str, +} + +impl PlatformRecognizer { + pub(super) fn new(locale: Option) -> Self { + let language = supported_language(locale.as_deref()) + .inspect_err(|error| log::warn!("speech: no dictation language: {error:#}")) + .ok() + .flatten(); + let separator = language + .as_ref() + .and_then(|language| language.LanguageTag().ok()) + .map_or(" ", |tag| phrase_separator(&tag.to_string_lossy())); + Self { + language, + separator, + } + } +} + +impl SpeechRecognizer for PlatformRecognizer { + fn audio_format(&self) -> AudioFormat { + AudioFormat::default() + } + + fn is_available(&self, _: &App) -> bool { + self.language.is_some() + } + + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + let Some(language) = &self.language else { + return Err(SpeechError::Unsupported); + }; + + let recognizer = WinSpeechRecognizer::Create(language).map_err(speech_error)?; + let continuous = recognizer + .ContinuousRecognitionSession() + .map_err(speech_error)?; + let (messages, rx) = unbounded(); + let mut session = Session { + recognizer: recognizer.clone(), + continuous: continuous.clone(), + hypothesis_token: None, + result_token: None, + completed_token: None, + messages: messages.clone(), + _task: None, + }; + + let constraint = SpeechRecognitionTopicConstraint::Create( + SpeechRecognitionScenario::Dictation, + &HSTRING::from("dictation"), + ) + .map_err(speech_error)?; + recognizer + .Constraints() + .and_then(|constraints| constraints.Append(&constraint)) + .map_err(speech_error)?; + + session.hypothesis_token = Some( + recognizer + .HypothesisGenerated(&TypedEventHandler::new({ + let messages = messages.clone(); + move |_, args: &Option| { + if let Some(args) = args { + let text = args.Hypothesis()?.Text()?; + _ = messages.try_send(Message::Hypothesis(text)); + } + Ok(()) + } + })) + .map_err(speech_error)?, + ); + session.result_token = Some( + continuous + .ResultGenerated(&TypedEventHandler::new({ + let messages = messages.clone(); + move |_, args: &Option| { + if let Some(args) = args { + _ = messages.try_send(Message::Result(args.Result()?)); + } + Ok(()) + } + })) + .map_err(speech_error)?, + ); + session.completed_token = Some( + continuous + .Completed(&TypedEventHandler::new({ + let messages = messages.clone(); + move |_, args: &Option| { + if let Some(args) = args { + _ = messages.try_send(Message::Completed(args.Status()?)); + } + Ok(()) + } + })) + .map_err(speech_error)?, + ); + + let separator = self.separator; + session._task = Some(cx.spawn(async move |cx| { + let result = dictate(&recognizer, &continuous, rx, &sink, separator, cx).await; + if let Err(error) = result { + cx.update(|cx| sink.error(error, cx)); + } + })); + Ok(Box::new(session)) + } +} + +/// What the WinRT threads and [`Session::finish`] tell the dictation task. +enum Message { + Hypothesis(HSTRING), + Result(SpeechRecognitionResult), + Completed(SpeechRecognitionResultStatus), + /// The user stopped talking. + Finish, +} + +struct Session { + recognizer: WinSpeechRecognizer, + continuous: SpeechContinuousRecognitionSession, + hypothesis_token: Option, + result_token: Option, + completed_token: Option, + messages: Sender, + _task: Option>, +} + +impl RecognitionSession for Session { + /// Ignored: the WinRT recognizer captures from the microphone itself. + fn push_audio(&mut self, _: &[i16], _: &mut App) {} + + fn finish(&mut self, _: &mut App) { + // Queued behind the start, so finishing while connecting stops the + // session as soon as it runs. + _ = self.messages.try_send(Message::Finish); + } +} + +impl Drop for Session { + fn drop(&mut self) { + if let Some(token) = self.hypothesis_token.take() { + _ = self.recognizer.RemoveHypothesisGenerated(token); + } + if let Some(token) = self.result_token.take() { + _ = self.continuous.RemoveResultGenerated(token); + } + if let Some(token) = self.completed_token.take() { + _ = self.continuous.RemoveCompleted(token); + } + + // Close the recognizer once the cancellation lands, or right away when + // there is nothing to cancel. + let recognizer = self.recognizer.clone(); + let closed = self.continuous.CancelAsync().and_then(|action| { + action.SetCompleted(&AsyncActionCompletedHandler::new(move |_, _| { + _ = recognizer.Close(); + Ok(()) + })) + }); + if closed.is_err() { + _ = self.recognizer.Close(); + } + } +} + +/// Compile the dictation constraint, start the continuous session and report +/// its results to `sink` until it completes. +async fn dictate( + recognizer: &WinSpeechRecognizer, + continuous: &SpeechContinuousRecognitionSession, + messages: Receiver, + sink: &SpeechSink, + separator: &'static str, + cx: &mut AsyncApp, +) -> Result<(), SpeechError> { + let compilation = operation(recognizer.CompileConstraintsAsync().map_err(speech_error)?) + .await + .map_err(speech_error)?; + let status = compilation.Status().map_err(speech_error)?; + if status != SpeechRecognitionResultStatus::Success { + return Err(status_error(status)); + } + action(continuous.StartAsync().map_err(speech_error)?) + .await + .map_err(speech_error)?; + cx.update(|cx| sink.ready(cx)); + + let mut separator_due = false; + while let Ok(message) = messages.recv().await { + match message { + Message::Hypothesis(text) => { + let text = joined(separator_due, separator, &text); + cx.update(|cx| sink.hypothesis(text, cx)); + } + Message::Result(result) => { + let Some(text) = phrase_text(&result) else { + continue; + }; + let text = joined(separator_due, separator, &text); + separator_due = true; + cx.update(|cx| sink.phrase(text, cx)); + } + Message::Completed(status) => { + return match status { + // Stopped, cancelled, or ended by the silence timeout. + SpeechRecognitionResultStatus::Success + | SpeechRecognitionResultStatus::UserCanceled + | SpeechRecognitionResultStatus::TimeoutExceeded => { + cx.update(|cx| sink.finish(cx)); + Ok(()) + } + status => Err(status_error(status)), + }; + } + // Stopping flushes the last phrase, then completes the session. + Message::Finish => action(continuous.StopAsync().map_err(speech_error)?) + .await + .map_err(speech_error)?, + } + } + Ok(()) +} + +/// The text of a recognized phrase, unless it was rejected or empty. +fn phrase_text(result: &SpeechRecognitionResult) -> Option { + if result.Status().ok()? != SpeechRecognitionResultStatus::Success + || result.Confidence().ok()? == SpeechRecognitionConfidence::Rejected + { + return None; + } + Some(result.Text().ok()?).filter(|text| !text.is_empty()) +} + +/// `text` preceded by `separator` once a phrase has been committed. +fn joined(separator_due: bool, separator: &str, text: &HSTRING) -> SharedString { + let text = text.to_string_lossy(); + if separator_due { + format!("{separator}{text}").into() + } else { + text.into() + } +} + +/// What goes between two phrases, which dictation returns without surrounding +/// whitespace: a space, except in languages written without spaces between +/// words (Chinese, Japanese, Thai, Lao, Khmer, Burmese). +fn phrase_separator(tag: &str) -> &'static str { + let primary = tag.split('-').next().unwrap_or_default(); + let unspaced = ["zh", "yue", "ja", "th", "lo", "km", "my"]; + if unspaced + .iter() + .any(|lang| primary.eq_ignore_ascii_case(lang)) + { + "" + } else { + " " + } +} + +/// The language to dictate `locale` (or, without one, the system's speech +/// language) in, if a speech pack supports it. +/// +/// A bare language such as `en` falls back to the first supported region. +fn supported_language(locale: Option<&str>) -> windows::core::Result> { + let requested = match locale { + Some(locale) => { + let tag = HSTRING::from(locale); + if !Language::IsWellFormed(&tag)? { + return Ok(None); + } + Language::CreateLanguage(&tag)? + } + None => WinSpeechRecognizer::SystemSpeechLanguage()?, + }; + let requested = requested.LanguageTag()?.to_string_lossy(); + let region_prefix = format!("{requested}-"); + + let mut fallback = None; + for language in WinSpeechRecognizer::SupportedTopicLanguages()? { + let tag = language.LanguageTag()?.to_string_lossy(); + if tag.eq_ignore_ascii_case(&requested) { + return Ok(Some(language)); + } + if fallback.is_none() + && tag.len() > region_prefix.len() + && tag[..region_prefix.len()].eq_ignore_ascii_case(®ion_prefix) + { + fallback = Some(language); + } + } + Ok(fallback) +} + +/// Await `operation` without blocking: its completion handler, which runs on a +/// WinRT thread, only wakes this future. +async fn operation( + operation: IAsyncOperation, +) -> windows::core::Result { + let (done, wait) = bounded(1); + operation.SetCompleted(&AsyncOperationCompletedHandler::new(move |_, _| { + _ = done.try_send(()); + Ok(()) + }))?; + _ = wait.recv().await; + operation.GetResults() +} + +/// Await `action` like [`operation`]. +async fn action(action: IAsyncAction) -> windows::core::Result<()> { + let (done, wait) = bounded(1); + action.SetCompleted(&AsyncActionCompletedHandler::new(move |_, _| { + _ = done.try_send(()); + Ok(()) + }))?; + _ = wait.recv().await; + action.GetResults() +} + +fn speech_error(error: windows::core::Error) -> SpeechError { + match error.code() { + E_ACCESSDENIED => SpeechError::PermissionDenied, + NO_CAPTURE_DEVICES => SpeechError::NoInputDevice, + PRIVACY_POLICY_NOT_ACCEPTED => SpeechError::recognizer(anyhow!( + "Online speech recognition is turned off; turn it on in Settings > \ + Privacy & security > Speech" + )), + _ => SpeechError::recognizer(error), + } +} + +fn status_error(status: SpeechRecognitionResultStatus) -> SpeechError { + match status { + SpeechRecognitionResultStatus::TopicLanguageNotSupported => SpeechError::Unsupported, + SpeechRecognitionResultStatus::MicrophoneUnavailable => SpeechError::NoInputDevice, + SpeechRecognitionResultStatus::NetworkFailure => { + SpeechError::recognizer(anyhow!("could not reach the online speech service")) + } + SpeechRecognitionResultStatus::AudioQualityFailure => { + SpeechError::recognizer(anyhow!("the audio was too poor to recognize")) + } + status => SpeechError::recognizer(anyhow!("recognition ended with status {}", status.0)), + } +} diff --git a/crates/component/src/speech/waveform.rs b/crates/component/src/speech/waveform.rs new file mode 100644 index 0000000000..874e4b8d06 --- /dev/null +++ b/crates/component/src/speech/waveform.rs @@ -0,0 +1,83 @@ +use gpui::{ + App, Entity, IntoElement, ParentElement as _, Pixels, RenderOnce, Styled as _, Window, div, px, +}; + +use crate::{ActiveTheme as _, Sizable, Size, h_flex}; + +use super::{SpeechState, state::LEVEL_HISTORY}; + +/// A live bar graph of a [`SpeechState`]'s recent input levels. +/// +/// The newest level is on the trailing end. The bars sit flat and muted while +/// no audio is captured. +#[derive(IntoElement)] +pub struct SpeechWaveform { + state: Entity, + bars: usize, + size: Size, +} + +impl SpeechWaveform { + /// A waveform for `state`. + pub fn new(state: &Entity) -> Self { + Self { + state: state.clone(), + bars: 24, + size: Size::default(), + } + } + + /// Set the number of bars, default 24, at most 48. + pub fn bars(mut self, bars: usize) -> Self { + self.bars = bars.clamp(1, LEVEL_HISTORY); + self + } + + fn height(&self) -> Pixels { + match self.size { + Size::Size(height) => height, + Size::XSmall => px(12.), + Size::Small => px(16.), + Size::Medium => px(20.), + Size::Large => px(24.), + } + } +} + +impl Sizable for SpeechWaveform { + fn with_size(mut self, size: impl Into) -> Self { + self.size = size.into(); + self + } +} + +impl RenderOnce for SpeechWaveform { + fn render(self, _: &mut Window, cx: &mut App) -> impl IntoElement { + let height = self.height(); + let bar_width = px(2.); + let state = self.state.read(cx); + let color = if state.status().is_capturing() { + cx.theme().primary + } else { + cx.theme().muted_foreground + }; + let levels = state.levels(); + // Right-align the history: pad the leading bars when fewer levels have + // arrived than there are bars. + let skip = levels.len().saturating_sub(self.bars); + let padding = self.bars.saturating_sub(levels.len()); + let levels = std::iter::repeat_n(0., padding).chain(levels.skip(skip)); + + h_flex() + .h(height) + .gap(bar_width) + .items_center() + .children(levels.map(|level| { + div() + .w(bar_width) + .h((height * level).max(bar_width)) + .rounded_full() + .bg(color) + })) + } +} diff --git a/crates/kit/Cargo.toml b/crates/kit/Cargo.toml index 9b77891044..3373991d1d 100644 --- a/crates/kit/Cargo.toml +++ b/crates/kit/Cargo.toml @@ -27,6 +27,8 @@ test-support = ["gpui/test-support", "gpui_platform/test-support", "gpui-base/te profiler = ["gpui/profiler"] inspector = ["gpui/inspector", "gpui-base/inspector", "gpui-component?/inspector"] decimal = ["component", "gpui-component/decimal"] +# Speech input: microphone capture and the system speech recognizer. +speech = ["component", "gpui-component/speech"] tree-sitter = ["component", "gpui-component/tree-sitter"] tree-sitter-languages = ["component", "gpui-component/tree-sitter-languages"] tree-sitter-astro = ["component", "gpui-component/tree-sitter-astro"] diff --git a/crates/story/Cargo.toml b/crates/story/Cargo.toml index 1fe54865d9..5e10ebc402 100644 --- a/crates/story/Cargo.toml +++ b/crates/story/Cargo.toml @@ -12,7 +12,7 @@ default = ["tree-sitter"] [dependencies] anyhow.workspace = true -gpui-kit.workspace = true +gpui-kit = { workspace = true, features = ["speech"] } gpui-fps.workspace = true async-channel = "2.3.1" diff --git a/crates/story/src/gallery.rs b/crates/story/src/gallery.rs index d2b43ec5c7..4c0b7c6bed 100644 --- a/crates/story/src/gallery.rs +++ b/crates/story/src/gallery.rs @@ -126,6 +126,7 @@ impl Gallery { StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), + StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), StoryContainer::panel::(window, cx), diff --git a/crates/story/src/stories/mod.rs b/crates/story/src/stories/mod.rs index 977ee8fec0..307e7f5fb4 100644 --- a/crates/story/src/stories/mod.rs +++ b/crates/story/src/stories/mod.rs @@ -64,6 +64,7 @@ mod shimmer_story; mod sidebar_story; mod skeleton_story; mod slider_story; +mod speech_story; mod spinner_story; mod status_bar_story; mod stepper_story; @@ -143,6 +144,7 @@ pub use shimmer_story::ShimmerStory; pub use sidebar_story::SidebarStory; pub use skeleton_story::SkeletonStory; pub use slider_story::SliderStory; +pub use speech_story::SpeechStory; pub use spinner_story::SpinnerStory; pub use status_bar_story::StatusBarStory; pub use stepper_story::StepperStory; diff --git a/crates/story/src/stories/speech_story.rs b/crates/story/src/stories/speech_story.rs new file mode 100644 index 0000000000..a3e25ed0a0 --- /dev/null +++ b/crates/story/src/stories/speech_story.rs @@ -0,0 +1,281 @@ +use gpui_kit::{ + App, AppContext, Context, Entity, FocusHandle, Focusable, IntoElement, ParentElement, Render, + Styled, Subscription, Window, div, prelude::FluentBuilder as _, +}; + +use gpui_kit::component::{ + ActiveTheme as _, Sizable as _, WindowExt as _, h_flex, + input::{Input, InputState}, + notification::Notification, + speech::{ + RecognitionSession, SpeechButton, SpeechError, SpeechEvent, SpeechRecognizer, SpeechSink, + SpeechState, SpeechWaveform, + }, + v_flex, +}; + +use crate::section; + +/// What [`DemoRecognizer`] "hears", one phrase at a time. +const SCRIPT: [&str; 2] = [ + "Speech input turns what you say into text.", + "Any recognizer plugs in through one trait.", +]; + +/// How much 16 kHz mono audio [`DemoRecognizer`] takes per word: 0.3 s. +const SAMPLES_PER_WORD: usize = 4_800; + +/// A recognizer that needs no service: it types [`SCRIPT`] one word per +/// [`SAMPLES_PER_WORD`] of audio, whatever the audio contains. +/// +/// A real recognizer has the same shape: `start` opens a connection, the +/// session streams audio to it and reports the service's results to the sink. +struct DemoRecognizer; + +impl SpeechRecognizer for DemoRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // Nothing to connect to, so audio is consumed at once. + sink.ready(cx); + Ok(Box::new(DemoSession { + sink, + phrase_ix: 0, + words: 0, + samples: 0, + })) + } +} + +struct DemoSession { + sink: SpeechSink, + phrase_ix: usize, + /// Words of the current phrase recognized so far. + words: usize, + /// Samples received since the last word. + samples: usize, +} + +impl DemoSession { + /// The recognized part of the current phrase, if any. + fn spoken(&self) -> Option { + let phrase = SCRIPT.get(self.phrase_ix)?; + let words = phrase.split(' ').take(self.words).collect::>(); + if words.is_empty() { + return None; + } + // Phrases are joined verbatim, so separate sentences here. + let separator = if self.phrase_ix > 0 { " " } else { "" }; + Some(format!("{separator}{}", words.join(" "))) + } + + fn next_word(&mut self, cx: &mut App) { + let Some(phrase) = SCRIPT.get(self.phrase_ix) else { + return; + }; + self.words += 1; + if self.words < phrase.split(' ').count() { + if let Some(spoken) = self.spoken() { + self.sink.hypothesis(spoken, cx); + } + } else { + self.commit(cx); + } + } + + fn commit(&mut self, cx: &mut App) { + if let Some(spoken) = self.spoken() { + self.sink.phrase(spoken, cx); + } + self.phrase_ix += 1; + self.words = 0; + } +} + +impl RecognitionSession for DemoSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + self.samples += samples.len(); + while self.samples >= SAMPLES_PER_WORD { + self.samples -= SAMPLES_PER_WORD; + self.next_word(cx); + } + } + + fn finish(&mut self, cx: &mut App) { + self.commit(cx); + self.sink.finish(cx); + } +} + +/// Stands in for the microphone on the web, which the story cannot capture: +/// pushes a tone whose loudness rises and falls like speech. +#[cfg(target_family = "wasm")] +struct GeneratedInput; + +#[cfg(target_family = "wasm")] +impl gpui_kit::component::speech::AudioInput for GeneratedInput { + fn start( + &self, + format: gpui_kit::component::speech::AudioFormat, + sink: gpui_kit::component::speech::AudioSink, + cx: &mut App, + ) -> Result { + const CHUNK: std::time::Duration = std::time::Duration::from_millis(100); + let len = (format.sample_rate() / 10) as usize * format.channels() as usize; + let task = cx.spawn(async move |cx| { + let mut tick = 0u8; + loop { + cx.background_executor().timer(CHUNK).await; + tick = tick.wrapping_add(1); + let loudness = 0.05 + 0.25 * (tick as f32 * 0.9).sin().abs(); + let samples = (0..len) + .map(|ix| ((ix as f32 * 0.07).sin() * loudness * i16::MAX as f32) as i16) + .collect(); + cx.update(|cx| sink.push(samples, cx)); + } + }); + Ok(Subscription::new(move || drop(task))) + } +} + +/// A text field the user can dictate into. +struct Dictation { + speech: Entity, + input: Entity, +} + +impl Dictation { + fn new( + speech: Entity, + window: &mut Window, + cx: &mut Context, + ) -> (Self, Subscription) { + let input = cx.new(|cx| InputState::new(window, cx).placeholder("Type or dictate")); + let subscription = cx.subscribe_in(&speech, window, { + let input = input.clone(); + move |_, _, event, window, cx| match event { + SpeechEvent::Final(text) if !text.is_empty() => { + input.update(cx, |input, cx| input.insert(text.clone(), window, cx)); + } + SpeechEvent::Error(error) => { + window.push_notification(Notification::error(error.to_string()), cx); + } + _ => {} + } + }); + (Self { speech, input }, subscription) + } + + fn render(&self, show_when_unsupported: bool, cx: &App) -> impl IntoElement { + let speech = self.speech.read(cx); + let status = speech.status(); + + v_flex() + .w_full() + .gap_2() + .child( + Input::new(&self.input).suffix( + h_flex() + .gap_2() + .when(status.is_capturing(), |this| { + this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + }) + .child( + SpeechButton::new(&self.speech) + .xsmall() + .show_when_unsupported(show_when_unsupported), + ), + ), + ) + .when(status.is_active(), |this| { + this.child( + div() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child(speech.transcript()), + ) + }) + } +} + +pub struct SpeechStory { + focus_handle: FocusHandle, + custom: Dictation, + system: Dictation, + _subscriptions: Vec, +} + +impl super::Story for SpeechStory { + fn title() -> &'static str { + "Speech" + } + + fn description() -> &'static str { + "Dictate text through the system recognizer or your own." + } + + fn new_view(window: &mut Window, cx: &mut App) -> Entity { + Self::view(window, cx) + } +} + +impl SpeechStory { + pub fn view(window: &mut Window, cx: &mut App) -> Entity { + cx.new(|cx| Self::new(window, cx)) + } + + fn new(window: &mut Window, cx: &mut Context) -> Self { + let custom = cx.new(|cx| { + let state = SpeechState::new(cx).recognizer(DemoRecognizer); + #[cfg(target_family = "wasm")] + let state = state.input(GeneratedInput); + state + }); + let system = cx.new(SpeechState::new); + + let (custom, custom_subscription) = Dictation::new(custom, window, cx); + let (system, system_subscription) = Dictation::new(system, window, cx); + + Self { + focus_handle: cx.focus_handle(), + custom, + system, + _subscriptions: vec![custom_subscription, system_subscription], + } + } +} + +impl Focusable for SpeechStory { + fn focus_handle(&self, _: &App) -> FocusHandle { + self.focus_handle.clone() + } +} + +impl Render for SpeechStory { + fn render(&mut self, _: &mut Window, cx: &mut Context) -> impl IntoElement { + v_flex() + .size_full() + .justify_start() + .gap_3() + .child( + section("With a custom recognizer") + .description( + "A recognizer defined in this story types a scripted sentence \ + while it receives audio.", + ) + .w_128() + .child(self.custom.render(false, cx)), + ) + .child( + section("System recognizer") + .description( + "Recognizes speech on macOS and Windows. Elsewhere, and in an app \ + without the required usage descriptions, the button is disabled.", + ) + .w_128() + .child(self.system.render(true, cx)), + ) + } +} diff --git a/examples/speech/Cargo.toml b/examples/speech/Cargo.toml new file mode 100644 index 0000000000..b03aca0292 --- /dev/null +++ b/examples/speech/Cargo.toml @@ -0,0 +1,14 @@ +[package] +name = "speech" +description = "A dictation notepad for trying out and testing GPUI Component speech input." +version = "0.7.0" +publish = false +edition.workspace = true + +[dependencies] +anyhow.workspace = true +cpal = "0.15.3" +gpui-kit = { workspace = true, features = ["speech"] } + +[lints] +workspace = true diff --git a/examples/speech/Info.plist b/examples/speech/Info.plist new file mode 100644 index 0000000000..6bd6d5e02b --- /dev/null +++ b/examples/speech/Info.plist @@ -0,0 +1,10 @@ + + + + + NSMicrophoneUsageDescription + Dictation listens to the microphone to turn what you say into text. + NSSpeechRecognitionUsageDescription + Dictation recognizes speech on this Mac to type what you say. + + diff --git a/examples/speech/README.md b/examples/speech/README.md new file mode 100644 index 0000000000..eea9bbd187 --- /dev/null +++ b/examples/speech/README.md @@ -0,0 +1,62 @@ +# Speech + +Dictation, a notepad you can talk into, built on GPUI Component's speech input. It is also +the test bench for the platform recognizers: the sidebar shows what this machine +supports, and the session log records every `SpeechEvent`. + +```sh +cargo run -p speech # open the app +cargo run -p speech -- --check # print the checks and exit +``` + +`--check` prints the platform, the default input device, and whether the system +recognizer is available for each language in the picker: + +```text +Platform macOS +Input device MacBook Pro Microphone +System recognizer, by language: + en-US available English (US) + zh-CN available 简体中文 + ja-JP unavailable 日本語 +``` + +## Trying it + +- **Demo** types a scripted passage while it hears audio. It needs no service, + network, or speech permission, so it works on every platform, Linux included. + Use it to check the capture, waveform, and event flow. +- **System** uses the operating system's recognizer in the language you pick. +- Click the microphone or press ⇧⌘D (Ctrl+Shift+D on + Windows and Linux) to start, and again to stop. The text goes into the note + at the cursor. **Discard** stops without inserting anything. + +## Platform notes + +### macOS + +`build.rs` links `Info.plist` into the executable, so `cargo run` can ask for +microphone and speech recognition access without an app bundle. The first +session asks for both. Recognition stays on the Mac, so a language the Mac can't +recognize offline shows **Unavailable**. + +When you run from a terminal, macOS attributes the microphone to the terminal +app. To ask again after denying access: + +```sh +tccutil reset Microphone +tccutil reset SpeechRecognition +``` + +### Windows + +Install the language's speech pack (Settings › Time & language › Speech) and +turn on **Online speech recognition** (Settings › Privacy & security › Speech). +Dictation runs through Microsoft's online service. The recognizer records from +the default input device itself, and the waveform follows the same device. + +### Linux + +There is no system recognizer, so **System** shows **Not supported**. Use +**Demo**, or plug in your own `SpeechRecognizer`. Building needs +`libasound2-dev`. diff --git a/examples/speech/build.rs b/examples/speech/build.rs new file mode 100644 index 0000000000..81adfb3106 --- /dev/null +++ b/examples/speech/build.rs @@ -0,0 +1,21 @@ +//! Embed `Info.plist` into the macOS executable. +//! +//! macOS asks for microphone and speech recognition access only when the +//! application describes why it needs them. An unbundled `cargo run` binary has +//! no bundle to carry that description, but the system also reads a property +//! list linked into the executable's `__TEXT,__info_plist` section. +//! +//! The list holds only the two usage descriptions. A bundle identifier would +//! make frameworks treat the binary as an application bundle, and GPUI's system +//! notifications then fail to start without one. + +fn main() { + println!("cargo:rerun-if-changed=Info.plist"); + if std::env::var("CARGO_CFG_TARGET_OS").as_deref() == Ok("macos") { + let plist = std::path::Path::new(env!("CARGO_MANIFEST_DIR")).join("Info.plist"); + println!( + "cargo:rustc-link-arg-bins=-Wl,-sectcreate,__TEXT,__info_plist,{}", + plist.display() + ); + } +} diff --git a/examples/speech/src/demo.rs b/examples/speech/src/demo.rs new file mode 100644 index 0000000000..4ed1a58e79 --- /dev/null +++ b/examples/speech/src/demo.rs @@ -0,0 +1,166 @@ +//! A recognizer that needs no service, permission or network: it types a +//! scripted passage while audio arrives, so the whole flow can be tried +//! anywhere, including on Linux. + +use gpui_kit::App; +use gpui_kit::component::speech::{RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink}; + +/// Audio per recognized token: 0.25 s at 16 kHz. +const SAMPLES_PER_TOKEN: usize = 4_000; + +pub struct DemoRecognizer { + script: &'static Script, +} + +impl DemoRecognizer { + /// A recognizer typing the passage for `language`, a BCP 47 tag. + pub fn new(language: &str) -> Self { + let script = SCRIPTS + .iter() + .find(|script| language.starts_with(script.language)) + .unwrap_or(&SCRIPTS[0]); + Self { script } + } +} + +impl SpeechRecognizer for DemoRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // There is nothing to connect to, so audio is consumed at once. + sink.ready(cx); + Ok(Box::new(DemoSession { + sink, + script: self.script, + sentence_ix: 0, + tokens: 0, + samples: 0, + })) + } +} + +struct Script { + language: &'static str, + /// Put between tokens and between sentences. + separator: &'static str, + sentences: &'static [&'static [&'static str]], +} + +const SCRIPTS: &[Script] = &[ + Script { + language: "en", + separator: " ", + sentences: &[ + &[ + "Speech", "input", "turns", "what", "you", "say", "into", "text.", + ], + &[ + "Stop", "whenever", "you", "like,", "and", "the", "words", "land", "in", "the", + "note.", + ], + &[ + "Any", "speech", "service", "plugs", "in", "through", "one", "trait.", + ], + ], + }, + Script { + language: "zh", + separator: "", + sentences: &[ + &["语音", "输入", "会把", "你说", "的话", "转成", "文字。"], + &["随时", "停止,", "文字", "就会", "写进", "笔记。"], + &[ + "任何", + "识别", + "服务", + "都能", + "通过", + "一个", + "接口", + "接进来。", + ], + ], + }, + Script { + language: "ja", + separator: "", + sentences: &[ + &["話した", "言葉が", "そのまま", "文字に", "なります。"], + &["止めると", "ノートに", "書き込まれます。"], + ], + }, +]; + +struct DemoSession { + sink: SpeechSink, + script: &'static Script, + sentence_ix: usize, + /// Tokens of the current sentence heard so far. + tokens: usize, + /// Samples received since the last token. + samples: usize, +} + +impl DemoSession { + /// The heard part of the current sentence, with the separator that joins it + /// to the sentences before. + fn heard(&self) -> Option { + let sentence = self.sentence()?; + if self.tokens == 0 { + return None; + } + let leading = if self.sentence_ix > 0 { + self.script.separator + } else { + "" + }; + Some(format!( + "{leading}{}", + sentence[..self.tokens].join(self.script.separator) + )) + } + + fn sentence(&self) -> Option<&'static [&'static str]> { + let sentences = self.script.sentences; + sentences.get(self.sentence_ix % sentences.len()).copied() + } + + fn next_token(&mut self, cx: &mut App) { + let Some(sentence) = self.sentence() else { + return; + }; + self.tokens += 1; + if self.tokens < sentence.len() { + if let Some(heard) = self.heard() { + self.sink.hypothesis(heard, cx); + } + } else { + self.commit(cx); + } + } + + fn commit(&mut self, cx: &mut App) { + if let Some(heard) = self.heard() { + self.sink.phrase(heard, cx); + } + self.sentence_ix += 1; + self.tokens = 0; + } +} + +impl RecognitionSession for DemoSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + self.samples += samples.len(); + while self.samples >= SAMPLES_PER_TOKEN { + self.samples -= SAMPLES_PER_TOKEN; + self.next_token(cx); + } + } + + fn finish(&mut self, cx: &mut App) { + self.commit(cx); + self.sink.finish(cx); + } +} diff --git a/examples/speech/src/main.rs b/examples/speech/src/main.rs new file mode 100644 index 0000000000..746f69ec55 --- /dev/null +++ b/examples/speech/src/main.rs @@ -0,0 +1,845 @@ +//! Dictation: a notepad you can talk into, built on GPUI Component's speech +//! input. It doubles as a test bench for the platform recognizers: the sidebar +//! reports what this machine supports and the session log records every event. +//! +//! `cargo run -p speech` opens the app; `cargo run -p speech -- --check` +//! prints the same checks to the terminal and exits. + +mod demo; + +use std::{ + borrow::Cow, + time::{Duration, Instant}, +}; + +use cpal::traits::{DeviceTrait as _, HostTrait as _}; +use gpui_kit::assets::{Assets, IconName as ExtraIcon, icon_assets}; +use gpui_kit::component::{ + ActiveTheme as _, Disableable as _, Icon, IndexPath, Selectable as _, Sizable as _, + StyledExt as _, TitleBar, WindowExt as _, + button::{Button, ButtonGroup, ButtonVariants as _}, + h_flex, + input::{Textarea, TextareaState}, + kbd::Kbd, + notification::Notification, + scroll::ScrollableElement as _, + select::{SearchableVec, Select, SelectEvent, SelectItem, SelectState}, + speech::{ + SpeechButton, SpeechEvent, SpeechState, SpeechStatus, SpeechWaveform, SystemRecognizer, + }, + tag::Tag, + v_flex, +}; +use gpui_kit::prelude::FluentBuilder as _; +use gpui_kit::*; + +use demo::DemoRecognizer; + +icon_assets!(ExtraIcons, [AudioLines, ScrollText, Eraser]); + +/// The default component icons plus the few extras this app uses. +struct AppAssets; + +impl AssetSource for AppAssets { + fn load(&self, path: &str) -> Result>> { + if let Some(bytes) = ExtraIcons.load(path)? { + return Ok(Some(bytes)); + } + Assets.load(path) + } + + fn list(&self, path: &str) -> Result> { + let mut paths = Assets.list(path)?; + paths.extend(ExtraIcons.list(path)?); + Ok(paths) + } +} + +actions!(dictation, [ToggleDictation]); + +const TOGGLE_KEYS: &str = "secondary-shift-d"; +/// Most log entries kept; older ones scroll away. +const LOG_LIMIT: usize = 200; + +/// Where the text comes from. +#[derive(Debug, Clone, Copy, PartialEq, Eq)] +enum Engine { + /// The operating system's recognizer. + System, + /// [`DemoRecognizer`]: a scripted passage, no service needed. + Demo, +} + +#[derive(Clone)] +struct Language { + tag: SharedString, + name: SharedString, +} + +impl SelectItem for Language { + type Value = SharedString; + + fn title(&self) -> SharedString { + self.name.clone() + } + + fn value(&self) -> &Self::Value { + &self.tag + } +} + +fn languages() -> Vec { + [ + ("en-US", "English (US)"), + ("en-GB", "English (UK)"), + ("zh-CN", "简体中文"), + ("zh-HK", "中文(香港)"), + ("ja-JP", "日本語"), + ] + .into_iter() + .map(|(tag, name)| Language { + tag: tag.into(), + name: name.into(), + }) + .collect() +} + +/// One line of the session log. +struct LogEntry { + at: Duration, + tone: Tone, + label: &'static str, + text: SharedString, +} + +#[derive(Clone, Copy, PartialEq, Eq)] +enum Tone { + Neutral, + Progress, + Success, + Danger, +} + +fn new_speech(engine: Engine, language: &str, cx: &mut App) -> Entity { + let language = language.to_string(); + cx.new(|cx| { + let state = SpeechState::new(cx); + match engine { + Engine::System => state.recognizer(SystemRecognizer::new().locale(language)), + Engine::Demo => state.recognizer(DemoRecognizer::new(&language)), + } + }) +} + +/// The name of the default audio input device, if there is one. +fn input_device_name() -> Option { + let device = cpal::default_host().default_input_device()?; + Some( + device + .name() + .unwrap_or_else(|_| "Unnamed device".into()) + .into(), + ) +} + +fn platform_name() -> &'static str { + if cfg!(target_os = "macos") { + "macOS" + } else if cfg!(target_os = "windows") { + "Windows" + } else if cfg!(target_os = "linux") { + "Linux" + } else { + "Other" + } +} + +struct DictationApp { + focus_handle: FocusHandle, + engine: Engine, + language: SharedString, + speech: Entity, + notes: Entity, + language_select: Entity>>, + input_device: Option, + /// When the current or last session started. + session_started: Option, + /// When the app opened; log times count from here. + opened: Instant, + log: Vec, + _speech_subscription: Subscription, + _subscriptions: Vec, +} + +impl DictationApp { + fn new(window: &mut Window, cx: &mut Context) -> Self { + let engine = Engine::System; + let language = SharedString::from("en-US"); + let speech = new_speech(engine, &language, cx); + let _speech_subscription = Self::subscribe_speech(&speech, window, cx); + + let notes = cx.new(|cx| { + TextareaState::new(window, cx) + .placeholder("Start typing, or dictate with the microphone below.") + }); + let language_select = cx.new(|cx| { + SelectState::new( + SearchableVec::new(languages()), + Some(IndexPath::default()), + window, + cx, + ) + }); + let _subscriptions = vec![cx.subscribe_in( + &language_select, + window, + |this, _, event: &SelectEvent>, window, cx| { + if let SelectEvent::Confirm(Some(tag)) = event { + this.language = tag.clone(); + this.rebuild_speech(window, cx); + } + }, + )]; + notes.update(cx, |notes, cx| notes.focus(window, cx)); + + Self { + focus_handle: cx.focus_handle(), + engine, + language, + speech, + notes, + language_select, + input_device: input_device_name(), + session_started: None, + opened: Instant::now(), + log: Vec::new(), + _speech_subscription, + _subscriptions, + } + } + + fn subscribe_speech( + speech: &Entity, + window: &mut Window, + cx: &mut Context, + ) -> Subscription { + cx.subscribe_in( + speech, + window, + |this, _, event: &SpeechEvent, window, cx| this.on_speech_event(event, window, cx), + ) + } + + /// Recreate the speech state for the chosen engine and language. + fn rebuild_speech(&mut self, window: &mut Window, cx: &mut Context) { + self.speech.update(cx, |speech, cx| speech.cancel(cx)); + self.speech = new_speech(self.engine, &self.language, cx); + self._speech_subscription = Self::subscribe_speech(&self.speech, window, cx); + self.input_device = input_device_name(); + cx.notify(); + } + + fn set_engine(&mut self, engine: Engine, window: &mut Window, cx: &mut Context) { + if self.engine != engine { + self.engine = engine; + self.rebuild_speech(window, cx); + } + } + + fn toggle_dictation(&mut self, _: &ToggleDictation, _: &mut Window, cx: &mut Context) { + self.speech.update(cx, |speech, cx| speech.toggle(cx)); + } + + fn on_speech_event( + &mut self, + event: &SpeechEvent, + window: &mut Window, + cx: &mut Context, + ) { + match event { + SpeechEvent::Started => { + self.session_started = Some(Instant::now()); + self.push_log(Tone::Progress, "Started", "Listening".into()); + } + SpeechEvent::Partial(text) => { + // Partial results arrive many times a second; keep one line per + // stretch of them. + if let Some(last) = self.log.last_mut() + && last.label == "Partial" + { + last.text = text.clone(); + last.at = self.opened.elapsed(); + } else { + self.push_log(Tone::Neutral, "Partial", text.clone()); + } + } + SpeechEvent::Final(text) => { + if text.is_empty() { + self.push_log(Tone::Neutral, "Final", "No speech recognized".into()); + } else { + self.push_log(Tone::Success, "Final", text.clone()); + self.insert_into_notes(text, window, cx); + } + } + SpeechEvent::Cancelled => { + self.push_log(Tone::Neutral, "Cancelled", "Transcript discarded".into()); + } + SpeechEvent::Error(error) => { + self.push_log(Tone::Danger, "Error", error.to_string().into()); + window.push_notification( + Notification::error(format!("Couldn’t dictate. {error}.")), + cx, + ); + } + } + cx.notify(); + } + + fn insert_into_notes( + &mut self, + text: &SharedString, + window: &mut Window, + cx: &mut Context, + ) { + self.notes.update(cx, |notes, cx| { + // Keep dictated passages apart from what is already there. + let value = notes.value(); + let needs_space = value + .chars() + .next_back() + .is_some_and(|c| !c.is_whitespace() && c.is_ascii()); + let text = if needs_space { + format!(" {text}") + } else { + text.to_string() + }; + notes.insert(text, window, cx); + notes.focus(window, cx); + }); + } + + fn push_log(&mut self, tone: Tone, label: &'static str, text: SharedString) { + if self.log.len() == LOG_LIMIT { + self.log.remove(0); + } + self.log.push(LogEntry { + at: self.opened.elapsed(), + tone, + label, + text, + }); + } + + fn render_sidebar(&self, cx: &mut Context) -> impl IntoElement { + let speech = self.speech.read(cx); + let active = speech.status().is_active(); + let engine_note = match self.engine { + Engine::System => { + "The operating system’s recognizer. On macOS it only recognizes on this Mac; \ + on Windows it uses Microsoft’s online service." + } + Engine::Demo => { + "Types a scripted passage while it hears audio. Needs no service, network or \ + speech permission." + } + }; + + v_flex() + .w(px(272.)) + .h_full() + .flex_shrink_0() + .gap_6() + .p_4() + .bg(cx.theme().sidebar) + .text_color(cx.theme().sidebar_foreground) + .border_r_1() + .border_color(cx.theme().sidebar_border) + .child( + sidebar_section("Recognizer", cx) + .child( + ButtonGroup::new("engine") + .small() + .outline() + .w_full() + .disabled(active) + .child( + Button::new("engine-system") + .flex_1() + .label("System") + .selected(self.engine == Engine::System), + ) + .child( + Button::new("engine-demo") + .flex_1() + .label("Demo") + .selected(self.engine == Engine::Demo), + ) + .on_click(cx.listener(|this, clicks: &Vec, window, cx| { + let engine = if clicks.contains(&1) { + Engine::Demo + } else { + Engine::System + }; + this.set_engine(engine, window, cx); + })), + ) + .child(caption(engine_note, cx)), + ) + .child( + sidebar_section("Language", cx) + .child(Select::new(&self.language_select).small().disabled(active)), + ) + .child(sidebar_section("Checks", cx).child(self.render_checks(cx))) + } + + fn render_checks(&self, cx: &mut Context) -> impl IntoElement { + let speech = self.speech.read(cx); + let recognizer = if !speech.has_recognizer() { + Tag::danger().outline().child("Not supported") + } else if speech.is_available(cx) { + Tag::success().outline().child("Available") + } else { + Tag::warning().outline().child("Unavailable") + }; + let session = match speech.status() { + SpeechStatus::Idle => Tag::secondary().outline().child("Idle"), + SpeechStatus::Connecting => Tag::info().outline().child("Connecting"), + SpeechStatus::Recording => Tag::info().outline().child("Recording"), + SpeechStatus::Stopping => Tag::info().outline().child("Finishing"), + }; + let input = match &self.input_device { + Some(name) => div() + .min_w_0() + .truncate() + .child(name.clone()) + .into_any_element(), + None => Tag::danger() + .small() + .outline() + .child("None") + .into_any_element(), + }; + + v_flex() + .gap_2() + .text_sm() + .child(check_row("Platform", div().child(platform_name()), cx)) + .child(check_row("Input device", input, cx)) + .child(check_row("Recognizer", recognizer.small(), cx)) + .child(check_row("Session", session.small(), cx)) + .when( + self.engine == Engine::System + && speech.has_recognizer() + && !speech.is_available(cx), + |this| this.child(caption(unavailable_hint(), cx)), + ) + } + + fn render_notes_header(&self, cx: &mut Context) -> impl IntoElement { + let characters = self.notes.read(cx).value().chars().count(); + + h_flex() + .items_center() + .justify_between() + .child( + v_flex() + .child(div().text_lg().font_semibold().child("Notes")) + .child( + div() + .text_xs() + .text_color(cx.theme().muted_foreground) + .child(match characters { + 0 => "Empty".to_string(), + 1 => "1 character".to_string(), + n => format!("{n} characters"), + }), + ), + ) + .child( + Button::new("clear-notes") + .ghost() + .small() + .icon(Icon::new(ExtraIcon::Eraser)) + .label("Clear") + .disabled(characters == 0) + .on_click(cx.listener(|this, _, window, cx| { + this.notes + .update(cx, |notes, cx| notes.set_value("", window, cx)); + cx.notify(); + })), + ) + } + + /// The bar that runs dictation: button, level, live text and controls. + fn render_dictation_bar(&self, cx: &mut Context) -> impl IntoElement { + let speech = self.speech.read(cx); + let status = speech.status(); + let transcript = speech.transcript(); + let supported = speech.has_recognizer(); + let available = speech.is_available(cx); + let elapsed = self + .session_started + .filter(|_| status.is_active()) + .map(|started| format_duration(started.elapsed())); + + let (title, detail): (SharedString, SharedString) = match status { + SpeechStatus::Idle if !supported => ( + "Not supported here".into(), + "This platform has no system recognizer. Switch to Demo to try dictation.".into(), + ), + SpeechStatus::Idle if !available => ("Not available".into(), unavailable_hint().into()), + SpeechStatus::Idle => ( + "Ready".into(), + "Click the microphone to dictate into the note at the cursor.".into(), + ), + SpeechStatus::Connecting => ( + "Connecting…".into(), + non_empty( + transcript, + "Start talking. What you say is kept while it connects.", + ), + ), + SpeechStatus::Recording => ( + format!("Listening · {}", elapsed.unwrap_or_default()).into(), + non_empty(transcript, "Start talking."), + ), + SpeechStatus::Stopping => ( + "Finishing…".into(), + non_empty(transcript, "Waiting for the last words."), + ), + }; + let capturing = status.is_capturing(); + let toggle_keys = Keystroke::parse(TOGGLE_KEYS).ok().map(Kbd::new); + + h_flex() + .gap_3() + .px_3() + .py_2p5() + .items_center() + .rounded(cx.theme().radius_lg) + .border_1() + .border_color(if capturing { + cx.theme().ring + } else { + cx.theme().border + }) + .bg(cx.theme().background) + .when(capturing, |this| this.shadow_sm()) + .child( + SpeechButton::new(&self.speech) + .show_when_unsupported(true) + .large(), + ) + .child( + v_flex() + .flex_1() + .min_w_0() + .gap_0p5() + .child( + div() + .text_sm() + .font_medium() + .text_color(if supported && available || status.is_active() { + cx.theme().foreground + } else { + cx.theme().muted_foreground + }) + .child(title), + ) + .child( + div() + .text_sm() + .truncate() + .text_color(if status.is_active() && !speech.transcript().is_empty() { + cx.theme().foreground + } else { + cx.theme().muted_foreground + }) + .child(detail), + ), + ) + .when(status.is_active(), |this| { + this.child(SpeechWaveform::new(&self.speech).bars(28).small()) + }) + .map(|this| { + if status.is_active() { + this.child( + Button::new("discard") + .ghost() + .small() + .label("Discard") + .tooltip("Stop without inserting the text") + .on_click(cx.listener(|this, _, _, cx| { + this.speech.update(cx, |speech, cx| speech.cancel(cx)); + })), + ) + } else { + this.when_some(toggle_keys.filter(|_| available), |this, kbd| { + this.child(kbd) + }) + } + }) + } + + fn render_log(&self, cx: &mut Context) -> impl IntoElement { + v_flex() + .h(px(168.)) + .flex_shrink_0() + .rounded(cx.theme().radius_lg) + .border_1() + .border_color(cx.theme().border) + .overflow_hidden() + .child( + h_flex() + .px_3() + .py_1p5() + .items_center() + .justify_between() + .border_b_1() + .border_color(cx.theme().border) + .child( + h_flex() + .gap_2() + .items_center() + .text_xs() + .font_medium() + .text_color(cx.theme().muted_foreground) + .child(Icon::new(ExtraIcon::ScrollText).xsmall()) + .child("Session log"), + ) + .child( + Button::new("clear-log") + .ghost() + .xsmall() + .label("Clear") + .disabled(self.log.is_empty()) + .on_click(cx.listener(|this, _, _, cx| { + this.log.clear(); + cx.notify(); + })), + ), + ) + .child( + div().flex_1().min_h_0().child( + v_flex() + .id("session-log") + .size_full() + .px_3() + .py_2() + .gap_1() + .overflow_y_scrollbar() + .when(self.log.is_empty(), |this| { + this.items_center().justify_center().child( + div() + .text_xs() + .text_color(cx.theme().muted_foreground) + .child("Events of each session appear here."), + ) + }) + .children(self.log.iter().rev().map(|entry| log_row(entry, cx))), + ), + ) + } +} + +impl Focusable for DictationApp { + fn focus_handle(&self, _: &App) -> FocusHandle { + self.focus_handle.clone() + } +} + +impl Render for DictationApp { + fn render(&mut self, _: &mut Window, cx: &mut Context) -> impl IntoElement { + v_flex() + .size_full() + .bg(cx.theme().background) + .text_color(cx.theme().foreground) + .on_action(cx.listener(Self::toggle_dictation)) + .child( + TitleBar::new().child( + h_flex() + .gap_2() + .items_center() + .text_sm() + .font_medium() + .child( + Icon::new(ExtraIcon::AudioLines) + .small() + .text_color(cx.theme().primary), + ) + .child("Dictation"), + ), + ) + .child( + h_flex() + .flex_1() + .min_h_0() + .child(self.render_sidebar(cx)) + .child( + v_flex() + .flex_1() + .min_w_0() + .h_full() + .gap_4() + .p_5() + .child(self.render_notes_header(cx)) + .child( + div() + .flex_1() + .min_h_0() + .child(Textarea::new(&self.notes).h_full()), + ) + .child(self.render_dictation_bar(cx)) + .child(self.render_log(cx)), + ), + ) + } +} + +fn sidebar_section(title: &'static str, cx: &App) -> Div { + v_flex().gap_2().child( + div() + .text_xs() + .font_medium() + .text_color(cx.theme().muted_foreground) + .child(title), + ) +} + +fn caption(text: &'static str, cx: &App) -> impl IntoElement { + div() + .text_xs() + .text_color(cx.theme().muted_foreground) + .child(text) +} + +fn check_row(label: &'static str, value: impl IntoElement, cx: &App) -> impl IntoElement { + h_flex() + .gap_3() + .items_center() + .justify_between() + .child( + div() + .flex_shrink_0() + .text_color(cx.theme().muted_foreground) + .child(label), + ) + .child(h_flex().flex_1().min_w_0().justify_end().child(value)) +} + +fn log_row(entry: &LogEntry, cx: &App) -> impl IntoElement { + let color = match entry.tone { + Tone::Neutral => cx.theme().muted_foreground, + Tone::Progress => cx.theme().info, + Tone::Success => cx.theme().success, + Tone::Danger => cx.theme().danger, + }; + + h_flex() + .gap_3() + .items_start() + .text_xs() + .child( + div() + .flex_shrink_0() + .font_family(cx.theme().mono_font_family.clone()) + .text_color(cx.theme().muted_foreground) + .child(format_timestamp(entry.at)), + ) + .child( + div() + .w(px(64.)) + .flex_shrink_0() + .font_medium() + .text_color(color) + .child(entry.label), + ) + .child(div().flex_1().min_w_0().child(entry.text.clone())) +} + +fn unavailable_hint() -> &'static str { + if cfg!(target_os = "macos") { + "This Mac can’t recognize the language offline, or speech recognition access is off \ + in System Settings › Privacy & Security." + } else if cfg!(target_os = "windows") { + "Install the language’s speech pack and turn on Online speech recognition in Settings › \ + Privacy & security › Speech." + } else { + "The recognizer can’t start right now." + } +} + +fn non_empty(text: SharedString, fallback: &'static str) -> SharedString { + if text.is_empty() { + fallback.into() + } else { + text + } +} + +fn format_duration(duration: Duration) -> String { + let seconds = duration.as_secs(); + format!("{}:{:02}", seconds / 60, seconds % 60) +} + +fn format_timestamp(duration: Duration) -> String { + let millis = duration.as_millis(); + format!( + "{:02}:{:02}.{:03}", + millis / 60_000, + millis / 1_000 % 60, + millis % 1_000 + ) +} + +/// Print what this machine supports and quit. +fn print_checks(cx: &mut App) { + println!("Platform {}", platform_name()); + println!( + "Input device {}", + input_device_name().as_deref().unwrap_or("none") + ); + println!("System recognizer, by language:"); + for language in languages() { + let recognizer = SystemRecognizer::new().locale(language.tag.clone()); + let available = + gpui_kit::component::speech::SpeechRecognizer::is_available(&recognizer, cx); + println!( + " {:<7} {:<12} {}", + language.tag.as_ref(), + if available { + "available" + } else { + "unavailable" + }, + language.name.as_ref(), + ); + } +} + +fn main() { + let check = std::env::args().any(|arg| arg == "--check"); + let app = gpui_kit::application().with_assets(AppAssets); + + app.run(move |cx| { + gpui_kit::init(cx); + if check { + print_checks(cx); + cx.quit(); + return; + } + + cx.bind_keys([KeyBinding::new(TOGGLE_KEYS, ToggleDictation, None)]); + cx.activate(true); + + let window_options = WindowOptions { + window_bounds: Some(WindowBounds::centered(size(px(980.), px(680.)), cx)), + window_min_size: Some(size(px(760.), px(520.))), + ..TitleBar::window_options() + }; + gpui_kit::open_window(window_options, cx, |window, cx| { + cx.new(|cx| DictationApp::new(window, cx)) + }) + .expect("Failed to open window"); + }); +} diff --git a/script/install-linux.sh b/script/install-linux.sh index e8a2cde716..594d7d3447 100755 --- a/script/install-linux.sh +++ b/script/install-linux.sh @@ -5,5 +5,5 @@ sudo apt update sudo apt install -y \ gcc g++ clang libfontconfig-dev libwayland-dev \ libwebkit2gtk-4.1-dev libxkbcommon-x11-dev libx11-xcb-dev \ - libssl-dev libzstd-dev \ + libssl-dev libzstd-dev libasound2-dev \ vulkan-validationlayers libvulkan1 diff --git a/skills/gpui-kit/SKILL.md b/skills/gpui-kit/SKILL.md index 6832cc2ccd..a67265bd57 100644 --- a/skills/gpui-kit/SKILL.md +++ b/skills/gpui-kit/SKILL.md @@ -133,6 +133,7 @@ fetch the component's `.md` doc. | `Editor` | `input::{Editor, EditorState}` | Stateful. Code editor, `tree-sitter` feature | | `NumberInput` | `input::{NumberInput, NumberInputEvent}` | Stateful. Numeric with step | | `OtpInput` | `input::OtpInput` | Stateful. One-time password | +| `SpeechButton` | `speech::{SpeechButton, SpeechState}` | Stateful. Dictation, `speech` feature | | `Select` | `select::{Select, SelectState}` | Stateful. Dropdown picker | | `Combobox` | `combobox::{Combobox, ComboboxState}` | Stateful. Searchable select | | `Checkbox` | `checkbox::Checkbox` | Stateless. `on_click` receives `&bool` | diff --git a/website/component/index.md b/website/component/index.md index 8ea400c11e..4bb6e390b5 100644 --- a/website/component/index.md +++ b/website/component/index.md @@ -52,6 +52,7 @@ collapsed: false - [DatePicker](date-picker) - Date selection with calendar - [TimeField](time-field) - Segmented time-of-day input - [OtpInput](otp-input) - One-time password input +- [Speech](speech) - Dictation through the system recognizer or your own - [ColorPicker](color-picker) - Color selection interface - [Form](form) - Form container and layout diff --git a/website/component/speech.md b/website/component/speech.md new file mode 100644 index 0000000000..102a83cfff --- /dev/null +++ b/website/component/speech.md @@ -0,0 +1,408 @@ +--- +title: Speech +description: Dictate text from the microphone through the system speech recognizer or any recognizer the application provides. +maturity: [experimental, platform-dependent] +--- + +# Speech + +The speech module turns what the user says into text. `SpeechState` owns a +dictation session: it captures audio from an `AudioInput`, feeds it to a +`SpeechRecognizer`, keeps the transcript, and emits `SpeechEvent`s. +`SpeechButton` starts and stops the session and `SpeechWaveform` shows the +input level while it captures. + +The application owns where the text goes. The state never edits an input on its +own; subscribe to its events and insert the final transcript where it belongs. + +Recognition and capture are both replaceable. Without further setup, the state +uses the recognizer built into macOS or Windows and the default microphone. +Implement `SpeechRecognizer` to use a cloud service or a local model, and +`AudioInput` to feed audio from elsewhere. + +## Enable the feature + +The microphone and the system recognizer are behind the `speech` feature: + +```toml +[dependencies] +gpui-kit = { version = "{{gpui_kit_version}}", features = ["speech"] } +``` + +The feature adds [cpal](https://crates.io/crates/cpal) for audio capture. On +Linux, building it needs the ALSA development package (`libasound2-dev` on +Debian and Ubuntu). + +Without the feature, and on the web, the types are still available but there is +no default input and no system recognizer. A state then works only with both an +application recognizer and an application input. + +## Import + +```rust +use gpui_kit::component::speech::{ + SpeechButton, SpeechEvent, SpeechState, SpeechStatus, SpeechWaveform, +}; +``` + +## Usage + +### Dictate into an input + +Create the state beside the input it fills, subscribe to it, and put the button +and waveform in the input's suffix: + +```rust +use gpui_kit::component::{ + WindowExt as _, h_flex, + input::{Input, InputState}, + notification::Notification, + speech::{SpeechButton, SpeechEvent, SpeechState, SpeechWaveform}, +}; + +let input = cx.new(|cx| InputState::new(window, cx)); +let speech = cx.new(SpeechState::new); + +let subscription = cx.subscribe_in(&speech, window, { + let input = input.clone(); + move |_, _, event, window, cx| match event { + SpeechEvent::Final(text) if !text.is_empty() => { + input.update(cx, |input, cx| input.insert(text.clone(), window, cx)); + } + SpeechEvent::Error(error) => { + window.push_notification(Notification::error(error.to_string()), cx); + } + _ => {} + } +}); +``` + +```rust +let capturing = self.speech.read(cx).status().is_capturing(); + +Input::new(&self.input).suffix( + h_flex() + .gap_2() + .when(capturing, |this| { + this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + }) + .child(SpeechButton::new(&self.speech).xsmall()), +) +``` + +Keep the subscription on the view that owns the input, for example in its +`_subscriptions` list. + +### Events + +| Event | When | +| --- | --- | +| `Started` | A session started and audio is being captured. | +| `Partial(text)` | The transcript changed while the user speaks. | +| `Final(text)` | The session ended normally. The text may be empty. | +| `Cancelled` | The session was cancelled and its transcript discarded. | +| `Error(error)` | The session failed and ended. | + +`Partial` and `Final` carry the **whole** transcript of the session so far: +every committed phrase followed by the current hypothesis. A later event +replaces an earlier one, so show the latest `Partial` text as a preview and +commit only the `Final` text. `SpeechState::transcript()` returns the same text +during render: + +```rust +let speech = self.speech.read(cx); + +v_flex() + .gap_2() + .child(Input::new(&self.input)) + .when(speech.status().is_active(), |this| { + this.child( + div() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child(speech.transcript()), + ) + }) +``` + +### Session control + +`SpeechButton` toggles the session. To drive it from an action, a key binding, +or another control, call the state directly: + +```rust +speech.update(cx, |speech, cx| speech.start(cx)); // Does nothing while running. +speech.update(cx, |speech, cx| speech.stop(cx)); // Emits `Final` once the result is in. +speech.update(cx, |speech, cx| speech.cancel(cx)); // Emits `Cancelled` at once. +speech.update(cx, |speech, cx| speech.toggle(cx)); // `start` when idle, otherwise `stop`. +``` + +`status()` reports where the session is: + +| `SpeechStatus` | Meaning | +| --- | --- | +| `Idle` | No session is running. | +| `Connecting` | Audio is captured while the recognizer connects. | +| `Recording` | Audio is captured and recognized. | +| `Stopping` | Capture stopped; waiting for the final result. | + +`is_active()` is true for every status but `Idle`, and `is_capturing()` for +`Connecting` and `Recording`. While `Stopping`, the button shows a spinner and +ignores clicks. If the recognizer does not deliver its final result within the +stop timeout, the session ends with the transcript so far. The timeout is +3 seconds by default: + +```rust +let speech = cx.new(|cx| SpeechState::new(cx).stop_timeout(Duration::from_secs(5))); +``` + +### Button states + +`SpeechButton` is a ghost icon `Button` with a localized tooltip and accessible +name: a microphone at rest, and a pressed stop glyph while capturing. + +- When the state has no recognizer or no input on this platform + (`has_recognizer()` is `false`), the button renders **nothing**, so an + application can place it unconditionally. Use `.show_when_unsupported(true)` + to render it disabled instead. +- When the recognizer reports itself unavailable (`is_available(cx)` is + `false`), for example because the language is not installed, the button is + disabled with an “unavailable” tooltip. +- `.disabled(true)` disables it for application reasons, such as a readonly + field. + +```rust +SpeechButton::new(&speech) + .small() + .show_when_unsupported(true) + .disabled(readonly) +``` + +## Choose a recognizer + +The state picks its recognizer in this order: + +1. the one passed to `.recognizer(...)`; +2. otherwise the platform's `SystemRecognizer`, unless `.system_fallback(false)` + turned it off; +3. otherwise none, and `has_recognizer()` is `false`. + +```rust +use gpui_kit::component::speech::SystemRecognizer; + +// Dictate Chinese regardless of the system language. +let speech = cx.new(|cx| { + SpeechState::new(cx).recognizer(SystemRecognizer::new().locale("zh-CN")) +}); + +// Use the application's own recognizer instead of the system's. +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(CloudRecognizer::new(client))); + +// Never fall back to the system recognizer. Without a recognizer of its own, +// `has_recognizer()` is false and the button renders nothing. +let speech = cx.new(|cx| SpeechState::new(cx).system_fallback(false)); +``` + +`CloudRecognizer` stands for an application type that implements +`SpeechRecognizer`; see [Implement a recognizer](#implement-a-recognizer). + +`SystemRecognizer::new()` recognizes the system's current language; +`.locale(...)` takes a BCP 47 tag such as `en-US` or `zh-CN`. A language the +platform cannot recognize makes the recognizer unavailable. + +## Implement a recognizer + +Implement `SpeechRecognizer` to use any speech service. `start` opens a +`RecognitionSession` and reports results through the `SpeechSink` it receives: + +```rust +use gpui_kit::App; +use gpui_kit::component::speech::{ + RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink, +}; + +/// Reports how much audio it has heard, as a stand-in for a real service. +struct DurationRecognizer; + +impl SpeechRecognizer for DurationRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // A service would connect here and call `ready` once connected. + sink.ready(cx); + Ok(Box::new(DurationSession { sink, samples: 0 })) + } +} + +struct DurationSession { + sink: SpeechSink, + samples: usize, +} + +impl RecognitionSession for DurationSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + // 16 kHz mono, the default `audio_format`. + self.samples += samples.len(); + let seconds = self.samples / 16_000; + self.sink.hypothesis(format!("{seconds} s of audio"), cx); + } + + fn finish(&mut self, cx: &mut App) { + let seconds = self.samples / 16_000; + self.sink.phrase(format!("{seconds} s of audio"), cx); + self.sink.finish(cx); + } +} +``` + +A session runs as follows: + +1. **`start`** opens the session. Connecting may take a while, so return at + once, buffer the audio pushed in the meantime, and call `sink.ready(cx)` + when the service accepts audio. The status moves from `Connecting` to + `Recording`. +2. **`push_audio`** delivers interleaved 16-bit PCM in the recognizer's + `audio_format()`, 16 kHz mono by default. Report the phrase being spoken + with `sink.hypothesis(...)`, which replaces the previous hypothesis, and a + recognized phrase with `sink.phrase(...)`, which commits it and clears the + hypothesis. +3. **`finish`** means the user stopped talking: send the remaining audio, wait + for the last result, then call `sink.finish(cx)`. The state then emits + `Final`. +4. **Dropping** the session cancels it. Close the connection there and report + nothing more. + +Report a failure with `sink.error(SpeechError::recognizer(error), cx)`; the +session ends with `SpeechEvent::Error`. + +A few rules keep a recognizer simple: + +- The sink is cheap to clone, so a task that reads the service's responses can + own a copy. Once its session is stopped, cancelled, or replaced, its calls are + ignored, so a late response cannot leak into the next session. +- Sink calls are applied after the current update. They may be made from + anywhere on the main thread, including from inside `start` or `push_audio`. +- Phrases are joined verbatim. Include any separator the language needs, such + as a leading space between English sentences and none between Chinese ones. +- Override `is_available` to return `false` while the recognizer cannot work, + for example before the user signs in. It is called on every render, so keep + it cheap. + +## Provide audio + +`AudioInput` is the audio seam. The built-in `Microphone` captures the default +input device and converts it to the recognizer's format. Replace it to feed +audio from elsewhere, such as a file in tests: + +```rust +use std::time::Duration; + +use gpui_kit::{App, Subscription}; +use gpui_kit::component::speech::{AudioFormat, AudioInput, AudioSink, SpeechError}; + +/// Pushes 100 ms of silence at a time. +struct Silence; + +impl AudioInput for Silence { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + let len = format.sample_rate() as usize / 10 * format.channels() as usize; + let task = cx.spawn(async move |cx| { + loop { + cx.background_executor().timer(Duration::from_millis(100)).await; + cx.update(|cx| sink.push(vec![0; len], cx)); + } + }); + // Capture runs until the state drops this subscription. + Ok(Subscription::new(move || drop(task))) + } +} + +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(recognizer).input(Silence)); +``` + +`levels()` exposes the recent input levels that `SpeechWaveform` draws, in +`0.0..=1.0` with the oldest first, for an application that renders its own +meter. + +## Platform support + +| Platform | `SystemRecognizer` | `Microphone` | +| --- | --- | --- | +| macOS | `SFSpeechRecognizer`, on the device only | Core Audio | +| Windows | `Windows.Media.SpeechRecognition`, through Microsoft's online service | WASAPI | +| Linux | None; provide a recognizer | ALSA | +| Web | None | None; provide an input | + +### macOS + +The application's `Info.plist` must describe why it uses the microphone and +speech recognition: + +```xml +NSMicrophoneUsageDescription +Dictate text into messages. +NSSpeechRecognitionUsageDescription +Turn your speech into text. +``` + +Without `NSMicrophoneUsageDescription`, the system refuses microphone access +without asking. Without `NSSpeechRecognitionUsageDescription`, the system +recognizer reports itself unavailable instead of asking, because asking without +it would terminate the process. A binary run outside an app bundle, such as +from `cargo run`, has neither key. + +Recognition runs on the device only, so audio never leaves the machine. A +language the Mac cannot recognize offline is unavailable. + +### Windows + +Dictation needs the language's speech pack and the **Online speech +recognition** setting in **Settings > Privacy & security > Speech**. Audio is +sent to Microsoft's online speech service. An application that must keep audio +on the device should use `.system_fallback(false)` with its own recognizer. + +The Windows recognizer listens to the default microphone itself. The state +still captures through its input, but only to drive the waveform, so a custom +`AudioInput` does not change what the system recognizer hears. + +### Linux + +Linux has no system recognizer. Speech input works there only with an +application recognizer; without one, `SpeechButton` renders nothing. The +`Microphone` captures through ALSA, and building it needs `libasound2-dev`. + +## API Reference + +- [SpeechState] — the session: `recognizer`, `input`, `system_fallback`, + `stop_timeout`, `start`, `stop`, `cancel`, `toggle`, `status`, + `has_recognizer`, `is_available`, `transcript`, `levels` +- [SpeechEvent] and [SpeechStatus] +- [SpeechButton] — `show_when_unsupported`, plus `Sizable` and `Disableable` +- [SpeechWaveform] — `bars` (default 24, at most 48), plus `Sizable` +- [SpeechRecognizer], [RecognitionSession] and [SpeechSink] — the recognition + seam +- [AudioInput], [AudioSink] and [AudioFormat] — the audio seam +- [SpeechError] +- [SystemRecognizer] and [Microphone] — the `speech` feature's defaults + +[SpeechState]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechState.html +[SpeechEvent]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechEvent.html +[SpeechStatus]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechStatus.html +[SpeechButton]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechButton.html +[SpeechWaveform]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechWaveform.html +[SpeechRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.SpeechRecognizer.html +[RecognitionSession]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.RecognitionSession.html +[SpeechSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechSink.html +[AudioInput]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.AudioInput.html +[AudioSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioSink.html +[AudioFormat]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioFormat.html +[SpeechError]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechError.html +[SystemRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SystemRecognizer.html +[Microphone]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.Microphone.html diff --git a/website/docs/installation.md b/website/docs/installation.md index 09aada789c..477fa2f819 100644 --- a/website/docs/installation.md +++ b/website/docs/installation.md @@ -31,8 +31,8 @@ Install the native toolchain for your operating system, then add the `gpui-kit`
sudo apt update
 sudo apt install -y gcc g++ clang libfontconfig-dev libwayland-dev \
   libwebkit2gtk-4.1-dev libxkbcommon-x11-dev libx11-xcb-dev \
-  libssl-dev libzstd-dev vulkan-validationlayers libvulkan1
-

This matches the repository's script/install-linux.sh for Ubuntu 24.04. Other distributions need equivalent development packages. To display a window, run in a graphical Wayland or X11 session with a working Vulkan driver; installing libvulkan1 alone does not install a GPU driver.

+ libssl-dev libzstd-dev libasound2-dev vulkan-validationlayers libvulkan1 +

This matches the repository's script/install-linux.sh for Ubuntu 24.04. Other distributions need equivalent development packages. libasound2-dev (ALSA) is needed only by the speech feature. To display a window, run in a graphical Wayland or X11 session with a working Vulkan driver; installing libvulkan1 alone does not install a GPU driver.

diff --git a/website/zh-CN/component/index.md b/website/zh-CN/component/index.md index 70fd697422..2ace1f5a94 100644 --- a/website/zh-CN/component/index.md +++ b/website/zh-CN/component/index.md @@ -36,6 +36,7 @@ collapsed: false - [DatePicker](date-picker) - 日期选择器 - [TimeField](time-field) - 分段时间输入 - [OtpInput](otp-input) - 一次性验证码输入 +- [Speech](speech) - 通过系统或自定义识别器进行语音输入 - [ColorPicker](color-picker) - 颜色选择器 - [Questionnaire](questionnaire) - 可组合的多步骤问卷与答案 - [Form](form) - 表单容器与布局 diff --git a/website/zh-CN/component/speech.md b/website/zh-CN/component/speech.md new file mode 100644 index 0000000000..f6b3936c7b --- /dev/null +++ b/website/zh-CN/component/speech.md @@ -0,0 +1,336 @@ +--- +title: Speech +description: 通过系统语音识别器或应用自己的识别器,把麦克风中的语音转成文本。 +maturity: [experimental, platform-dependent] +--- + +# Speech + +Speech 模块把用户说的话转成文本。`SpeechState` 管理一次语音输入会话:它从 `AudioInput` 采集音频,交给 `SpeechRecognizer` 识别,保存转写文本并发出 `SpeechEvent`。`SpeechButton` 负责开始和停止会话,`SpeechWaveform` 在采集期间显示输入音量。 + +文本写到哪里由应用决定。状态本身不会修改任何输入框;应用订阅它的事件,再把最终文本插入到合适的位置。 + +识别和采集都可以替换。不做额外配置时,状态使用 macOS 或 Windows 自带的识别器和默认麦克风。实现 `SpeechRecognizer` 可以接入云端服务或本地模型,实现 `AudioInput` 可以从其他来源提供音频。 + +## 启用 feature + +麦克风和系统识别器位于 `speech` feature 之后: + +```toml +[dependencies] +gpui-kit = { version = "{{gpui_kit_version}}", features = ["speech"] } +``` + +该 feature 会引入 [cpal](https://crates.io/crates/cpal) 采集音频。在 Linux 上构建需要 ALSA 开发包(Debian 和 Ubuntu 上为 `libasound2-dev`)。 + +未启用该 feature 时,以及在 Web 上,这些类型依然可用,但没有默认输入,也没有系统识别器。此时状态只有在应用同时提供识别器和输入时才能工作。 + +## 导入 + +```rust +use gpui_kit::component::speech::{ + SpeechButton, SpeechEvent, SpeechState, SpeechStatus, SpeechWaveform, +}; +``` + +## 用法 + +### 语音输入到 Input + +在要填充的输入框旁边创建状态并订阅它,再把按钮和波形放到输入框的 suffix 中: + +```rust +use gpui_kit::component::{ + WindowExt as _, h_flex, + input::{Input, InputState}, + notification::Notification, + speech::{SpeechButton, SpeechEvent, SpeechState, SpeechWaveform}, +}; + +let input = cx.new(|cx| InputState::new(window, cx)); +let speech = cx.new(SpeechState::new); + +let subscription = cx.subscribe_in(&speech, window, { + let input = input.clone(); + move |_, _, event, window, cx| match event { + SpeechEvent::Final(text) if !text.is_empty() => { + input.update(cx, |input, cx| input.insert(text.clone(), window, cx)); + } + SpeechEvent::Error(error) => { + window.push_notification(Notification::error(error.to_string()), cx); + } + _ => {} + } +}); +``` + +```rust +let capturing = self.speech.read(cx).status().is_capturing(); + +Input::new(&self.input).suffix( + h_flex() + .gap_2() + .when(capturing, |this| { + this.child(SpeechWaveform::new(&self.speech).bars(12).xsmall()) + }) + .child(SpeechButton::new(&self.speech).xsmall()), +) +``` + +把订阅保存在拥有该输入框的视图上,例如放进它的 `_subscriptions` 列表。 + +### 事件 + +| 事件 | 触发时机 | +| --- | --- | +| `Started` | 会话已开始,正在采集音频。 | +| `Partial(text)` | 用户说话期间转写文本发生变化。 | +| `Final(text)` | 会话正常结束。文本可能为空。 | +| `Cancelled` | 会话被取消,转写文本被丢弃。 | +| `Error(error)` | 会话失败并结束。 | + +`Partial` 和 `Final` 携带的是本次会话到目前为止的**完整**转写文本:所有已确认的短语,加上当前的识别假设。后一个事件会取代前一个,因此把最新的 `Partial` 文本作为预览显示,只提交 `Final` 文本。在 render 中,`SpeechState::transcript()` 返回同样的文本: + +```rust +let speech = self.speech.read(cx); + +v_flex() + .gap_2() + .child(Input::new(&self.input)) + .when(speech.status().is_active(), |this| { + this.child( + div() + .text_sm() + .text_color(cx.theme().muted_foreground) + .child(speech.transcript()), + ) + }) +``` + +### 控制会话 + +`SpeechButton` 会切换会话。如果要从 Action、快捷键或其他控件触发,直接调用状态的方法: + +```rust +speech.update(cx, |speech, cx| speech.start(cx)); // 会话进行中时不做任何事。 +speech.update(cx, |speech, cx| speech.stop(cx)); // 结果到达后发出 `Final`。 +speech.update(cx, |speech, cx| speech.cancel(cx)); // 立即发出 `Cancelled`。 +speech.update(cx, |speech, cx| speech.toggle(cx)); // 空闲时 `start`,否则 `stop`。 +``` + +`status()` 表示会话所处的阶段: + +| `SpeechStatus` | 含义 | +| --- | --- | +| `Idle` | 没有进行中的会话。 | +| `Connecting` | 正在采集音频,识别器仍在连接。 | +| `Recording` | 正在采集并识别音频。 | +| `Stopping` | 已停止采集,等待最终结果。 | + +除 `Idle` 外,`is_active()` 都为 true;`is_capturing()` 只在 `Connecting` 和 `Recording` 时为 true。处于 `Stopping` 时,按钮显示 spinner 并忽略点击。如果识别器在停止超时内没有给出最终结果,会话会以当前已有的文本结束。超时默认为 3 秒: + +```rust +let speech = cx.new(|cx| SpeechState::new(cx).stop_timeout(Duration::from_secs(5))); +``` + +### 按钮状态 + +`SpeechButton` 是一个 ghost 样式的图标 `Button`,带有本地化的 tooltip 和无障碍名称:空闲时显示麦克风,采集时显示按下状态的停止图标。 + +- 当状态在当前平台上没有识别器或没有输入(`has_recognizer()` 为 `false`)时,按钮**不渲染任何内容**,应用可以无条件放置它。使用 `.show_when_unsupported(true)` 可以改为渲染禁用的按钮。 +- 当识别器报告自己不可用(`is_available(cx)` 为 `false`),例如语言未安装时,按钮会禁用,tooltip 提示不可用。 +- `.disabled(true)` 用于应用自身的禁用原因,例如 readonly 的输入框。 + +```rust +SpeechButton::new(&speech) + .small() + .show_when_unsupported(true) + .disabled(readonly) +``` + +## 选择识别器 + +状态按以下顺序选择识别器: + +1. 通过 `.recognizer(...)` 传入的识别器; +2. 否则使用平台的 `SystemRecognizer`,除非 `.system_fallback(false)` 关闭了回退; +3. 否则没有识别器,`has_recognizer()` 为 `false`。 + +```rust +use gpui_kit::component::speech::SystemRecognizer; + +// 无论系统语言是什么,都识别中文。 +let speech = cx.new(|cx| { + SpeechState::new(cx).recognizer(SystemRecognizer::new().locale("zh-CN")) +}); + +// 使用应用自己的识别器代替系统识别器。 +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(CloudRecognizer::new(client))); + +// 不回退到系统识别器。没有自己的识别器时, +// `has_recognizer()` 为 false,按钮不渲染任何内容。 +let speech = cx.new(|cx| SpeechState::new(cx).system_fallback(false)); +``` + +`CloudRecognizer` 代表应用中实现了 `SpeechRecognizer` 的类型,参见[实现识别器](#实现识别器)。 + +`SystemRecognizer::new()` 识别系统当前语言;`.locale(...)` 接受 BCP 47 语言标签,例如 `en-US` 或 `zh-CN`。平台无法识别的语言会使识别器不可用。 + +## 实现识别器 + +实现 `SpeechRecognizer` 即可接入任意语音服务。`start` 打开一个 `RecognitionSession`,并通过收到的 `SpeechSink` 报告结果: + +```rust +use gpui_kit::App; +use gpui_kit::component::speech::{ + RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink, +}; + +/// 报告已收到多少音频,用来代替真实服务。 +struct DurationRecognizer; + +impl SpeechRecognizer for DurationRecognizer { + fn start( + &self, + sink: SpeechSink, + cx: &mut App, + ) -> Result, SpeechError> { + // 真实服务在这里建立连接,连接成功后再调用 `ready`。 + sink.ready(cx); + Ok(Box::new(DurationSession { sink, samples: 0 })) + } +} + +struct DurationSession { + sink: SpeechSink, + samples: usize, +} + +impl RecognitionSession for DurationSession { + fn push_audio(&mut self, samples: &[i16], cx: &mut App) { + // 16 kHz 单声道,即默认的 `audio_format`。 + self.samples += samples.len(); + let seconds = self.samples / 16_000; + self.sink.hypothesis(format!("{seconds} s of audio"), cx); + } + + fn finish(&mut self, cx: &mut App) { + let seconds = self.samples / 16_000; + self.sink.phrase(format!("{seconds} s of audio"), cx); + self.sink.finish(cx); + } +} +``` + +一次会话的流程如下: + +1. **`start`** 打开会话。连接可能需要一段时间,因此应立即返回,缓存这期间推送的音频,并在服务开始接收音频时调用 `sink.ready(cx)`。状态随之从 `Connecting` 变为 `Recording`。 +2. **`push_audio`** 按识别器的 `audio_format()`(默认 16 kHz 单声道)传入交错的 16 位 PCM。用 `sink.hypothesis(...)` 报告正在说的短语,它会替换上一个识别假设;用 `sink.phrase(...)` 报告识别完成的短语,它会确认该短语并清空识别假设。 +3. **`finish`** 表示用户已停止说话:发送剩余音频,等待最后的结果,然后调用 `sink.finish(cx)`。状态随后发出 `Final`。 +4. **drop** 会话即取消会话。在 drop 时关闭连接,之后不再报告任何内容。 + +出错时调用 `sink.error(SpeechError::recognizer(error), cx)`,会话以 `SpeechEvent::Error` 结束。 + +以下几点能让识别器保持简单: + +- sink 可以低成本 clone,读取服务响应的任务可以持有一份。会话停止、取消或被替换后,它的调用会被忽略,迟到的响应不会混入下一次会话。 +- sink 的调用在当前 update 结束后生效。可以在主线程的任何位置调用,包括在 `start` 或 `push_audio` 内部。 +- 短语按原样拼接。请带上语言需要的分隔符,例如英文句子之间的前导空格;中文句子之间则不需要。 +- 识别器暂时无法工作时(例如用户登录之前),重写 `is_available` 返回 `false`。它在每次 render 时都会调用,应保持轻量。 + +## 提供音频 + +`AudioInput` 是音频的扩展点。内置的 `Microphone` 从默认输入设备采集,并转换为识别器需要的格式。如需从其他来源提供音频(例如测试中的文件),替换它即可: + +```rust +use std::time::Duration; + +use gpui_kit::{App, Subscription}; +use gpui_kit::component::speech::{AudioFormat, AudioInput, AudioSink, SpeechError}; + +/// 每次推送 100 ms 的静音。 +struct Silence; + +impl AudioInput for Silence { + fn start( + &self, + format: AudioFormat, + sink: AudioSink, + cx: &mut App, + ) -> Result { + let len = format.sample_rate() as usize / 10 * format.channels() as usize; + let task = cx.spawn(async move |cx| { + loop { + cx.background_executor().timer(Duration::from_millis(100)).await; + cx.update(|cx| sink.push(vec![0; len], cx)); + } + }); + // 采集一直持续到状态 drop 这个 subscription。 + Ok(Subscription::new(move || drop(task))) + } +} + +let speech = cx.new(|cx| SpeechState::new(cx).recognizer(recognizer).input(Silence)); +``` + +`levels()` 返回 `SpeechWaveform` 绘制所用的近期输入音量,取值 `0.0..=1.0`,按时间从旧到新排列,适合需要自绘音量表的应用。 + +## 平台支持 + +| 平台 | `SystemRecognizer` | `Microphone` | +| --- | --- | --- | +| macOS | `SFSpeechRecognizer`,仅在本机识别 | Core Audio | +| Windows | `Windows.Media.SpeechRecognition`,经由 Microsoft 在线服务 | WASAPI | +| Linux | 无,需由应用提供识别器 | ALSA | +| Web | 无 | 无,需由应用提供输入 | + +### macOS + +应用的 `Info.plist` 必须说明使用麦克风和语音识别的原因: + +```xml +NSMicrophoneUsageDescription +用于在消息中语音输入文字。 +NSSpeechRecognitionUsageDescription +用于把你的语音转成文字。 +``` + +缺少 `NSMicrophoneUsageDescription` 时,系统会直接拒绝麦克风访问,不会询问用户。缺少 `NSSpeechRecognitionUsageDescription` 时,系统识别器会报告不可用,而不是发起询问,因为缺少该键时发起询问会导致进程终止。在 app bundle 之外运行的二进制(例如通过 `cargo run` 启动)不包含这两个键。 + +识别只在本机进行,音频不会离开设备。Mac 无法离线识别的语言不可用。 + +### Windows + +语音输入需要安装对应语言的语音包,并在**设置 > 隐私和安全性 > 语音**中打开**联机语音识别**。音频会发送到 Microsoft 的在线语音服务。音频必须保留在本机的应用应使用 `.system_fallback(false)`,并提供自己的识别器。 + +Windows 识别器会自行监听默认麦克风。状态仍通过其输入采集音频,但只用于驱动波形,因此自定义 `AudioInput` 不会改变系统识别器听到的内容。 + +### Linux + +Linux 没有系统识别器,只有在应用提供识别器时才能使用语音输入;否则 `SpeechButton` 不渲染任何内容。`Microphone` 通过 ALSA 采集,构建时需要 `libasound2-dev`。 + +## API 参考 + +- [SpeechState]:会话本身,包括 `recognizer`、`input`、`system_fallback`、`stop_timeout`、`start`、`stop`、`cancel`、`toggle`、`status`、`has_recognizer`、`is_available`、`transcript`、`levels` +- [SpeechEvent] 与 [SpeechStatus] +- [SpeechButton]:`show_when_unsupported`,以及 `Sizable` 和 `Disableable` +- [SpeechWaveform]:`bars`(默认 24,最多 48),以及 `Sizable` +- [SpeechRecognizer]、[RecognitionSession] 与 [SpeechSink]:识别的扩展点 +- [AudioInput]、[AudioSink] 与 [AudioFormat]:音频的扩展点 +- [SpeechError] +- [SystemRecognizer] 与 [Microphone]:`speech` feature 提供的默认实现 + +[SpeechState]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechState.html +[SpeechEvent]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechEvent.html +[SpeechStatus]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechStatus.html +[SpeechButton]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechButton.html +[SpeechWaveform]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechWaveform.html +[SpeechRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.SpeechRecognizer.html +[RecognitionSession]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.RecognitionSession.html +[SpeechSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SpeechSink.html +[AudioInput]: https://docs.rs/gpui-component/latest/gpui_component/speech/trait.AudioInput.html +[AudioSink]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioSink.html +[AudioFormat]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.AudioFormat.html +[SpeechError]: https://docs.rs/gpui-component/latest/gpui_component/speech/enum.SpeechError.html +[SystemRecognizer]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.SystemRecognizer.html +[Microphone]: https://docs.rs/gpui-component/latest/gpui_component/speech/struct.Microphone.html diff --git a/website/zh-CN/docs/installation.md b/website/zh-CN/docs/installation.md index ddaba31b1e..4c6d0e3b9d 100644 --- a/website/zh-CN/docs/installation.md +++ b/website/zh-CN/docs/installation.md @@ -31,8 +31,8 @@ order: -1
sudo apt update
 sudo apt install -y gcc g++ clang libfontconfig-dev libwayland-dev \
   libwebkit2gtk-4.1-dev libxkbcommon-x11-dev libx11-xcb-dev \
-  libssl-dev libzstd-dev vulkan-validationlayers libvulkan1
-

此清单与仓库的 script/install-linux.sh 一致,适用于 Ubuntu 24.04;其他发行版需要安装对应的开发包。显示窗口还需要可用的 Wayland 或 X11 图形会话及 Vulkan 驱动;单独安装 libvulkan1 并不会安装 GPU 驱动。

+ libssl-dev libzstd-dev libasound2-dev vulkan-validationlayers libvulkan1 +

此清单与仓库的 script/install-linux.sh 一致,适用于 Ubuntu 24.04;其他发行版需要安装对应的开发包。libasound2-dev(ALSA)仅在启用 speech feature 时需要。显示窗口还需要可用的 Wayland 或 X11 图形会话及 Vulkan 驱动;单独安装 libvulkan1 并不会安装 GPU 驱动。