gpui-component 0.7.1

GPUI Component: the styled component library of GPUI Kit, with 60+ desktop UI components for GPUI.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
//! The Windows recognizer: continuous dictation with
//! `Windows.Media.SpeechRecognition`.
//!
//! The WinRT recognizer captures from the default microphone itself and has no
//! way to consume audio from elsewhere, so the pushed PCM is ignored; the
//! [`Microphone`](crate::speech::Microphone) keeps capturing alongside it only
//! to drive the waveform, which WASAPI's shared mode allows.
//!
//! The speech objects are agile, so they are called from the main thread, an
//! STA that GPUI initializes with `OleInitialize`, without further apartment
//! setup. Their completions and events arrive on WinRT threads, which only
//! forward them over a channel to a foreground task that reports to the sink.

use anyhow::anyhow;
use gpui::{App, AsyncApp, SharedString, Task};
use smol::channel::{Receiver, Sender, bounded, unbounded};
use windows::{
    Foundation::{
        AsyncActionCompletedHandler, AsyncOperationCompletedHandler, EventRegistrationToken,
        IAsyncAction, IAsyncOperation, TypedEventHandler,
    },
    Globalization::Language,
    Media::SpeechRecognition::{
        SpeechContinuousRecognitionCompletedEventArgs,
        SpeechContinuousRecognitionResultGeneratedEventArgs, SpeechContinuousRecognitionSession,
        SpeechRecognitionConfidence, SpeechRecognitionHypothesisGeneratedEventArgs,
        SpeechRecognitionResult, SpeechRecognitionResultStatus, SpeechRecognitionScenario,
        SpeechRecognitionTopicConstraint, SpeechRecognizer as WinSpeechRecognizer,
    },
    Win32::Foundation::E_ACCESSDENIED,
    core::{HRESULT, HSTRING, RuntimeType},
};

use crate::speech::{AudioFormat, RecognitionSession, SpeechError, SpeechRecognizer, SpeechSink};

/// `SPERR_SPEECH_PRIVACY_POLICY_NOT_ACCEPTED`: "Online speech recognition" is
/// turned off in the privacy settings, which dictation requires.
const PRIVACY_POLICY_NOT_ACCEPTED: HRESULT = HRESULT(0x80045509_u32 as i32);
/// `MF_E_NO_CAPTURE_DEVICES_AVAILABLE`.
const NO_CAPTURE_DEVICES: HRESULT = HRESULT(0xC00DABE0_u32 as i32);

pub(super) struct PlatformRecognizer {
    /// The dictation language, or `None` when the requested one is malformed or
    /// has no speech pack installed.
    language: Option<Language>,
    separator: &'static str,
}

impl PlatformRecognizer {
    pub(super) fn new(locale: Option<SharedString>) -> Self {
        let language = supported_language(locale.as_deref())
            .inspect_err(|error| log::warn!("speech: no dictation language: {error:#}"))
            .ok()
            .flatten();
        let separator = language
            .as_ref()
            .and_then(|language| language.LanguageTag().ok())
            .map_or(" ", |tag| phrase_separator(&tag.to_string_lossy()));
        Self {
            language,
            separator,
        }
    }
}

impl SpeechRecognizer for PlatformRecognizer {
    fn audio_format(&self) -> AudioFormat {
        AudioFormat::default()
    }

    fn is_available(&self, _: &App) -> bool {
        self.language.is_some()
    }

    fn start(
        &self,
        sink: SpeechSink,
        cx: &mut App,
    ) -> Result<Box<dyn RecognitionSession>, SpeechError> {
        let Some(language) = &self.language else {
            return Err(SpeechError::Unsupported);
        };

        let recognizer = WinSpeechRecognizer::Create(language).map_err(speech_error)?;
        let continuous = recognizer
            .ContinuousRecognitionSession()
            .map_err(speech_error)?;
        let (messages, rx) = unbounded();
        let mut session = Session {
            recognizer: recognizer.clone(),
            continuous: continuous.clone(),
            hypothesis_token: None,
            result_token: None,
            completed_token: None,
            messages: messages.clone(),
            _task: None,
        };

        let constraint = SpeechRecognitionTopicConstraint::Create(
            SpeechRecognitionScenario::Dictation,
            &HSTRING::from("dictation"),
        )
        .map_err(speech_error)?;
        recognizer
            .Constraints()
            .and_then(|constraints| constraints.Append(&constraint))
            .map_err(speech_error)?;

        session.hypothesis_token = Some(
            recognizer
                .HypothesisGenerated(&TypedEventHandler::new({
                    let messages = messages.clone();
                    move |_, args: &Option<SpeechRecognitionHypothesisGeneratedEventArgs>| {
                        if let Some(args) = args {
                            let text = args.Hypothesis()?.Text()?;
                            _ = messages.try_send(Message::Hypothesis(text));
                        }
                        Ok(())
                    }
                }))
                .map_err(speech_error)?,
        );
        session.result_token = Some(
            continuous
                .ResultGenerated(&TypedEventHandler::new({
                    let messages = messages.clone();
                    move |_, args: &Option<SpeechContinuousRecognitionResultGeneratedEventArgs>| {
                        if let Some(args) = args {
                            _ = messages.try_send(Message::Result(args.Result()?));
                        }
                        Ok(())
                    }
                }))
                .map_err(speech_error)?,
        );
        session.completed_token = Some(
            continuous
                .Completed(&TypedEventHandler::new({
                    let messages = messages.clone();
                    move |_, args: &Option<SpeechContinuousRecognitionCompletedEventArgs>| {
                        if let Some(args) = args {
                            _ = messages.try_send(Message::Completed(args.Status()?));
                        }
                        Ok(())
                    }
                }))
                .map_err(speech_error)?,
        );

        let separator = self.separator;
        session._task = Some(cx.spawn(async move |cx| {
            let result = dictate(&recognizer, &continuous, rx, &sink, separator, cx).await;
            if let Err(error) = result {
                cx.update(|cx| sink.error(error, cx));
            }
        }));
        Ok(Box::new(session))
    }
}

/// What the WinRT threads and [`Session::finish`] tell the dictation task.
enum Message {
    Hypothesis(HSTRING),
    Result(SpeechRecognitionResult),
    Completed(SpeechRecognitionResultStatus),
    /// The user stopped talking.
    Finish,
}

struct Session {
    recognizer: WinSpeechRecognizer,
    continuous: SpeechContinuousRecognitionSession,
    hypothesis_token: Option<EventRegistrationToken>,
    result_token: Option<EventRegistrationToken>,
    completed_token: Option<EventRegistrationToken>,
    messages: Sender<Message>,
    _task: Option<Task<()>>,
}

impl RecognitionSession for Session {
    /// Ignored: the WinRT recognizer captures from the microphone itself.
    fn push_audio(&mut self, _: &[i16], _: &mut App) {}

    fn finish(&mut self, _: &mut App) {
        // Queued behind the start, so finishing while connecting stops the
        // session as soon as it runs.
        _ = self.messages.try_send(Message::Finish);
    }
}

impl Drop for Session {
    fn drop(&mut self) {
        if let Some(token) = self.hypothesis_token.take() {
            _ = self.recognizer.RemoveHypothesisGenerated(token);
        }
        if let Some(token) = self.result_token.take() {
            _ = self.continuous.RemoveResultGenerated(token);
        }
        if let Some(token) = self.completed_token.take() {
            _ = self.continuous.RemoveCompleted(token);
        }

        // Close the recognizer once the cancellation lands, or right away when
        // there is nothing to cancel.
        let recognizer = self.recognizer.clone();
        let closed = self.continuous.CancelAsync().and_then(|action| {
            action.SetCompleted(&AsyncActionCompletedHandler::new(move |_, _| {
                _ = recognizer.Close();
                Ok(())
            }))
        });
        if closed.is_err() {
            _ = self.recognizer.Close();
        }
    }
}

/// Compile the dictation constraint, start the continuous session and report
/// its results to `sink` until it completes.
async fn dictate(
    recognizer: &WinSpeechRecognizer,
    continuous: &SpeechContinuousRecognitionSession,
    messages: Receiver<Message>,
    sink: &SpeechSink,
    separator: &'static str,
    cx: &mut AsyncApp,
) -> Result<(), SpeechError> {
    let compilation = operation(recognizer.CompileConstraintsAsync().map_err(speech_error)?)
        .await
        .map_err(speech_error)?;
    let status = compilation.Status().map_err(speech_error)?;
    if status != SpeechRecognitionResultStatus::Success {
        return Err(status_error(status));
    }
    action(continuous.StartAsync().map_err(speech_error)?)
        .await
        .map_err(speech_error)?;
    cx.update(|cx| sink.ready(cx));

    let mut separator_due = false;
    while let Ok(message) = messages.recv().await {
        match message {
            Message::Hypothesis(text) => {
                let text = joined(separator_due, separator, &text);
                cx.update(|cx| sink.hypothesis(text, cx));
            }
            Message::Result(result) => {
                let Some(text) = phrase_text(&result) else {
                    continue;
                };
                let text = joined(separator_due, separator, &text);
                separator_due = true;
                cx.update(|cx| sink.phrase(text, cx));
            }
            Message::Completed(status) => {
                return match status {
                    // Stopped, cancelled, or ended by the silence timeout.
                    SpeechRecognitionResultStatus::Success
                    | SpeechRecognitionResultStatus::UserCanceled
                    | SpeechRecognitionResultStatus::TimeoutExceeded => {
                        cx.update(|cx| sink.finish(cx));
                        Ok(())
                    }
                    status => Err(status_error(status)),
                };
            }
            // Stopping flushes the last phrase, then completes the session.
            Message::Finish => {
                let stopped = match continuous.StopAsync() {
                    Ok(stop) => action(stop).await,
                    Err(error) => Err(error),
                };
                // The session may have ended by itself, e.g. after the silence
                // timeout, with its `Completed` still queued behind this; stopping
                // it then fails, and that queued event ends the session instead.
                if let Err(error) = stopped {
                    log::debug!("speech: stopping dictation failed: {error}");
                }
            }
        }
    }
    Ok(())
}

/// The text of a recognized phrase, unless it was rejected or empty.
fn phrase_text(result: &SpeechRecognitionResult) -> Option<HSTRING> {
    if result.Status().ok()? != SpeechRecognitionResultStatus::Success
        || result.Confidence().ok()? == SpeechRecognitionConfidence::Rejected
    {
        return None;
    }
    Some(result.Text().ok()?).filter(|text| !text.is_empty())
}

/// `text` preceded by `separator` once a phrase has been committed.
fn joined(separator_due: bool, separator: &str, text: &HSTRING) -> SharedString {
    let text = text.to_string_lossy();
    if separator_due {
        format!("{separator}{text}").into()
    } else {
        text.into()
    }
}

/// What goes between two phrases, which dictation returns without surrounding
/// whitespace: a space, except in languages written without spaces between
/// words (Chinese, Japanese, Thai, Lao, Khmer, Burmese).
fn phrase_separator(tag: &str) -> &'static str {
    let primary = tag.split('-').next().unwrap_or_default();
    let unspaced = ["zh", "yue", "ja", "th", "lo", "km", "my"];
    if unspaced
        .iter()
        .any(|lang| primary.eq_ignore_ascii_case(lang))
    {
        ""
    } else {
        " "
    }
}

/// The language to dictate `locale` (or, without one, the system's speech
/// language) in, if a speech pack supports it.
///
/// A bare language such as `en` falls back to the first supported region.
fn supported_language(locale: Option<&str>) -> windows::core::Result<Option<Language>> {
    let requested = match locale {
        Some(locale) => {
            let tag = HSTRING::from(locale);
            if !Language::IsWellFormed(&tag)? {
                return Ok(None);
            }
            Language::CreateLanguage(&tag)?
        }
        None => WinSpeechRecognizer::SystemSpeechLanguage()?,
    };
    let requested = requested.LanguageTag()?.to_string_lossy();
    let region_prefix = format!("{requested}-");

    let mut fallback = None;
    for language in WinSpeechRecognizer::SupportedTopicLanguages()? {
        let tag = language.LanguageTag()?.to_string_lossy();
        if tag.eq_ignore_ascii_case(&requested) {
            return Ok(Some(language));
        }
        if fallback.is_none()
            && tag.len() > region_prefix.len()
            && tag[..region_prefix.len()].eq_ignore_ascii_case(&region_prefix)
        {
            fallback = Some(language);
        }
    }
    Ok(fallback)
}

/// Await `operation` without blocking: its completion handler, which runs on a
/// WinRT thread, only wakes this future.
async fn operation<T: RuntimeType + 'static>(
    operation: IAsyncOperation<T>,
) -> windows::core::Result<T> {
    let (done, wait) = bounded(1);
    operation.SetCompleted(&AsyncOperationCompletedHandler::new(move |_, _| {
        _ = done.try_send(());
        Ok(())
    }))?;
    _ = wait.recv().await;
    operation.GetResults()
}

/// Await `action` like [`operation`].
async fn action(action: IAsyncAction) -> windows::core::Result<()> {
    let (done, wait) = bounded(1);
    action.SetCompleted(&AsyncActionCompletedHandler::new(move |_, _| {
        _ = done.try_send(());
        Ok(())
    }))?;
    _ = wait.recv().await;
    action.GetResults()
}

fn speech_error(error: windows::core::Error) -> SpeechError {
    match error.code() {
        E_ACCESSDENIED => SpeechError::PermissionDenied,
        NO_CAPTURE_DEVICES => SpeechError::NoInputDevice,
        PRIVACY_POLICY_NOT_ACCEPTED => SpeechError::recognizer(anyhow!(
            "Online speech recognition is turned off; turn it on in Settings > \
             Privacy & security > Speech"
        )),
        _ => SpeechError::recognizer(error),
    }
}

fn status_error(status: SpeechRecognitionResultStatus) -> SpeechError {
    match status {
        SpeechRecognitionResultStatus::TopicLanguageNotSupported => SpeechError::Unsupported,
        SpeechRecognitionResultStatus::MicrophoneUnavailable => SpeechError::NoInputDevice,
        SpeechRecognitionResultStatus::NetworkFailure => {
            SpeechError::recognizer(anyhow!("could not reach the online speech service"))
        }
        SpeechRecognitionResultStatus::AudioQualityFailure => {
            SpeechError::recognizer(anyhow!("the audio was too poor to recognize"))
        }
        status => SpeechError::recognizer(anyhow!("recognition ended with status {}", status.0)),
    }
}