Skip to main content

vtcode_llm/providers/shared/
utf8.rs

1//! Incremental UTF-8 decoding across network chunk boundaries.
2
3/// Incrementally decodes a byte stream into UTF-8 text.
4///
5/// Network chunks can split a multibyte code point (CJK, emoji, accented
6/// characters, smart quotes) across boundaries. Decoding each chunk
7/// independently with `String::from_utf8_lossy` corrupts such code points into
8/// `U+FFFD` replacement characters. This decoder buffers any trailing incomplete
9/// sequence until the rest of its bytes arrive, while still replacing genuinely
10/// invalid bytes with `U+FFFD` (matching `from_utf8_lossy` semantics).
11#[derive(Debug, Default)]
12pub struct Utf8StreamDecoder {
13    pending: Vec<u8>,
14}
15
16impl Utf8StreamDecoder {
17    pub(crate) fn new() -> Self {
18        Self::default()
19    }
20
21    /// Appends `bytes` and returns the decodable UTF-8 prefix. A trailing
22    /// incomplete multibyte sequence is retained for the next call.
23    pub(crate) fn push(&mut self, bytes: &[u8]) -> String {
24        self.pending.extend_from_slice(bytes);
25        let mut out = String::with_capacity(bytes.len());
26        loop {
27            match std::str::from_utf8(&self.pending) {
28                Ok(text) => {
29                    out.push_str(text);
30                    self.pending.clear();
31                    break;
32                }
33                Err(err) => {
34                    let valid = err.valid_up_to();
35                    if let Some(valid_bytes) = self.pending.get(..valid) {
36                        // `valid_bytes` is guaranteed valid UTF-8 by `valid_up_to`.
37                        out.push_str(&String::from_utf8_lossy(valid_bytes));
38                    }
39                    match err.error_len() {
40                        // Genuinely invalid sequence: emit replacement and skip it.
41                        Some(invalid_len) => {
42                            out.push('\u{FFFD}');
43                            self.pending.drain(..valid + invalid_len);
44                        }
45                        // Incomplete trailing sequence: keep it for the next push.
46                        None => {
47                            self.pending.drain(..valid);
48                            break;
49                        }
50                    }
51                }
52            }
53        }
54        out
55    }
56
57    /// Appends `bytes` and writes the decodable UTF-8 prefix directly into
58    /// `out`. A trailing incomplete multibyte sequence is retained for the
59    /// next call.
60    ///
61    /// This avoids the intermediate `String` allocation that `push` creates
62    /// when the caller only needs bytes (e.g. feeding an SSE byte buffer).
63    /// The output is identical to `push(bytes).into_bytes()`.
64    pub(crate) fn push_bytes(&mut self, bytes: &[u8], out: &mut Vec<u8>) {
65        self.pending.extend_from_slice(bytes);
66        loop {
67            match std::str::from_utf8(&self.pending) {
68                Ok(text) => {
69                    out.extend_from_slice(text.as_bytes());
70                    self.pending.clear();
71                    break;
72                }
73                Err(err) => {
74                    let valid = err.valid_up_to();
75                    if let Some(valid_bytes) = self.pending.get(..valid) {
76                        // `valid_bytes` is guaranteed valid UTF-8 by `valid_up_to`,
77                        // so a direct byte copy is safe and avoids `from_utf8_lossy`.
78                        out.extend_from_slice(valid_bytes);
79                    }
80                    match err.error_len() {
81                        // Genuinely invalid sequence: emit replacement and skip it.
82                        Some(invalid_len) => {
83                            out.extend_from_slice("\u{FFFD}".as_bytes());
84                            self.pending.drain(..valid + invalid_len);
85                        }
86                        // Incomplete trailing sequence: keep it for the next push.
87                        None => {
88                            self.pending.drain(..valid);
89                            break;
90                        }
91                    }
92                }
93            }
94        }
95    }
96}
97
98#[cfg(test)]
99mod tests {
100    use super::*;
101
102    #[test]
103    fn utf8_stream_decoder_push_bytes_matches_push() {
104        // Complete valid UTF-8 across multiple chunks.
105        let full = "data: {\"hello\":\"world\"}\n\n".as_bytes();
106        let mut dec_str = Utf8StreamDecoder::new();
107        let mut dec_bytes = Utf8StreamDecoder::new();
108        for (i, &byte) in full.iter().enumerate() {
109            let mut out = Vec::new();
110            dec_bytes.push_bytes(std::slice::from_ref(&byte), &mut out);
111            let s = dec_str.push(std::slice::from_ref(&byte));
112            assert_eq!(out, s.into_bytes(), "byte {i}: push_bytes must match push");
113        }
114        // Both decoders should have empty pending buffers after complete input.
115        let mut tail = Vec::new();
116        dec_bytes.push_bytes(&[], &mut tail);
117        assert!(tail.is_empty());
118    }
119
120    #[test]
121    fn utf8_stream_decoder_push_bytes_handles_split_multibyte() {
122        // Split a multibyte character (U+00E9 = 0xC3 0xA9) across chunks.
123        let mut dec = Utf8StreamDecoder::new();
124        let mut out = Vec::new();
125        dec.push_bytes(&[0xC3], &mut out);
126        assert!(out.is_empty(), "incomplete multibyte should produce no output");
127        dec.push_bytes(&[0xA9, b'h', b'i'], &mut out);
128        assert_eq!(std::str::from_utf8(&out).unwrap(), "\u{00E9}hi");
129    }
130
131    #[test]
132    fn utf8_stream_decoder_push_bytes_emits_replacement_for_invalid() {
133        // 0xFF is never a valid UTF-8 lead byte.
134        let mut dec = Utf8StreamDecoder::new();
135        let mut out = Vec::new();
136        dec.push_bytes(&[b'a', 0xFF, b'b'], &mut out);
137        assert_eq!(std::str::from_utf8(&out).unwrap(), "a\u{FFFD}b");
138    }
139}