Skip to main content

nodejs/stdlib/
string_decoder.rs

1//! Node `string_decoder` core module: `new StringDecoder(encoding)` with
2//! `.write(buffer)` / `.end([buffer])`. A StringDecoder turns byte chunks into a
3//! string, holding back an incomplete trailing multibyte sequence until the next
4//! chunk completes it.
5//!
6//! Every encoding that has a chunk boundary buffers across it: UTF-8 holds an
7//! incomplete trailing sequence, UTF-16LE holds an odd byte and a trailing high
8//! surrogate, and base64 holds up to two bytes so it only ever emits whole
9//! 3-byte groups. The single-byte encodings consume everything.
10
11use crate::host::{with_host, JsObj};
12use fusevm::Value;
13use indexmap::IndexMap;
14
15/// The methods a `StringDecoder` instance carries. Also the method set its real
16/// prototype object is built from, so a read (`sd.write`), a call (`sd.write(b)`)
17/// and a prototype lookup (`StringDecoder.prototype.write`) cannot disagree.
18pub const INSTANCE_METHODS: &[&str] = &["write", "end"];
19
20/// Node's `normalizeEncoding`: the `encoding` property reports the CANONICAL
21/// name, not the spelling that was passed — `new StringDecoder('ucs2').encoding`
22/// is `'utf16le'` and `new StringDecoder('UTF-8').encoding` is `'utf8'`. Code
23/// that branches on `decoder.encoding` (iconv-lite does) sees the canonical set.
24fn normalize_encoding(enc: &str) -> String {
25    match enc.to_ascii_lowercase().as_str() {
26        "utf8" | "utf-8" => "utf8",
27        "ucs2" | "ucs-2" | "utf16le" | "utf-16le" => "utf16le",
28        "latin1" | "binary" => "latin1",
29        other => return other.to_string(),
30    }
31    .to_string()
32}
33
34/// `new StringDecoder([encoding])`.
35pub fn construct(args: &[Value]) -> Result<Value, String> {
36    let enc = if args.is_empty() {
37        "utf8".to_string()
38    } else {
39        super::arg_str(args, 0)
40    };
41    Ok(with_host(|h| {
42        let mut m = IndexMap::new();
43        m.insert("@@native".into(), h.new_str("StringDecoder"));
44        m.insert("encoding".into(), h.new_str(normalize_encoding(&enc)));
45        // Held-back bytes from a UTF-8 sequence split across chunks.
46        let empty = h.new_array(Vec::new());
47        m.insert("@@pending".into(), empty);
48        h.new_object(m)
49    }))
50}
51
52/// The byte content of a Buffer / typed array / array argument.
53fn bytes_of(v: &Value) -> Vec<u8> {
54    with_host(|h| match h.get(v) {
55        Some(JsObj::Object(p)) => {
56            let field = p.get("@@bytes").or_else(|| p.get("@@elems"));
57            match field.and_then(|a| h.get(a)) {
58                Some(JsObj::Array(items)) => items.iter().map(|x| h.to_number(x) as u8).collect(),
59                _ => Vec::new(),
60            }
61        }
62        Some(JsObj::Array(items)) => items.iter().map(|x| h.to_number(x) as u8).collect(),
63        _ => Vec::new(),
64    })
65}
66
67fn encoding_of(recv: &Value) -> String {
68    with_host(|h| match h.get(recv) {
69        Some(JsObj::Object(p)) => p
70            .get("encoding")
71            .map(|v| h.str_of(v))
72            .unwrap_or_else(|| "utf8".into()),
73        _ => "utf8".into(),
74    })
75}
76
77fn pending_of(recv: &Value) -> Vec<u8> {
78    with_host(|h| match h.get(recv) {
79        Some(JsObj::Object(p)) => match p.get("@@pending").and_then(|a| h.get(a)) {
80            Some(JsObj::Array(items)) => items.iter().map(|x| h.to_number(x) as u8).collect(),
81            _ => Vec::new(),
82        },
83        _ => Vec::new(),
84    })
85}
86
87fn set_pending(recv: &Value, bytes: &[u8]) {
88    with_host(|h| {
89        let arr = h.new_array(bytes.iter().map(|b| Value::Float(*b as f64)).collect());
90        if let Some(JsObj::Object(p)) = h.get_mut(recv) {
91            p.insert("@@pending".into(), arr);
92        }
93    });
94}
95
96pub fn instance_call(recv: &Value, method: &str, args: &[Value]) -> Result<Value, String> {
97    let enc = encoding_of(recv);
98    match method {
99        "write" => {
100            let mut buf = pending_of(recv);
101            buf.extend(bytes_of(&args.first().cloned().unwrap_or(Value::Undef)));
102            let (decoded, tail) = decode(&enc, &buf);
103            set_pending(recv, &tail);
104            Ok(with_host(|h| h.new_str(decoded)))
105        }
106        "end" => {
107            let mut buf = pending_of(recv);
108            if let Some(v) = args.first() {
109                buf.extend(bytes_of(v));
110            }
111            set_pending(recv, &[]);
112            let (mut decoded, tail) = decode(&enc, &buf);
113            decoded.push_str(&flush(&enc, &tail));
114            Ok(with_host(|h| h.new_str(decoded)))
115        }
116        _ => Err(crate::host::type_error(&format!(
117            "{method} is not a function"
118        ))),
119    }
120}
121
122/// What `end()` emits for bytes still held when the stream closes. Each encoding
123/// resolves its own remainder, so this is not one blanket replacement char:
124///
125/// * UTF-8 — one `U+FFFD` for the whole truncated sequence (not one per byte).
126/// * UTF-16LE — a held HIGH SURROGATE is emitted as a code unit; a dangling odd
127///   byte is dropped silently, with no replacement char (measured on node
128///   v26.7.0: `d.write(Buffer.from([0x61])); d.end()` is `''`). The lone
129///   surrogate itself becomes `U+FFFD` here — the `utf16` storage boundary.
130/// * base64 — the short group is padded and emitted (`AQ==`), which is the whole
131///   reason the bytes were held rather than encoded early.
132fn flush(enc: &str, tail: &[u8]) -> String {
133    if tail.is_empty() {
134        return String::new();
135    }
136    match enc {
137        "base64" => super::to_base64(tail),
138        "base64url" => super::to_base64url(tail),
139        "utf16le" => {
140            let units: Vec<u16> = tail
141                .chunks_exact(2)
142                .map(|c| u16::from_le_bytes([c[0], c[1]]))
143                .collect();
144            crate::utf16::to_string_lossy(&units)
145        }
146        _ => "\u{FFFD}".to_string(),
147    }
148}
149
150/// Decode `buf` in `enc`, returning (decoded string, held-back trailing bytes).
151/// Single-byte encodings consume everything; the rest hold back the partial tail
152/// that only the next chunk can complete.
153fn decode(enc: &str, buf: &[u8]) -> (String, Vec<u8>) {
154    match enc {
155        // `ascii` masks the high bit, `latin1`/`binary` keep the whole byte.
156        "ascii" => (
157            buf.iter().map(|b| (*b & 0x7f) as char).collect(),
158            Vec::new(),
159        ),
160        "latin1" => (buf.iter().map(|b| *b as char).collect(), Vec::new()),
161        "hex" => (super::to_hex(buf), Vec::new()),
162        // Base64 is 3 bytes → 4 characters. Emitting a short group would pad it
163        // mid-stream (`AQ==` then `AgM=` instead of `AQID`), so the remainder is
164        // held until the group closes or `end()` pads it.
165        "base64" | "base64url" => {
166            let keep = buf.len() % 3;
167            let (head, tail) = buf.split_at(buf.len() - keep);
168            let s = if enc == "base64url" {
169                super::to_base64url(head)
170            } else {
171                super::to_base64(head)
172            };
173            (s, tail.to_vec())
174        }
175        // UTF-16LE: an odd trailing byte is half a code unit, and a trailing HIGH
176        // surrogate is half a code point — both wait for the next chunk.
177        "utf16le" => {
178            let mut keep = buf.len() % 2;
179            let whole = buf.len() - keep;
180            if whole >= 2 {
181                let last = u16::from_le_bytes([buf[whole - 2], buf[whole - 1]]);
182                if (0xD800..0xDC00).contains(&last) {
183                    keep += 2;
184                }
185            }
186            let (head, tail) = buf.split_at(buf.len() - keep);
187            let units: Vec<u16> = head
188                .chunks_exact(2)
189                .map(|c| u16::from_le_bytes([c[0], c[1]]))
190                .collect();
191            (crate::utf16::to_string_lossy(&units), tail.to_vec())
192        }
193        // utf8 / utf-8 (and anything else): keep a split multibyte tail pending.
194        _ => {
195            let split = incomplete_utf8_tail(buf);
196            let (head, tail) = buf.split_at(buf.len() - split);
197            (String::from_utf8_lossy(head).into_owned(), tail.to_vec())
198        }
199    }
200}
201
202/// Number of trailing bytes that form an incomplete UTF-8 sequence (0..=3).
203fn incomplete_utf8_tail(buf: &[u8]) -> usize {
204    // Walk back over continuation bytes (10xxxxxx) to the lead byte.
205    let mut i = buf.len();
206    let mut cont = 0;
207    while i > 0 && buf[i - 1] & 0b1100_0000 == 0b1000_0000 && cont < 3 {
208        i -= 1;
209        cont += 1;
210    }
211    if i == 0 {
212        return 0;
213    }
214    let lead = buf[i - 1];
215    let needed = if lead & 0b1000_0000 == 0 {
216        1
217    } else if lead & 0b1110_0000 == 0b1100_0000 {
218        2
219    } else if lead & 0b1111_0000 == 0b1110_0000 {
220        3
221    } else if lead & 0b1111_1000 == 0b1111_0000 {
222        4
223    } else {
224        1
225    };
226    // If the lead + its continuations are all present, nothing is pending.
227    if cont + 1 >= needed {
228        0
229    } else {
230        cont + 1
231    }
232}