Skip to main content

rustpython_common/encodings/
wide.rs

1//! Shared utf-16 / utf-32 unit helpers.
2
3use super::*;
4use crate::wtf8::CodePoint;
5
6/// Byte order for a wide Unicode encoding.
7#[derive(Clone, Copy, Debug, PartialEq, Eq)]
8pub enum ByteOrder {
9    /// Platform endianness; a leading BOM selects the other side on decode.
10    Native,
11    Little,
12    Big,
13}
14
15impl ByteOrder {
16    #[must_use]
17    pub const fn is_big_endian(self) -> bool {
18        match self {
19            Self::Native => cfg!(target_endian = "big"),
20            Self::Little => false,
21            Self::Big => true,
22        }
23    }
24}
25
26pub(super) fn push_u16(out: &mut Vec<u8>, unit: u16, big_endian: bool) {
27    out.extend_from_slice(&if big_endian {
28        unit.to_be_bytes()
29    } else {
30        unit.to_le_bytes()
31    });
32}
33
34pub(super) fn push_u32(out: &mut Vec<u8>, unit: u32, big_endian: bool) {
35    out.extend_from_slice(&if big_endian {
36        unit.to_be_bytes()
37    } else {
38        unit.to_le_bytes()
39    });
40}
41
42pub(super) fn emit_utf16(out: &mut Vec<u8>, cp: u32, big_endian: bool) {
43    if cp <= 0xffff {
44        push_u16(out, cp as u16, big_endian);
45    } else {
46        let v = cp - 0x10000;
47        push_u16(out, 0xd800 | ((v >> 10) as u16), big_endian);
48        push_u16(out, 0xdc00 | ((v & 0x3ff) as u16), big_endian);
49    }
50}
51
52pub(super) fn emit_utf32(out: &mut Vec<u8>, cp: u32, big_endian: bool) {
53    push_u32(out, cp, big_endian);
54}
55
56pub(super) fn read_u16(data: &[u8], big_endian: bool) -> u16 {
57    let arr = [data[0], data[1]];
58    if big_endian {
59        u16::from_be_bytes(arr)
60    } else {
61        u16::from_le_bytes(arr)
62    }
63}
64
65pub(super) fn read_u32(data: &[u8], big_endian: bool) -> u32 {
66    let arr = [data[0], data[1], data[2], data[3]];
67    if big_endian {
68        u32::from_be_bytes(arr)
69    } else {
70        u32::from_le_bytes(arr)
71    }
72}
73
74pub(super) const fn is_surrogate(cp: u32) -> bool {
75    matches!(cp, 0xd800..=0xdfff)
76}
77
78/// `(big_endian, bom_len, reported_byteorder)` for a native or fixed decode.
79///
80/// `reported_byteorder` is CPython's `-1` / `0` / `1` (`le` / native-no-BOM / `be`).
81pub(super) fn resolve_bom(data: &[u8], order: ByteOrder, unit: usize) -> (bool, usize, i32) {
82    match order {
83        ByteOrder::Little => (false, 0, -1),
84        ByteOrder::Big => (true, 0, 1),
85        ByteOrder::Native if unit == 2 && data.starts_with(&[0xff, 0xfe]) => (false, 2, -1),
86        ByteOrder::Native if unit == 2 && data.starts_with(&[0xfe, 0xff]) => (true, 2, 1),
87        ByteOrder::Native if unit == 4 && data.starts_with(&[0xff, 0xfe, 0x00, 0x00]) => {
88            (false, 4, -1)
89        }
90        ByteOrder::Native if unit == 4 && data.starts_with(&[0x00, 0x00, 0xfe, 0xff]) => {
91            (true, 4, 1)
92        }
93        ByteOrder::Native => (cfg!(target_endian = "big"), 0, 0),
94    }
95}
96
97pub(super) fn encode_wide<Ctx, E>(
98    mut ctx: Ctx,
99    errors: &E,
100    big_endian: bool,
101    bom: bool,
102    unit: usize,
103    reason: &str,
104) -> Result<Vec<u8>, Ctx::Error>
105where
106    Ctx: EncodeContext,
107    E: EncodeErrorHandler<Ctx>,
108{
109    let emit = if unit == 2 { emit_utf16 } else { emit_utf32 };
110    let mut out = Vec::new();
111    if bom {
112        emit(&mut out, 0xfeff, big_endian);
113    }
114    loop {
115        let data = ctx.remaining_data();
116        let mut iter = iter_code_points(data);
117        let Some((i, _)) = iter.find(|(_, c)| is_surrogate(c.to_u32())) else {
118            for (_, c) in iter_code_points(data) {
119                emit(&mut out, c.to_u32(), big_endian);
120            }
121            break;
122        };
123        for (_, c) in iter_code_points(&data[..i.bytes]) {
124            emit(&mut out, c.to_u32(), big_endian);
125        }
126        let err_end = match { iter }.find(|(_, c)| !is_surrogate(c.to_u32())) {
127            Some((j, _)) => ctx.position() + j,
128            None => ctx.data_len(),
129        };
130        let range = (ctx.position() + i)..err_end;
131        let replace = ctx.handle_error(errors, range.clone(), Some(reason))?;
132        match replace {
133            EncodeReplace::Str(s) => {
134                for c in s.as_ref().code_points() {
135                    let cp = c.to_u32();
136                    if is_surrogate(cp) {
137                        return Err(ctx.error_encoding(range, Some(reason)));
138                    }
139                    emit(&mut out, cp, big_endian);
140                }
141            }
142            EncodeReplace::Bytes(b) => {
143                if b.as_ref().len() % unit != 0 {
144                    return Err(ctx.error_encoding(range, Some(reason)));
145                }
146                out.extend_from_slice(b.as_ref());
147            }
148        }
149    }
150    Ok(out)
151}
152
153pub(super) fn push_codepoint(out: &mut Wtf8Buf, cp: u32) {
154    if let Some(c) = CodePoint::from_u32(cp) {
155        out.push(c);
156    }
157}