rustpython_common/encodings/
wide.rs1use super::*;
4use crate::wtf8::CodePoint;
5
6#[derive(Clone, Copy, Debug, PartialEq, Eq)]
8pub enum ByteOrder {
9 Native,
11 Little,
12 Big,
13}
14
15impl ByteOrder {
16 #[must_use]
17 pub const fn is_big_endian(self) -> bool {
18 match self {
19 Self::Native => cfg!(target_endian = "big"),
20 Self::Little => false,
21 Self::Big => true,
22 }
23 }
24}
25
26pub(super) fn push_u16(out: &mut Vec<u8>, unit: u16, big_endian: bool) {
27 out.extend_from_slice(&if big_endian {
28 unit.to_be_bytes()
29 } else {
30 unit.to_le_bytes()
31 });
32}
33
34pub(super) fn push_u32(out: &mut Vec<u8>, unit: u32, big_endian: bool) {
35 out.extend_from_slice(&if big_endian {
36 unit.to_be_bytes()
37 } else {
38 unit.to_le_bytes()
39 });
40}
41
42pub(super) fn emit_utf16(out: &mut Vec<u8>, cp: u32, big_endian: bool) {
43 if cp <= 0xffff {
44 push_u16(out, cp as u16, big_endian);
45 } else {
46 let v = cp - 0x10000;
47 push_u16(out, 0xd800 | ((v >> 10) as u16), big_endian);
48 push_u16(out, 0xdc00 | ((v & 0x3ff) as u16), big_endian);
49 }
50}
51
52pub(super) fn emit_utf32(out: &mut Vec<u8>, cp: u32, big_endian: bool) {
53 push_u32(out, cp, big_endian);
54}
55
56pub(super) fn read_u16(data: &[u8], big_endian: bool) -> u16 {
57 let arr = [data[0], data[1]];
58 if big_endian {
59 u16::from_be_bytes(arr)
60 } else {
61 u16::from_le_bytes(arr)
62 }
63}
64
65pub(super) fn read_u32(data: &[u8], big_endian: bool) -> u32 {
66 let arr = [data[0], data[1], data[2], data[3]];
67 if big_endian {
68 u32::from_be_bytes(arr)
69 } else {
70 u32::from_le_bytes(arr)
71 }
72}
73
74pub(super) const fn is_surrogate(cp: u32) -> bool {
75 matches!(cp, 0xd800..=0xdfff)
76}
77
78pub(super) fn resolve_bom(data: &[u8], order: ByteOrder, unit: usize) -> (bool, usize, i32) {
82 match order {
83 ByteOrder::Little => (false, 0, -1),
84 ByteOrder::Big => (true, 0, 1),
85 ByteOrder::Native if unit == 2 && data.starts_with(&[0xff, 0xfe]) => (false, 2, -1),
86 ByteOrder::Native if unit == 2 && data.starts_with(&[0xfe, 0xff]) => (true, 2, 1),
87 ByteOrder::Native if unit == 4 && data.starts_with(&[0xff, 0xfe, 0x00, 0x00]) => {
88 (false, 4, -1)
89 }
90 ByteOrder::Native if unit == 4 && data.starts_with(&[0x00, 0x00, 0xfe, 0xff]) => {
91 (true, 4, 1)
92 }
93 ByteOrder::Native => (cfg!(target_endian = "big"), 0, 0),
94 }
95}
96
97pub(super) fn encode_wide<Ctx, E>(
98 mut ctx: Ctx,
99 errors: &E,
100 big_endian: bool,
101 bom: bool,
102 unit: usize,
103 reason: &str,
104) -> Result<Vec<u8>, Ctx::Error>
105where
106 Ctx: EncodeContext,
107 E: EncodeErrorHandler<Ctx>,
108{
109 let emit = if unit == 2 { emit_utf16 } else { emit_utf32 };
110 let mut out = Vec::new();
111 if bom {
112 emit(&mut out, 0xfeff, big_endian);
113 }
114 loop {
115 let data = ctx.remaining_data();
116 let mut iter = iter_code_points(data);
117 let Some((i, _)) = iter.find(|(_, c)| is_surrogate(c.to_u32())) else {
118 for (_, c) in iter_code_points(data) {
119 emit(&mut out, c.to_u32(), big_endian);
120 }
121 break;
122 };
123 for (_, c) in iter_code_points(&data[..i.bytes]) {
124 emit(&mut out, c.to_u32(), big_endian);
125 }
126 let err_end = match { iter }.find(|(_, c)| !is_surrogate(c.to_u32())) {
127 Some((j, _)) => ctx.position() + j,
128 None => ctx.data_len(),
129 };
130 let range = (ctx.position() + i)..err_end;
131 let replace = ctx.handle_error(errors, range.clone(), Some(reason))?;
132 match replace {
133 EncodeReplace::Str(s) => {
134 for c in s.as_ref().code_points() {
135 let cp = c.to_u32();
136 if is_surrogate(cp) {
137 return Err(ctx.error_encoding(range, Some(reason)));
138 }
139 emit(&mut out, cp, big_endian);
140 }
141 }
142 EncodeReplace::Bytes(b) => {
143 if b.as_ref().len() % unit != 0 {
144 return Err(ctx.error_encoding(range, Some(reason)));
145 }
146 out.extend_from_slice(b.as_ref());
147 }
148 }
149 }
150 Ok(out)
151}
152
153pub(super) fn push_codepoint(out: &mut Wtf8Buf, cp: u32) {
154 if let Some(c) = CodePoint::from_u32(cp) {
155 out.push(c);
156 }
157}