Skip to main content

rustpython_common/encodings/
utf32.rs

1//! UTF-32 encode / incremental decode.
2
3use super::wide::{self, emit_utf32, is_surrogate, push_codepoint, read_u32, resolve_bom};
4use super::*;
5use crate::wtf8::Wtf8Buf;
6
7pub use super::wide::ByteOrder;
8
9pub const ENCODING_NAME: &str = "utf-32";
10pub const ENCODING_NAME_LE: &str = "utf-32-le";
11pub const ENCODING_NAME_BE: &str = "utf-32-be";
12
13const ERR_REASON: &str = "surrogates not allowed";
14
15pub fn encode<Ctx, E>(
16    ctx: Ctx,
17    errors: &E,
18    order: ByteOrder,
19    bom: bool,
20) -> Result<Vec<u8>, Ctx::Error>
21where
22    Ctx: EncodeContext,
23    E: EncodeErrorHandler<Ctx>,
24{
25    wide::encode_wide(ctx, errors, order.is_big_endian(), bom, 4, ERR_REASON)
26}
27
28/// Decode one incremental chunk.
29///
30/// Returns `(text, consumed, byteorder)` where `byteorder` is CPython's
31/// `-1` / `0` / `1`.
32pub fn decode<Ctx, E>(
33    mut ctx: Ctx,
34    errors: &E,
35    order: ByteOrder,
36    final_decode: bool,
37) -> Result<(Wtf8Buf, usize, i32), Ctx::Error>
38where
39    Ctx: DecodeContext,
40    E: DecodeErrorHandler<Ctx>,
41{
42    let (big_endian, skip, byteorder) = resolve_bom(ctx.remaining_data(), order, 4);
43    ctx.advance(skip);
44    let mut out = Wtf8Buf::new();
45    loop {
46        let rest = ctx.remaining_data();
47        if rest.len() < 4 {
48            if rest.is_empty() || !final_decode {
49                break;
50            }
51            let start = ctx.position();
52            let end = ctx.full_data().len();
53            let replace = ctx.handle_error(errors, start..end, Some("truncated data"))?;
54            out.push_wtf8(replace.as_ref());
55            continue;
56        }
57        let ch = read_u32(rest, big_endian);
58        if is_surrogate(ch) {
59            let start = ctx.position();
60            let replace = ctx.handle_error(
61                errors,
62                start..start + 4,
63                Some("code point in surrogate code point range(0xd800, 0xe000)"),
64            )?;
65            out.push_wtf8(replace.as_ref());
66            continue;
67        }
68        if ch >= 0x110000 {
69            let start = ctx.position();
70            let replace = ctx.handle_error(
71                errors,
72                start..start + 4,
73                Some("code point not in range(0x110000)"),
74            )?;
75            out.push_wtf8(replace.as_ref());
76            continue;
77        }
78        push_codepoint(&mut out, ch);
79        ctx.advance(4);
80    }
81    Ok((out, ctx.position(), byteorder))
82}
83
84pub fn encode_codepoint(out: &mut Vec<u8>, cp: u32, order: ByteOrder) {
85    emit_utf32(out, cp, order.is_big_endian());
86}