Skip to main content

edifact_primitives/
charset.rs

1//! The character set of an interchange, as its `UNB` syntax identifier
2//! (S001/0001) declares it.
3//!
4//! An interchange is bytes, not text: `UNOC` — the level the Allgemeine
5//! Festlegungen require unless agreed otherwise, and the only one the MIGs
6//! list — is ISO 8859-1, where `ü` is the single byte `0xFC`. Handing such
7//! bytes around as a Rust `String` (UTF-8) silently changes them. These
8//! functions convert at the boundary: [`decode_interchange`] reads bytes in
9//! the declared character set, [`encode_interchange`] writes text in it and
10//! refuses a character the set cannot hold rather than sending it in another
11//! encoding under the declared one.
12//!
13//! Supported: `UNOA`/`UNOB` (ASCII), `UNOC` (ISO 8859-1), `UNOW`/`UNOY`
14//! (UTF-8). The other ISO 8859 levels (`UNOD`…`UNOK`) are reported as
15//! unsupported.
16
17use crate::EdifactDelimiters;
18
19/// A character set an interchange can be encoded in.
20#[derive(Debug, Clone, Copy, PartialEq, Eq)]
21pub enum Charset {
22    /// `UNOA`, `UNOB`: 7-bit ASCII.
23    Ascii,
24    /// `UNOC`: ISO 8859-1 (Latin-1).
25    Latin1,
26    /// `UNOW`, `UNOY`: UTF-8.
27    Utf8,
28}
29
30/// Why an interchange could not be decoded or encoded.
31#[derive(Debug, Clone, PartialEq, Eq)]
32pub enum CharsetError {
33    /// The input has no `UNB` segment with a syntax identifier.
34    NoSyntaxIdentifier,
35    /// A syntax identifier this crate has no character set for.
36    Unsupported { syntax_identifier: String },
37    /// A byte sequence that is not valid in the declared character set.
38    InvalidBytes {
39        syntax_identifier: String,
40        offset: usize,
41    },
42    /// A character the declared character set cannot represent.
43    Unrepresentable {
44        syntax_identifier: String,
45        character: char,
46        /// Character index in the text.
47        position: usize,
48    },
49}
50
51impl std::fmt::Display for CharsetError {
52    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
53        match self {
54            Self::NoSyntaxIdentifier => {
55                write!(f, "the interchange has no UNB syntax identifier")
56            }
57            Self::Unsupported { syntax_identifier } => write!(
58                f,
59                "unsupported UNB syntax identifier {syntax_identifier:?} \
60                 (supported: UNOA, UNOB, UNOC, UNOW, UNOY)"
61            ),
62            Self::InvalidBytes {
63                syntax_identifier,
64                offset,
65            } => write!(
66                f,
67                "byte {offset} is not valid in the character set of {syntax_identifier}"
68            ),
69            Self::Unrepresentable {
70                syntax_identifier,
71                character,
72                position,
73            } => write!(
74                f,
75                "character {character:?} (U+{:04X}) at position {position} cannot be \
76                 encoded in {syntax_identifier}",
77                *character as u32
78            ),
79        }
80    }
81}
82
83impl std::error::Error for CharsetError {}
84
85impl Charset {
86    /// The character set a `UNB` syntax identifier (`UNOC`) declares.
87    pub fn for_syntax_identifier(syntax_identifier: &str) -> Result<Self, CharsetError> {
88        match syntax_identifier.to_ascii_uppercase().as_str() {
89            "UNOA" | "UNOB" => Ok(Self::Ascii),
90            "UNOC" => Ok(Self::Latin1),
91            "UNOW" | "UNOY" => Ok(Self::Utf8),
92            _ => Err(CharsetError::Unsupported {
93                syntax_identifier: syntax_identifier.to_string(),
94            }),
95        }
96    }
97
98    /// Read `bytes` as text in this character set.
99    pub fn decode(self, bytes: &[u8], syntax_identifier: &str) -> Result<String, CharsetError> {
100        let invalid = |offset| CharsetError::InvalidBytes {
101            syntax_identifier: syntax_identifier.to_string(),
102            offset,
103        };
104        match self {
105            Self::Ascii => match bytes.iter().position(|b| !b.is_ascii()) {
106                Some(offset) => Err(invalid(offset)),
107                None => Ok(bytes.iter().map(|&b| b as char).collect()),
108            },
109            // Every byte is a character, and ISO 8859-1 is the first 256 code
110            // points of Unicode, so the byte value is the code point.
111            Self::Latin1 => Ok(bytes.iter().map(|&b| b as char).collect()),
112            Self::Utf8 => {
113                String::from_utf8(bytes.to_vec()).map_err(|e| invalid(e.utf8_error().valid_up_to()))
114            }
115        }
116    }
117
118    /// Write `text` in this character set, refusing a character it cannot hold.
119    pub fn encode(self, text: &str, syntax_identifier: &str) -> Result<Vec<u8>, CharsetError> {
120        let limit = match self {
121            Self::Ascii => 0x7F,
122            Self::Latin1 => 0xFF,
123            Self::Utf8 => return Ok(text.as_bytes().to_vec()),
124        };
125        text.chars()
126            .enumerate()
127            .map(|(position, c)| {
128                if (c as u32) <= limit {
129                    Ok(c as u8)
130                } else {
131                    Err(CharsetError::Unrepresentable {
132                        syntax_identifier: syntax_identifier.to_string(),
133                        character: c,
134                        position,
135                    })
136                }
137            })
138            .collect()
139    }
140}
141
142/// The syntax identifier of the interchange's `UNB` (`UNOC` in
143/// `UNB+UNOC:3+…`), read from the raw bytes. The service segments are ASCII
144/// in every syntax level, so this works before the character set is known.
145pub fn syntax_identifier(input: &[u8]) -> Option<String> {
146    let (has_una, delimiters) = EdifactDelimiters::detect(input);
147    let mut rest = if has_una && input.len() >= 9 {
148        &input[9..]
149    } else {
150        input
151    };
152    while let Some((first, tail)) = rest.split_first() {
153        if first.is_ascii_whitespace() {
154            rest = tail;
155        } else {
156            break;
157        }
158    }
159    let rest = rest.strip_prefix(b"UNB")?;
160    let rest = rest.strip_prefix(&[delimiters.element])?;
161    let end = rest
162        .iter()
163        .position(|&b| {
164            b == delimiters.component || b == delimiters.element || b == delimiters.segment
165        })
166        .unwrap_or(rest.len());
167    let id = std::str::from_utf8(&rest[..end]).ok()?.trim();
168    (!id.is_empty()).then(|| id.to_string())
169}
170
171/// Read an interchange's bytes as text, in the character set its `UNB`
172/// syntax identifier declares.
173pub fn decode_interchange(bytes: &[u8]) -> Result<String, CharsetError> {
174    let id = syntax_identifier(bytes).ok_or(CharsetError::NoSyntaxIdentifier)?;
175    Charset::for_syntax_identifier(&id)?.decode(bytes, &id)
176}
177
178/// Write an interchange's text as bytes, in the character set its `UNB`
179/// syntax identifier declares. A character outside that set is an error.
180pub fn encode_interchange(text: &str) -> Result<Vec<u8>, CharsetError> {
181    let id = syntax_identifier(text.as_bytes()).ok_or(CharsetError::NoSyntaxIdentifier)?;
182    Charset::for_syntax_identifier(&id)?.encode(text, &id)
183}
184
185#[cfg(test)]
186mod tests {
187    use super::*;
188
189    const UNOC: &str = "UNA:+.? 'UNB+UNOC:3+S:500+R:500+250401:1200+REF'FTX+ACB+++Müller'";
190
191    #[test]
192    fn reads_the_syntax_identifier_with_and_without_una() {
193        assert_eq!(syntax_identifier(UNOC.as_bytes()).as_deref(), Some("UNOC"));
194        assert_eq!(
195            syntax_identifier(b"UNB+UNOY:4+S+R+250401:1200+REF'").as_deref(),
196            Some("UNOY")
197        );
198        assert_eq!(
199            syntax_identifier(b"UNA:+.? '\nUNB+UNOA:3+S'").as_deref(),
200            Some("UNOA")
201        );
202        assert_eq!(syntax_identifier(b"UNH+1+UTILMD:D:11A:UN:S2.1'"), None);
203    }
204
205    #[test]
206    fn unoc_is_latin1_on_the_wire() {
207        let bytes = encode_interchange(UNOC).unwrap();
208        assert!(bytes.windows(2).any(|w| w == [b'M', 0xFC]), "ü is 0xFC");
209        assert_eq!(bytes.len(), UNOC.chars().count());
210        assert_eq!(decode_interchange(&bytes).unwrap(), UNOC);
211    }
212
213    #[test]
214    fn unoc_refuses_a_character_outside_latin1() {
215        let text = UNOC.replace("Müller", "Müller €");
216        let err = encode_interchange(&text).unwrap_err();
217        assert!(
218            matches!(
219                err,
220                CharsetError::Unrepresentable {
221                    character: '€', ..
222                }
223            ),
224            "{err}"
225        );
226    }
227
228    #[test]
229    fn unoy_is_utf8() {
230        let text = "UNB+UNOY:4+S+R+250401:1200+REF'FTX+ACB+++Müller €'";
231        let bytes = encode_interchange(text).unwrap();
232        assert_eq!(bytes, text.as_bytes());
233        assert_eq!(decode_interchange(&bytes).unwrap(), text);
234        let mut broken = bytes.clone();
235        broken.push(0xFF);
236        assert!(matches!(
237            decode_interchange(&broken),
238            Err(CharsetError::InvalidBytes { .. })
239        ));
240    }
241
242    #[test]
243    fn unoa_is_ascii() {
244        assert!(matches!(
245            encode_interchange("UNB+UNOA:3+S+R+250401:1200+REF'FTX+ACB+++Müller'"),
246            Err(CharsetError::Unrepresentable {
247                character: 'ü', ..
248            })
249        ));
250        assert!(matches!(
251            decode_interchange(b"UNB+UNOA:3+S'FTX+ACB+++M\xFCller'"),
252            Err(CharsetError::InvalidBytes { offset: 24, .. })
253        ));
254    }
255
256    #[test]
257    fn other_levels_and_missing_unb_are_errors() {
258        assert!(matches!(
259            decode_interchange(b"UNB+UNOD:3+S'"),
260            Err(CharsetError::Unsupported { .. })
261        ));
262        assert_eq!(
263            decode_interchange(b"UNH+1'"),
264            Err(CharsetError::NoSyntaxIdentifier)
265        );
266    }
267}