1use crate::EdifactDelimiters;
18
19#[derive(Debug, Clone, Copy, PartialEq, Eq)]
21pub enum Charset {
22 Ascii,
24 Latin1,
26 Utf8,
28}
29
30#[derive(Debug, Clone, PartialEq, Eq)]
32pub enum CharsetError {
33 NoSyntaxIdentifier,
35 Unsupported { syntax_identifier: String },
37 InvalidBytes {
39 syntax_identifier: String,
40 offset: usize,
41 },
42 Unrepresentable {
44 syntax_identifier: String,
45 character: char,
46 position: usize,
48 },
49}
50
51impl std::fmt::Display for CharsetError {
52 fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
53 match self {
54 Self::NoSyntaxIdentifier => {
55 write!(f, "the interchange has no UNB syntax identifier")
56 }
57 Self::Unsupported { syntax_identifier } => write!(
58 f,
59 "unsupported UNB syntax identifier {syntax_identifier:?} \
60 (supported: UNOA, UNOB, UNOC, UNOW, UNOY)"
61 ),
62 Self::InvalidBytes {
63 syntax_identifier,
64 offset,
65 } => write!(
66 f,
67 "byte {offset} is not valid in the character set of {syntax_identifier}"
68 ),
69 Self::Unrepresentable {
70 syntax_identifier,
71 character,
72 position,
73 } => write!(
74 f,
75 "character {character:?} (U+{:04X}) at position {position} cannot be \
76 encoded in {syntax_identifier}",
77 *character as u32
78 ),
79 }
80 }
81}
82
83impl std::error::Error for CharsetError {}
84
85impl Charset {
86 pub fn for_syntax_identifier(syntax_identifier: &str) -> Result<Self, CharsetError> {
88 match syntax_identifier.to_ascii_uppercase().as_str() {
89 "UNOA" | "UNOB" => Ok(Self::Ascii),
90 "UNOC" => Ok(Self::Latin1),
91 "UNOW" | "UNOY" => Ok(Self::Utf8),
92 _ => Err(CharsetError::Unsupported {
93 syntax_identifier: syntax_identifier.to_string(),
94 }),
95 }
96 }
97
98 pub fn decode(self, bytes: &[u8], syntax_identifier: &str) -> Result<String, CharsetError> {
100 let invalid = |offset| CharsetError::InvalidBytes {
101 syntax_identifier: syntax_identifier.to_string(),
102 offset,
103 };
104 match self {
105 Self::Ascii => match bytes.iter().position(|b| !b.is_ascii()) {
106 Some(offset) => Err(invalid(offset)),
107 None => Ok(bytes.iter().map(|&b| b as char).collect()),
108 },
109 Self::Latin1 => Ok(bytes.iter().map(|&b| b as char).collect()),
112 Self::Utf8 => {
113 String::from_utf8(bytes.to_vec()).map_err(|e| invalid(e.utf8_error().valid_up_to()))
114 }
115 }
116 }
117
118 pub fn encode(self, text: &str, syntax_identifier: &str) -> Result<Vec<u8>, CharsetError> {
120 let limit = match self {
121 Self::Ascii => 0x7F,
122 Self::Latin1 => 0xFF,
123 Self::Utf8 => return Ok(text.as_bytes().to_vec()),
124 };
125 text.chars()
126 .enumerate()
127 .map(|(position, c)| {
128 if (c as u32) <= limit {
129 Ok(c as u8)
130 } else {
131 Err(CharsetError::Unrepresentable {
132 syntax_identifier: syntax_identifier.to_string(),
133 character: c,
134 position,
135 })
136 }
137 })
138 .collect()
139 }
140}
141
142pub fn syntax_identifier(input: &[u8]) -> Option<String> {
146 let (has_una, delimiters) = EdifactDelimiters::detect(input);
147 let mut rest = if has_una && input.len() >= 9 {
148 &input[9..]
149 } else {
150 input
151 };
152 while let Some((first, tail)) = rest.split_first() {
153 if first.is_ascii_whitespace() {
154 rest = tail;
155 } else {
156 break;
157 }
158 }
159 let rest = rest.strip_prefix(b"UNB")?;
160 let rest = rest.strip_prefix(&[delimiters.element])?;
161 let end = rest
162 .iter()
163 .position(|&b| {
164 b == delimiters.component || b == delimiters.element || b == delimiters.segment
165 })
166 .unwrap_or(rest.len());
167 let id = std::str::from_utf8(&rest[..end]).ok()?.trim();
168 (!id.is_empty()).then(|| id.to_string())
169}
170
171pub fn decode_interchange(bytes: &[u8]) -> Result<String, CharsetError> {
174 let id = syntax_identifier(bytes).ok_or(CharsetError::NoSyntaxIdentifier)?;
175 Charset::for_syntax_identifier(&id)?.decode(bytes, &id)
176}
177
178pub fn encode_interchange(text: &str) -> Result<Vec<u8>, CharsetError> {
181 let id = syntax_identifier(text.as_bytes()).ok_or(CharsetError::NoSyntaxIdentifier)?;
182 Charset::for_syntax_identifier(&id)?.encode(text, &id)
183}
184
185#[cfg(test)]
186mod tests {
187 use super::*;
188
189 const UNOC: &str = "UNA:+.? 'UNB+UNOC:3+S:500+R:500+250401:1200+REF'FTX+ACB+++Müller'";
190
191 #[test]
192 fn reads_the_syntax_identifier_with_and_without_una() {
193 assert_eq!(syntax_identifier(UNOC.as_bytes()).as_deref(), Some("UNOC"));
194 assert_eq!(
195 syntax_identifier(b"UNB+UNOY:4+S+R+250401:1200+REF'").as_deref(),
196 Some("UNOY")
197 );
198 assert_eq!(
199 syntax_identifier(b"UNA:+.? '\nUNB+UNOA:3+S'").as_deref(),
200 Some("UNOA")
201 );
202 assert_eq!(syntax_identifier(b"UNH+1+UTILMD:D:11A:UN:S2.1'"), None);
203 }
204
205 #[test]
206 fn unoc_is_latin1_on_the_wire() {
207 let bytes = encode_interchange(UNOC).unwrap();
208 assert!(bytes.windows(2).any(|w| w == [b'M', 0xFC]), "ü is 0xFC");
209 assert_eq!(bytes.len(), UNOC.chars().count());
210 assert_eq!(decode_interchange(&bytes).unwrap(), UNOC);
211 }
212
213 #[test]
214 fn unoc_refuses_a_character_outside_latin1() {
215 let text = UNOC.replace("Müller", "Müller €");
216 let err = encode_interchange(&text).unwrap_err();
217 assert!(
218 matches!(
219 err,
220 CharsetError::Unrepresentable {
221 character: '€', ..
222 }
223 ),
224 "{err}"
225 );
226 }
227
228 #[test]
229 fn unoy_is_utf8() {
230 let text = "UNB+UNOY:4+S+R+250401:1200+REF'FTX+ACB+++Müller €'";
231 let bytes = encode_interchange(text).unwrap();
232 assert_eq!(bytes, text.as_bytes());
233 assert_eq!(decode_interchange(&bytes).unwrap(), text);
234 let mut broken = bytes.clone();
235 broken.push(0xFF);
236 assert!(matches!(
237 decode_interchange(&broken),
238 Err(CharsetError::InvalidBytes { .. })
239 ));
240 }
241
242 #[test]
243 fn unoa_is_ascii() {
244 assert!(matches!(
245 encode_interchange("UNB+UNOA:3+S+R+250401:1200+REF'FTX+ACB+++Müller'"),
246 Err(CharsetError::Unrepresentable {
247 character: 'ü', ..
248 })
249 ));
250 assert!(matches!(
251 decode_interchange(b"UNB+UNOA:3+S'FTX+ACB+++M\xFCller'"),
252 Err(CharsetError::InvalidBytes { offset: 24, .. })
253 ));
254 }
255
256 #[test]
257 fn other_levels_and_missing_unb_are_errors() {
258 assert!(matches!(
259 decode_interchange(b"UNB+UNOD:3+S'"),
260 Err(CharsetError::Unsupported { .. })
261 ));
262 assert_eq!(
263 decode_interchange(b"UNH+1'"),
264 Err(CharsetError::NoSyntaxIdentifier)
265 );
266 }
267}