use crate::EdifactDelimiters;
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Charset {
Ascii,
Latin1,
Utf8,
}
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum CharsetError {
NoSyntaxIdentifier,
Unsupported { syntax_identifier: String },
InvalidBytes {
syntax_identifier: String,
offset: usize,
},
Unrepresentable {
syntax_identifier: String,
character: char,
position: usize,
},
}
impl std::fmt::Display for CharsetError {
fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
match self {
Self::NoSyntaxIdentifier => {
write!(f, "the interchange has no UNB syntax identifier")
}
Self::Unsupported { syntax_identifier } => write!(
f,
"unsupported UNB syntax identifier {syntax_identifier:?} \
(supported: UNOA, UNOB, UNOC, UNOW, UNOY)"
),
Self::InvalidBytes {
syntax_identifier,
offset,
} => write!(
f,
"byte {offset} is not valid in the character set of {syntax_identifier}"
),
Self::Unrepresentable {
syntax_identifier,
character,
position,
} => write!(
f,
"character {character:?} (U+{:04X}) at position {position} cannot be \
encoded in {syntax_identifier}",
*character as u32
),
}
}
}
impl std::error::Error for CharsetError {}
impl Charset {
pub fn for_syntax_identifier(syntax_identifier: &str) -> Result<Self, CharsetError> {
match syntax_identifier.to_ascii_uppercase().as_str() {
"UNOA" | "UNOB" => Ok(Self::Ascii),
"UNOC" => Ok(Self::Latin1),
"UNOW" | "UNOY" => Ok(Self::Utf8),
_ => Err(CharsetError::Unsupported {
syntax_identifier: syntax_identifier.to_string(),
}),
}
}
pub fn decode(self, bytes: &[u8], syntax_identifier: &str) -> Result<String, CharsetError> {
let invalid = |offset| CharsetError::InvalidBytes {
syntax_identifier: syntax_identifier.to_string(),
offset,
};
match self {
Self::Ascii => match bytes.iter().position(|b| !b.is_ascii()) {
Some(offset) => Err(invalid(offset)),
None => Ok(bytes.iter().map(|&b| b as char).collect()),
},
Self::Latin1 => Ok(bytes.iter().map(|&b| b as char).collect()),
Self::Utf8 => {
String::from_utf8(bytes.to_vec()).map_err(|e| invalid(e.utf8_error().valid_up_to()))
}
}
}
pub fn encode(self, text: &str, syntax_identifier: &str) -> Result<Vec<u8>, CharsetError> {
let limit = match self {
Self::Ascii => 0x7F,
Self::Latin1 => 0xFF,
Self::Utf8 => return Ok(text.as_bytes().to_vec()),
};
text.chars()
.enumerate()
.map(|(position, c)| {
if (c as u32) <= limit {
Ok(c as u8)
} else {
Err(CharsetError::Unrepresentable {
syntax_identifier: syntax_identifier.to_string(),
character: c,
position,
})
}
})
.collect()
}
}
pub fn syntax_identifier(input: &[u8]) -> Option<String> {
let (has_una, delimiters) = EdifactDelimiters::detect(input);
let mut rest = if has_una && input.len() >= 9 {
&input[9..]
} else {
input
};
while let Some((first, tail)) = rest.split_first() {
if first.is_ascii_whitespace() {
rest = tail;
} else {
break;
}
}
let rest = rest.strip_prefix(b"UNB")?;
let rest = rest.strip_prefix(&[delimiters.element])?;
let end = rest
.iter()
.position(|&b| {
b == delimiters.component || b == delimiters.element || b == delimiters.segment
})
.unwrap_or(rest.len());
let id = std::str::from_utf8(&rest[..end]).ok()?.trim();
(!id.is_empty()).then(|| id.to_string())
}
pub fn decode_interchange(bytes: &[u8]) -> Result<String, CharsetError> {
let id = syntax_identifier(bytes).ok_or(CharsetError::NoSyntaxIdentifier)?;
Charset::for_syntax_identifier(&id)?.decode(bytes, &id)
}
pub fn encode_interchange(text: &str) -> Result<Vec<u8>, CharsetError> {
let id = syntax_identifier(text.as_bytes()).ok_or(CharsetError::NoSyntaxIdentifier)?;
Charset::for_syntax_identifier(&id)?.encode(text, &id)
}
#[cfg(test)]
mod tests {
use super::*;
const UNOC: &str = "UNA:+.? 'UNB+UNOC:3+S:500+R:500+250401:1200+REF'FTX+ACB+++Müller'";
#[test]
fn reads_the_syntax_identifier_with_and_without_una() {
assert_eq!(syntax_identifier(UNOC.as_bytes()).as_deref(), Some("UNOC"));
assert_eq!(
syntax_identifier(b"UNB+UNOY:4+S+R+250401:1200+REF'").as_deref(),
Some("UNOY")
);
assert_eq!(
syntax_identifier(b"UNA:+.? '\nUNB+UNOA:3+S'").as_deref(),
Some("UNOA")
);
assert_eq!(syntax_identifier(b"UNH+1+UTILMD:D:11A:UN:S2.1'"), None);
}
#[test]
fn unoc_is_latin1_on_the_wire() {
let bytes = encode_interchange(UNOC).unwrap();
assert!(bytes.windows(2).any(|w| w == [b'M', 0xFC]), "ü is 0xFC");
assert_eq!(bytes.len(), UNOC.chars().count());
assert_eq!(decode_interchange(&bytes).unwrap(), UNOC);
}
#[test]
fn unoc_refuses_a_character_outside_latin1() {
let text = UNOC.replace("Müller", "Müller €");
let err = encode_interchange(&text).unwrap_err();
assert!(
matches!(
err,
CharsetError::Unrepresentable {
character: '€', ..
}
),
"{err}"
);
}
#[test]
fn unoy_is_utf8() {
let text = "UNB+UNOY:4+S+R+250401:1200+REF'FTX+ACB+++Müller €'";
let bytes = encode_interchange(text).unwrap();
assert_eq!(bytes, text.as_bytes());
assert_eq!(decode_interchange(&bytes).unwrap(), text);
let mut broken = bytes.clone();
broken.push(0xFF);
assert!(matches!(
decode_interchange(&broken),
Err(CharsetError::InvalidBytes { .. })
));
}
#[test]
fn unoa_is_ascii() {
assert!(matches!(
encode_interchange("UNB+UNOA:3+S+R+250401:1200+REF'FTX+ACB+++Müller'"),
Err(CharsetError::Unrepresentable {
character: 'ü', ..
})
));
assert!(matches!(
decode_interchange(b"UNB+UNOA:3+S'FTX+ACB+++M\xFCller'"),
Err(CharsetError::InvalidBytes { offset: 24, .. })
));
}
#[test]
fn other_levels_and_missing_unb_are_errors() {
assert!(matches!(
decode_interchange(b"UNB+UNOD:3+S'"),
Err(CharsetError::Unsupported { .. })
));
assert_eq!(
decode_interchange(b"UNH+1'"),
Err(CharsetError::NoSyntaxIdentifier)
);
}
}