edifact-primitives 0.30.2

Shared EDIFACT primitive types — zero dependencies
Documentation
//! The character set of an interchange, as its `UNB` syntax identifier
//! (S001/0001) declares it.
//!
//! An interchange is bytes, not text: `UNOC` — the level the Allgemeine
//! Festlegungen require unless agreed otherwise, and the only one the MIGs
//! list — is ISO 8859-1, where `ü` is the single byte `0xFC`. Handing such
//! bytes around as a Rust `String` (UTF-8) silently changes them. These
//! functions convert at the boundary: [`decode_interchange`] reads bytes in
//! the declared character set, [`encode_interchange`] writes text in it and
//! refuses a character the set cannot hold rather than sending it in another
//! encoding under the declared one.
//!
//! Supported: `UNOA`/`UNOB` (ASCII), `UNOC` (ISO 8859-1), `UNOW`/`UNOY`
//! (UTF-8). The other ISO 8859 levels (`UNOD`…`UNOK`) are reported as
//! unsupported.

use crate::EdifactDelimiters;

/// A character set an interchange can be encoded in.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub enum Charset {
    /// `UNOA`, `UNOB`: 7-bit ASCII.
    Ascii,
    /// `UNOC`: ISO 8859-1 (Latin-1).
    Latin1,
    /// `UNOW`, `UNOY`: UTF-8.
    Utf8,
}

/// Why an interchange could not be decoded or encoded.
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum CharsetError {
    /// The input has no `UNB` segment with a syntax identifier.
    NoSyntaxIdentifier,
    /// A syntax identifier this crate has no character set for.
    Unsupported { syntax_identifier: String },
    /// A byte sequence that is not valid in the declared character set.
    InvalidBytes {
        syntax_identifier: String,
        offset: usize,
    },
    /// A character the declared character set cannot represent.
    Unrepresentable {
        syntax_identifier: String,
        character: char,
        /// Character index in the text.
        position: usize,
    },
}

impl std::fmt::Display for CharsetError {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        match self {
            Self::NoSyntaxIdentifier => {
                write!(f, "the interchange has no UNB syntax identifier")
            }
            Self::Unsupported { syntax_identifier } => write!(
                f,
                "unsupported UNB syntax identifier {syntax_identifier:?} \
                 (supported: UNOA, UNOB, UNOC, UNOW, UNOY)"
            ),
            Self::InvalidBytes {
                syntax_identifier,
                offset,
            } => write!(
                f,
                "byte {offset} is not valid in the character set of {syntax_identifier}"
            ),
            Self::Unrepresentable {
                syntax_identifier,
                character,
                position,
            } => write!(
                f,
                "character {character:?} (U+{:04X}) at position {position} cannot be \
                 encoded in {syntax_identifier}",
                *character as u32
            ),
        }
    }
}

impl std::error::Error for CharsetError {}

impl Charset {
    /// The character set a `UNB` syntax identifier (`UNOC`) declares.
    pub fn for_syntax_identifier(syntax_identifier: &str) -> Result<Self, CharsetError> {
        match syntax_identifier.to_ascii_uppercase().as_str() {
            "UNOA" | "UNOB" => Ok(Self::Ascii),
            "UNOC" => Ok(Self::Latin1),
            "UNOW" | "UNOY" => Ok(Self::Utf8),
            _ => Err(CharsetError::Unsupported {
                syntax_identifier: syntax_identifier.to_string(),
            }),
        }
    }

    /// Read `bytes` as text in this character set.
    pub fn decode(self, bytes: &[u8], syntax_identifier: &str) -> Result<String, CharsetError> {
        let invalid = |offset| CharsetError::InvalidBytes {
            syntax_identifier: syntax_identifier.to_string(),
            offset,
        };
        match self {
            Self::Ascii => match bytes.iter().position(|b| !b.is_ascii()) {
                Some(offset) => Err(invalid(offset)),
                None => Ok(bytes.iter().map(|&b| b as char).collect()),
            },
            // Every byte is a character, and ISO 8859-1 is the first 256 code
            // points of Unicode, so the byte value is the code point.
            Self::Latin1 => Ok(bytes.iter().map(|&b| b as char).collect()),
            Self::Utf8 => {
                String::from_utf8(bytes.to_vec()).map_err(|e| invalid(e.utf8_error().valid_up_to()))
            }
        }
    }

    /// Write `text` in this character set, refusing a character it cannot hold.
    pub fn encode(self, text: &str, syntax_identifier: &str) -> Result<Vec<u8>, CharsetError> {
        let limit = match self {
            Self::Ascii => 0x7F,
            Self::Latin1 => 0xFF,
            Self::Utf8 => return Ok(text.as_bytes().to_vec()),
        };
        text.chars()
            .enumerate()
            .map(|(position, c)| {
                if (c as u32) <= limit {
                    Ok(c as u8)
                } else {
                    Err(CharsetError::Unrepresentable {
                        syntax_identifier: syntax_identifier.to_string(),
                        character: c,
                        position,
                    })
                }
            })
            .collect()
    }
}

/// The syntax identifier of the interchange's `UNB` (`UNOC` in
/// `UNB+UNOC:3+…`), read from the raw bytes. The service segments are ASCII
/// in every syntax level, so this works before the character set is known.
pub fn syntax_identifier(input: &[u8]) -> Option<String> {
    let (has_una, delimiters) = EdifactDelimiters::detect(input);
    let mut rest = if has_una && input.len() >= 9 {
        &input[9..]
    } else {
        input
    };
    while let Some((first, tail)) = rest.split_first() {
        if first.is_ascii_whitespace() {
            rest = tail;
        } else {
            break;
        }
    }
    let rest = rest.strip_prefix(b"UNB")?;
    let rest = rest.strip_prefix(&[delimiters.element])?;
    let end = rest
        .iter()
        .position(|&b| {
            b == delimiters.component || b == delimiters.element || b == delimiters.segment
        })
        .unwrap_or(rest.len());
    let id = std::str::from_utf8(&rest[..end]).ok()?.trim();
    (!id.is_empty()).then(|| id.to_string())
}

/// Read an interchange's bytes as text, in the character set its `UNB`
/// syntax identifier declares.
pub fn decode_interchange(bytes: &[u8]) -> Result<String, CharsetError> {
    let id = syntax_identifier(bytes).ok_or(CharsetError::NoSyntaxIdentifier)?;
    Charset::for_syntax_identifier(&id)?.decode(bytes, &id)
}

/// Write an interchange's text as bytes, in the character set its `UNB`
/// syntax identifier declares. A character outside that set is an error.
pub fn encode_interchange(text: &str) -> Result<Vec<u8>, CharsetError> {
    let id = syntax_identifier(text.as_bytes()).ok_or(CharsetError::NoSyntaxIdentifier)?;
    Charset::for_syntax_identifier(&id)?.encode(text, &id)
}

#[cfg(test)]
mod tests {
    use super::*;

    const UNOC: &str = "UNA:+.? 'UNB+UNOC:3+S:500+R:500+250401:1200+REF'FTX+ACB+++Müller'";

    #[test]
    fn reads_the_syntax_identifier_with_and_without_una() {
        assert_eq!(syntax_identifier(UNOC.as_bytes()).as_deref(), Some("UNOC"));
        assert_eq!(
            syntax_identifier(b"UNB+UNOY:4+S+R+250401:1200+REF'").as_deref(),
            Some("UNOY")
        );
        assert_eq!(
            syntax_identifier(b"UNA:+.? '\nUNB+UNOA:3+S'").as_deref(),
            Some("UNOA")
        );
        assert_eq!(syntax_identifier(b"UNH+1+UTILMD:D:11A:UN:S2.1'"), None);
    }

    #[test]
    fn unoc_is_latin1_on_the_wire() {
        let bytes = encode_interchange(UNOC).unwrap();
        assert!(bytes.windows(2).any(|w| w == [b'M', 0xFC]), "ü is 0xFC");
        assert_eq!(bytes.len(), UNOC.chars().count());
        assert_eq!(decode_interchange(&bytes).unwrap(), UNOC);
    }

    #[test]
    fn unoc_refuses_a_character_outside_latin1() {
        let text = UNOC.replace("Müller", "Müller €");
        let err = encode_interchange(&text).unwrap_err();
        assert!(
            matches!(
                err,
                CharsetError::Unrepresentable {
                    character: '€', ..
                }
            ),
            "{err}"
        );
    }

    #[test]
    fn unoy_is_utf8() {
        let text = "UNB+UNOY:4+S+R+250401:1200+REF'FTX+ACB+++Müller €'";
        let bytes = encode_interchange(text).unwrap();
        assert_eq!(bytes, text.as_bytes());
        assert_eq!(decode_interchange(&bytes).unwrap(), text);
        let mut broken = bytes.clone();
        broken.push(0xFF);
        assert!(matches!(
            decode_interchange(&broken),
            Err(CharsetError::InvalidBytes { .. })
        ));
    }

    #[test]
    fn unoa_is_ascii() {
        assert!(matches!(
            encode_interchange("UNB+UNOA:3+S+R+250401:1200+REF'FTX+ACB+++Müller'"),
            Err(CharsetError::Unrepresentable {
                character: 'ü', ..
            })
        ));
        assert!(matches!(
            decode_interchange(b"UNB+UNOA:3+S'FTX+ACB+++M\xFCller'"),
            Err(CharsetError::InvalidBytes { offset: 24, .. })
        ));
    }

    #[test]
    fn other_levels_and_missing_unb_are_errors() {
        assert!(matches!(
            decode_interchange(b"UNB+UNOD:3+S'"),
            Err(CharsetError::Unsupported { .. })
        ));
        assert_eq!(
            decode_interchange(b"UNH+1'"),
            Err(CharsetError::NoSyntaxIdentifier)
        );
    }
}