use super::*;
fn nonzero(value: u32) -> NonZeroU32 {
NonZeroU32::new(value).expect("nonzero")
}
const PIPE_UPPER: Tokenization = Tokenization::new(
WordDelimiter::Pipe,
LetterCase::Upper,
Granularity::Character,
&[],
);
const AS_WRITTEN: Tokenization = Tokenization::new(
WordDelimiter::Pipe,
LetterCase::AsWritten,
Granularity::Character,
&[],
);
fn table(tokens: &[&str]) -> Vocabulary {
let entries: Vec<String> = tokens
.iter()
.enumerate()
.map(|(id, token)| format!("{}: {id}", serde_json::to_string(token).expect("a token")))
.collect();
Vocabulary::from_json(format!("{{{}}}", entries.join(", ")).as_bytes()).expect("a table")
}
#[test]
fn the_staged_contract_is_the_staged_artifacts() {
let contract = AcousticContract::BASE960H;
assert_eq!(contract.blank(), 0);
assert_eq!(contract.blank(), BLANK_ID);
assert_eq!(contract.geometry(), AcousticGeometry::WAV2VEC2);
assert_eq!(contract.tokenization(), PIPE_UPPER);
assert_eq!(contract.output(), OutputKind::LogProbabilities);
assert_eq!(contract.sentinel_band(), Some(SentinelBand::Fp16Saturation));
assert_eq!(
check_tokenization(
contract.blank(),
contract.tokenization(),
&Vocabulary::bundled(),
true
),
Ok(())
);
let geometry = AcousticGeometry::WAV2VEC2;
assert_eq!(geometry.sample_rate().get(), 16_000);
assert_eq!(geometry.receptive_field().get(), 400);
assert_eq!(geometry.stride().get(), 320);
assert_eq!(
geometry.frames(960_000),
2_999,
"the staged window's frames"
);
}
#[test]
fn a_models_own_contract_carries_no_band() {
let geometry = AcousticGeometry::new(16_000, nonzero(640), nonzero(320)).expect("a geometry");
let contract = AcousticContract::new(3, geometry, PIPE_UPPER, OutputKind::Logits);
assert_eq!(contract.blank(), 3);
assert_eq!(contract.geometry(), geometry);
assert_eq!(contract.tokenization(), PIPE_UPPER);
assert_eq!(contract.output(), OutputKind::Logits);
assert_eq!(contract.sentinel_band(), None);
}
#[test]
fn a_declared_frame_count_fits_more_than_one_front_end() {
let wide = AcousticGeometry::new(16_000, nonzero(640), nonzero(320)).expect("a geometry");
assert_eq!(
wide.frames(960_000),
AcousticGeometry::WAV2VEC2.frames(960_000)
);
assert_ne!(wide, AcousticGeometry::WAV2VEC2);
assert_eq!(wide.frames(720), 1);
assert_eq!(AcousticGeometry::WAV2VEC2.frames(720), 2);
}
#[test]
fn frames_is_the_conv_output_length() {
let geometry = AcousticGeometry::WAV2VEC2;
for (samples, frames) in [
(0, 0),
(399, 0),
(400, 1),
(719, 1),
(720, 2),
(48_000, 149),
] {
assert_eq!(geometry.frames(samples), frames, "{samples} samples");
}
}
#[test]
fn a_rate_other_than_16_khz_is_refused_by_name() {
for rate in [0u32, 8_000, 22_050, 44_100, 48_000] {
assert_eq!(
AcousticGeometry::new(rate, nonzero(400), nonzero(320)),
Err(GeometryError::SampleRate(rate)),
"{rate} Hz"
);
}
}
#[test]
fn a_geometry_at_16_khz_is_refused_for_nothing_else() {
for (receptive_field, stride) in [
(200u32, 100u32),
(299, 100),
(1, 1),
(320, 79),
(80, 320),
(640, 320),
(u32::MAX, u32::MAX),
] {
let geometry = AcousticGeometry::new(16_000, nonzero(receptive_field), nonzero(stride))
.unwrap_or_else(|err| panic!("{receptive_field}/{stride}: {err}"));
assert_eq!(
(geometry.receptive_field().get(), geometry.stride().get()),
(receptive_field, stride)
);
}
}
#[test]
fn the_band_holds_the_saturated_log_zero_and_nothing_computed() {
let band = SentinelBand::Fp16Saturation;
assert_eq!(band.ceiling(), -32_768.0);
for value in [-45_440.0f32, -65_504.0, -32_768.0, f32::NEG_INFINITY] {
assert!(band.holds(value), "{value}");
}
for value in [-32_767.0f32, -30.81, -1.0, 0.0, f32::NAN] {
assert!(!band.holds(value), "{value}");
}
}
fn declaring(specials: &'static [&'static str]) -> Tokenization {
Tokenization::new(
WordDelimiter::Pipe,
LetterCase::Upper,
Granularity::Character,
specials,
)
}
#[test]
fn a_blank_spelled_as_a_space_is_no_whitespace_token() {
assert_eq!(
check_tokenization(0, PIPE_UPPER, &table(&[" ", "|", "A", "B"]), true),
Ok(())
);
}
#[test]
fn a_declared_lowercase_special_is_no_letter() {
assert_eq!(
check_tokenization(
0,
declaring(&["a"]),
&table(&["<pad>", "|", "A", "B", "a"]),
true
),
Ok(())
);
}
#[test]
fn a_declared_tab_special_is_no_whitespace_token() {
assert_eq!(
check_tokenization(
0,
declaring(&["\t"]),
&table(&["<pad>", "|", "A", "\t"]),
true
),
Ok(())
);
}
#[test]
fn a_lexical_lowercase_letter_or_whitespace_token_is_still_refused() {
assert_eq!(
check_tokenization(
0,
declaring(&["a"]),
&table(&["<pad>", "|", "A", "a", "b"]),
true
),
Err(TokenizationError::UpperWithLowercase('b'))
);
assert_eq!(
check_tokenization(0, PIPE_UPPER, &table(&["<pad>", "|", "A", "\t"]), true),
Err(TokenizationError::WhitespaceToken("\t".to_owned()))
);
}
fn declaring_with(delimiter: WordDelimiter, specials: &'static [&'static str]) -> Tokenization {
Tokenization::new(
delimiter,
LetterCase::Upper,
Granularity::Character,
specials,
)
}
#[test]
fn an_empty_named_special_is_refused_by_name() {
assert_eq!(
check_tokenization(
0,
declaring(&[""]),
&table(&["<pad>", "|", "A", "B", ""]),
true
),
Err(TokenizationError::EmptySpecial(4))
);
assert_eq!(
check_tokenization(
0,
declaring_with(WordDelimiter::Space, &[""]),
&table(&["<pad>", " ", "A", ""]),
true
),
Err(TokenizationError::EmptySpecial(3))
);
}
#[test]
fn an_empty_special_the_blank_or_the_delimiter_reserves_is_accepted() {
assert_eq!(
check_tokenization(0, declaring(&[""]), &table(&["", "|", "A", "B"]), true),
Ok(())
);
assert_eq!(
check_tokenization(
0,
declaring_with(WordDelimiter::Absent, &[""]),
&table(&["<pad>", "A", "B", ""]),
false
),
Ok(())
);
}
const fn upper(delimiter: WordDelimiter) -> Tokenization {
Tokenization::new(delimiter, LetterCase::Upper, Granularity::Character, &[])
}
#[test]
fn a_space_in_the_table_is_the_stated_delimiter_or_refused_by_name() {
let spaced = table(&["<pad>", " ", "|", "A", "B"]);
for tokenization in [PIPE_UPPER, upper(WordDelimiter::Absent)] {
for word_delimited in [true, false] {
assert_eq!(
check_tokenization(0, tokenization, &spaced, word_delimited),
Err(TokenizationError::WhitespaceToken(" ".to_owned())),
"{tokenization:?}, word_delimited {word_delimited}"
);
}
}
assert_eq!(WordDelimiter::from_token(" "), Ok(WordDelimiter::Space));
assert_eq!(WordDelimiter::from_token("|"), Ok(WordDelimiter::Pipe));
for token in ["_", "\t", "<sp>", "||"] {
assert_eq!(
WordDelimiter::from_token(token),
Err(TokenizationError::UnsupportedDelimiter(token.to_owned())),
"{token:?}"
);
}
let space = upper(WordDelimiter::Space);
assert_eq!(check_tokenization(0, space, &spaced, true), Ok(()));
let tabbed = table(&["<pad>", " ", "\t", "A"]);
assert_eq!(
check_tokenization(0, space, &tabbed, true),
Err(TokenizationError::WhitespaceToken("\t".to_owned()))
);
}
#[test]
fn the_a_b_b_table_without_a_is_refused_under_upper_case_and_read_as_written() {
let mixed = table(&["<pad>", "|", "A", "B", "b"]);
assert_eq!(
check_tokenization(0, PIPE_UPPER, &mixed, true),
Err(TokenizationError::UpperWithLowercase('b'))
);
assert_eq!(check_tokenization(0, AS_WRITTEN, &mixed, true), Ok(()));
}
#[test]
fn every_case_statement_is_checked_against_the_table() {
let as_written = Tokenization::new(
WordDelimiter::Pipe,
LetterCase::AsWritten,
Granularity::Character,
&[],
);
let upper = table(&["<pad>", "|", "A", "B"]);
let lower = table(&["<pad>", "|", "a", "b"]);
let both = table(&["<pad>", "|", "A", "a", "B", "b"]);
let han = table(&["<pad>", "|", "中", "文"]);
assert_eq!(check_tokenization(0, PIPE_UPPER, &upper, true), Ok(()));
assert_eq!(check_tokenization(0, as_written, &lower, true), Ok(()));
assert_eq!(check_tokenization(0, as_written, &both, true), Ok(()));
assert_eq!(check_tokenization(0, as_written, &han, true), Ok(()));
assert_eq!(check_tokenization(0, PIPE_UPPER, &han, true), Ok(()));
assert_eq!(
check_tokenization(0, as_written, &upper, true),
Err(TokenizationError::ProjectedAsWritten)
);
assert_eq!(
check_tokenization(0, PIPE_UPPER, &lower, true),
Err(TokenizationError::UpperWithLowercase('a'))
);
assert_eq!(
check_tokenization(0, PIPE_UPPER, &both, true),
Err(TokenizationError::UpperWithLowercase('a'))
);
}
#[test]
fn a_stated_upper_case_needs_no_lexical_a() {
assert_eq!(
check_tokenization(0, PIPE_UPPER, &table(&["A", "|", "B", "C"]), true),
Ok(())
);
assert_eq!(
check_tokenization(
0,
declaring(&["A"]),
&table(&["<pad>", "|", "B", "C", "A"]),
true
),
Ok(())
);
}
#[test]
fn a_lexical_lowercase_letter_is_refused_under_upper_case_without_a_lexical_a() {
assert_eq!(
check_tokenization(
0,
declaring(&["a"]),
&table(&["A", "|", "B", "a", "b"]),
true
),
Err(TokenizationError::UpperWithLowercase('b'))
);
}
fn as_written_declaring(specials: &'static [&'static str]) -> Tokenization {
Tokenization::new(
WordDelimiter::Pipe,
LetterCase::AsWritten,
Granularity::Character,
specials,
)
}
#[test]
fn a_mixed_case_table_whose_a_is_the_blank_is_read_as_written() {
assert_eq!(
check_tokenization(0, AS_WRITTEN, &table(&["a", "|", "A", "B", "b"]), true),
Ok(())
);
}
#[test]
fn a_mixed_case_table_whose_upper_a_is_a_declared_special_is_read_as_written() {
assert_eq!(
check_tokenization(
0,
as_written_declaring(&["A"]),
&table(&["<pad>", "|", "B", "a", "b", "A"]),
true
),
Ok(())
);
}
#[test]
fn an_upper_case_table_is_refused_as_written_without_a_lexical_a() {
assert_eq!(
check_tokenization(0, AS_WRITTEN, &table(&["A", "|", "B", "C"]), true),
Err(TokenizationError::ProjectedAsWritten)
);
}
#[test]
fn the_delimiter_statement_is_checked_against_the_table_and_the_normalizer() {
let absent = Tokenization::new(
WordDelimiter::Absent,
LetterCase::Upper,
Granularity::Character,
&[],
);
let with_pipe = table(&["<pad>", "|", "A"]);
let without_pipe = table(&["<pad>", "A", "B"]);
assert_eq!(check_tokenization(0, PIPE_UPPER, &with_pipe, true), Ok(()));
assert_eq!(check_tokenization(0, absent, &without_pipe, false), Ok(()));
assert_eq!(check_tokenization(0, absent, &with_pipe, false), Ok(()));
assert_eq!(
check_tokenization(0, PIPE_UPPER, &without_pipe, true),
Err(TokenizationError::DelimiterMissing)
);
assert_eq!(
check_tokenization(0, PIPE_UPPER, &with_pipe, false),
Err(TokenizationError::DelimiterUnused)
);
assert_eq!(
check_tokenization(0, absent, &without_pipe, true),
Err(TokenizationError::DelimiterRequired)
);
let space = upper(WordDelimiter::Space);
let with_space = table(&["<pad>", " ", "A"]);
assert_eq!(check_tokenization(0, space, &with_space, true), Ok(()));
assert_eq!(
check_tokenization(0, space, &with_pipe, true),
Err(TokenizationError::DelimiterMissing)
);
assert_eq!(
check_tokenization(0, space, &with_space, false),
Err(TokenizationError::DelimiterUnused)
);
}
#[test]
fn a_subword_class_beside_its_own_characters_is_refused_by_name() {
let subword = table(&["<pad>", "|", "A", "B", "AB"]);
assert_eq!(
check_tokenization(0, PIPE_UPPER, &subword, true),
Err(TokenizationError::NotCharacterLevel("AB".to_owned()))
);
}
#[test]
fn a_multicharacter_token_is_accepted_only_when_declared_special() {
let with_pad = table(&["-", "|", "A", "B", "<pad>"]);
let undeclared = Tokenization::new(
WordDelimiter::Pipe,
LetterCase::Upper,
Granularity::Character,
&[],
);
let declared = Tokenization::new(
WordDelimiter::Pipe,
LetterCase::Upper,
Granularity::Character,
&["<pad>"],
);
assert_eq!(
check_tokenization(0, undeclared, &with_pad, true),
Err(TokenizationError::NotCharacterLevel("<pad>".to_owned()))
);
assert_eq!(check_tokenization(0, declared, &with_pad, true), Ok(()));
}
#[test]
fn a_multicharacter_blank_is_exempt_without_being_declared_special() {
let hf_style = table(&["<pad>", "|", "A", "B"]);
assert_eq!(
check_tokenization(0, PIPE_UPPER, &hf_style, true),
Ok(()),
"id 0, `<pad>`, is the stated blank and needs no `specials` entry"
);
}
#[test]
fn a_single_scalar_non_ascii_letter_passes_as_lexical() {
let accented = table(&["-", "|", "\u{e9}", "\u{df}"]);
let as_written = Tokenization::new(
WordDelimiter::Pipe,
LetterCase::AsWritten,
Granularity::Character,
&[],
);
assert_eq!(check_tokenization(0, as_written, &accented, true), Ok(()));
}