use super::*;
fn compressed_len_of_tokens(tokens: &[u32]) -> usize {
(tokens.len() as f32 * 4.0 / compression_ratio_of_tokens(tokens)).round() as usize
}
fn compressed_len_of_text(text: &str) -> usize {
(text.len() as f32 / compression_ratio_of_text(text)).round() as usize
}
#[test]
fn compression_ratio_detects_repetition() {
let unique: Vec<u32> = (0..200).collect();
let repeated = vec![42u32; 200];
assert!(compression_ratio_of_tokens(&repeated) > compression_ratio_of_tokens(&unique));
assert!(compression_ratio_of_tokens(&repeated) > 2.4); assert_eq!(compressed_len_of_tokens(&repeated), 13);
}
#[test]
fn compression_ratio_of_tokens_matches_apple_libcompression_golden() {
let doc_example: Vec<u32> = [1212, 318, 257, 1332, 13].repeat(2);
assert_eq!(doc_example.len(), 10, "40 raw i32-LE bytes");
assert_eq!(compressed_len_of_tokens(&doc_example), 24);
let phrase_x8: Vec<u32> = [1212, 318, 257, 1332, 13].repeat(8);
assert_eq!(compressed_len_of_tokens(&phrase_x8), 24); }
#[test]
fn compression_ratio_of_tokens_empty_is_zero_matching_swift() {
assert_eq!(compression_ratio_of_tokens(&[]), 0.0);
}
#[test]
fn compression_ratio_of_text_empty_is_infinite() {
assert_eq!(compression_ratio_of_text(""), f32::INFINITY);
}
#[test]
fn compression_ratio_of_text_matches_apple_libcompression_golden() {
let the_x20 = "the the the the the the the the the the the the the the the the the the the the";
assert_eq!(the_x20.len(), 79);
assert_eq!(compressed_len_of_text(the_x20), 9);
let fox_x5 = "the quick brown fox jumps over the lazy dog ".repeat(5);
assert_eq!(fox_x5.len(), 220);
assert_eq!(compressed_len_of_text(&fox_x5), 48); }
#[test]
fn compression_ratio_of_text_detects_repetition() {
let unique = "the quick brown fox jumps over the lazy dog";
let repeated = "the the the the the the the the the the the the the the the the the the the the";
assert!(compression_ratio_of_text(repeated) > compression_ratio_of_text(unique));
}
#[test]
fn normalized_matches_swift_semantics() {
assert_eq!(normalized("Hello, World!"), "hello world");
assert_eq!(normalized("multi-word_test"), "multi wordtest");
assert_eq!(normalized(" a b "), "a b");
}
#[test]
fn normalized_deletes_underscores_fusing_the_surrounding_word() {
assert_eq!(normalized("under_score"), "underscore");
assert_eq!(normalized("a_b_c"), "abc");
}
#[test]
fn normalized_only_the_literal_ascii_hyphen_becomes_a_space() {
assert_eq!(normalized("em\u{2014}dash"), "emdash"); assert_eq!(normalized("en\u{2013}dash"), "endash"); assert_eq!(normalized("a-b"), "a b");
}
#[test]
fn normalized_preserves_unicode_letters() {
assert_eq!(normalized("Café Résumé"), "café résumé");
}
#[test]
fn normalized_collapses_multi_hyphen_runs_and_drops_other_punctuation() {
assert_eq!(normalized("a--b"), "a b"); assert_eq!(normalized("100%"), "100"); }
#[test]
fn normalized_empty_and_blank_inputs() {
assert_eq!(normalized(""), "");
assert_eq!(normalized(" "), "");
assert_eq!(normalized("!!!"), "");
}
#[test]
fn trims_special_token_wrapping() {
assert_eq!(trim_special_token_chars("<|endoftext|>"), "endoftext");
assert_eq!(trim_special_token_chars("plain"), "plain");
}
#[test]
fn trim_special_token_chars_is_a_character_class_trim_not_a_fixed_affix() {
assert_eq!(trim_special_token_chars("<<|x|>"), "x");
assert_eq!(trim_special_token_chars("<|a|><|b|>"), "a|><|b");
}
#[test]
fn normalized_deletes_apostrophes_and_curly_quotes() {
assert_eq!(normalized("don't"), "dont");
assert_eq!(
normalized("\u{2018}quoted\u{2019} \u{201C}text\u{201D}"),
"quoted text"
);
}
use crate::audio::whisper::result::WordTiming;
fn word(text: &str, start: f32, end: f32) -> WordTiming {
WordTiming::new(text, vec![1], start, end, 0.9)
}
#[test]
fn common_prefix_compares_normalized_and_returns_current_elements() {
let previous = [
word(" Hey", 0.0, 0.2),
word(" you", 0.2, 0.4),
word(" there", 0.4, 0.6),
];
let current = [
word(" hey,", 0.1, 0.3),
word(" You", 0.3, 0.5),
word(" friend", 0.5, 0.7),
];
let prefix = find_longest_common_prefix(&previous, ¤t);
assert_eq!(prefix.len(), 2);
assert_eq!(prefix[0].word(), " hey,");
assert!((prefix[0].start() - 0.1).abs() < 1e-6, "newer timings kept");
assert_eq!(
find_longest_common_prefix(&previous[..1], ¤t).len(),
1
);
assert!(find_longest_common_prefix(&[], ¤t).is_empty());
}
#[test]
fn different_suffix_is_current_past_the_common_prefix() {
let previous = [word(" Hey", 0.0, 0.2), word(" you", 0.2, 0.4)];
let current = [
word(" hey", 0.0, 0.2),
word(" you", 0.2, 0.4),
word(" friend", 0.4, 0.7),
];
let suffix = find_longest_different_suffix(&previous, ¤t);
assert_eq!(suffix.len(), 1);
assert_eq!(suffix[0].word(), " friend");
let disjoint = [word(" But", 0.0, 0.2)];
assert_eq!(find_longest_different_suffix(&disjoint, ¤t).len(), 3);
}