freeswitch_log_parser/
decode.rs1use std::io::{self, BufRead};
12
13#[derive(Debug, Clone, PartialEq, Eq)]
15#[non_exhaustive]
16pub enum Utf8Decode {
17 Clean,
19 TruncatedCodepoint { at: usize },
23 InvalidBytes { at: usize },
26}
27
28#[derive(Debug, Clone, PartialEq, Eq)]
30pub struct DecodedLine {
31 pub text: String,
33 pub decode: Utf8Decode,
34}
35
36pub fn classify_utf8(line: &[u8]) -> Utf8Decode {
39 let mut rest = line;
40 let mut base = 0;
41 let mut truncated_at: Option<usize> = None;
44 loop {
45 match std::str::from_utf8(rest) {
46 Ok(_) => {
47 return match truncated_at {
48 Some(at) => Utf8Decode::TruncatedCodepoint { at },
49 None => Utf8Decode::Clean,
50 };
51 }
52 Err(e) => {
53 let valid = e.valid_up_to();
54 let at = base + valid;
55 match e.error_len() {
56 None => {
57 return Utf8Decode::TruncatedCodepoint {
58 at: truncated_at.unwrap_or(at),
59 };
60 }
61 Some(n) => {
62 let bad = &rest[valid..valid + n];
63 if !is_incomplete_multibyte(bad) {
64 return Utf8Decode::InvalidBytes { at };
65 }
66 truncated_at.get_or_insert(at);
67 base = at + n;
68 rest = &rest[valid + n..];
69 }
70 }
71 }
72 }
73 }
74}
75
76fn is_incomplete_multibyte(seq: &[u8]) -> bool {
79 let Some((&lead, cont)) = seq.split_first() else {
80 return false;
81 };
82 let need = match lead {
83 0xC2..=0xDF => 2,
84 0xE0..=0xEF => 3,
85 0xF0..=0xF4 => 4,
86 _ => return false,
87 };
88 seq.len() < need && cont.iter().all(|&b| (0x80..=0xBF).contains(&b))
89}
90
91pub fn truncate_at_char_boundary(s: &str, max_bytes: usize) -> &str {
98 if s.len() <= max_bytes {
99 return s;
100 }
101 let mut end = max_bytes;
102 while !s.is_char_boundary(end) {
103 end -= 1;
104 }
105 &s[..end]
106}
107
108pub fn decode_log_line(bytes: &[u8]) -> DecodedLine {
114 let mut buf = bytes;
115 if let Some((&b'\n', rest)) = buf.split_last() {
116 buf = match rest.split_last() {
117 Some((&b'\r', rest)) => rest,
118 _ => rest,
119 };
120 }
121 DecodedLine {
122 text: String::from_utf8_lossy(buf).into_owned(),
123 decode: classify_utf8(buf),
124 }
125}
126
127pub fn read_log_lines<R: BufRead>(mut r: R) -> impl Iterator<Item = io::Result<DecodedLine>> {
134 std::iter::from_fn(move || {
135 let mut buf = Vec::new();
136 match r.read_until(b'\n', &mut buf) {
137 Ok(0) => None,
138 Ok(_) => Some(Ok(decode_log_line(&buf))),
139 Err(e) => Some(Err(e)),
140 }
141 })
142}
143
144#[cfg(test)]
145mod tests {
146 use super::*;
147 use std::io::Cursor;
148
149 #[test]
150 fn production_pattern_is_truncated() {
151 assert_eq!(
153 classify_utf8(b"\xe2\x80\x32"),
154 Utf8Decode::TruncatedCodepoint { at: 0 }
155 );
156 }
157
158 #[test]
159 fn truncation_then_clean_tail_latches_truncated() {
160 let line = b"CHANNEL_UN\xe2\x80then a long clean ASCII tail follows here";
162 assert_eq!(
163 classify_utf8(line),
164 Utf8Decode::TruncatedCodepoint { at: 10 }
165 );
166 }
167
168 #[test]
169 fn incomplete_at_end_of_line_is_truncated() {
170 assert_eq!(
171 classify_utf8(b"\xe2\x80"),
172 Utf8Decode::TruncatedCodepoint { at: 0 }
173 );
174 assert_eq!(
175 classify_utf8(b"\xe2"),
176 Utf8Decode::TruncatedCodepoint { at: 0 }
177 );
178 }
179
180 #[test]
181 fn malformed_bytes_are_invalid() {
182 assert_eq!(classify_utf8(b"\xff"), Utf8Decode::InvalidBytes { at: 0 });
183 assert_eq!(classify_utf8(b"\x80"), Utf8Decode::InvalidBytes { at: 0 });
184 assert_eq!(
185 classify_utf8(b"\xc0\x80"),
186 Utf8Decode::InvalidBytes { at: 0 }
187 );
188 }
189
190 #[test]
191 fn genuine_wins_over_earlier_truncation() {
192 match classify_utf8(b"\xe2\x80\x32\xff") {
194 Utf8Decode::InvalidBytes { .. } => {}
195 other => panic!("expected InvalidBytes, got {other:?}"),
196 }
197 }
198
199 #[test]
200 fn clean_line_is_clean() {
201 assert_eq!(classify_utf8("héllo wörld".as_bytes()), Utf8Decode::Clean);
202 assert_eq!(classify_utf8(b"plain ascii"), Utf8Decode::Clean);
203 }
204
205 #[test]
206 fn truncate_at_char_boundary_shorter_input_unchanged() {
207 assert_eq!(truncate_at_char_boundary("abc", 80), "abc");
208 assert_eq!(truncate_at_char_boundary("abc", 3), "abc");
209 assert_eq!(truncate_at_char_boundary("", 0), "");
210 }
211
212 #[test]
213 fn truncate_at_char_boundary_ascii_cut() {
214 assert_eq!(truncate_at_char_boundary("abcdef", 4), "abcd");
215 }
216
217 #[test]
218 fn truncate_at_char_boundary_backs_off_multibyte() {
219 assert_eq!(truncate_at_char_boundary("abcéf", 4), "abc");
221 let s = "ab\u{fffd}cd";
223 assert_eq!(truncate_at_char_boundary(s, 3), "ab");
224 assert_eq!(truncate_at_char_boundary(s, 4), "ab");
225 assert_eq!(truncate_at_char_boundary(s, 5), "ab\u{fffd}");
226 }
227
228 #[test]
229 fn read_log_lines_reports_per_line_verdict() {
230 let mut buf = Vec::new();
231 buf.extend_from_slice(b"first clean line\n");
232 buf.extend_from_slice(b"bad\xe2\x80stuff\n");
233 buf.extend_from_slice(b"third clean line\n");
234
235 let lines: Vec<DecodedLine> = read_log_lines(Cursor::new(buf))
236 .map(|d| d.expect("no io error"))
237 .collect();
238
239 assert_eq!(lines.len(), 3);
240 assert_eq!(lines[0].decode, Utf8Decode::Clean);
241 assert_eq!(lines[2].decode, Utf8Decode::Clean);
242 match lines[1].decode {
243 Utf8Decode::TruncatedCodepoint { .. } => {}
244 ref other => panic!("expected TruncatedCodepoint, got {other:?}"),
245 }
246 assert!(lines[1].text.starts_with("bad"));
248 assert!(lines[1].text.contains('\u{fffd}'));
249 assert!(lines[1].text.ends_with("stuff"));
250 }
251}