urna_format/encoding/
txt_streams.rs1use super::intpack::{IntpackReader, pack_u64s};
34use super::zstd_codec::{zstd_decode, zstd_encode};
35use crate::bytes::le_u64;
36use crate::error::UrnaError;
37use crate::layout::{
38 SECTION_CHUNKS_CANONICAL, SECTION_PAYLOAD_PREFIX_SIZE, SECTION_PAYLOAD_VERSION,
39};
40
41pub const TXT_STREAMS_V1: u8 = 0;
46
47pub(super) const HEADER: usize = 9;
49
50pub(super) fn malformed(reason: impl Into<String>) -> UrnaError {
51 UrnaError::MalformedSectionPayload {
52 section_id: SECTION_CHUNKS_CANONICAL,
53 reason: reason.into(),
54 }
55}
56
57pub(super) fn write_container(kind: u8, count: usize, table: &[u8], streams: &[u8]) -> Vec<u8> {
61 let mut out = Vec::with_capacity(HEADER + table.len() + streams.len());
62 out.push(kind);
63 out.extend_from_slice(&(count as u64).to_le_bytes());
64 out.extend_from_slice(table);
65 out.extend_from_slice(streams);
66 out
67}
68
69pub(super) fn build_canonical(count: usize, bodies: &[Vec<u8>]) -> crate::Result<Vec<u8>> {
74 let total: usize = bodies.iter().map(|b| b.len()).sum();
75 let mut out = Vec::with_capacity(SECTION_PAYLOAD_PREFIX_SIZE + count * 4 + total);
76 out.extend_from_slice(&SECTION_PAYLOAD_VERSION.to_le_bytes());
77 out.extend_from_slice(&(count as u64).to_le_bytes());
78 for raw in bodies {
79 let len = u32::try_from(raw.len())
80 .map_err(|_| malformed("txt_streams: stream longer than u32"))?;
81 out.extend_from_slice(&len.to_le_bytes());
82 out.extend_from_slice(raw);
83 }
84 Ok(out)
85}
86
87pub fn encode_txt_streams(texts: &[String]) -> crate::Result<Vec<u8>> {
91 let mut streams: Vec<u8> = Vec::new();
92 let mut offsets: Vec<u64> = Vec::with_capacity(texts.len() + 1);
95 offsets.push(0);
96 for t in texts {
97 streams.extend_from_slice(&zstd_encode(t.as_bytes())?);
98 offsets.push(streams.len() as u64);
99 }
100 let table = pack_u64s(&offsets);
101 Ok(write_container(
102 TXT_STREAMS_V1,
103 texts.len(),
104 &table,
105 &streams,
106 ))
107}
108
109pub struct TxtStreams<'a> {
114 streams: &'a [u8],
115 offsets: Vec<u64>,
116 count: usize,
117}
118
119impl<'a> TxtStreams<'a> {
120 pub fn parse(bytes: &'a [u8]) -> crate::Result<Self> {
121 let (kind, rest) = bytes
122 .split_first()
123 .ok_or_else(|| malformed("txt_streams: empty"))?;
124 if *kind != TXT_STREAMS_V1 {
125 return Err(malformed(format!("txt_streams: unknown kind {}", *kind)));
126 }
127 if rest.len() < 8 {
128 return Err(malformed("txt_streams: truncated count"));
129 }
130 let declared = le_u64(&rest[0..8])?;
131 let table_bytes = &rest[8..];
132 let reader = IntpackReader::parse(table_bytes)?;
136 if reader.is_empty() {
137 return Err(malformed("txt_streams: offset table must hold n+1 >= 1"));
138 }
139 let count = reader.len() - 1;
140 if declared != count as u64 {
143 return Err(malformed("txt_streams: declared count != offset count - 1"));
144 }
145 let mut offsets = Vec::with_capacity(reader.len());
146 for i in 0..reader.len() {
147 offsets.push(reader.get(i)?);
148 }
149 if offsets[0] != 0 {
150 return Err(malformed("txt_streams: first offset must be 0"));
151 }
152 let table_len = pack_u64s(&offsets).len();
157 if table_len > table_bytes.len() {
158 return Err(malformed("txt_streams: truncated offset table"));
159 }
160 let streams = &table_bytes[table_len..];
161 for w in offsets.windows(2) {
164 if w[1] < w[0] {
165 return Err(malformed("txt_streams: non-monotonic offsets"));
166 }
167 }
168 if offsets.last().copied() != Some(streams.len() as u64) {
169 return Err(malformed("txt_streams: final offset != streams length"));
170 }
171 Ok(Self {
172 streams,
173 offsets,
174 count,
175 })
176 }
177
178 #[inline]
179 pub fn len(&self) -> usize {
180 self.count
181 }
182
183 #[inline]
184 pub fn is_empty(&self) -> bool {
185 self.count == 0
186 }
187
188 fn stream(&self, i: usize) -> crate::Result<&'a [u8]> {
190 if i >= self.count {
191 return Err(malformed("txt_streams: index out of range"));
192 }
193 let start = self.offsets[i] as usize;
194 let end = self.offsets[i + 1] as usize;
195 self.streams
196 .get(start..end)
197 .ok_or_else(|| malformed("txt_streams: stream slice out of bounds"))
198 }
199
200 pub fn text(&self, i: usize) -> crate::Result<String> {
204 let raw = zstd_decode(self.stream(i)?).map_err(|e| match e {
205 UrnaError::MalformedSectionPayload { reason, .. } => malformed(reason),
206 other => other,
207 })?;
208 String::from_utf8(raw).map_err(|e| malformed(format!("txt_streams: invalid utf-8: {}", e)))
209 }
210}
211
212pub fn decode(bytes: &[u8]) -> crate::Result<Vec<u8>> {
217 let parsed = TxtStreams::parse(bytes)?;
218 let mut bodies: Vec<Vec<u8>> = Vec::with_capacity(parsed.count);
219 for i in 0..parsed.count {
220 let raw = zstd_decode(parsed.stream(i)?).map_err(|e| match e {
221 UrnaError::MalformedSectionPayload { reason, .. } => malformed(reason),
222 other => other,
223 })?;
224 std::str::from_utf8(&raw)
227 .map_err(|e| malformed(format!("txt_streams: invalid utf-8: {}", e)))?;
228 bodies.push(raw);
229 }
230 build_canonical(parsed.count, &bodies)
231}
232
233#[cfg(test)]
234mod tests {
235 use super::*;
236 use crate::sections::encode_chunks_canonical;
237
238 fn texts(items: &[&str]) -> Vec<String> {
239 items.iter().map(|s| s.to_string()).collect()
240 }
241
242 fn assert_byte_identical(items: &[&str]) {
243 let t = texts(items);
244 let packed = encode_txt_streams(&t).unwrap();
245 assert_eq!(packed[0], TXT_STREAMS_V1);
246 let raw = encode_chunks_canonical(&t).unwrap();
247 assert_eq!(decode(&packed).unwrap(), raw, "decode must rebuild raw");
248 }
249
250 #[test]
251 #[cfg_attr(miri, ignore)] fn byte_identical_across_corpora() {
253 assert_byte_identical(&[]);
254 assert_byte_identical(&["only one"]);
255 assert_byte_identical(&["primeiro", "segundo", "terceiro"]);
256 assert_byte_identical(&["coração", "informação", "açaí é ótimo", ""]);
258 }
259
260 #[test]
261 #[cfg_attr(miri, ignore)] fn o1_seek_returns_the_right_stream() {
263 let t = texts(&["alpha", "beta", "gama", "delta"]);
264 let packed = encode_txt_streams(&t).unwrap();
265 let parsed = TxtStreams::parse(&packed).unwrap();
266 assert_eq!(parsed.len(), 4);
267 for (i, s) in t.iter().enumerate() {
268 assert_eq!(&parsed.text(i).unwrap(), s, "text({}) mismatch", i);
269 }
270 assert!(parsed.text(4).is_err(), "oob index must error");
271 }
272
273 #[test]
274 #[cfg_attr(miri, ignore)] fn determinism_two_encodes_byte_identical() {
276 let t = texts(&["a", "bb", "ccc", "coração"]);
277 assert_eq!(
278 encode_txt_streams(&t).unwrap(),
279 encode_txt_streams(&t).unwrap()
280 );
281 }
282}