paircomp_core/within_line.rs
1use crate::file::open_regular_file;
2use crate::{Error, Fingerprint, LineInfo};
3use std::io::{self, Read};
4use std::path::Path;
5
6/// Inspects a 1-based line of a regular file.
7///
8/// Returns `Some` with its raw byte length, including any terminating LF, or
9/// `None` beyond EOF. CR is ordinary content; a trailing LF does not create an
10/// additional empty line. Invalid UTF-8 is accepted.
11///
12/// Reopens the path and requires [file stability](crate#file-stability).
13///
14/// # Errors
15///
16/// Returns [`Error::InvalidLineNumber`] for line zero, [`Error::Io`] if
17/// metadata lookup, opening, or reading fails, or [`Error::NotRegularFile`]
18/// for a non-regular input.
19///
20/// # Examples
21///
22/// See [`fingerprint_line_prefix`] for an example inspecting and hashing a line.
23pub fn inspect_line(path: &Path, line: u64) -> Result<Option<LineInfo>, Error> {
24 let byte_len = scan_line(open_regular_file(path)?, line, u64::MAX, |_| {})?;
25 Ok((byte_len > 0).then_some(LineInfo { byte_len }))
26}
27
28/// Hashes only the first `byte_count` raw bytes of a 1-based line.
29///
30/// Zero bytes or an absent line hashes the empty sequence. Requests beyond
31/// the line include its entire content, including LF if present, without
32/// entering the next line or adding an EOF marker. Invalid UTF-8 is accepted.
33/// The path must identify a readable regular file even when `byte_count` is
34/// zero.
35///
36/// Reopens the path and requires [file stability](crate#file-stability).
37///
38/// # Errors
39///
40/// Returns [`Error::InvalidLineNumber`] for line zero, [`Error::Io`] if
41/// metadata lookup, opening, or reading fails, or [`Error::NotRegularFile`]
42/// for a non-regular input.
43///
44/// # Examples
45///
46/// A line-local prefix excludes preceding lines and may end inside a UTF-8
47/// code point. Longer requests stop at the selected line's LF.
48///
49/// ```
50/// use paircomp_core::{fingerprint_line_prefix, fingerprint_through_line, inspect_line};
51/// use std::fs;
52///
53/// let directory = std::env::temp_dir()
54/// .join(format!("paircomp-prefix-example-{}", std::process::id()));
55/// fs::create_dir(&directory)?;
56/// let path = directory.join("sample.txt");
57/// fs::write(&path, "header\ncafé\nnext\n")?;
58///
59/// assert_eq!(inspect_line(&path, 2)?.map(|line| line.byte_len), Some(6));
60/// let prefix = fingerprint_line_prefix(&path, 2, 4)?;
61/// assert_eq!(prefix.as_bytes(), blake3::hash(b"caf\xc3").as_bytes());
62/// let whole_line = fingerprint_line_prefix(&path, 2, 100)?;
63/// assert_eq!(whole_line.as_bytes(), blake3::hash("café\n".as_bytes()).as_bytes());
64///
65/// // A file prefix also includes every preceding line.
66/// let file_prefix = fingerprint_through_line(&path, 2)?;
67/// assert_eq!(file_prefix.as_bytes(), blake3::hash("header\ncafé\n".as_bytes()).as_bytes());
68///
69/// // Zero bytes and an absent line both hash the empty sequence.
70/// assert_eq!(fingerprint_line_prefix(&path, 2, 0)?, fingerprint_line_prefix(&path, 4, 100)?);
71/// fs::remove_dir_all(&directory)?;
72/// # Ok::<(), Box<dyn std::error::Error>>(())
73/// ```
74pub fn fingerprint_line_prefix(
75 path: &Path,
76 line: u64,
77 byte_count: u64,
78) -> Result<Fingerprint, Error> {
79 let mut hasher = blake3::Hasher::new();
80 scan_line(open_regular_file(path)?, line, byte_count, |bytes| {
81 hasher.update(bytes);
82 })?;
83 Ok(Fingerprint(*hasher.finalize().as_bytes()))
84}
85
86/// Maps a 1-based byte position to a 1-based Unicode code-point position.
87///
88/// Every byte of a multibyte character maps to the same character. Immediately
89/// after an existing line, returns the next character position. CR and LF each
90/// count as a code point; these positions are not visual editor columns.
91///
92/// Validates the entire selected line, returning `None` if any part is invalid
93/// UTF-8, even after the requested byte. An absent line permits byte position
94/// 1 and returns `None`. Other lines' encodings do not affect the result.
95///
96/// Reopens the path and requires [file stability](crate#file-stability).
97///
98/// # Errors
99///
100/// Returns [`Error::InvalidLineNumber`] for line zero and
101/// [`Error::InvalidBytePosition`] for byte zero or a position more than one
102/// past the line. Coordinate validation also applies to invalid UTF-8 lines.
103/// Returns [`Error::Io`] if metadata lookup, opening, or reading fails,
104/// [`Error::NotRegularFile`] for a non-regular input, or [`Error::FileTooLarge`]
105/// if the next character position cannot fit in `u64`.
106///
107/// # Examples
108///
109/// Both bytes of `é` map to character 4. The terminating LF counts as a
110/// separate code point, and the position immediately after it is also valid.
111///
112/// ```
113/// use paircomp_core::utf8_character_position;
114/// use std::fs;
115///
116/// let directory = std::env::temp_dir()
117/// .join(format!("paircomp-utf8-example-{}", std::process::id()));
118/// fs::create_dir(&directory)?;
119/// let path = directory.join("sample.txt");
120/// fs::write(&path, b"caf\xc3\xa9\nvalid prefix\xff\n")?;
121///
122/// assert_eq!(utf8_character_position(&path, 1, 4)?, Some(4));
123/// assert_eq!(utf8_character_position(&path, 1, 5)?, Some(4));
124/// assert_eq!(utf8_character_position(&path, 1, 6)?, Some(5)); // LF
125/// assert_eq!(utf8_character_position(&path, 1, 7)?, Some(6)); // After LF
126///
127/// // Invalid UTF-8 later in line 2 suppresses even its first position.
128/// assert_eq!(utf8_character_position(&path, 2, 1)?, None);
129/// // A trailing LF does not create a third line.
130/// assert_eq!(utf8_character_position(&path, 3, 1)?, None);
131/// fs::remove_dir_all(&directory)?;
132/// # Ok::<(), Box<dyn std::error::Error>>(())
133/// ```
134pub fn utf8_character_position(path: &Path, line: u64, byte: u64) -> Result<Option<u64>, Error> {
135 if byte == 0 {
136 return Err(Error::InvalidBytePosition);
137 }
138 character_position(open_regular_file(path)?, line, byte)
139}
140
141/// Maps a byte position using a reader positioned at the start of the file.
142///
143/// The caller must reject byte zero before calling this helper.
144fn character_position(reader: impl Read, line: u64, byte: u64) -> Result<Option<u64>, Error> {
145 // Preserve an incomplete code point across chunks; UTF-8 needs at most four bytes.
146 let mut pending = [0_u8; 4];
147 let mut pending_len = 0;
148 let mut valid = true;
149 let mut seen = 0_u64;
150 let mut characters = 0_u64;
151 let mut position = None;
152 let byte_len = scan_line(reader, line, u64::MAX, |bytes| {
153 for &value in bytes {
154 if !valid {
155 // Keep scanning after invalid UTF-8 so the final byte bounds and
156 // any later I/O errors are still checked.
157 break;
158 }
159 seen += 1;
160 pending[pending_len] = value;
161 pending_len += 1;
162 match std::str::from_utf8(&pending[..pending_len]) {
163 Ok(_) => {
164 characters += 1;
165 if byte > seen - pending_len as u64 && byte <= seen {
166 position = Some(characters);
167 }
168 pending_len = 0;
169 }
170 Err(error) if error.error_len().is_none() && pending_len < 4 => {}
171 Err(_) => valid = false,
172 }
173 }
174 })?;
175 // Subtraction also handles a hypothetical u64::MAX-byte line without overflow.
176 if byte - 1 > byte_len {
177 return Err(Error::InvalidBytePosition);
178 }
179 if !valid || pending_len != 0 || byte_len == 0 {
180 return Ok(None);
181 }
182 if byte - 1 == byte_len {
183 return Ok(Some(characters.checked_add(1).ok_or(Error::FileTooLarge)?));
184 }
185 Ok(position)
186}
187
188/// Visits bounded chunks from a single line, without retaining its contents.
189///
190/// The reader must start at the beginning of the file. Visits at most
191/// `byte_limit` bytes, including the terminating LF if reached, and returns the
192/// number visited. An absent line or a zero limit visits nothing and returns
193/// zero. Line zero is invalid even when the limit is zero.
194fn scan_line(
195 mut reader: impl Read,
196 line: u64,
197 byte_limit: u64,
198 mut visit: impl FnMut(&[u8]),
199) -> Result<u64, Error> {
200 if line == 0 {
201 return Err(Error::InvalidLineNumber);
202 }
203 let mut lines_to_skip = line - 1;
204 let mut byte_len = 0_u64;
205 let mut buffer = [0_u8; 8192];
206 while byte_len < byte_limit {
207 let read = match reader.read(&mut buffer) {
208 Ok(0) => break,
209 Ok(read) => read,
210 Err(error) if error.kind() == io::ErrorKind::Interrupted => continue,
211 Err(error) => return Err(error.into()),
212 };
213 let mut start = 0;
214 while lines_to_skip > 0 && start < read {
215 if buffer[start] == b'\n' {
216 lines_to_skip -= 1;
217 }
218 start += 1;
219 }
220 if start == read {
221 continue;
222 }
223 let bytes = &buffer[start..read];
224 let line_end = bytes.iter().position(|&value| value == b'\n');
225 let available = line_end.map_or(bytes.len(), |index| index + 1);
226 let included = (available as u64).min(byte_limit - byte_len) as usize;
227 byte_len = byte_len
228 .checked_add(included as u64)
229 .ok_or(Error::FileTooLarge)?;
230 visit(&bytes[..included]);
231 if line_end.is_some() {
232 break;
233 }
234 }
235 Ok(byte_len)
236}
237
238#[cfg(test)]
239mod tests;