Skip to main content

cosh_tools/fs/
read.rs

1//! Read file(s) and format the output as hashline sections.
2//!
3//! Each call to [`read`] carries one or more [`Target`] entries.  A target can
4//! request the whole file, a syntactic block at a given line, a definition
5//! block matching a name (symbol, struct, class, …), or one or more exact line
6//! ranges (a plain slice — no AST involved) via the advertised
7//! `offset`/`limit` shape, or via the legacy `line_range` string.  When the
8//! name search targets a directory the entire tree is walked recursively.
9//!
10//! Output is hashline-numbered (`N| text`) so the model can anchor edits
11//! directly.  Token-saving behaviors:
12//!
13//! - **Exact range reads** — `offset: 50, limit: 51` (or the legacy
14//!   `line_range: "50-100"` string, which also supports comma-separated
15//!   disjoint ranges like `"10-20,200-220"`) returns only the requested
16//!   lines, with a footer reporting how many lines remain and how to continue.
17//! - **Elided blocks** — blocks larger than [`ELIDE_MIN_BLOCK_LINES`] are
18//!   shown as their head/tail with a `…` marker and a footer telling the model
19//!   which `offset`/`limit` to re-read when it needs the elided body.
20//! - **Column truncation** — lines longer than [`MAX_COLUMN`] characters are
21//!   cut with a `...` suffix and a notice.
22//! - **Seen lines** — every surfaced line is recorded against the snapshot
23//!   tag ([`rollback::record_seen_lines`]), so the recovery path warns when an
24//!   edit anchors lines the model never saw (including elided interiors).
25use super::types::{FsMetadata, FsRead, Target};
26
27use cosh_sdk::hashline::{
28    format,
29    fs::{DiskFilesystem, Filesystem},
30    normalize,
31};
32use cosh_sdk::rollback;
33use schemars::JsonSchema;
34use serde::{Deserialize, Serialize};
35use std::path::Path;
36
37/// Blocks at or below this many lines are returned whole; larger blocks are
38/// elided to their head/tail.
39const ELIDE_MIN_BLOCK_LINES: usize = 24;
40/// Lines kept from the head of an elided block.
41const ELIDE_KEEP_HEAD: usize = 3;
42/// Lines kept from the tail of an elided block.
43const ELIDE_KEEP_TAIL: usize = 2;
44/// Lines longer than this (characters) are truncated with a `...` suffix.
45const MAX_COLUMN: usize = 200;
46
47#[derive(Debug, Serialize, Deserialize, JsonSchema)]
48pub struct ReadResult {
49    pub path: String,
50    pub file_hash: String,
51    pub header: String,
52    pub content: String,
53    pub warnings: Option<String>,
54}
55
56/// Run every target and return hashline-formatted results.
57///
58/// Each target produces one or more [`ReadResult`] entries. Errors and
59/// incomplete reads are recorded as warnings inside each result instead of
60/// aborting the entire operation.
61pub async fn read(metadata: FsMetadata, tg: FsRead) -> Vec<ReadResult> {
62    let fs = DiskFilesystem::new();
63    let mut results: Vec<ReadResult> = Vec::new();
64
65    for target in tg.targets {
66        results.extend(read_target(fs.clone(), target, &metadata).await);
67    }
68
69    results
70}
71
72async fn read_target(fs: DiskFilesystem, target: Target, metadata: &FsMetadata) -> Vec<ReadResult> {
73    match metadata.fs_guard(&target.path) {
74        Ok(validated_path) => {
75            let path = validated_path.to_string_lossy().to_string();
76            let target = Target {
77                path,
78                line: target.line,
79                symbol: target.symbol,
80                line_range: target.line_range,
81                offset: target.offset,
82                limit: target.limit,
83            };
84            read_target_impl(fs, target).await
85        }
86        Err(e) => vec![ReadResult {
87            path: target.path.clone(),
88            file_hash: String::new(),
89            header: String::new(),
90            content: String::new(),
91            warnings: Some(e),
92        }],
93    }
94}
95
96/// Split file text into lines, dropping the empty tail element left by a
97/// trailing newline so line N maps to `lines[N - 1]`.
98fn file_lines(text: &str) -> Vec<String> {
99    let mut lines: Vec<String> = text.split('\n').map(str::to_string).collect();
100    if lines.last().is_some_and(|l| l.is_empty()) {
101        lines.pop();
102    }
103    lines
104}
105
106/// Parse a comma-separated list of 1-based inclusive line ranges (`"start-end"`).
107fn parse_line_ranges(raw: &str) -> Result<Vec<(u32, u32)>, String> {
108    let mut ranges = Vec::new();
109    for part in raw.split(',') {
110        let part = part.trim();
111        if part.is_empty() {
112            return Err(format!(
113                "line_range must be \"start-end\" (1-based, inclusive), got: {raw}"
114            ));
115        }
116        let (start, end) = part.split_once('-').ok_or_else(|| {
117            format!("line_range must be \"start-end\" (1-based, inclusive), got: {raw}")
118        })?;
119        let start: u32 = start.trim().parse().map_err(|_| {
120            format!("line_range must be \"start-end\" (1-based, inclusive), got: {raw}")
121        })?;
122        let end: u32 = end.trim().parse().map_err(|_| {
123            format!("line_range must be \"start-end\" (1-based, inclusive), got: {raw}")
124        })?;
125        if start < 1 || end < start {
126            return Err(format!(
127                "line_range must satisfy 1 <= start <= end, got: {raw}"
128            ));
129        }
130        ranges.push((start, end));
131    }
132    if ranges.is_empty() {
133        return Err("line_range must not be empty".to_string());
134    }
135    Ok(ranges)
136}
137
138/// Truncate a line to [`MAX_COLUMN`] chars with a `...` suffix when longer.
139/// Returns the display text and whether truncation happened.
140fn truncate_column(line: &str) -> (String, bool) {
141    let count = line.chars().count();
142    if count <= MAX_COLUMN {
143        return (line.to_string(), false);
144    }
145    let mut s: String = line.chars().take(MAX_COLUMN - 3).collect();
146    s.push_str("...");
147    (s, true)
148}
149
150/// Render lines `start..=end` (1-based, clamped to the file) as numbered
151/// hashline lines, truncating over-long columns. Records every surfaced line
152/// (with its ORIGINAL text) into `seen` so recovery can distinguish shown
153/// anchors from elided ones.
154fn render_lines(
155    lines: &[String],
156    start: u32,
157    end: u32,
158    seen: &mut Vec<(u32, String)>,
159) -> (Vec<String>, bool) {
160    let mut out = Vec::new();
161    let mut truncated_any = false;
162    for n in start..=end.min(lines.len() as u32) {
163        let text = &lines[(n - 1) as usize];
164        let (display, was_truncated) = truncate_column(text);
165        truncated_any |= was_truncated;
166        out.push(format::format_numbered_line(n, &display));
167        seen.push((n, text.clone()));
168    }
169    (out, truncated_any)
170}
171
172/// Render a block-relative line slice as numbered hashline lines, labelling
173/// each line with its absolute 1-based number: `lines[0]` is `start_line`.
174///
175/// Unlike [`render_lines`] (whose input is the whole file, indexed by absolute
176/// line number), this indexes `lines` relatively so blocks that start past line
177/// 1 render their body correctly instead of clamping to an empty range.
178fn render_block_lines(
179    lines: &[String],
180    start_line: u32,
181    end: u32,
182    seen: &mut Vec<(u32, String)>,
183) -> (Vec<String>, bool) {
184    let mut out = Vec::new();
185    let mut truncated_any = false;
186    for (i, text) in lines.iter().enumerate() {
187        let n = start_line + i as u32;
188        if n > end {
189            break;
190        }
191        let (display, was_truncated) = truncate_column(text);
192        truncated_any |= was_truncated;
193        out.push(format::format_numbered_line(n, &display));
194        seen.push((n, text.clone()));
195    }
196    (out, truncated_any)
197}
198
199/// Output of [`read_range_body`]: numbered body, seen lines, content notices,
200/// and warnings (in that order).
201type RangeBody = (String, Vec<(u32, String)>, Vec<String>, Vec<String>);
202
203/// Build the numbered body + seen lines + content notices + warnings for one
204/// or more line ranges.
205///
206/// - Content notices (appended to the body the model reads): elision markers
207///   for gaps between ranges, column-truncation, and the remaining-lines hint.
208/// - Warnings (delivery problems): ranges past EOF, which were skipped.
209fn read_range_body(lines: &[String], ranges: &[(u32, u32)]) -> RangeBody {
210    let total = lines.len() as u32;
211    let mut parts: Vec<String> = Vec::new();
212    let mut seen: Vec<(u32, String)> = Vec::new();
213    let mut notices: Vec<String> = Vec::new();
214    let mut warnings: Vec<String> = Vec::new();
215    let mut truncated_any = false;
216    let mut prev_end: Option<u32> = None;
217
218    for &(start, end) in ranges {
219        if start > total {
220            warnings.push(format!(
221                "[Range {start}-{end} is beyond end of file ({total} lines total); skipped]"
222            ));
223            continue;
224        }
225        let eff_end = end.min(total);
226        if let Some(pe) = prev_end
227            && start > pe + 1
228        {
229            parts.push(format!("[…{} lines between ranges elided]", start - pe - 1));
230        }
231        let (rendered, tr) = render_lines(lines, start, eff_end, &mut seen);
232        truncated_any |= tr;
233        parts.extend(rendered);
234        prev_end = Some(eff_end);
235    }
236
237    if truncated_any {
238        notices.push(format!(
239            "[Some lines were truncated to {MAX_COLUMN} columns; use offset/limit to re-read specific lines for the full content]"
240        ));
241    }
242    if let Some(last_shown) = seen.iter().map(|(n, _)| *n).max()
243        && last_shown < total
244    {
245        notices.push(format!(
246            "[{} more lines in file; continue with offset {}, limit {}]",
247            total - last_shown,
248            last_shown + 1,
249            total - last_shown
250        ));
251    }
252    (parts.join("\n"), seen, notices, warnings)
253}
254
255/// Elide a block's interior: blocks at or below [`ELIDE_MIN_BLOCK_LINES`] are
256/// returned whole; larger ones keep [`ELIDE_KEEP_HEAD`] head lines, a `…`
257/// marker, and [`ELIDE_KEEP_TAIL`] tail lines. Returns the body lines, the
258/// elided line range (absolute, 1-based) when interior lines were dropped, and
259/// whether any line was column-truncated. Only the KEPT lines are recorded as
260/// seen — an edit anchoring an elided interior correctly warns on recovery.
261fn elide_block(
262    lines: &[String],
263    start_line: u32,
264    seen: &mut Vec<(u32, String)>,
265) -> (Vec<String>, Option<(u32, u32)>, bool) {
266    let n = lines.len();
267    if n <= ELIDE_MIN_BLOCK_LINES {
268        let (out, truncated_any) =
269            render_block_lines(lines, start_line, start_line + n as u32 - 1, seen);
270        return (out, None, truncated_any);
271    }
272    let head_n = ELIDE_KEEP_HEAD.min(n);
273    let tail_n = ELIDE_KEEP_TAIL.min(n.saturating_sub(head_n));
274    let mut out = Vec::new();
275    let mut truncated_any = false;
276    let (head, tr1) = render_block_lines(lines, start_line, start_line + head_n as u32 - 1, seen);
277    out.extend(head);
278    truncated_any |= tr1;
279    out.push("…".to_string());
280    let tail_start = n - tail_n; // 0-based index of the first kept tail line
281    let (tail, tr2) = render_block_lines(
282        &lines[tail_start..],
283        start_line + tail_start as u32,
284        start_line + n as u32 - 1,
285        seen,
286    );
287    out.extend(tail);
288    truncated_any |= tr2;
289    let elided = (
290        start_line + head_n as u32,
291        start_line + tail_start as u32 - 1,
292    );
293    (out, Some(elided), truncated_any)
294}
295
296/// Record the lines a read surfaced against the snapshot tag for `path`.
297fn record_seen(path: &str, file_hash: &str, seen: &[(u32, String)]) {
298    rollback::record_seen_lines(path, file_hash, seen);
299}
300
301/// Append informational notice lines to `content` (they are read by the model
302/// as part of the output, not surfaced as warnings — warnings are reserved for
303/// delivery problems).
304fn append_content_notices(content: &mut String, notices: &[String]) {
305    for notice in notices {
306        content.push('\n');
307        content.push_str(notice);
308    }
309}
310
311async fn read_target_impl(fs: DiskFilesystem, target: Target) -> Vec<ReadResult> {
312    if let Some(name) = target.symbol {
313        return search_symbol(&fs, &target.path, &name).await;
314    }
315
316    // Normalize the advertised `offset`/`limit` shape (the CC-trained
317    // form) into a single contiguous range before the legacy `line_range`
318    // handling: `offset` alone degrades to the legacy `line` behavior
319    // (syntactic block), `offset`+`limit` becomes one plain slice.
320    // Explicit `offset`+`limit` wins over a legacy `line_range` string.
321    let target = if target.limit.is_some() && target.offset.is_none() {
322        let mut t = target;
323        t.limit = None; // `limit` without `offset` is meaningless — drop it.
324        t
325    } else if let Some(offset) = target.offset {
326        let mut t = target;
327        match t.limit {
328            Some(limit) => {
329                if limit < 1 {
330                    return vec![ReadResult {
331                        path: t.path.clone(),
332                        file_hash: String::new(),
333                        header: String::new(),
334                        content: String::new(),
335                        warnings: Some(format!("`limit` must be >= 1, got: {limit}")),
336                    }];
337                }
338                let start = offset.max(1).min(u32::MAX as usize);
339                // Saturate at u32::MAX (the legacy range parser's domain):
340                // huge `limit` values must clamp, never overflow — the
341                // slice is clamped to EOF later anyway.
342                let end = start.saturating_add(limit - 1).min(u32::MAX as usize);
343                t.line_range = Some(format!("{start}-{end}"));
344            }
345            // `offset` without `limit` reads the syntactic block containing
346            // that line — the legacy `line` behavior.
347            None => t.line = Some(offset),
348        }
349        t.offset = None;
350        t.limit = None;
351        t
352    } else {
353        target
354    };
355
356    if Path::new(&target.path).is_dir() {
357        return vec![ReadResult {
358            path: target.path.clone(),
359            file_hash: String::new(),
360            header: String::new(),
361            content: String::new(),
362            warnings: Some(format!(
363                "cannot read directory `{path}` without a `symbol` filter. \
364                 When `path` is a directory, a `symbol` (e.g., a function or struct name) must be \
365                 provided so the tool searches for matching definitions across all supported source \
366                 files in that tree. \
367                 To read entire files, pass the file path with no `offset`, `limit`, or `symbol` fields.",
368                path = target.path
369            )),
370        }];
371    }
372
373    let text = match read_normalized(&fs, &target.path).await {
374        Ok(t) => t,
375        Err(e) => {
376            return vec![ReadResult {
377                path: target.path.clone(),
378                file_hash: String::new(),
379                header: String::new(),
380                content: String::new(),
381                warnings: Some(e),
382            }];
383        }
384    };
385    let file_lines = file_lines(&text);
386
387    let _ = rollback::record(&target.path, &text);
388    let hash = format::compute_file_hash(&text);
389    let header = format::format_hashline_header(&target.path, &hash);
390
391    // Exact line ranges — a plain slice, no AST: the model reads only the
392    // lines it asked for instead of a whole syntactic block.
393    if let Some(range_raw) = target.line_range {
394        let ranges = match parse_line_ranges(&range_raw) {
395            Ok(r) => r,
396            Err(e) => {
397                return vec![ReadResult {
398                    path: target.path.clone(),
399                    file_hash: hash,
400                    header: header.clone(),
401                    content: format!("{header}\n{}", format::format_numbered_lines(&text, 1)),
402                    warnings: Some(e),
403                }];
404            }
405        };
406        let (body, seen, notices, warnings) = read_range_body(&file_lines, &ranges);
407        record_seen(&target.path, &hash, &seen);
408        let mut content = format!("{header}\n{body}");
409        append_content_notices(&mut content, &notices);
410        let warnings = (!warnings.is_empty()).then(|| warnings.join("\n"));
411        return vec![ReadResult {
412            path: target.path.clone(),
413            file_hash: hash,
414            header: header.clone(),
415            content,
416            warnings,
417        }];
418    }
419
420    if let Some(line) = target.line {
421        let ts = cosh_sdk::tree_sitter::tree_sitter();
422        let ln: u32 = match line.try_into() {
423            Ok(l) => l,
424            Err(e) => {
425                return vec![ReadResult {
426                    path: target.path.clone(),
427                    file_hash: hash,
428                    header: header.clone(),
429                    content: format!("{header}\n{}", format::format_numbered_lines(&text, 1)),
430                    warnings: Some(format!("cannot convert line {line}: {e}")),
431                }];
432            }
433        };
434        match ts.resolve_block(&target.path, &text, ln) {
435            Some(span) => {
436                let end_idx = (span.end as usize).min(file_lines.len());
437                let start_idx = (span.start as usize - 1).min(end_idx);
438                let block_lines = file_lines[start_idx..end_idx].to_vec();
439                let mut seen = Vec::new();
440                let (body_lines, elided, truncated_any) =
441                    elide_block(&block_lines, span.start, &mut seen);
442                record_seen(&target.path, &hash, &seen);
443                let mut notices: Vec<String> = Vec::new();
444                if let Some((s, e)) = elided {
445                    notices.push(format!(
446                        "[…{} lines elided; re-read with offset {}, limit {}]",
447                        e - s + 1,
448                        s,
449                        e - s + 1
450                    ));
451                }
452                if truncated_any {
453                    notices.push(format!(
454                        "[Some lines were truncated to {MAX_COLUMN} columns; use offset/limit to re-read specific lines for the full content]"
455                    ));
456                }
457                let mut content = format!("{header}\n{}", body_lines.join("\n"));
458                append_content_notices(&mut content, &notices);
459                vec![ReadResult {
460                    path: target.path.clone(),
461                    file_hash: hash,
462                    header: header.clone(),
463                    content,
464                    warnings: None,
465                }]
466            }
467            None => {
468                let content = format!("{header}\n{}", format::format_numbered_lines(&text, 1));
469                vec![ReadResult {
470                    path: target.path.clone(),
471                    file_hash: hash,
472                    header: header.clone(),
473                    content,
474                    warnings: Some(format!(
475                        "could not resolve a syntactic block starting at line {line} in `{path}`. \
476                         Possible causes: the line does not begin a valid block (e.g. fn, struct, \
477                         impl, enum, trait, mod), the line number exceeds the file length, or the \
478                         line falls inside a string or comment. \
479                         Try a different line number, an `offset`/`limit` range, or read the whole file instead.",
480                        path = target.path
481                    )),
482                }]
483            }
484        }
485    } else {
486        // Whole-file read: record every line as seen.
487        let seen: Vec<(u32, String)> = file_lines
488            .iter()
489            .enumerate()
490            .map(|(i, l)| (i as u32 + 1, l.clone()))
491            .collect();
492        record_seen(&target.path, &hash, &seen);
493        vec![ReadResult {
494            path: target.path.clone(),
495            file_hash: hash,
496            header: header.clone(),
497            content: format!("{header}\n{}", format::format_numbered_lines(&text, 1)),
498            warnings: None,
499        }]
500    }
501}
502
503async fn search_symbol(fs: &DiskFilesystem, path: &str, name: &str) -> Vec<ReadResult> {
504    let paths = match if Path::new(path).is_dir() {
505        Ok(collect_source_files(Path::new(path)))
506    } else if cosh_sdk::tree_sitter::language::detect_language(path).is_some() {
507        Ok(vec![path.to_string()])
508    } else {
509        Err(format!(
510            "no tree-sitter grammar available for `{path}`; use `offset` targeting or read the whole file instead"
511        ))
512    } {
513        Ok(p) => p,
514        Err(e) => {
515            return vec![ReadResult {
516                path: path.to_string(),
517                file_hash: String::new(),
518                header: String::new(),
519                content: String::new(),
520                warnings: Some(e),
521            }];
522        }
523    };
524
525    let ts = cosh_sdk::tree_sitter::tree_sitter();
526    let mut results: Vec<ReadResult> = Vec::new();
527
528    for p in &paths {
529        let text = match read_normalized(fs, p).await {
530            Ok(t) => t,
531            Err(e) => {
532                results.push(ReadResult {
533                    path: p.clone(),
534                    file_hash: String::new(),
535                    header: String::new(),
536                    content: String::new(),
537                    warnings: Some(e),
538                });
539                continue;
540            }
541        };
542        if let Some(span) = ts.resolve_symbol(p, &text, name) {
543            let _ = rollback::record(p, &text);
544            let hash = format::compute_file_hash(&text);
545            let header = format::format_hashline_header(p, &hash);
546            let file_lines = file_lines(&text);
547            let end_idx = (span.end as usize).min(file_lines.len());
548            let start_idx = (span.start as usize - 1).min(end_idx);
549            let block_lines = file_lines[start_idx..end_idx].to_vec();
550            let mut seen = Vec::new();
551            let (body_lines, elided, truncated_any) =
552                elide_block(&block_lines, span.start, &mut seen);
553            record_seen(p, &hash, &seen);
554            let mut notices: Vec<String> = Vec::new();
555            if let Some((s, e)) = elided {
556                notices.push(format!(
557                    "[…{} lines elided; re-read with offset {}, limit {}]",
558                    e - s + 1,
559                    s,
560                    e - s + 1
561                ));
562            }
563            if truncated_any {
564                notices.push(format!(
565                    "[Some lines were truncated to {MAX_COLUMN} columns; use offset/limit to re-read specific lines for the full content]"
566                ));
567            }
568            let mut content = format!("{header}\n{}", body_lines.join("\n"));
569            append_content_notices(&mut content, &notices);
570            results.push(ReadResult {
571                path: p.clone(),
572                file_hash: hash,
573                header: header.clone(),
574                content,
575                warnings: None,
576            });
577        }
578    }
579
580    if results.is_empty() {
581        return vec![ReadResult {
582            path: path.to_string(),
583            file_hash: String::new(),
584            header: String::new(),
585            content: String::new(),
586            warnings: Some(format!(
587                "symbol `{name}` not found in `{path}`. \
588                 Verify the symbol name is spelled exactly as defined in source code. \
589                 If `{path}` is a directory, it may contain no files with a supported \
590                 tree-sitter grammar. \
591                 Try using `offset` targeting to read specific sections, or read the whole \
592                 file to inspect its contents.",
593            )),
594        }];
595    }
596
597    results
598}
599
600/// Recursively collects files within a specified directory.
601///
602/// Returns only files with a supported Tree-sitter grammar.
603fn collect_source_files(path: &Path) -> Vec<String> {
604    let mut files = Vec::new();
605    if let Ok(entries) = std::fs::read_dir(path) {
606        for entry in entries.flatten() {
607            let p = entry.path();
608            if p.is_dir() {
609                files.extend(collect_source_files(&p));
610            } else if p.is_file() {
611                let s = p.to_string_lossy().to_string();
612                if cosh_sdk::tree_sitter::language::detect_language(&s).is_some() {
613                    files.push(s);
614                }
615            }
616        }
617    }
618    files
619}
620
621async fn read_normalized(fs: &DiskFilesystem, path: &str) -> Result<String, String> {
622    let file_text = fs.read_text(path).await.map_err(|e| e.to_string())?;
623    let bom_result = normalize::strip_bom(&file_text);
624    Ok(normalize::normalize_to_lf(&bom_result.text))
625}