cosh-tools 0.1.4

Tools crate for cosh - coding agent for the terminal
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
//! Severity contract for sub-agent code-review reports.
//!
//! A code-review sub-agent task (`code_review: true` on
//! [`SubAgentCallInput`](crate::subagent::types::SubAgentCallInput)) carries
//! a prompt contract: the FINAL REPORT — the sub-agent's last message, the
//! one written after its final tool call and the only text returned to the
//! caller (see [`crate::subagent::closure`]) — must START with an HTML
//! comment header declaring the outcome: `<!-- severity: green -->`,
//! `<!-- severity: yellow -->` (minor issues / bad practice at most) or
//! `<!-- severity: red -->` (something critical was found).
//!
//! The marker lives on the final report — not on the turn's first message —
//! for a logical reason: the agent can only declare the review outcome once
//! it has DONE the analysis. Earlier narration (progress notes between tool
//! calls) belongs to the live TUI timeline and is never shown to the caller,
//! so coloring the turn by the first line would tint the box from a message
//! written before any work happened.
//!
//! An HTML comment was chosen deliberately: it is a shape every model
//! already knows how to produce, it is inert in markdown rendering, and it
//! survives copy-through without being reformatted.
//!
//! The header is CONSUMED by the client: it is stripped from the rendered
//! report and only drives the sub-agent box color (green/orange/red — the
//! wire's middle value stays named `yellow`; only the rendered color is
//! orange). A report without a header (or a non-review task) leaves the
//! box's neutral per-agent color untouched.

use serde::{Deserialize, Serialize};

/// The review outcome declared by the severity header.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum Severity {
    /// All good — at most cosmetic details.
    Green,
    /// Minor issues / bad practice at most.
    Yellow,
    /// Something critical was found.
    Red,
}

impl Severity {
    /// Parse the DSL keyword (case-insensitive). Returns `None` for any
    /// other word — an unknown severity is treated as "no header", leaving
    /// the box color untouched, rather than guessing.
    pub fn parse(word: &str) -> Option<Self> {
        match word.trim().to_ascii_lowercase().as_str() {
            "green" => Some(Self::Green),
            "yellow" => Some(Self::Yellow),
            "red" => Some(Self::Red),
            _ => None,
        }
    }
}

/// The prompt contract appended to the input when `code_review` is set. The
/// marker must open the FINAL REPORT — the sub-agent's last message, written
/// after its final tool call, which is the only text returned to the caller
/// ([`crate::subagent::closure`]). Earlier narration is not the report and
/// carries no marker.
pub(crate) const SEVERITY_CONTRACT: &str = "\n\n---\nREPORT FORMAT CONTRACT (mandatory): your final report — your LAST message, the one you write AFTER your final tool calls, which is the only text returned to the caller — must START with an HTML comment header declaring the review outcome, on its own first line, exactly one of:\n<!-- severity: green -->   (all good — at most cosmetic details)\n<!-- severity: yellow --> (minor issues or bad practice found, nothing critical)\n<!-- severity: red -->    (something critical was found)\nDo NOT put the header on earlier progress messages: intermediate narration between tool calls is shown live but is not your report. The header is consumed by the client tooling and never shown; everything after it is the report itself. Do not put anything before the header.";

/// Extract the severity from the first line of a report and return
/// `(severity, report_without_the_header)`.
///
/// The header must be the report's first non-empty line and must match
/// `<!-- severity: WORD -->` (whitespace inside the comment is tolerated;
/// the match is case-insensitive). Anything else — no comment, wrong word,
/// header not on the first line — means "no severity": the report is
/// returned untouched with `None`.
pub fn extract_severity(report: &str) -> (Option<Severity>, &str) {
    // Tolerate leading blank lines before the header.
    let trimmed = report.trim_start_matches(['\n', '\r', ' ', '\t']);
    let Some(first_line_end) = trimmed.find('\n') else {
        // Single-line report: the whole thing is the candidate line. With a
        // header the body is empty by definition; without one the report is
        // returned UNTOUCHED (dropping the only line would lose the text).
        return match parse_header_line(trimmed) {
            Some(severity) => (Some(severity), ""),
            None => (None, report),
        };
    };
    let (first_line, after) = trimmed.split_at(first_line_end);
    match parse_header_line(first_line) {
        Some(severity) => (
            Some(severity),
            // CRLF outputs leave a leading `\r` after splitting on `\n`
            // (and Windows bodies may pad blank lines with `\r`): strip
            // both so the body never starts with stray carriage returns.
            after.trim_start_matches(['\n', '\r']),
        ),
        None => (None, report),
    }
}

/// `<!-- severity: WORD -->` on a single line, case-insensitive, whitespace
/// tolerant. Returns `None` for anything else.
fn parse_header_line(line: &str) -> Option<Severity> {
    let line = line.trim();
    let inner = line.strip_prefix("<!--")?.strip_suffix("-->")?;
    let inner = inner.trim();
    let word = inner.strip_prefix("severity:")?;
    Severity::parse(word)
}

/// Append the severity contract to a task input (review tasks only).
///
/// The explicit `code_review` flag wins, but it is NOT trusted blindly:
/// session evidence showed the orchestrating model routinely forgets to set
/// it on review dispatches, so the contract was never appended and the
/// sub-agent never produced the `<!-- severity: ... -->` header (the box
/// stayed untinted — the reported "agent ignores the DSL"). When the flag
/// is absent, a conservative heuristic infers a review task from the input
/// itself and appends the contract anyway; the header is consumed (stripped
/// and never shown) so a false positive costs the task nothing but a
/// harmless first line, and the extraction treats a report without one as
/// "no severity".
pub fn with_severity_contract(input: &str, code_review: bool) -> String {
    if is_review_task(input, code_review) {
        format!("{input}{SEVERITY_CONTRACT}")
    } else {
        input.to_string()
    }
}

/// Whether this dispatch is a code-review task: the explicit `code_review`
/// flag, or the same conservative heuristic [`with_severity_contract`]
/// applies when the caller forgot the flag. Both the contract append (input
/// side) and the header enforcement (report side) must ask THIS so a task
/// that got the contract is also the task whose report is enforced.
pub fn is_review_task(input: &str, code_review: bool) -> bool {
    code_review || looks_like_code_review(input)
}

/// Enforce the severity header on a FINISHED review report.
///
/// The prompt contract asks the sub-agent to open its final report with the
/// `<!-- severity: ... -->` header, but prompting is not enforcement — a
/// model can end its report without one (or bury it after narration, which
/// `extract_severity` then ignores). On a completed review turn the header
/// is MANDATORY: it is the only color signal the report box gets. When the
/// report lacks a valid first-line header, the client INJECTS the header
/// that best matches the report's content ([`infer_severity`]) — the DSL is
/// honored even when the sub-agent ignored it. Reports that already carry
/// the header pass through untouched, and non-review tasks are never
/// touched. The header is never RENDERED (the visible report strips it);
/// on the external path the enforced text is what the orchestrator's tool
/// call returns, but as an inert HTML comment it is inert transcript
/// metadata, not displayable text.
pub fn enforce_severity_header(report: &str, is_review: bool) -> String {
    if !is_review || report.trim().is_empty() || extract_severity(report).0.is_some() {
        return report.to_string();
    }
    let word = match infer_severity(report) {
        Severity::Green => "green",
        Severity::Yellow => "yellow",
        Severity::Red => "red",
    };
    format!("<!-- severity: {word} -->\n{report}")
}

/// Words that flip a NEARBY finding keyword into "absent": a review saying
/// "no critical findings" is CLEAN, not red. Only the three tokens before
/// the keyword are considered — enough to reach a negation separated by
/// short filler ("nothing is a major concern"), short enough that ordinary
/// prose does not extend the shadow across sentences.
const NEGATIONS: [&str; 6] = ["no", "not", "never", "none", "without", "nothing"];

/// Best-effort severity inference from a review report's own findings, used
/// only when the sub-agent failed to declare the header. Scans for the
/// finding vocabulary the review protocol prescribes — CRITICAL findings are
/// red, MAJOR findings (or an explicit "changes required" verdict) are
/// yellow, anything else reads as green. The scan is token-based (word
/// boundaries), so "majority" does not read as "major", and a negation right
/// before a keyword ("no critical findings", "no changes required") marks
/// the finding ABSENT instead of present: a false RED on a clean report is
/// the most misleading failure this inference could produce, so cleanliness
/// outranks coverage. The header is consumed by the client either way, so a
/// mis-inference costs a tint shade, never text.
fn infer_severity(report: &str) -> Severity {
    let tokens: Vec<String> = report
        .split(|c: char| !c.is_alphanumeric())
        .filter(|t| !t.is_empty())
        .map(|t| t.to_ascii_lowercase())
        .collect();
    let negated = |i: usize| {
        tokens[..i]
            .iter()
            .rev()
            .take(3)
            .any(|t| NEGATIONS.contains(&t.as_str()))
    };
    if tokens
        .iter()
        .enumerate()
        .any(|(i, t)| t == "critical" && !negated(i))
    {
        return Severity::Red;
    }
    if tokens.iter().enumerate().any(|(i, t)| {
        (t == "major"
            || (t == "changes" && tokens.get(i + 1).is_some_and(|next| next == "required")))
            && !negated(i)
    }) {
        return Severity::Yellow;
    }
    Severity::Green
}

/// Conservative heuristic for review tasks the caller forgot to flag.
///
/// Matches only unmistakable review phrasing ("code review", the VERDICT
/// protocol used by this repo's own review dispatches, and
/// review-only/re-review wording). It deliberately does NOT match generic
/// "review the changes" prose — a false positive is cheap (the header is
/// stripped from the rendered report) but appending a report-format
/// contract to an unrelated implementation task would still be noise.
fn looks_like_code_review(input: &str) -> bool {
    let lower = input.to_ascii_lowercase();
    lower.contains("code review")
        || lower.contains("re-review")
        || lower.contains("review only")
        || lower.contains("review-only")
        || lower.contains("verdict:")
}

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn parse_accepts_the_three_keywords_case_insensitive() {
        assert_eq!(Severity::parse("green"), Some(Severity::Green));
        assert_eq!(Severity::parse("YELLOW"), Some(Severity::Yellow));
        assert_eq!(Severity::parse("  Red "), Some(Severity::Red));
        assert_eq!(Severity::parse("orange"), None);
        assert_eq!(Severity::parse(""), None);
    }

    #[test]
    fn extract_reads_the_header_and_strips_it() {
        let report = "<!-- severity: red -->\n\n## Critical\n\nBug found.";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, Some(Severity::Red));
        assert_eq!(rest, "## Critical\n\nBug found.");
    }

    #[test]
    fn extract_tolerates_leading_blank_lines_and_whitespace() {
        let report = "\n\n  <!--severity: GREEN-->  \nbody";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, Some(Severity::Green));
        assert_eq!(rest, "body");
    }

    /// CRLF reports: the header line's trailing `\r` is already trimmed by
    /// `parse_header_line`, but the split leaves `"\r\n"` before the body —
    /// the body must not start with stray carriage returns.
    #[test]
    fn extract_strips_crlf_blank_lines_after_the_header() {
        let report = "<!-- severity: red -->\r\n\r\n## Critical\r\nBug found.";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, Some(Severity::Red));
        assert!(rest.starts_with("## Critical"));
        assert!(!rest.starts_with('\r'));
    }

    #[test]
    fn extract_without_header_returns_the_report_untouched() {
        let report = "## Findings\n\nnothing wrong";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, None);
        // Bit-identical round trip: no header means no mutation.
        assert_eq!(rest, report);
    }

    #[test]
    fn extract_ignores_headers_not_on_the_first_line() {
        let report = "## Report\n<!-- severity: red -->\nbody";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, None);
        assert_eq!(rest, report);
    }

    #[test]
    fn extract_ignores_unknown_severity_words() {
        let report = "<!-- severity: purple -->\nbody";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, None);
        assert_eq!(rest, report);
    }

    #[test]
    fn extract_handles_single_line_header_only_report() {
        let report = "<!-- severity: yellow -->";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, Some(Severity::Yellow));
        assert_eq!(rest, "");
    }

    /// Regression: a single-line report WITHOUT a header must come back
    /// untouched — an earlier version returned an empty body, dropping the
    /// only line of the report.
    #[test]
    fn extract_keeps_a_single_line_report_without_header() {
        let report = "just one line of findings";
        let (severity, rest) = extract_severity(report);
        assert_eq!(severity, None);
        assert_eq!(rest, report);
    }

    #[test]
    fn contract_is_appended_only_for_review_tasks() {
        let plain = with_severity_contract("do the task", false);
        assert_eq!(plain, "do the task");
        assert!(!plain.contains("severity"));

        let review = with_severity_contract("review this PR", true);
        assert!(review.starts_with("review this PR"));
        assert!(review.contains("<!-- severity: green -->"));
        assert!(review.contains("<!-- severity: yellow -->"));
        assert!(review.contains("<!-- severity: red -->"));
    }

    /// Regression (live session evidence): the orchestrating model routinely
    /// dispatches code reviews WITHOUT setting `code_review: true`, so the
    /// contract was never appended and the sub-agent never emitted the
    /// `<!-- severity: ... -->` header — the box stayed untinted ("agent
    /// ignores the DSL"). The heuristic must catch those unflagged review
    /// dispatches while leaving ordinary implementation tasks untouched.
    #[test]
    fn unflagged_review_dispatches_get_the_contract_via_heuristic() {
        // Real unflagged dispatch shapes from the live session / review flow.
        for input in [
            "Faça um CODE REVIEW do último commit deste repositório. Rode git show e leia o código.",
            "Faça um RE-REVIEW do commit mais recente deste repositório.",
            "Code review task (review only — do NOT modify anything).",
            "You are a code reviewer. Review ONLY; modify nothing. End with VERDICT: ACCEPT.",
        ] {
            let out = with_severity_contract(input, false);
            assert!(
                out.starts_with(input),
                "the input text itself must stay untouched"
            );
            assert!(
                out.contains("<!-- severity: red -->"),
                "unflagged review dispatch must get the contract: {input:?}"
            );
        }

        // Ordinary implementation tasks stay contract-free.
        for input in [
            "Fix the failing test in src/lib.rs and run the suite.",
            "review the changes and apply the refactor to the module",
            "Add a caching layer to the renderer.",
        ] {
            let out = with_severity_contract(input, false);
            assert_eq!(out, input, "non-review task must not get the contract");
        }
    }

    /// The explicit flag wins even when the input carries no review wording
    /// (the caller knows the task kind better than the heuristic).
    #[test]
    fn the_explicit_flag_always_appends_the_contract() {
        let out = with_severity_contract("audit the crate thoroughly", true);
        assert!(out.contains("<!-- severity: green -->"));
    }

    /// `is_review_task` must answer exactly what `with_severity_contract`
    /// asked when it decided to append: the same task that got the contract
    /// is the task whose report gets the header enforced.
    #[test]
    fn is_review_task_agrees_with_the_contract_append() {
        assert!(is_review_task("plain task", true));
        assert!(is_review_task(
            "Faça um CODE REVIEW do último commit.",
            false
        ));
        assert!(!is_review_task("plain task", false));
        assert!(!is_review_task(
            "review the changes and apply the refactor",
            false
        ));
    }

    /// Enforcement (the point of the fix): a review report WITHOUT the
    /// header gets one injected — inferred from its own findings — so the
    /// box tint works even when the sub-agent ignored the DSL.
    #[test]
    fn a_headerless_review_report_gets_the_header_injected() {
        // The report's text is untouched: the injected header is PREPENDED,
        // never merged into the body.
        let report = "## Findings\n\n- MAJOR: the parser drops comments.\n";
        let out = enforce_severity_header(report, true);
        assert!(out.starts_with("<!-- severity: yellow -->\n"));
        assert!(out.ends_with(report));
        assert_eq!(extract_severity(&out).0, Some(Severity::Yellow));
    }

    /// The inference maps the report vocabulary to the three severities.
    #[test]
    fn the_inference_reads_the_findings_vocabulary() {
        let word = |report: &str| {
            let out = enforce_severity_header(report, true);
            extract_severity(&out).0
        };
        assert_eq!(
            word("CRITICAL: use-after-free in the render loop."),
            Some(Severity::Red)
        );
        assert_eq!(
            word("MAJOR: the lock is held across the await point."),
            Some(Severity::Yellow)
        );
        assert_eq!(
            word("VERDICT: CHANGES REQUIRED (naming only)."),
            Some(Severity::Yellow)
        );
        assert_eq!(
            word("All checks pass; only cosmetic notes remain."),
            Some(Severity::Green)
        );
    }

    /// Regression (review round 2): the scan is token-based and
    /// negation-aware. A CLEAN report must never be tinted red/yellow —
    /// that is the loudest mis-color this inference could produce.
    #[test]
    fn the_inference_never_colors_a_clean_report_by_substring_or_negation() {
        let word = |report: &str| {
            let out = enforce_severity_header(report, true);
            extract_severity(&out).0
        };
        // "majority" is not a MAJOR finding (substring false positive).
        assert_eq!(
            word("For the majority of the changes the code follows the conventions."),
            Some(Severity::Green)
        );
        // Negated findings are ABSENT, not present.
        assert_eq!(
            word("No critical findings; no major issues were raised."),
            Some(Severity::Green)
        );
        assert_eq!(
            word("Nothing is a major concern; VERDICT: no changes required."),
            Some(Severity::Green)
        );
        assert_eq!(
            word("There is without any critical defect in the patch."),
            Some(Severity::Green)
        );
        // A real finding still wins even after earlier negated prose.
        assert_eq!(
            word("No critical findings in module A. CRITICAL: data loss in module B."),
            Some(Severity::Red)
        );
        // The negation shadow does not extend across intervening prose.
        assert_eq!(
            word("No blockers this round. MAJOR: the parser drops comments."),
            Some(Severity::Yellow)
        );
    }

    /// An empty (or whitespace-only) report is never minted into a
    /// header-only report by a hypothetical caller that skipped the
    /// call-site guard.
    #[test]
    fn an_empty_report_is_never_enforced() {
        assert_eq!(enforce_severity_header("", true), "");
        assert_eq!(enforce_severity_header("   \n  ", true), "   \n  ");
    }

    /// A report that ALREADY carries a valid header passes through
    /// untouched — enforcement never overrides the sub-agent's own verdict.
    #[test]
    fn a_report_with_a_header_passes_through_untouched() {
        let report = "<!-- severity: red -->\nCRITICAL: data loss.\n";
        assert_eq!(enforce_severity_header(report, true), report);
    }

    /// Non-review tasks are never touched, even when their text mentions
    /// findings-like vocabulary (a build log can contain "CRITICAL").
    #[test]
    fn a_non_review_report_is_never_touched() {
        let report = "CRITICAL: the build log shows a warning.\n";
        assert_eq!(enforce_severity_header(report, false), report);
    }
}