Skip to main content

scc_cli/
benchatlas.rs

1//! Atlas recall benchmark v2 (Wave 8 §57): structured recall of
2//! independently documented ground truth against the startup System Atlas,
3//! with precision and token-density metrics and per-gap diagnosis.
4//!
5//! Ground truth is organized into seven layers (the v2 ontology):
6//! `architecture` (components/subsystems the agent must know at startup),
7//! `entrypoints` (invokable surfaces), `behavior` (flows/lifecycles),
8//! `state_authority` (state owners), `contracts` (HTTP/CLI/API contracts),
9//! `landmarks` (symbols one zoom level deeper — informational), and
10//! `tests` (informational). The quality gate is the equal-weighted mean of
11//! the FIVE startup-required layers only (architecture, entrypoints,
12//! behavior, state_authority, contracts) — landmarks/tests are excluded,
13//! which is the anti-bloat guarantee: dumping implementation symbols into
14//! the atlas can no longer inflate the score.
15//!
16//! Scoring is STRUCTURED, not text-substring: each layer is matched against
17//! the machine model (`scc_context::atlas::build_atlas`), not the rendered
18//! text. Item/haystack normalization applies the documented aliases
19//! (`::` -> `.`, `fn X` -> `X`, `./p` -> `p`) so e.g. `Controller::run`
20//! matches a flow step rendered as `Controller.run`.
21//!
22//! v3 metrics:
23//! - `precision` (startup_required_precision): per startup-required layer,
24//!   |atlas entries in that layer that match a ground-truth item| /
25//!   |atlas entries in that layer| — a too-much-architecture detector: an
26//!   atlas bloated with facts the ground truth never describes scores low.
27//!   `RepoRecall.precision` is the equal-weighted mean over the five
28//!   startup-required layers.
29//! - `F2` per layer: (5 * P * R) / (4 * P + R) from that layer's precision
30//!   P and recall R (zero when P + R == 0); the gate still uses recall.
31//! - `density` (architecture_density): matched startup-required items per
32//!   1000 atlas tokens; `atlas_tokens` per repo is reported too.
33//!
34//! The v1/v2 `--holdout` protocol is now labelled **development** vs
35//! **validation** (the "holdout" corpus has been inspected and tuned
36//! against — calling it blind would be dishonest; the on-disk dirs
37//! `benchmarks/holdout` stay as they are).
38//!
39//! `--blind` scores the NEW frozen corpus (`benchmarks/blind-test` +
40//! `benchmarks/blind-test-ground-truth`), never used by tuning, and prints
41//! ONLY aggregates (overall, per-section means, the validation-vs-blind
42//! generalization gap, precision, density) — no per-repo rows, no missed
43//! keys, no filenames. blind-test failures are never shown to tuning agents.
44//!
45//! `--diagnose` classifies every missed item by WHERE it disappeared
46//! (PARSER/EXTRACTOR/WRITER/RESOLUTION/COMPILER/PROJECTION/ALIAS) via a
47//! deterministic store->flows->components->text ladder, and prints a
48//! per-kind histogram plus per-repo gap lines (the regeneration source for
49//! `benchmarks/results/ground-truth-gaps.md`).
50//!
51//! When `benchmarks/corpus/` is absent (or empty), the harness falls back to
52//! the golden `fixtures/`: ground truth is synthesized from
53//! `benchmarks/tasks.json`, fixture copies are indexed in a temp dir (the
54//! golden fixtures are never written into), and the same recall pipeline runs.
55
56use crate::benchctx::{BenchmarkCorpus, BenchTask};
57use crate::Compiler;
58use scc_context::atlas;
59use scc_context::ContextCompiler;
60use scc_core::Entity;
61use scc_core::SystemAtlas;
62use scc_indexer::scan::Language;
63use scc_store::Store;
64use serde::{Deserialize, Serialize};
65use std::collections::{BTreeMap, BTreeSet};
66use std::path::{Path, PathBuf};
67
68/// Quality gate: overall mean recall must be >= this floor (Wave 8 §57).
69/// The floor is over the five startup-required layers ONLY.
70// trace:exempt reason=internal-detail
71pub const ATLAS_GATE: f64 = 0.5;
72
73/// Holdout verdict tolerance: the validation corpus may lag the development
74/// corpus by up to this much (overall recall, absolute) before the run is
75/// called OVERFIT. The band absorbs corpus-difficulty, LOC-mix, and
76/// ground-truth-strictness differences; a lag beyond it means the
77/// development-tuned rules do not generalize to unseen repos.
78// trace:exempt reason=internal-detail
79pub const HOLDOUT_TOLERANCE: f64 = 0.05;
80
81/// The five startup-required layers that count toward the overall score
82/// (architecture, entrypoints, behavior, state_authority, contracts — see
83/// `ALL_SECTIONS`; landmarks + tests are informational).
84// trace:exempt reason=internal-detail
85const ALL_SECTIONS: [&str; 7] = [
86    "architecture",
87    "entrypoints",
88    "behavior",
89    "state_authority",
90    "contracts",
91    "landmarks",
92    "tests",
93];
94
95/// Default per-section regression guard (`--guard-section-delta`): any
96/// startup-required section dropping by more than this between two compared
97/// runs fails the Wave-11 guard.
98// trace:exempt reason=internal-detail
99pub const DEFAULT_SECTION_GUARD: f64 = 0.05;
100
101/// Minimal pure-Rust SHA-256 (FIPS 180-4) for the blind-test manifest hash.
102/// Deterministic, dependency-free (scc-cli has no crypto dep), panic-free.
103/// Public only so the roundtrip unit test can exercise it directly.
104// trace:exempt reason=internal-detail
105pub mod sha256 {
106    // trace:exempt reason=internal-detail
107    const K: [u32; 64] = [
108        0x428a2f98, 0x71374491, 0xb5c0fbcf, 0xe9b5dba5, 0x3956c25b, 0x59f111f1, 0x923f82a4,
109        0xab1c5ed5, 0xd807aa98, 0x12835b01, 0x243185be, 0x550c7dc3, 0x72be5d74, 0x80deb1fe,
110        0x9bdc06a7, 0xc19bf174, 0xe49b69c1, 0xefbe4786, 0x0fc19dc6, 0x240ca1cc, 0x2de92c6f,
111        0x4a7484aa, 0x5cb0a9dc, 0x76f988da, 0x983e5152, 0xa831c66d, 0xb00327c8, 0xbf597fc7,
112        0xc6e00bf3, 0xd5a79147, 0x06ca6351, 0x14292967, 0x27b70a85, 0x2e1b2138, 0x4d2c6dfc,
113        0x53380d13, 0x650a7354, 0x766a0abb, 0x81c2c92e, 0x92722c85, 0xa2bfe8a1, 0xa81a664b,
114        0xc24b8b70, 0xc76c51a3, 0xd192e819, 0xd6990624, 0xf40e3585, 0x106aa070, 0x19a4c116,
115        0x1e376c08, 0x2748774c, 0x34b0bcb5, 0x391c0cb3, 0x4ed8aa4a, 0x5b9cca4f, 0x682e6ff3,
116        0x748f82ee, 0x78a5636f, 0x84c87814, 0x8cc70208, 0x90befffa, 0xa4506ceb, 0xbef9a3f7,
117        0xc67178f2,
118    ];
119
120    // trace:exempt reason=internal-detail
121    const H0: [u32; 8] = [
122        0x6a09e667, 0xbb67ae85, 0x3c6ef372, 0xa54ff53a, 0x510e527f, 0x9b05688c, 0x1f83d9ab,
123        0x5be0cd19,
124    ];
125
126    /// SHA-256 digest of `data` as 32 raw bytes.
127    // trace:exempt reason=internal-detail
128    pub fn digest(data: &[u8]) -> [u8; 32] {
129        let mut h = H0;
130        let bit_len: u64 = (data.len() as u64).wrapping_mul(8);
131        // pad: 0x80, zeros to 56 mod 64, then the 64-bit big-endian bit length
132        let mut buf: Vec<u8> = Vec::with_capacity(((data.len() + 72) / 64) * 64);
133        buf.extend_from_slice(data);
134        buf.push(0x80);
135        while buf.len() % 64 != 56 {
136            buf.push(0);
137        }
138        buf.extend_from_slice(&bit_len.to_be_bytes());
139
140        let mut w = [0u32; 64];
141        for chunk in buf.as_chunks::<64>().0 {
142            for (i, word) in w.iter_mut().enumerate().take(16) {
143                let o = i * 4;
144                *word = u32::from_be_bytes([
145                    chunk[o],
146                    chunk[o + 1],
147                    chunk[o + 2],
148                    chunk[o + 3],
149                ]);
150            }
151            for i in 16..64 {
152                let s0 = w[i - 15].rotate_right(7) ^ w[i - 15].rotate_right(18) ^ (w[i - 15] >> 3);
153                let s1 = w[i - 2].rotate_right(17) ^ w[i - 2].rotate_right(19) ^ (w[i - 2] >> 10);
154                w[i] = w[i - 16]
155                    .wrapping_add(s0)
156                    .wrapping_add(w[i - 7])
157                    .wrapping_add(s1);
158            }
159            let [mut a, mut b, mut c, mut d, mut e, mut f, mut g, mut hh] = h;
160            for i in 0..64 {
161                let s1 = e.rotate_right(6) ^ e.rotate_right(11) ^ e.rotate_right(25);
162                let ch = (e & f) ^ (!e & g);
163                let t1 = hh
164                    .wrapping_add(s1)
165                    .wrapping_add(ch)
166                    .wrapping_add(K[i])
167                    .wrapping_add(w[i]);
168                let s0 = a.rotate_right(2) ^ a.rotate_right(13) ^ a.rotate_right(22);
169                let maj = (a & b) ^ (a & c) ^ (b & c);
170                let t2 = s0.wrapping_add(maj);
171                hh = g;
172                g = f;
173                f = e;
174                e = d.wrapping_add(t1);
175                d = c;
176                c = b;
177                b = a;
178                a = t1.wrapping_add(t2);
179            }
180            h[0] = h[0].wrapping_add(a);
181            h[1] = h[1].wrapping_add(b);
182            h[2] = h[2].wrapping_add(c);
183            h[3] = h[3].wrapping_add(d);
184            h[4] = h[4].wrapping_add(e);
185            h[5] = h[5].wrapping_add(f);
186            h[6] = h[6].wrapping_add(g);
187            h[7] = h[7].wrapping_add(hh);
188        }
189        let mut out = [0u8; 32];
190        for (i, v) in h.iter().enumerate() {
191            out[i * 4..i * 4 + 4].copy_from_slice(&v.to_be_bytes());
192        }
193        out
194    }
195
196    /// Lowercase hex of the SHA-256 digest.
197    // trace:exempt reason=internal-detail
198    pub fn hex(data: &[u8]) -> String {
199        let mut out = String::with_capacity(64);
200        for b in digest(data) {
201            out.push_str(&format!("{b:02x}"));
202        }
203        out
204    }
205}
206
207// trace:exempt reason=internal-detail
208
209/// Ground-truth sections parsed from `benchmarks/ground-truth/<name>.md`
210/// (one `- <key string>` bullet per item). The v2 ontology; legacy section
211/// names (components/flows/ownership) are accepted and normalized.
212#[derive(Debug, Clone, Default)]
213// trace:exempt reason=internal-detail
214pub struct GroundTruthDoc {
215    pub architecture: Vec<String>,
216    pub entrypoints: Vec<String>,
217    pub behavior: Vec<String>,
218    pub state_authority: Vec<String>,
219    pub contracts: Vec<String>,
220    pub landmarks: Vec<String>,
221    pub tests: Vec<String>,
222}
223
224// trace:exempt reason=internal-detail
225impl GroundTruthDoc {
226    // trace:exempt reason=internal-detail
227    pub fn section(&self, name: &str) -> &Vec<String> {
228        match name {
229            "architecture" => &self.architecture,
230            "entrypoints" => &self.entrypoints,
231            "behavior" => &self.behavior,
232            "state_authority" => &self.state_authority,
233            "contracts" => &self.contracts,
234            "landmarks" => &self.landmarks,
235            "tests" => &self.tests,
236            _ => unreachable!("unknown section {name}"),
237        }
238    }
239
240    // trace:exempt reason=internal-detail
241    fn section_mut(&mut self, name: &str) -> &mut Vec<String> {
242        match name {
243            "architecture" => &mut self.architecture,
244            "entrypoints" => &mut self.entrypoints,
245            "behavior" => &mut self.behavior,
246            "state_authority" => &mut self.state_authority,
247            "contracts" => &mut self.contracts,
248            "landmarks" => &mut self.landmarks,
249            "tests" => &mut self.tests,
250            _ => unreachable!("unknown section {name}"),
251        }
252    }
253
254    /// Remove duplicates, preserving first-seen order.
255    // trace:exempt reason=internal-detail
256    fn dedupe(&mut self) {
257        for name in ALL_SECTIONS {
258            let mut seen: BTreeSet<String> = BTreeSet::new();
259            self.section_mut(name).retain(|item| seen.insert(item.clone()));
260        }
261    }
262
263    // trace:exempt reason=internal-detail
264    fn to_markdown(&self) -> String {
265        let mut out = String::from("# fixtures fallback (synthesized from benchmarks/tasks.json)\n");
266        for name in ALL_SECTIONS {
267            out.push_str(&format!("## {name}\n"));
268            for item in self.section(name) {
269                out.push_str(&format!("- {item}\n"));
270            }
271        }
272        out
273    }
274}
275
276// trace:exempt reason=internal-detail
277
278/// Gap-kind classification for a missed ground-truth item (`--diagnose`):
279/// where the fact disappeared between source and the rendered atlas.
280#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord, Serialize, Deserialize)]
281#[serde(rename_all = "SCREAMING_SNAKE_CASE")]
282// trace:exempt reason=internal-detail
283pub enum GapKind {
284    /// The symbol/string is not parseable by any enabled extractor
285    /// (language disabled, ignored path, or a format no extractor reads).
286    Parser,
287    /// Parsed, but no semantic fact was emitted for it (e.g. no route /
288    /// registration / export extraction for the construct).
289    Extractor,
290    /// The fact exists in the ExtractedFile but was not written to the
291    /// store. Not observable from the store side (the heuristic ladder
292    /// below maps parsed-but-absent facts to `Extractor`); the kind exists
293    /// so the taxonomy is complete.
294    Writer,
295    /// Exists in the store but was never wired into the graph (the matched
296    /// symbol has zero relationships — resolution/compilation never reached
297    /// it, so no flow or component can carry it).
298    Resolution,
299    /// Exists in the store but was not compiled into components/flows.
300    Compiler,
301    /// Compiled into a component but dropped by budget/policy/rendering.
302    Projection,
303    /// Likely present under a different spelling; the aliases did not match.
304    Alias,
305}
306
307// trace:exempt reason=internal-detail
308impl GapKind {
309    // trace:exempt reason=internal-detail
310    pub fn as_str(&self) -> &'static str {
311        match self {
312            GapKind::Parser => "PARSER",
313            GapKind::Extractor => "EXTRACTOR",
314            GapKind::Writer => "WRITER",
315            GapKind::Resolution => "RESOLUTION",
316            GapKind::Compiler => "COMPILER",
317            GapKind::Projection => "PROJECTION",
318            GapKind::Alias => "ALIAS",
319        }
320    }
321}
322
323#[derive(Debug, Clone, Serialize, Deserialize)]
324// trace:exempt reason=internal-detail
325pub struct GapFinding {
326    pub section: String,
327    pub item: String,
328    pub kind: GapKind,
329    pub detail: String,
330}
331
332// trace:exempt reason=internal-detail
333
334#[derive(Debug, Clone, Serialize, Deserialize)]
335// trace:exempt reason=internal-detail
336pub struct RepoRecall {
337    pub repo: String,
338    pub architecture: f64,
339    pub entrypoints: f64,
340    pub behavior: f64,
341    pub state_authority: f64,
342    pub contracts: f64,
343    pub landmarks: f64,
344    pub tests: f64,
345    /// Equal-weighted mean of the five startup-required layers (the gate).
346    pub overall: f64,
347    /// Mean startup-required precision over the five startup-required
348    /// layers: per layer, |atlas entries that match a ground-truth item| /
349    /// |atlas entries in the layer| (v3 — a too-much-architecture
350    /// detector; see `layer_precision`).
351    pub precision: f64,
352    /// Per-layer startup-required precision (the five startup layers only).
353    pub layer_precision: BTreeMap<String, f64>,
354    /// Per-layer F2 = (5*P*R)/(4*P+R) over the five startup-required
355    /// layers (zero when P+R==0); `f2` is the equal-weighted mean.
356    pub layer_f2: BTreeMap<String, f64>,
357    /// Equal-weighted mean F2 over the five startup-required layers.
358    pub f2: f64,
359    /// Matched startup-required items per 1000 atlas tokens
360    /// (architecture_density).
361    pub density: f64,
362    /// Rendered atlas token count.
363    pub atlas_tokens: usize,
364    /// Number of call edges upgraded to RESOLVED by the semantic backends
365    /// (pyright/tsserver) before scoring; 0 when `--no-resolve`.
366    pub resolved_calls: usize,
367    /// Semantic backends that were available and used for resolution
368    /// (pyright/tsserver); empty when unavailable or `--no-resolve`.
369    #[serde(default, skip_serializing_if = "Vec::is_empty")]
370    pub backends_used: Vec<String>,
371    /// Semantic backends unavailable (not installed) — resolution degraded,
372    /// not fatal.
373    #[serde(default, skip_serializing_if = "Vec::is_empty")]
374    pub backends_missing: Vec<String>,
375    /// Resolution error text when the backend was found but the resolve
376    /// pass failed (e.g. LSP handshake) — degraded, not fatal.
377    #[serde(default, skip_serializing_if = "Option::is_none")]
378    pub resolve_error: Option<String>,
379    /// Number of ground-truth items in the (informational) landmarks layer.
380    pub landmark_items: usize,
381    /// When set, the repo was not scored (missing dir / missing ground
382    /// truth / index failure) and the recall fields are meaningless.
383    pub skipped_reason: Option<String>,
384    /// Ground-truth key strings the structured matcher missed
385    /// (`section:key`), for diagnosing misses.
386    pub missed: Vec<String>,
387    /// Per-item gap classification (populated when `--diagnose`).
388    #[serde(default, skip_serializing_if = "Vec::is_empty")]
389    pub gaps: Vec<GapFinding>,
390}
391
392// trace:exempt reason=internal-detail
393impl RepoRecall {
394    // trace:exempt reason=internal-detail
395    fn skipped(repo: &str, reason: impl Into<String>) -> Self {
396        RepoRecall {
397            repo: repo.to_string(),
398            skipped_reason: Some(reason.into()),
399            ..Default::default()
400        }
401    }
402}
403
404// trace:exempt reason=internal-detail
405impl Default for RepoRecall {
406    // trace:exempt reason=internal-detail
407    fn default() -> Self {
408        RepoRecall {
409            repo: String::new(),
410            architecture: 0.0,
411            entrypoints: 0.0,
412            behavior: 0.0,
413            state_authority: 0.0,
414            contracts: 0.0,
415            landmarks: 0.0,
416            tests: 0.0,
417            overall: 0.0,
418            precision: 0.0,
419            layer_precision: BTreeMap::new(),
420            layer_f2: BTreeMap::new(),
421            f2: 0.0,
422            density: 0.0,
423            atlas_tokens: 0,
424            resolved_calls: 0,
425            backends_used: Vec::new(),
426            backends_missing: Vec::new(),
427            resolve_error: None,
428            landmark_items: 0,
429            skipped_reason: None,
430            missed: Vec::new(),
431            gaps: Vec::new(),
432        }
433    }
434}
435
436// trace:exempt reason=internal-detail
437
438#[derive(Debug, Clone, Default, Serialize, Deserialize)]
439// trace:exempt reason=internal-detail
440pub struct AtlasRecallReport {
441    /// One row per requested repo, in sorted order; skipped repos carry
442    /// `skipped_reason`.
443    pub repos: Vec<RepoRecall>,
444    /// Where the run came from: "benchmarks/corpus" or "fixtures fallback".
445    pub mode: String,
446    pub mean_architecture: f64,
447    pub mean_entrypoints: f64,
448    pub mean_behavior: f64,
449    pub mean_state_authority: f64,
450    pub mean_contracts: f64,
451    pub mean_landmarks: f64,
452    pub mean_tests: f64,
453    /// Equal-weighted mean of the five startup-required layers over scored
454    /// repos (the gate).
455    pub mean_overall: f64,
456    pub mean_precision: f64,
457    pub mean_f2: f64,
458    pub mean_density: f64,
459    pub mean_atlas_tokens: f64,
460    pub scored: usize,
461    pub skipped: usize,
462    pub gate_passed: bool,
463    /// Gap-kind histogram over all diagnosed items (kind -> count).
464    #[serde(default, skip_serializing_if = "BTreeMap::is_empty")]
465    pub gap_histogram: BTreeMap<String, usize>,
466}
467
468// trace:exempt reason=internal-detail
469impl AtlasRecallReport {
470    /// Clone with all per-repo detail stripped (repos, gaps, histogram):
471    /// only the aggregates survive. The blind protocol keeps this invariant
472    /// end to end — blind-test failures are never shown to tuning agents,
473    /// and the blind JSON / blind-v1.txt output is aggregates-only.
474    // trace:exempt reason=internal-detail
475    pub fn aggregates_only(&self) -> Self {
476        let mut c = self.clone();
477        c.repos.clear();
478        c.gap_histogram.clear();
479        c
480    }
481
482    /// Mean per-layer precision over scored repos (the five startup layers).
483    // trace:exempt reason=internal-detail
484    fn mean_layer_precision(&self) -> BTreeMap<String, f64> {
485        self.mean_layer_map(|r| &r.layer_precision)
486    }
487
488    /// Mean per-layer F2 over scored repos (the five startup layers).
489    // trace:exempt reason=internal-detail
490    fn mean_layer_f2(&self) -> BTreeMap<String, f64> {
491        self.mean_layer_map(|r| &r.layer_f2)
492    }
493
494    // trace:exempt reason=internal-detail
495    fn mean_layer_map(
496        &self,
497        // trace:exempt reason=internal-detail
498        pick: impl Fn(&RepoRecall) -> &BTreeMap<String, f64>,
499    ) -> BTreeMap<String, f64> {
500        let mut sums: BTreeMap<String, f64> = BTreeMap::new();
501        let mut counts: BTreeMap<String, usize> = BTreeMap::new();
502        for r in &self.repos {
503            if r.skipped_reason.is_some() {
504                continue;
505            }
506            for (layer, v) in pick(r) {
507                *sums.entry(layer.clone()).or_insert(0.0) += v;
508                *counts.entry(layer.clone()).or_insert(0) += 1;
509            }
510        }
511        sums.into_iter()
512            .map(|(layer, sum)| {
513                let n = counts.get(&layer).copied().unwrap_or(0).max(1) as f64;
514                (layer, sum / n)
515            })
516            .collect()
517    }
518}
519
520/// Parse a ground-truth markdown doc into per-section key strings.
521///
522/// Accepts the Wave 8 corpus format (`## section` heading + `- item` bullets)
523/// for both the v2 ontology (architecture/entrypoints/behavior/
524/// state_authority/contracts/landmarks/tests) and the legacy names
525/// (components -> architecture, flows -> behavior, ownership ->
526/// state_authority). A bullet is either the bare key string or
527/// `<key string> — explanation`; the explanation is not expected in atlas
528/// output, so only the key string (before ` — `) is kept. Inline-code
529/// backticks are stripped.
530// trace:exempt reason=internal-detail
531pub fn parse_ground_truth(md: &str) -> GroundTruthDoc {
532    let mut doc = GroundTruthDoc::default();
533    let mut current: Option<&'static str> = None;
534    let mut seen: BTreeSet<String> = BTreeSet::new();
535    for raw in md.lines() {
536        let line = raw.trim();
537        if let Some(rest) = line.strip_prefix("## ") {
538            current = match rest.trim().to_ascii_lowercase().as_str() {
539                "architecture" | "components" => Some("architecture"),
540                "entrypoints" => Some("entrypoints"),
541                "behavior" | "flows" => Some("behavior"),
542                "state_authority" | "ownership" => Some("state_authority"),
543                "contracts" => Some("contracts"),
544                "landmarks" => Some("landmarks"),
545                "tests" => Some("tests"),
546                _ => None,
547            };
548            continue;
549        }
550        let Some(section) = current else { continue };
551        let Some(item) = line
552            .strip_prefix("- ")
553            .or_else(|| line.strip_prefix("* "))
554        else {
555            continue;
556        };
557        let item = item.trim().trim_matches('`').trim();
558        if item.is_empty() {
559            continue;
560        }
561        let key = match item.split_once(" — ") {
562            Some((k, _)) => k.trim().trim_matches('`').trim(),
563            None => item,
564        };
565        if key.is_empty() {
566            continue;
567        }
568        // duplicates (same key bulleted twice in one section, e.g. a route
569        // listed both as decorator and as canonical form) would skew the
570        // denominator — keep the first occurrence only.
571        if !seen.insert(format!("{section}:{key}")) {
572            continue;
573        }
574        doc.section_mut(section).push(key.to_string());
575    }
576    doc
577}
578
579/// Normalize a ground-truth item / atlas string for matching, applying the
580/// documented aliases: `::` -> `.` (so `Controller::run` matches
581/// `Controller.run`), `fn X` -> `X`, and `./p` -> `p` (path prefix).
582// trace:exempt reason=internal-detail
583fn norm(s: &str) -> String {
584    let mut out = s.to_ascii_lowercase();
585    out = out.replace("::", ".");
586    if let Some(rest) = out.strip_prefix("fn ") {
587        out = rest.to_string();
588    }
589    if let Some(rest) = out.strip_prefix("./") {
590        out = rest.to_string();
591    }
592    out
593}
594
595/// Normalize and join parts into one haystack (newline-separated).
596// trace:exempt reason=internal-detail
597fn norm_join(parts: impl IntoIterator<Item = String>) -> String {
598    let mut out: Vec<String> = parts.into_iter().map(|p| norm(&p)).collect();
599    out.sort();
600    out.dedup();
601    out.join("\n")
602}
603
604/// Structured atlas haystacks, one per ontology layer, built from the
605/// machine model (`SystemAtlas`) plus the rendered pack (for the
606/// informational `tests` layer and the ALIAS gap check).
607// trace:exempt reason=internal-detail
608struct AtlasLayers {
609    architecture: String,
610    entrypoints: String,
611    behavior: String,
612    state_authority: String,
613    contracts: String,
614    landmarks: String,
615    /// Normalized rendered atlas content (tests layer + alias check).
616    text: String,
617    /// Normalized flow inventory: derived flows + canonical flow graphs
618    /// (names, triggers, step/node operations) — used by gap diagnosis.
619    flows: String,
620    /// Normalized component inventory (names, purposes, implementations,
621    /// owns targets) — used by gap diagnosis.
622    components: String,
623}
624
625// trace:exempt reason=internal-detail
626fn build_layers(
627    ctx: &ContextCompiler<'_>,
628    pack: &scc_context::ContextPack,
629    atlas: &SystemAtlas,
630) -> AtlasLayers {
631    let mut arch_parts: Vec<String> = Vec::new();
632    let mut sa_parts: Vec<String> = Vec::new();
633    let mut land_parts: Vec<String> = Vec::new();
634    let mut comp_parts: Vec<String> = Vec::new();
635    for c in &atlas.components {
636        comp_parts.push(c.name.clone());
637        comp_parts.push(c.purpose.clone());
638        comp_parts.extend(c.implementation.iter().cloned());
639        comp_parts.extend(c.owns.iter().map(|o| o.target.clone()));
640        arch_parts.push(c.name.clone());
641        if !c.purpose.is_empty() {
642            arch_parts.push(c.purpose.clone());
643        }
644        arch_parts.extend(c.implementation.iter().cloned());
645        for o in &c.owns {
646            sa_parts.push(o.target.clone());
647        }
648        land_parts.extend(c.implementation.iter().cloned());
649    }
650    sa_parts.extend(atlas.data_stores.iter().cloned());
651
652    let mut ep_parts: Vec<String> = Vec::new();
653    for e in &atlas.entrypoints {
654        ep_parts.push(e.name.clone());
655        ep_parts.push(e.trigger.clone());
656        // the symbol id's last segment is the symbol name (e.g.
657        // repo://x/symbol/fastapi/applications.py/FastAPI -> FastAPI)
658        if let Some(seg) = e.symbol.rsplit('/').next() {
659            if !seg.is_empty() {
660                ep_parts.push(seg.to_string());
661            }
662        }
663    }
664
665    let mut bh_parts: Vec<String> = Vec::new();
666    let mut flow_parts: Vec<String> = Vec::new();
667    for f in &atlas.flows {
668        bh_parts.push(f.name.clone());
669        flow_parts.push(f.name.clone());
670        if let Some(t) = &f.trigger {
671            bh_parts.push(t.clone());
672            flow_parts.push(t.clone());
673        }
674        for s in &f.steps {
675            bh_parts.push(s.clone());
676            flow_parts.push(s.clone());
677            land_parts.push(s.clone());
678        }
679    }
680    // canonical flow graphs (the flow edge source) — ops join the flow
681    // inventory for diagnosis even when the flow was not rendered
682    if let Ok(graphs) = ctx.store.flow_graphs() {
683        for g in &graphs {
684            flow_parts.push(g.name.clone());
685            if let Some(t) = &g.trigger {
686                flow_parts.push(t.clone());
687            }
688            for n in &g.nodes {
689                flow_parts.push(n.operation.clone());
690                flow_parts.push(n.actor.clone());
691            }
692        }
693    }
694    // derived (non-sequence) flows: view flows carry name/trigger/steps
695    for f in ctx.view.flows() {
696        flow_parts.push(f.name.clone());
697        if let Some(t) = &f.trigger {
698            flow_parts.push(t.clone());
699        }
700        for s in &f.steps {
701            flow_parts.push(s.actor.clone());
702            flow_parts.push(s.operation.clone());
703        }
704    }
705
706    // Wave 9: contracts are first-class `Contract` records; the layer
707    // haystack keeps the contract strings (operations) exactly as before.
708    let contracts: Vec<String> = atlas
709        .contracts
710        .iter()
711        .flat_map(|c| c.operations.iter().cloned())
712        .collect();
713
714    AtlasLayers {
715        architecture: norm_join(arch_parts),
716        entrypoints: norm_join(ep_parts),
717        behavior: norm_join(bh_parts),
718        state_authority: norm_join(sa_parts),
719        contracts: norm_join(contracts),
720        landmarks: norm_join(land_parts),
721        text: norm(&pack.content),
722        flows: norm_join(flow_parts),
723        components: norm_join(comp_parts),
724    }
725}
726
727/// The haystack a layer matches against.
728// trace:exempt reason=internal-detail
729fn layer_haystack<'a>(section: &str, layers: &'a AtlasLayers, text_norm: &'a str) -> &'a str {
730    match section {
731        "architecture" => &layers.architecture,
732        "entrypoints" => &layers.entrypoints,
733        "behavior" => &layers.behavior,
734        "state_authority" => &layers.state_authority,
735        "contracts" => &layers.contracts,
736        "landmarks" => &layers.landmarks,
737        "tests" => text_norm,
738        _ => unreachable!("unknown section {section}"),
739    }
740}
741
742/// Recall for one layer: fraction of ground-truth key strings found
743/// (case-insensitive, aliases applied) in the layer's structured haystack.
744/// An empty ground truth scores 1.0 (nothing to miss).
745/// Whether a ground-truth chain item (`A -> B -> C`) matches the layer's
746/// haystack: each step must appear — in order — in the per-step lines.
747/// Chain items never match a plain substring test (the haystack is one
748/// step per line), so this is the honest interpretation of a chain.
749// trace:exempt reason=internal-detail
750fn chain_matches(chain: &str, haystack: &str) -> bool {
751    let steps: Vec<String> = chain.split(" -> ").map(norm).collect();
752    if steps.len() < 2 {
753        return false;
754    }
755    let lines: Vec<&str> = haystack.lines().collect();
756    let mut pos = 0usize;
757    for step in steps {
758        let mut found = false;
759        while pos < lines.len() {
760            if norm(lines[pos]).contains(&step) {
761                found = true;
762                pos += 1;
763                break;
764            }
765            pos += 1;
766        }
767        if !found {
768            return false;
769        }
770    }
771    true
772}
773
774// trace:exempt reason=internal-detail
775fn item_matches(item: &str, haystack: &str) -> bool {
776    if item.contains(" -> ") {
777        chain_matches(item, haystack)
778    } else {
779        haystack.contains(&norm(item))
780    }
781}
782
783// trace:exempt reason=internal-detail
784fn layer_recall(items: &[String], haystack: &str) -> (f64, usize, usize) {
785    if items.is_empty() {
786        return (1.0, 0, 0);
787    }
788    let mut hit = 0usize;
789    for item in items {
790        if item_matches(item, haystack) {
791            hit += 1;
792        }
793    }
794    (hit as f64 / items.len() as f64, hit, items.len())
795}
796
797/// Whether one ground-truth item matches its layer's structured haystack.
798// trace:exempt reason=internal-detail
799fn item_matched(section: &str, item: &str, layers: &AtlasLayers, text_norm: &str) -> bool {
800    item_matches(item, layer_haystack(section, layers, text_norm))
801}
802
803/// Startup-required precision for one layer: |atlas entries in the layer
804/// that match a ground-truth item| / |atlas entries in the layer|. The
805/// haystack is one normalized entry per line, so entries are countable.
806/// An entry matches when some ground-truth item (chain items included,
807/// against a single line they virtually never match) is contained in it.
808/// An empty layer haystack scores 1.0 (nothing spurious to report).
809// trace:exempt reason=internal-detail
810fn layer_precision(items: &[String], haystack: &str) -> f64 {
811    let entries: Vec<&str> = haystack.lines().filter(|l| !l.is_empty()).collect();
812    if entries.is_empty() {
813        return 1.0;
814    }
815    let matched = entries
816        .iter()
817        .filter(|line| items.iter().any(|item| item_matches(item, line)))
818        .count();
819    matched as f64 / entries.len() as f64
820}
821
822/// F2 score from precision P and recall R: (5*P*R)/(4*P+R), zero when
823/// P + R == 0. Recall-weighting (beta=2) rewards recall over precision,
824/// matching the gate's recall-first stance while still penalizing bloat.
825// trace:exempt reason=internal-detail
826fn f2_score(p: f64, r: f64) -> f64 {
827    if p + r == 0.0 {
828        return 0.0;
829    }
830    (5.0 * p * r) / (4.0 * p + r)
831}
832
833/// Startup facts per 1000 atlas tokens (architecture_density): matched
834/// startup-required items over the rendered pack's token count.
835// trace:exempt reason=internal-detail
836fn token_density(matched_startup: usize, tokens: usize) -> f64 {
837    if tokens == 0 {
838        return 0.0;
839    }
840    matched_startup as f64 / (tokens as f64 / 1000.0)
841}
842
843/// Index one repo in place and score its ground truth against the atlas.
844///
845/// v2 scoring is structured: each layer is matched against the machine
846/// `SystemAtlas` (component name/purpose/implementation for architecture;
847/// entrypoint name/trigger/symbol; flow name/trigger/step ops; owns claims +
848/// data stores; contracts), except the informational `tests` layer which is
849/// matched against the rendered text. `diagnose` additionally classifies
850/// every missed item by gap kind.
851///
852/// `resolve` runs the language-aware semantic backends (pyright + tsserver)
853/// on the freshly indexed repo before the atlas is built, so call chains
854/// upgrade from EXTRACTED candidates to RESOLVED edges and the behavior
855/// flows are seeded from resolved paths; the per-repo `resolved_calls`
856/// reports how many edges were upgraded (0 with `--no-resolve` or when the
857/// backends are unavailable — resolution degrades, never fails the run).
858// trace:exempt reason=internal-detail
859pub fn score_repo(
860    repo_dir: &Path,
861    gt: &GroundTruthDoc,
862    diagnose: bool,
863    resolve: bool,
864) -> Result<RepoRecall, String> {
865    crate::commands::cmd_index(repo_dir, true).map_err(|e| format!("index failed: {e}"))?;
866    let mut backends_used: Vec<String> = Vec::new();
867    let mut backends_missing: Vec<String> = Vec::new();
868    let mut resolve_error: Option<String> = None;
869    let resolved_calls = if resolve {
870        match crate::resolve_and_recompile(repo_dir) {
871            Ok(rep) => {
872                backends_used = rep.backends_used;
873                backends_missing = rep.backends_missing;
874                rep.upgraded
875            }
876            Err(e) => {
877                eprintln!(
878                    "benchatlas: semantic resolution skipped for {}: {e}",
879                    repo_dir.display()
880                );
881                resolve_error = Some(e.to_string());
882                0
883            }
884        }
885    } else {
886        0
887    };
888    let store = crate::open_store(repo_dir).map_err(|e| format!("store: {e}"))?;
889    let config = crate::load_config(repo_dir).map_err(|e| format!("config: {e}"))?;
890    let stale = crate::stale_paths(&store).map_err(|e| format!("stale: {e}"))?;
891    let comp = crate::compiler(&store, &config, stale).map_err(|e| format!("compiler: {e}"))?;
892    // Wave 11: build the machine atlas once, order its sections by
893    // evidence-backed confidence (highest first — precision via ordering,
894    // no entries dropped), and render the agent-facing pack from the ranked
895    // atlas so agents read the strongest facts first under the token budget.
896    let mut atlas = atlas::build_atlas(&comp.ctx());
897    scc_context::rank::rank_startup_atlas(&mut atlas);
898    let pack = atlas::render_atlas(&comp.ctx(), &atlas, comp.ctx().settings.atlas_tokens, false);
899    let layers = build_layers(&comp.ctx(), &pack, &atlas);
900    let text_norm = &layers.text;
901
902    let (architecture, arch_hit, _) = layer_recall(&gt.architecture, &layers.architecture);
903    let (entrypoints, ep_hit, _) = layer_recall(&gt.entrypoints, &layers.entrypoints);
904    let (behavior, bh_hit, _) = layer_recall(&gt.behavior, &layers.behavior);
905    let (state_authority, sa_hit, _) = layer_recall(&gt.state_authority, &layers.state_authority);
906    let (contracts, ct_hit, _) = layer_recall(&gt.contracts, &layers.contracts);
907    let (landmarks, _, _) = layer_recall(&gt.landmarks, &layers.landmarks);
908    let (tests, _, _) = layer_recall(&gt.tests, text_norm);
909    let overall = (architecture + entrypoints + behavior + state_authority + contracts) / 5.0;
910
911    // v3 startup-required precision + F2 per layer (the five startup layers
912    // only — landmarks/tests are informational and excluded, mirroring the
913    // recall gate's anti-bloat stance).
914    let startup_layers = [
915        ("architecture", architecture),
916        ("entrypoints", entrypoints),
917        ("behavior", behavior),
918        ("state_authority", state_authority),
919        ("contracts", contracts),
920    ];
921    let mut layer_precision_map: BTreeMap<String, f64> = BTreeMap::new();
922    let mut layer_f2: BTreeMap<String, f64> = BTreeMap::new();
923    let mut precision_sum = 0.0;
924    let mut f2_sum = 0.0;
925    for (name, recall) in startup_layers {
926        let p = layer_precision(gt.section(name), layer_haystack(name, &layers, text_norm));
927        let f2 = f2_score(p, recall);
928        layer_precision_map.insert(name.to_string(), p);
929        layer_f2.insert(name.to_string(), f2);
930        precision_sum += p;
931        f2_sum += f2;
932    }
933    let precision = precision_sum / startup_layers.len() as f64;
934    let f2 = f2_sum / startup_layers.len() as f64;
935
936    let matched_startup = arch_hit + ep_hit + bh_hit + sa_hit + ct_hit;
937    let density = token_density(matched_startup, pack.tokens);
938
939    let mut missed: Vec<String> = Vec::new();
940    for section in ALL_SECTIONS {
941        for item in gt.section(section) {
942            if !item_matched(section, item, &layers, text_norm) {
943                missed.push(format!("{section}:{item}"));
944            }
945        }
946    }
947
948    let gaps = if diagnose {
949        let store_cands = store_candidates(&comp);
950        missed
951            .iter()
952            .map(|m| {
953                let (section, item) = m
954                    .split_once(':')
955                    .map(|(s, i)| (s.to_string(), i.to_string()))
956                    .unwrap_or_else(|| ("?".into(), m.clone()));
957                classify_gap(
958                    &section,
959                    &item,
960                    repo_dir,
961                    &store,
962                    &config,
963                    &comp,
964                    &layers,
965                    &store_cands,
966                )
967            })
968            .collect()
969    } else {
970        Vec::new()
971    };
972
973    let repo = repo_dir
974        .file_name()
975        .map(|s| s.to_string_lossy().to_string())
976        .unwrap_or_else(|| repo_dir.display().to_string());
977    Ok(RepoRecall {
978        repo,
979        architecture,
980        entrypoints,
981        behavior,
982        state_authority,
983        contracts,
984        landmarks,
985        tests,
986        overall,
987        precision,
988        layer_precision: layer_precision_map,
989        layer_f2,
990        f2,
991        density,
992        atlas_tokens: pack.tokens,
993        resolved_calls,
994        backends_used,
995        backends_missing,
996        resolve_error,
997        landmark_items: gt.landmarks.len(),
998        skipped_reason: None,
999        missed,
1000        gaps,
1001    })
1002}
1003
1004/// Run the recall benchmark over `repo_names` (sorted for deterministic
1005/// output). Repos whose corpus dir is missing, whose ground-truth doc is
1006/// missing, or whose index/atlas fails are recorded with `skipped_reason` —
1007/// this function never panics on missing dirs.
1008// trace:exempt reason=internal-detail
1009pub fn run_atlas_recall(
1010    corpus_dir: &Path,
1011    ground_truth_dir: &Path,
1012    repo_names: &[String],
1013    diagnose: bool,
1014    resolve: bool,
1015) -> Result<AtlasRecallReport, String> {
1016    let mut names: Vec<&String> = repo_names.iter().collect();
1017    names.sort();
1018
1019    let mut report = AtlasRecallReport {
1020        mode: format!("corpus: {}", corpus_dir.display()),
1021        ..Default::default()
1022    };
1023    for name in names {
1024        let repo_dir = corpus_dir.join(name);
1025        if !repo_dir.is_dir() {
1026            report.skipped += 1;
1027            report
1028                .repos
1029                .push(RepoRecall::skipped(name, "corpus dir missing"));
1030            continue;
1031        }
1032        let gt_path = ground_truth_dir.join(format!("{name}.md"));
1033        let gt = match std::fs::read_to_string(&gt_path) {
1034            Ok(text) => parse_ground_truth(&text),
1035            Err(_) => {
1036                report.skipped += 1;
1037                report.repos.push(RepoRecall::skipped(
1038                    name,
1039                    format!("ground truth missing: {}", gt_path.display()),
1040                ));
1041                continue;
1042            }
1043        };
1044        match score_repo(&repo_dir, &gt, diagnose, resolve) {
1045            Ok(r) => {
1046                report.mean_architecture += r.architecture;
1047                report.mean_entrypoints += r.entrypoints;
1048                report.mean_behavior += r.behavior;
1049                report.mean_state_authority += r.state_authority;
1050                report.mean_contracts += r.contracts;
1051                report.mean_landmarks += r.landmarks;
1052                report.mean_tests += r.tests;
1053                report.mean_precision += r.precision;
1054                report.mean_f2 += r.f2;
1055                report.mean_density += r.density;
1056                report.mean_atlas_tokens += r.atlas_tokens as f64;
1057                if diagnose {
1058                    for g in &r.gaps {
1059                        *report
1060                            .gap_histogram
1061                            .entry(g.kind.as_str().to_string())
1062                            .or_insert(0) += 1;
1063                    }
1064                }
1065                report.scored += 1;
1066                report.repos.push(r);
1067            }
1068            Err(e) => {
1069                report.skipped += 1;
1070                report.repos.push(RepoRecall::skipped(name, e));
1071            }
1072        }
1073    }
1074
1075    if report.scored > 0 {
1076        let n = report.scored as f64;
1077        report.mean_architecture /= n;
1078        report.mean_entrypoints /= n;
1079        report.mean_behavior /= n;
1080        report.mean_state_authority /= n;
1081        report.mean_contracts /= n;
1082        report.mean_landmarks /= n;
1083        report.mean_tests /= n;
1084        report.mean_precision /= n;
1085        report.mean_f2 /= n;
1086        report.mean_density /= n;
1087        report.mean_atlas_tokens /= n;
1088        report.mean_overall = (report.mean_architecture
1089            + report.mean_entrypoints
1090            + report.mean_behavior
1091            + report.mean_state_authority
1092            + report.mean_contracts)
1093            / 5.0;
1094    }
1095    report.gate_passed = report.mean_overall >= ATLAS_GATE;
1096    Ok(report)
1097}
1098/// Top-level entry for `scc bench atlas`: locate the workspace, resolve the
1099/// corpus/ground-truth directories (or the fixtures fallback), and run.
1100// trace:v1 id=impl.scc.bench.atlas work=WORK-SCC-003 verifies=REQ-SCC-TEST
1101pub fn run_atlas_bench(
1102    corpus: Option<&Path>,
1103    ground_truth: Option<&Path>,
1104    diagnose: bool,
1105    resolve: bool,
1106) -> Result<AtlasRecallReport, String> {
1107    let cwd = std::env::current_dir().map_err(|e| e.to_string())?;
1108    let root = crate::find_root(&cwd);
1109    let default_corpus = root.join("benchmarks").join("corpus");
1110
1111    // Fixtures fallback: no corpus dir (or an empty one) -> run over the
1112    // golden fixtures with ground truth synthesized from tasks.json.
1113    if corpus.is_none() && repo_dirs(&default_corpus).is_empty() {
1114        return fixtures_fallback(&root, diagnose, resolve);
1115    }
1116    if let Some(p) = corpus {
1117        if !p.is_dir() {
1118            return Err(format!("corpus dir not found: {}", p.display()));
1119        }
1120    }
1121
1122    let corpus_dir = corpus.map(Path::to_path_buf).unwrap_or(default_corpus);
1123    let gt_dir = match ground_truth {
1124        Some(p) => p.to_path_buf(),
1125        None => root.join("benchmarks").join("ground-truth"),
1126    };
1127    let names = repo_dirs(&corpus_dir);
1128    run_atlas_recall(&corpus_dir, &gt_dir, &names, diagnose, resolve)
1129}
1130
1131// trace:exempt reason=internal-detail
1132
1133/// Overfit verdict over the dev-vs-holdout overall gap.
1134#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
1135#[serde(rename_all = "SCREAMING_SNAKE_CASE")]
1136// trace:exempt reason=internal-detail
1137pub enum HoldoutVerdict {
1138    /// holdout >= dev: the dev-tuned rules generalize at least as well.
1139    NoOverfit,
1140    /// dev - tolerance <= holdout < dev: lag inside the noise band.
1141    Borderline,
1142    /// holdout < dev - tolerance: dev-tuned rules do not generalize.
1143    Overfit,
1144}
1145
1146// trace:exempt reason=internal-detail
1147impl HoldoutVerdict {
1148    // trace:exempt reason=internal-detail
1149    pub fn as_str(&self) -> &'static str {
1150        match self {
1151            HoldoutVerdict::NoOverfit => "NO OVERFIT",
1152            HoldoutVerdict::Borderline => "BORDERLINE",
1153            HoldoutVerdict::Overfit => "OVERFIT",
1154        }
1155    }
1156}
1157
1158/// Verdict over the dev-vs-holdout overall gap (`dev`, `holdout` are the
1159/// equal-weighted mean recalls of the five startup-required layers).
1160// trace:exempt reason=internal-detail
1161pub fn holdout_verdict(dev: f64, holdout: f64) -> HoldoutVerdict {
1162    if holdout >= dev {
1163        HoldoutVerdict::NoOverfit
1164    } else if holdout >= dev - HOLDOUT_TOLERANCE {
1165        HoldoutVerdict::Borderline
1166    } else {
1167        HoldoutVerdict::Overfit
1168    }
1169}
1170
1171// trace:exempt reason=internal-detail
1172
1173/// Dev-vs-holdout comparison for `scc bench atlas --holdout`.
1174#[derive(Debug, Clone, Serialize, Deserialize)]
1175// trace:exempt reason=internal-detail
1176pub struct HoldoutComparison {
1177    pub dev: AtlasRecallReport,
1178    /// The validation corpus report (`benchmarks/holdout` — the inspected
1179    /// corpus that tuning has seen; the on-disk dir name is kept, only the
1180    /// output labels say "validation").
1181    pub holdout: AtlasRecallReport,
1182    /// Per-layer gap = validation mean - development mean (fraction,
1183    /// negative = lag).
1184    pub gap_architecture: f64,
1185    pub gap_entrypoints: f64,
1186    pub gap_behavior: f64,
1187    pub gap_state_authority: f64,
1188    pub gap_contracts: f64,
1189    pub gap_overall: f64,
1190    pub verdict: HoldoutVerdict,
1191    /// Path of the written comparison file.
1192    pub results_file: String,
1193}
1194
1195// trace:exempt reason=internal-detail
1196impl HoldoutComparison {
1197    // trace:exempt reason=internal-detail
1198    fn layer_gap(dev: f64, holdout: f64) -> f64 {
1199        (holdout - dev).clamp(-1.0, 1.0)
1200    }
1201}
1202
1203/// Run the holdout protocol: score the development corpus and the
1204/// validation corpus with the same recall pipeline, compute per-layer gaps,
1205/// write `benchmarks/results/holdout-v3.txt`, and return the comparison.
1206///
1207/// `corpus`/`ground_truth` (when given) select the DEVELOPMENT corpus,
1208/// exactly as in `run_atlas_bench` (defaults: `benchmarks/corpus` +
1209/// `benchmarks/ground-truth`). The validation dirs are fixed protocol
1210/// paths: `benchmarks/holdout` + `benchmarks/holdout-ground-truth` (the
1211/// on-disk names are kept from v1; only the output labels say validation).
1212/// The validation corpus must exist — a missing dir is an error, not a
1213/// silent empty run.
1214///
1215/// `resolve` applies to BOTH corpora (the same pipeline must score
1216/// development and validation identically).
1217// trace:exempt reason=internal-detail
1218pub fn run_atlas_holdout(
1219    corpus: Option<&Path>,
1220    ground_truth: Option<&Path>,
1221    diagnose: bool,
1222    resolve: bool,
1223) -> Result<HoldoutComparison, String> {
1224    let cwd = std::env::current_dir().map_err(|e| e.to_string())?;
1225    let root = crate::find_root(&cwd);
1226    let results_dir = root.join("benchmarks").join("results");
1227    let results_file = results_dir.join("holdout-v3.txt");
1228
1229    let dev = run_atlas_bench(corpus, ground_truth, diagnose, resolve)?;
1230
1231    let holdout_corpus = root.join("benchmarks").join("holdout");
1232    let holdout_gt = root.join("benchmarks").join("holdout-ground-truth");
1233    if !holdout_corpus.is_dir() {
1234        return Err(format!(
1235            "holdout corpus dir not found (run --holdout from the workspace): {}",
1236            holdout_corpus.display()
1237        ));
1238    }
1239    let names = repo_dirs(&holdout_corpus);
1240    if names.is_empty() {
1241        return Err(format!(
1242            "holdout corpus dir is empty: {}",
1243            holdout_corpus.display()
1244        ));
1245    }
1246    let mut holdout = run_atlas_recall(&holdout_corpus, &holdout_gt, &names, diagnose, resolve)?;
1247    holdout.mode = format!("validation: {}", holdout_corpus.display());
1248
1249    let c = HoldoutComparison {
1250        gap_architecture: HoldoutComparison::layer_gap(dev.mean_architecture, holdout.mean_architecture),
1251        gap_entrypoints: HoldoutComparison::layer_gap(dev.mean_entrypoints, holdout.mean_entrypoints),
1252        gap_behavior: HoldoutComparison::layer_gap(dev.mean_behavior, holdout.mean_behavior),
1253        gap_state_authority: HoldoutComparison::layer_gap(
1254            dev.mean_state_authority,
1255            holdout.mean_state_authority,
1256        ),
1257        gap_contracts: HoldoutComparison::layer_gap(dev.mean_contracts, holdout.mean_contracts),
1258        gap_overall: HoldoutComparison::layer_gap(dev.mean_overall, holdout.mean_overall),
1259        verdict: holdout_verdict(dev.mean_overall, holdout.mean_overall),
1260        results_file: results_file.display().to_string(),
1261        dev,
1262        holdout,
1263    };
1264
1265    std::fs::create_dir_all(&results_dir).map_err(|e| e.to_string())?;
1266    std::fs::write(&results_file, c.to_results_text()).map_err(|e| e.to_string())?;
1267    Ok(c)
1268}
1269
1270// trace:exempt reason=internal-detail
1271impl HoldoutComparison {
1272    /// Deterministic markdown text for `benchmarks/results/holdout-v3.txt`.
1273    // trace:exempt reason=internal-detail
1274    fn to_results_text(&self) -> String {
1275        let mut out = String::new();
1276        out.push_str("# Holdout v3 — development corpus vs validation corpus\n");
1277        out.push_str(&format!("development corpus: {}\n", self.dev.mode));
1278        out.push_str(&format!("validation corpus:  {}\n", self.holdout.mode));
1279        out.push_str(&format!(
1280            "results:            {}\n",
1281            self.results_file
1282        ));
1283        out.push('\n');
1284
1285        let rows: [(&str, f64, f64); 6] = [
1286            ("architecture", self.dev.mean_architecture, self.holdout.mean_architecture),
1287            ("entrypoints", self.dev.mean_entrypoints, self.holdout.mean_entrypoints),
1288            ("behavior", self.dev.mean_behavior, self.holdout.mean_behavior),
1289            ("state_authority", self.dev.mean_state_authority, self.holdout.mean_state_authority),
1290            ("contracts", self.dev.mean_contracts, self.holdout.mean_contracts),
1291            ("overall (gate)", self.dev.mean_overall, self.holdout.mean_overall),
1292        ];
1293        out.push_str(&format!(
1294            "{:<18} {:>12} {:>12} {:>10}\n",
1295            "layer", "development", "validation", "gap"
1296        ));
1297        for (layer, dev, ho) in rows {
1298            let gap = HoldoutComparison::layer_gap(dev, ho);
1299            out.push_str(&format!(
1300                "{:<18} {:>12.3} {:>12.3} {:>+10.3}\n",
1301                layer, dev, ho, gap
1302            ));
1303        }
1304        out.push('\n');
1305        out.push_str(&format!(
1306            "scored: development {} (skipped {}) | validation {} (skipped {})\n",
1307            self.dev.scored, self.dev.skipped, self.holdout.scored, self.holdout.skipped
1308        ));
1309        out.push_str(&format!(
1310            "precision: development {:.3} | validation {:.3}\n",
1311            self.dev.mean_precision, self.holdout.mean_precision
1312        ));
1313        out.push_str(&format!(
1314            "F2: development {:.3} | validation {:.3}\n",
1315            self.dev.mean_f2, self.holdout.mean_f2
1316        ));
1317        let dev_resolved: usize = self.dev.repos.iter().map(|r| r.resolved_calls).sum();
1318        let holdout_resolved: usize = self.holdout.repos.iter().map(|r| r.resolved_calls).sum();
1319        out.push_str(&format!(
1320            "resolved calls (upgraded): development {dev_resolved} | validation {holdout_resolved}\n"
1321        ));
1322        out.push_str(&format!(
1323            "density (facts/1k tokens): development {:.2} | validation {:.2}\n",
1324            self.dev.mean_density, self.holdout.mean_density
1325        ));
1326        out.push_str(&format!(
1327            "atlas tokens: development {:.0} | validation {:.0}\n",
1328            self.dev.mean_atlas_tokens, self.holdout.mean_atlas_tokens
1329        ));
1330        out.push('\n');
1331        out.push_str("## per-layer precision + F2 (startup-required layers)\n");
1332        out.push_str(&format!(
1333            "{:<18} {:>12} {:>12}   {:>12} {:>12}\n",
1334            "layer", "P:development", "P:validation", "F2:development", "F2:validation"
1335        ));
1336        for (layer, _, _) in rows {
1337            let p_dev = self.dev.mean_layer_precision().get(layer).copied().unwrap_or(0.0);
1338            let p_ho = self.holdout.mean_layer_precision().get(layer).copied().unwrap_or(0.0);
1339            let f_dev = self.dev.mean_layer_f2().get(layer).copied().unwrap_or(0.0);
1340            let f_ho = self.holdout.mean_layer_f2().get(layer).copied().unwrap_or(0.0);
1341            out.push_str(&format!(
1342                "{:<18} {:>12.3} {:>12.3}   {:>12.3} {:>12.3}\n",
1343                layer, p_dev, p_ho, f_dev, f_ho
1344            ));
1345        }
1346        out.push('\n');
1347        out.push_str(&format!(
1348            "## verdict: {} (gap = {:.3}; tolerance = {:.3})\n",
1349            self.verdict.as_str(),
1350            self.gap_overall,
1351            HOLDOUT_TOLERANCE
1352        ));
1353        match self.verdict {
1354            HoldoutVerdict::NoOverfit => out.push_str(
1355                "The validation corpus scores at least as well as the development \
1356                 corpus; the development-tuned rules generalize to unseen repos.\n",
1357            ),
1358            HoldoutVerdict::Borderline => out.push_str(
1359                "The validation corpus lags the development corpus, but by less than \
1360                 the tolerance band; the gap is consistent with \
1361                 corpus-difficulty/ground-truth noise, not demonstrated overfitting.\n",
1362            ),
1363            HoldoutVerdict::Overfit => out.push_str(
1364                "The validation corpus lags the development corpus by more than the \
1365                 tolerance band; rules tuned on the development corpus do not \
1366                 generalize to unseen repos.\n",
1367            ),
1368        }
1369        out.push('\n');
1370        out.push_str("## validation repo overall recall (sorted)\n");
1371        for r in &self.holdout.repos {
1372            match &r.skipped_reason {
1373                Some(reason) => out.push_str(&format!("  {:<24} skipped: {reason}\n", r.repo)),
1374                None => out.push_str(&format!(
1375                    "  {:<24} {:>8.3}\n",
1376                    r.repo, r.overall
1377                )),
1378            }
1379        }
1380        out
1381    }
1382}
1383
1384/// Print the development and validation reports side by side plus the gap
1385/// summary.
1386// trace:exempt reason=internal-detail
1387pub fn print_holdout_report(c: &HoldoutComparison, diagnose: bool) {
1388    println!("scc bench atlas --holdout — development corpus vs validation corpus (v1)");
1389    println!("\n=== DEVELOPMENT corpus ===");
1390    print_report(&c.dev, diagnose);
1391    println!("\n=== VALIDATION corpus ===");
1392    print_report(&c.holdout, diagnose);
1393    println!("\n=== gap (validation - development) ===");
1394    println!(
1395        "  {:<18} {:>10}\n  {:<18} {:>+10.3}\n  {:<18} {:>+10.3}\n  {:<18} {:>+10.3}\n  {:<18} {:>+10.3}\n  {:<18} {:>+10.3}\n  {:<18} {:>+10.3}",
1396        "layer", "gap",
1397        "architecture", c.gap_architecture,
1398        "entrypoints", c.gap_entrypoints,
1399        "behavior", c.gap_behavior,
1400        "state_authority", c.gap_state_authority,
1401        "contracts", c.gap_contracts,
1402        "overall", c.gap_overall,
1403    );
1404    println!(
1405        "  verdict: {} (validation {:.3} vs development {:.3}; tolerance {:.3})",
1406        c.verdict.as_str(),
1407        c.holdout.mean_overall,
1408        c.dev.mean_overall,
1409        HOLDOUT_TOLERANCE
1410    );
1411    println!(
1412        "  F2: development {:.3} | validation {:.3}",
1413        c.dev.mean_f2, c.holdout.mean_f2
1414    );
1415    println!("  results written to: {}", c.results_file);
1416}
1417
1418// ---------------------------------------------------------------------------
1419// Wave 11 — GENERALIZATION II gates (--compare OLD NEW)
1420// ---------------------------------------------------------------------------
1421
1422/// Generalization efficiency (GE): how much of the development improvement
1423/// between two runs transferred to validation.
1424///
1425/// `GE = validation_delta / development_delta` over the overall recall
1426/// means. A positive GE means validation moved the same direction as
1427/// development (semantic waves generalize); negative GE means validation
1428/// regressed while development improved (overfit). When development did not
1429/// move (`dev_delta == 0`) the ratio is degenerate: a validation-only
1430/// improvement counts as pure generalization (`1.0`), anything else as
1431/// `0.0` — never NaN/inf.
1432// trace:exempt reason=internal-detail
1433pub fn generalization_efficiency(dev_delta: f64, validation_delta: f64) -> f64 {
1434    if dev_delta == 0.0 {
1435        return if validation_delta > 0.0 { 1.0 } else { 0.0 };
1436    }
1437    validation_delta / dev_delta
1438}
1439
1440// trace:exempt reason=internal-detail
1441
1442/// Wave-11 gate report over two saved holdout result files (JSON
1443/// `HoldoutComparison`s): the GE gate (`--gate-ge MIN`, default 0.0 — fails
1444/// when `GE <= MIN`; semantic waves must generalize) and the per-section
1445/// validation regression guard (`--guard-section-delta MAX`, default 0.05 —
1446/// fails when ANY startup-required section regresses by more than MAX
1447/// between the two runs, in development or validation).
1448#[derive(Debug, Clone, Serialize, Deserialize)]
1449// trace:exempt reason=internal-detail
1450pub struct CompareReport {
1451    pub old_file: String,
1452    pub new_file: String,
1453    /// new.dev.mean_overall - old.dev.mean_overall.
1454    pub dev_delta_overall: f64,
1455    /// new.holdout.mean_overall - old.holdout.mean_overall.
1456    pub validation_delta_overall: f64,
1457    /// validation_delta_overall / dev_delta_overall (guarded).
1458    pub generalization_efficiency: f64,
1459    /// Per-section delta = new - old, development corpus.
1460    pub dev_deltas: BTreeMap<String, f64>,
1461    /// Per-section delta = new - old, validation corpus.
1462    pub validation_deltas: BTreeMap<String, f64>,
1463    /// Largest startup-required section drop (new - old) across both
1464    /// corpora; 0.0 when nothing regressed.
1465    pub max_section_regression: f64,
1466    /// The section with the largest drop (`section@corpus`), e.g.
1467    /// `contracts@validation`; deterministic (first at the max in
1468    /// `STARTUP_SECTIONS` order, development before validation).
1469    pub max_regression_section: String,
1470    pub gate_ge: f64,
1471    pub guard_section_delta: f64,
1472    pub ge_passed: bool,
1473    pub guard_passed: bool,
1474    /// Human-readable failure reasons; empty when the report passes.
1475    pub failures: Vec<String>,
1476}
1477
1478/// The five startup-required sections the regression guard watches.
1479// trace:exempt reason=internal-detail
1480pub const STARTUP_SECTIONS: [&str; 5] = [
1481    "architecture",
1482    "entrypoints",
1483    "behavior",
1484    "state_authority",
1485    "contracts",
1486];
1487
1488// trace:exempt reason=internal-detail
1489impl CompareReport {
1490    // trace:exempt reason=internal-detail
1491    pub fn passed(&self) -> bool {
1492        self.failures.is_empty()
1493    }
1494}
1495
1496/// Compare two saved holdout result files (JSON `HoldoutComparison`s, e.g.
1497/// `scc bench atlas --holdout --json` output) and apply the Wave-11 gates.
1498/// `old` is the earlier run (the pre-wave baseline), `new` the current one;
1499/// deltas are new - old.
1500// trace:exempt reason=internal-detail
1501pub fn compare_runs(
1502    old: &HoldoutComparison,
1503    new: &HoldoutComparison,
1504    gate_ge: f64,
1505    guard_section_delta: f64,
1506) -> CompareReport {
1507    let dev_delta_overall = new.dev.mean_overall - old.dev.mean_overall;
1508    let validation_delta_overall = new.holdout.mean_overall - old.holdout.mean_overall;
1509    let ge = generalization_efficiency(dev_delta_overall, validation_delta_overall);
1510
1511    let section = |r: &AtlasRecallReport, name: &str| -> f64 {
1512        match name {
1513            "architecture" => r.mean_architecture,
1514            "entrypoints" => r.mean_entrypoints,
1515            "behavior" => r.mean_behavior,
1516            "state_authority" => r.mean_state_authority,
1517            "contracts" => r.mean_contracts,
1518            _ => 0.0,
1519        }
1520    };
1521    let mut dev_deltas: BTreeMap<String, f64> = BTreeMap::new();
1522    let mut validation_deltas: BTreeMap<String, f64> = BTreeMap::new();
1523    let mut max_section_regression: f64 = 0.0;
1524    let mut max_regression_section = String::new();
1525    for name in STARTUP_SECTIONS {
1526        let d_dev = section(&new.dev, name) - section(&old.dev, name);
1527        let d_val = section(&new.holdout, name) - section(&old.holdout, name);
1528        dev_deltas.insert(name.to_string(), d_dev);
1529        validation_deltas.insert(name.to_string(), d_val);
1530        for (d, corpus) in [(d_dev, "development"), (d_val, "validation")] {
1531            if -d > max_section_regression {
1532                max_section_regression = -d;
1533                max_regression_section = format!("{name}@{corpus}");
1534            }
1535        }
1536    }
1537
1538    let ge_passed = ge > gate_ge;
1539    let guard_passed = max_section_regression <= guard_section_delta;
1540    let mut failures: Vec<String> = Vec::new();
1541    if !ge_passed {
1542        failures.push(format!(
1543            "generalization efficiency {ge:.3} <= gate {gate_ge:.3} \
1544             (validation delta {validation_delta_overall:+.3} vs development delta {dev_delta_overall:+.3}): \
1545             the semantic wave did not generalize"
1546        ));
1547    }
1548    if !guard_passed {
1549        failures.push(format!(
1550            "startup-required section regressed by {max_section_regression:.3} > guard {guard_section_delta:.3} \
1551             (worst: {max_regression_section}; new vs old run, development or validation)"
1552        ));
1553    }
1554
1555    CompareReport {
1556        old_file: String::new(),
1557        new_file: String::new(),
1558        dev_delta_overall,
1559        validation_delta_overall,
1560        generalization_efficiency: ge,
1561        dev_deltas,
1562        validation_deltas,
1563        max_section_regression,
1564        max_regression_section,
1565        gate_ge,
1566        guard_section_delta,
1567        ge_passed,
1568        guard_passed,
1569        failures,
1570    }
1571}
1572
1573/// Load a saved holdout JSON result file into a `HoldoutComparison`.
1574// trace:exempt reason=internal-detail
1575pub fn load_holdout_result(path: &Path) -> Result<HoldoutComparison, String> {
1576    let text = std::fs::read_to_string(path)
1577        .map_err(|e| format!("cannot read {}: {e}", path.display()))?;
1578    serde_json::from_str(&text)
1579        .map_err(|e| format!("cannot parse {} as a holdout JSON result: {e}", path.display()))
1580}
1581
1582/// Print the Wave-11 compare report (deltas, GE, per-section guard).
1583// trace:exempt reason=internal-detail
1584pub fn print_compare_report(r: &CompareReport) {
1585    println!("scc bench atlas --compare — Wave-11 generalization gates");
1586    println!("  old: {}", r.old_file);
1587    println!("  new: {}", r.new_file);
1588    println!(
1589        "  development delta:  {:+.3} (new - old, overall)",
1590        r.dev_delta_overall
1591    );
1592    println!(
1593        "  validation delta:   {:+.3} (new - old, overall)",
1594        r.validation_delta_overall
1595    );
1596    println!(
1597        "  generalization efficiency: {:.3} (validation delta / development delta)",
1598        r.generalization_efficiency
1599    );
1600    println!(
1601        "  GE gate (--gate-ge {:.3}): {}",
1602        r.gate_ge,
1603        if r.ge_passed { "PASS" } else { "FAIL" }
1604    );
1605    println!("  per-section deltas (new - old):");
1606    println!(
1607        "  {:<18} {:>12} {:>12}",
1608        "section", "development", "validation"
1609    );
1610    for name in STARTUP_SECTIONS {
1611        println!(
1612            "  {:<18} {:>+12.3} {:>+12.3}",
1613            name,
1614            r.dev_deltas.get(name).copied().unwrap_or(0.0),
1615            r.validation_deltas.get(name).copied().unwrap_or(0.0)
1616        );
1617    }
1618    println!(
1619        "  max section regression: {:.3} (guard --guard-section-delta {:.3}): {}",
1620        r.max_section_regression,
1621        r.guard_section_delta,
1622        if r.guard_passed { "PASS" } else { "FAIL" }
1623    );
1624    if !r.max_regression_section.is_empty() {
1625        println!("    worst regression: {}", r.max_regression_section);
1626    }
1627    if r.passed() {
1628        println!("  verdict: PASS (all Wave-11 generalization gates)");
1629    } else {
1630        println!("  verdict: FAIL");
1631        for f in &r.failures {
1632            println!("    - {f}");
1633        }
1634    }
1635}
1636
1637// trace:exempt reason=internal-detail
1638
1639/// One pinned blind-test clone in `benchmarks/blind-lock.json`: the
1640/// upstream URL plus the exact commit the on-disk clone must sit at.
1641#[derive(Debug, Clone, Default, Serialize, Deserialize)]
1642// trace:exempt reason=internal-detail
1643pub struct BlindLockEntry {
1644    /// Upstream clone URL (informational — the commit is what is enforced).
1645    pub url: String,
1646    /// Pinned commit (full sha) the on-disk clone must be at.
1647    pub commit: String,
1648}
1649
1650// trace:exempt reason=internal-detail
1651
1652/// Load the committed blind commit lock (`benchmarks/blind-lock.json`,
1653/// shaped `{"blind-test": {<name>: {"url": ..., "commit": ...}}}`) as a
1654/// sorted name -> entry map. A missing or malformed lock is a hard error —
1655/// the lock is a protocol artifact, and skipping it would silently disable
1656/// the commit-pin guard.
1657// trace:exempt reason=internal-detail
1658fn load_blind_lock(root: &Path) -> Result<BTreeMap<String, BlindLockEntry>, String> {
1659    let path = root.join("benchmarks").join("blind-lock.json");
1660    let text = std::fs::read_to_string(&path)
1661        .map_err(|e| format!("cannot read blind lock {}: {e}", path.display()))?;
1662    #[derive(Deserialize)]
1663    // trace:exempt reason=internal-detail
1664    struct LockFile {
1665        #[serde(rename = "blind-test")]
1666        blind_test: BTreeMap<String, BlindLockEntry>,
1667    }
1668    let lock: LockFile =
1669        serde_json::from_str(&text).map_err(|e| format!("parse blind lock {}: {e}", path.display()))?;
1670    Ok(lock.blind_test)
1671}
1672
1673// trace:exempt reason=internal-detail
1674
1675/// The on-disk HEAD commit of a clone dir (`git -C <dir> rev-parse HEAD`),
1676/// or an error when the dir is not a git checkout.
1677// trace:exempt reason=internal-detail
1678fn git_head(dir: &Path) -> Result<String, String> {
1679    let out = std::process::Command::new("git")
1680        .args(["rev-parse", "HEAD"])
1681        .current_dir(dir)
1682        .output()
1683        .map_err(|e| format!("cannot run git in {}: {e}", dir.display()))?;
1684    if !out.status.success() {
1685        return Err(format!(
1686            "git rev-parse HEAD failed in {}: {}",
1687            dir.display(),
1688            String::from_utf8_lossy(&out.stderr).trim()
1689        ));
1690    }
1691    let sha = String::from_utf8_lossy(&out.stdout).trim().to_string();
1692    if sha.is_empty() {
1693        return Err(format!(
1694            "git rev-parse HEAD returned nothing in {}",
1695            dir.display()
1696        ));
1697    }
1698    Ok(sha)
1699}
1700
1701// trace:exempt reason=internal-detail
1702
1703/// Pure HEAD-vs-lock comparison (testable without git): Ok when the
1704/// on-disk clone HEAD equals the pinned commit, else a hard error naming
1705/// the repo, the actual commit, and the pin.
1706// trace:exempt reason=internal-detail
1707fn verify_head(name: &str, actual: &str, expected: &str) -> Result<(), String> {
1708    if actual == expected {
1709        Ok(())
1710    } else {
1711        Err(format!("blind-test repo {name} at {actual}, lock pins {expected}"))
1712    }
1713}
1714
1715/// Blind-test manifest (Wave 11 — GENERALIZATION II): a sha256 fingerprint
1716/// of the frozen blind set — the ground-truth answer keys
1717/// (`benchmarks/blind-test-ground-truth/**`), the clone list
1718/// (the committed `benchmarks/blind-test/README.md` manifest plus the
1719/// on-disk repo dirs — the git-ls-files equivalent for the gitignored
1720/// clones), and the commit pins from `benchmarks/blind-lock.json` (a
1721/// `lock <name> <sha>` line per repo, so the digest covers the pinned
1722/// commits). Written into the blind results header; `--blind` verifies the
1723/// hash matches the previous run before scoring and errors on mismatch, so
1724/// a changed blind set can never silently re-score different keys.
1725#[derive(Debug, Clone, Default, Serialize, Deserialize)]
1726// trace:exempt reason=internal-detail
1727pub struct BlindManifest {
1728    /// sha256 hex of the deterministic manifest text.
1729    pub sha256: String,
1730    /// Number of ground-truth key files hashed.
1731    pub ground_truth_files: usize,
1732    /// Number of blind-test repo dirs in the clone list.
1733    pub repos: usize,
1734    /// The deterministic manifest text (paths + per-file hashes + clone
1735    /// list); the sha256 is over exactly this.
1736    pub text: String,
1737}
1738
1739// trace:exempt reason=internal-detail
1740
1741/// Deterministic manifest text + sha256 over the blind set under `root`:
1742/// every file in `benchmarks/blind-test-ground-truth/**` (path + content
1743/// hash), the clone list (the committed `benchmarks/blind-test/README.md`
1744/// content hash + the sorted top-level repo dir names — the git-ls-files
1745/// equivalent for the gitignored clones), and a `lock <name> <sha>` line
1746/// per repo from the committed `benchmarks/blind-lock.json` — the digest
1747/// covers the pinned commits. Missing ground-truth dir is an error (the
1748/// protocol requires it); a missing README is tolerated (the clone list
1749/// then reduces to the repo dirs); a missing lock is an error (the commit
1750/// pins are a protocol artifact).
1751// trace:exempt reason=internal-detail
1752pub fn blind_manifest(root: &Path) -> Result<BlindManifest, String> {
1753    let gt_dir = root.join("benchmarks").join("blind-test-ground-truth");
1754    if !gt_dir.is_dir() {
1755        return Err(format!(
1756            "blind-test ground-truth dir not found (run --blind from the workspace): {}",
1757            gt_dir.display()
1758        ));
1759    }
1760    let blind_dir = root.join("benchmarks").join("blind-test");
1761    let mut out = String::from("# scc blind-test manifest (deterministic)\n");
1762
1763    let mut gt_files: Vec<PathBuf> = Vec::new();
1764    collect_files(&gt_dir, &mut gt_files);
1765    gt_files.sort();
1766    for f in &gt_files {
1767        let rel = f
1768            .strip_prefix(&gt_dir)
1769            .map(|p| p.display().to_string())
1770            .unwrap_or_else(|_| f.display().to_string());
1771        let content = std::fs::read(f).map_err(|e| format!("read {}: {e}", f.display()))?;
1772        out.push_str(&format!("ground-truth {rel} {}\n", sha256::hex(&content)));
1773    }
1774
1775    // clone list: the committed README manifest + the on-disk repo dirs
1776    // (git ls-files of benchmarks/blind-test is just README.md — the
1777    // clones are gitignored — so the dir names are the machine-level
1778    // clone-set fingerprint).
1779    let readme = blind_dir.join("README.md");
1780    if readme.is_file() {
1781        let content =
1782            std::fs::read(&readme).map_err(|e| format!("read {}: {e}", readme.display()))?;
1783        out.push_str(&format!("README.md {}\n", sha256::hex(&content)));
1784    }
1785    let dirs = repo_dirs(&blind_dir);
1786    for d in &dirs {
1787        out.push_str(&format!("clone {d}\n"));
1788    }
1789    // commit pins: the committed benchmarks/blind-lock.json (url + commit
1790    // per blind-test repo). The digest covers the pinned commits, so a
1791    // re-pin (or a reclone that updates the lock) changes the hash —
1792    // and run_atlas_blind additionally verifies each on-disk clone HEAD
1793    // against the pin before scoring.
1794    let lock = load_blind_lock(root)?;
1795    for (name, entry) in &lock {
1796        out.push_str(&format!("lock {name} {}\n", entry.commit));
1797    }
1798
1799    let digest = sha256::hex(out.as_bytes());
1800    Ok(BlindManifest {
1801        sha256: digest,
1802        ground_truth_files: gt_files.len(),
1803        repos: dirs.len(),
1804        text: out,
1805    })
1806}
1807
1808/// Recursively collect regular files under `dir` into `out`.
1809// trace:exempt reason=internal-detail
1810fn collect_files(dir: &Path, out: &mut Vec<PathBuf>) {
1811    let Ok(entries) = std::fs::read_dir(dir) else {
1812        return;
1813    };
1814    for entry in entries.flatten() {
1815        let p = entry.path();
1816        if p.is_dir() {
1817            collect_files(&p, out);
1818        } else if p.is_file() {
1819            out.push(p);
1820        }
1821    }
1822}
1823
1824/// The manifest hash recorded in a previous blind results-file header
1825/// (`blind manifest sha256: <hex>`), for the change-detection check.
1826/// The header line also carries a human summary after the hex
1827/// (`(20 ground-truth files, ...)`); only the first whitespace token is
1828/// the hash.
1829// trace:exempt reason=internal-detail
1830fn manifest_hash_from_header(header: &str) -> Option<String> {
1831    header.lines().find_map(|l| {
1832        l.trim()
1833            .strip_prefix("blind manifest sha256:")
1834            .map(str::trim)
1835            .and_then(|h| h.split_whitespace().next())
1836            .filter(|h| !h.is_empty())
1837            .map(|h| h.to_string())
1838    })
1839}
1840
1841/// Guarded division for the blind transfer ratio: `numerator / denominator`,
1842/// 0.0 when the denominator is 0 (nothing transferred onto nothing).
1843// trace:exempt reason=internal-detail
1844fn safe_ratio(numerator: f64, denominator: f64) -> f64 {
1845    if denominator == 0.0 {
1846        0.0
1847    } else {
1848        numerator / denominator
1849    }
1850}
1851
1852/// Per-section blind transfer ratios (blind mean / validation mean) over
1853/// the seven layers plus overall — deterministic, 0.0-guarded.
1854// trace:exempt reason=internal-detail
1855fn blind_transfer_ratios(c: &BlindComparison) -> Vec<(&'static str, f64)> {
1856    let v = &c.validation;
1857    let b = &c.blind;
1858    vec![
1859        ("architecture", safe_ratio(b.mean_architecture, v.mean_architecture)),
1860        ("entrypoints", safe_ratio(b.mean_entrypoints, v.mean_entrypoints)),
1861        ("behavior", safe_ratio(b.mean_behavior, v.mean_behavior)),
1862        ("state_authority", safe_ratio(b.mean_state_authority, v.mean_state_authority)),
1863        ("contracts", safe_ratio(b.mean_contracts, v.mean_contracts)),
1864        ("landmarks", safe_ratio(b.mean_landmarks, v.mean_landmarks)),
1865        ("tests", safe_ratio(b.mean_tests, v.mean_tests)),
1866        ("overall", safe_ratio(b.mean_overall, v.mean_overall)),
1867    ]
1868}
1869
1870// trace:exempt reason=internal-detail
1871
1872/// Blind-test protocol comparison: validation-vs-blind generalization.
1873#[derive(Debug, Clone, Serialize, Deserialize)]
1874// trace:exempt reason=internal-detail
1875pub struct BlindComparison {
1876    /// Validation corpus aggregates (`benchmarks/holdout`) — per-repo
1877    /// detail stripped.
1878    pub validation: AtlasRecallReport,
1879    /// Blind corpus aggregates (`benchmarks/blind-test`) — per-repo detail
1880    /// stripped: blind-test failures are never shown to tuning agents.
1881    pub blind: AtlasRecallReport,
1882    /// Per-layer gap = blind mean - validation mean (fraction, negative = lag).
1883    pub gap_architecture: f64,
1884    pub gap_entrypoints: f64,
1885    pub gap_behavior: f64,
1886    pub gap_state_authority: f64,
1887    pub gap_contracts: f64,
1888    pub gap_landmarks: f64,
1889    pub gap_tests: f64,
1890    pub gap_overall: f64,
1891    /// The sha256 manifest fingerprint of the frozen blind set scored in
1892    /// this run (ground-truth keys + clone list); verified against the
1893    /// previous run before scoring.
1894    pub manifest: BlindManifest,
1895    /// Path of the written results file.
1896    pub results_file: String,
1897}
1898
1899/// Score one fixed-protocol corpus with the same recall pipeline; missing
1900/// or empty dirs are errors, not silent empty runs.
1901// trace:exempt reason=internal-detail
1902fn score_protocol_corpus(
1903    corpus: &Path,
1904    ground_truth: &Path,
1905    resolve: bool,
1906    mode_label: &str,
1907) -> Result<AtlasRecallReport, String> {
1908    if !corpus.is_dir() {
1909        return Err(format!(
1910            "corpus dir not found (run from the workspace): {}",
1911            corpus.display()
1912        ));
1913    }
1914    let names = repo_dirs(corpus);
1915    if names.is_empty() {
1916        return Err(format!("corpus dir is empty: {}", corpus.display()));
1917    }
1918    let mut report = run_atlas_recall(corpus, ground_truth, &names, false, resolve)?;
1919    report.mode = format!("{mode_label}: {}", corpus.display());
1920    Ok(report)
1921}
1922
1923// trace:exempt reason=internal-detail
1924
1925/// Run the blind protocol: verify the frozen blind clones are at their
1926/// pinned commits (`benchmarks/blind-lock.json`), score the validation
1927/// corpus (`benchmarks/holdout`) and the blind-test corpus
1928/// (`benchmarks/blind-test`) with the same recall pipeline, keep ONLY
1929/// aggregates (per-repo rows, missed keys, and filenames are stripped —
1930/// blind-test failures are never shown to tuning agents), compute the
1931/// validation-vs-blind generalization gap, write
1932/// `benchmarks/results/blind-v1.txt`, and return the comparison.
1933///
1934/// `--diagnose` is refused: diagnosis prints per-repo miss lines, which
1935/// would leak the blind misses ("blind corpus is not diagnosable").
1936// trace:exempt reason=internal-detail
1937pub fn run_atlas_blind(diagnose: bool, resolve: bool) -> Result<BlindComparison, String> {
1938    if diagnose {
1939        return Err("blind corpus is not diagnosable".into());
1940    }
1941    let cwd = std::env::current_dir().map_err(|e| e.to_string())?;
1942    let root = crate::find_root(&cwd);
1943    let results_dir = root.join("benchmarks").join("results");
1944    let results_file = results_dir.join("blind-v1.txt");
1945
1946    // Wave 11: verify the frozen blind set did not change since the last
1947    // recorded run BEFORE scoring — a mismatched manifest is a protocol
1948    // error, not a silent re-score of different keys.
1949    let manifest = blind_manifest(&root)?;
1950    if results_file.is_file() {
1951        let previous = std::fs::read_to_string(&results_file).map_err(|e| e.to_string())?;
1952        if let Some(prev_hash) = manifest_hash_from_header(&previous) {
1953            if prev_hash != manifest.sha256 {
1954                return Err(format!(
1955                    "blind-test set changed: manifest sha256 {prev_hash} (previous run) != {} (current); \
1956                     the blind ground truth or clone list was modified — refusing to score",
1957                    manifest.sha256
1958                ));
1959            }
1960        }
1961    }
1962
1963    let validation_corpus = root.join("benchmarks").join("holdout");
1964    let validation_gt = root.join("benchmarks").join("holdout-ground-truth");
1965    let blind_corpus = root.join("benchmarks").join("blind-test");
1966    let blind_gt = root.join("benchmarks").join("blind-test-ground-truth");
1967
1968    // Wave 13: the blind corpus is frozen at pinned commits — verify every
1969    // clone's on-disk HEAD against benchmarks/blind-lock.json BEFORE
1970    // scoring. A reclone at a different commit (or a missing clone) is a
1971    // protocol error, not a silent re-score of different code.
1972    let lock = load_blind_lock(&root)?;
1973    for (name, entry) in &lock {
1974        let clone = blind_corpus.join(name);
1975        let actual = git_head(&clone)
1976            .map_err(|e| format!("blind-test repo {name} missing or not a git checkout: {e}"))?;
1977        verify_head(name, &actual, &entry.commit)?;
1978    }
1979
1980    let validation = score_protocol_corpus(&validation_corpus, &validation_gt, resolve, "validation")?;
1981    let blind = score_protocol_corpus(&blind_corpus, &blind_gt, resolve, "blind-test")?;
1982    // Aggregates only, end to end: drop every per-repo row, missed key,
1983    // and filename before the comparison leaves this function.
1984    let validation = validation.aggregates_only();
1985    let blind = blind.aggregates_only();
1986
1987    let c = BlindComparison {
1988        gap_architecture: HoldoutComparison::layer_gap(
1989            validation.mean_architecture,
1990            blind.mean_architecture,
1991        ),
1992        gap_entrypoints: HoldoutComparison::layer_gap(
1993            validation.mean_entrypoints,
1994            blind.mean_entrypoints,
1995        ),
1996        gap_behavior: HoldoutComparison::layer_gap(validation.mean_behavior, blind.mean_behavior),
1997        gap_state_authority: HoldoutComparison::layer_gap(
1998            validation.mean_state_authority,
1999            blind.mean_state_authority,
2000        ),
2001        gap_contracts: HoldoutComparison::layer_gap(
2002            validation.mean_contracts,
2003            blind.mean_contracts,
2004        ),
2005        gap_landmarks: HoldoutComparison::layer_gap(validation.mean_landmarks, blind.mean_landmarks),
2006        gap_tests: HoldoutComparison::layer_gap(validation.mean_tests, blind.mean_tests),
2007        gap_overall: HoldoutComparison::layer_gap(validation.mean_overall, blind.mean_overall),
2008        manifest,
2009        results_file: results_file.display().to_string(),
2010        validation,
2011        blind,
2012    };
2013
2014    std::fs::create_dir_all(&results_dir).map_err(|e| e.to_string())?;
2015    std::fs::write(&results_file, c.to_blind_text()).map_err(|e| e.to_string())?;
2016    Ok(c)
2017}
2018
2019// trace:exempt reason=internal-detail
2020impl BlindComparison {
2021    /// Deterministic aggregates-only text for
2022    /// `benchmarks/results/blind-v1.txt` (aggregates + gap only).
2023    // trace:exempt reason=internal-detail
2024    fn to_blind_text(&self) -> String {
2025        let mut out = String::new();
2026        out.push_str("# Blind v1 — validation vs blind (aggregates only)\n");
2027        out.push_str(&format!(
2028            "blind manifest sha256: {} ({} ground-truth files, {} blind-test repos)\n",
2029            self.manifest.sha256, self.manifest.ground_truth_files, self.manifest.repos
2030        ));
2031        out.push_str(&format!("validation corpus: {}\n", self.validation.mode));
2032        out.push_str(&format!("blind corpus:      {}\n", self.blind.mode));
2033        out.push_str(&format!("results:           {}\n", self.results_file));
2034        out.push_str(
2035            "# blind-test failures are never shown to tuning agents: no per-repo rows, no filenames, no missed keys\n",
2036        );
2037        out.push('\n');
2038
2039        let rows: [(&str, f64, f64); 8] = [
2040            ("architecture", self.validation.mean_architecture, self.blind.mean_architecture),
2041            ("entrypoints", self.validation.mean_entrypoints, self.blind.mean_entrypoints),
2042            ("behavior", self.validation.mean_behavior, self.blind.mean_behavior),
2043            ("state_authority", self.validation.mean_state_authority, self.blind.mean_state_authority),
2044            ("contracts", self.validation.mean_contracts, self.blind.mean_contracts),
2045            ("landmarks", self.validation.mean_landmarks, self.blind.mean_landmarks),
2046            ("tests", self.validation.mean_tests, self.blind.mean_tests),
2047            ("overall (gate)", self.validation.mean_overall, self.blind.mean_overall),
2048        ];
2049        out.push_str(&format!(
2050            "{:<18} {:>12} {:>12} {:>10}\n",
2051            "layer", "validation", "blind", "gap"
2052        ));
2053        for (layer, v, b) in rows {
2054            let gap = HoldoutComparison::layer_gap(v, b);
2055            out.push_str(&format!(
2056                "{:<18} {:>12.3} {:>12.3} {:>+10.3}\n",
2057                layer, v, b, gap
2058            ));
2059        }
2060        out.push('\n');
2061        out.push_str(&format!(
2062            "scored: validation {} (skipped {}) | blind {} (skipped {})\n",
2063            self.validation.scored,
2064            self.validation.skipped,
2065            self.blind.scored,
2066            self.blind.skipped
2067        ));
2068        out.push_str(&format!(
2069            "precision: validation {:.3} | blind {:.3}\n",
2070            self.validation.mean_precision, self.blind.mean_precision
2071        ));
2072        out.push_str(&format!(
2073            "F2: validation {:.3} | blind {:.3}\n",
2074            self.validation.mean_f2, self.blind.mean_f2
2075        ));
2076        out.push_str(&format!(
2077            "density (facts/1k tokens): validation {:.2} | blind {:.2}\n",
2078            self.validation.mean_density, self.blind.mean_density
2079        ));
2080        out.push_str(&format!(
2081            "atlas tokens: validation {:.0} | blind {:.0}\n",
2082            self.validation.mean_atlas_tokens, self.blind.mean_atlas_tokens
2083        ));
2084        out.push('\n');
2085        out.push_str("## blind transfer ratio (blind / validation) — how much of the validation recall transfers to the unseen blind corpus\n");
2086        for (layer, ratio) in blind_transfer_ratios(self) {
2087            out.push_str(&format!("  {:<18} {:>10.3}\n", layer, ratio));
2088        }
2089        out.push('\n');
2090        out.push_str("## generalization gap (blind - validation) — informational, not gating\n");
2091        for (layer, gap) in [
2092            ("architecture", self.gap_architecture),
2093            ("entrypoints", self.gap_entrypoints),
2094            ("behavior", self.gap_behavior),
2095            ("state_authority", self.gap_state_authority),
2096            ("contracts", self.gap_contracts),
2097            ("landmarks", self.gap_landmarks),
2098            ("tests", self.gap_tests),
2099            ("overall", self.gap_overall),
2100        ] {
2101            out.push_str(&format!("  {:<18} {:>+10.3}\n", layer, gap));
2102        }
2103        out.push_str(&format!(
2104            "  gate (recall >= {ATLAS_GATE}): blind {} | validation {}\n",
2105            if self.blind.gate_passed { "PASS" } else { "FAIL" },
2106            if self.validation.gate_passed { "PASS" } else { "FAIL" }
2107        ));
2108        out
2109    }
2110}
2111
2112/// Print the blind protocol: aggregates ONLY (no per-repo rows, no missed
2113/// keys, no filenames) plus the validation-vs-blind generalization gap.
2114// trace:exempt reason=internal-detail
2115pub fn print_blind_report(c: &BlindComparison) {
2116    println!("scc bench atlas --blind — validation vs blind (aggregates only)");
2117    println!("  validation corpus: {}", c.validation.mode);
2118    println!("  blind corpus:      {}", c.blind.mode);
2119    println!(
2120        "  blind manifest sha256: {} ({} ground-truth files, {} blind-test repos)",
2121        c.manifest.sha256, c.manifest.ground_truth_files, c.manifest.repos
2122    );
2123    println!("  blind-test failures are never shown to tuning agents.");
2124    println!("\n=== per-section means ===");
2125    println!(
2126        "  {:<18} {:>12} {:>12} {:>10}",
2127        "layer", "validation", "blind", "gap"
2128    );
2129    let rows: [(&str, f64, f64); 8] = [
2130        ("architecture", c.validation.mean_architecture, c.blind.mean_architecture),
2131        ("entrypoints", c.validation.mean_entrypoints, c.blind.mean_entrypoints),
2132        ("behavior", c.validation.mean_behavior, c.blind.mean_behavior),
2133        ("state_authority", c.validation.mean_state_authority, c.blind.mean_state_authority),
2134        ("contracts", c.validation.mean_contracts, c.blind.mean_contracts),
2135        ("landmarks", c.validation.mean_landmarks, c.blind.mean_landmarks),
2136        ("tests", c.validation.mean_tests, c.blind.mean_tests),
2137        ("overall (gate)", c.validation.mean_overall, c.blind.mean_overall),
2138    ];
2139    for (layer, v, b) in rows {
2140        let gap = HoldoutComparison::layer_gap(v, b);
2141        println!("  {:<18} {:>12.3} {:>12.3} {:>+10.3}", layer, v, b, gap);
2142    }
2143    println!(
2144        "  scored: validation {} (skipped {}) | blind {} (skipped {})",
2145        c.validation.scored, c.validation.skipped, c.blind.scored, c.blind.skipped
2146    );
2147    println!(
2148        "  precision: validation {:.3} | blind {:.3}   F2: validation {:.3} | blind {:.3}",
2149        c.validation.mean_precision, c.blind.mean_precision, c.validation.mean_f2, c.blind.mean_f2
2150    );
2151    println!(
2152        "  density (facts/1k tokens): validation {:.2} | blind {:.2}   atlas tokens: validation {:.0} | blind {:.0}",
2153        c.validation.mean_density, c.blind.mean_density,
2154        c.validation.mean_atlas_tokens, c.blind.mean_atlas_tokens
2155    );
2156    println!("\n=== blind transfer ratio (blind / validation) ===");
2157    for (layer, ratio) in blind_transfer_ratios(c) {
2158        println!("  {:<18} {:>10.3}", layer, ratio);
2159    }
2160    println!("\n=== generalization gap (blind - validation) — informational, not gating ===");
2161    for (layer, gap) in [
2162        ("architecture", c.gap_architecture),
2163        ("entrypoints", c.gap_entrypoints),
2164        ("behavior", c.gap_behavior),
2165        ("state_authority", c.gap_state_authority),
2166        ("contracts", c.gap_contracts),
2167        ("landmarks", c.gap_landmarks),
2168        ("tests", c.gap_tests),
2169        ("overall", c.gap_overall),
2170    ] {
2171        println!("  {:<18} {:>+10.3}", layer, gap);
2172    }
2173    println!("  results written to: {}", c.results_file);
2174}
2175
2176/// Sorted non-hidden subdirectory names of `dir` (the corpus listing).
2177// trace:exempt reason=internal-detail
2178fn repo_dirs(dir: &Path) -> Vec<String> {
2179    let mut out: Vec<String> = Vec::new();
2180    let Ok(entries) = std::fs::read_dir(dir) else {
2181        return out;
2182    };
2183    for entry in entries.flatten() {
2184        let name = entry.file_name().to_string_lossy().to_string();
2185        if name.starts_with('.') {
2186            continue;
2187        }
2188        if entry.path().is_dir() {
2189            out.push(name);
2190        }
2191    }
2192    out.sort();
2193    out
2194}
2195
2196/// Hermetic fixtures run: copy the golden fixtures into a temp corpus (the
2197/// real fixtures are never indexed into) and synthesize ground-truth docs
2198/// from `benchmarks/tasks.json`.
2199// trace:exempt reason=internal-detail
2200fn fixtures_fallback(root: &Path, diagnose: bool, resolve: bool) -> Result<AtlasRecallReport, String> {
2201    let fixtures = locate_fixtures_dir().ok_or("cannot locate fixtures/ directory")?;
2202    let tasks_path = root.join("benchmarks").join("tasks.json");
2203    let text = std::fs::read_to_string(&tasks_path)
2204        .map_err(|e| format!("cannot read {}: {e}", tasks_path.display()))?;
2205    let corpus: BenchmarkCorpus =
2206        serde_json::from_str(&text).map_err(|e| format!("tasks.json parse: {e}"))?;
2207
2208    let names: Vec<String> = corpus
2209        .tasks
2210        .iter()
2211        .map(|t| t.repo.clone())
2212        .collect::<BTreeSet<_>>()
2213        .into_iter()
2214        .collect();
2215
2216    let tmp = tempfile::TempDir::new().map_err(|e| e.to_string())?;
2217    let tmp_corpus = tmp.path().join("corpus");
2218    let tmp_gt = tmp.path().join("ground-truth");
2219    std::fs::create_dir_all(&tmp_corpus).map_err(|e| e.to_string())?;
2220    std::fs::create_dir_all(&tmp_gt).map_err(|e| e.to_string())?;
2221    for name in &names {
2222        let src = fixtures.join(name);
2223        if src.is_dir() {
2224            copy_tree_skip_scc(&src, &tmp_corpus.join(name));
2225        }
2226        let doc = ground_truth_from_tasks(&corpus.tasks, name);
2227        std::fs::write(tmp_gt.join(format!("{name}.md")), doc.to_markdown())
2228            .map_err(|e| e.to_string())?;
2229    }
2230
2231    let mut report = run_atlas_recall(&tmp_corpus, &tmp_gt, &names, diagnose, resolve)?;
2232    report.mode = "fixtures fallback (ground truth from benchmarks/tasks.json)".to_string();
2233    Ok(report)
2234}
2235
2236/// Fixtures-fallback ground truth per repo, synthesized from the task
2237/// corpus. Mapping: components -> architecture; routes -> entrypoints AND
2238/// contracts (HTTP routes are both); symbols proxy flow steps (behavior);
2239/// stores + data -> state_authority (who owns the store/DB); tests -> tests
2240/// (informational).
2241// trace:exempt reason=internal-detail
2242fn ground_truth_from_tasks(tasks: &[BenchTask], repo: &str) -> GroundTruthDoc {
2243    let mut doc = GroundTruthDoc::default();
2244    for t in tasks {
2245        if t.repo != repo {
2246            continue;
2247        }
2248        doc.architecture
2249            .extend(t.ground_truth.components.iter().cloned());
2250        for r in &t.ground_truth.routes {
2251            doc.entrypoints.push(r.clone());
2252            doc.contracts.push(r.clone());
2253        }
2254        doc.behavior
2255            .extend(t.ground_truth.symbols.iter().cloned());
2256        doc.state_authority
2257            .extend(t.ground_truth.stores.iter().cloned());
2258        doc.state_authority
2259            .extend(t.ground_truth.data.iter().cloned());
2260        doc.tests.extend(t.ground_truth.tests.iter().cloned());
2261    }
2262    doc.dedupe();
2263    doc
2264}
2265
2266/// Locate the fixtures directory: walk up from cwd; fall back to the
2267/// workspace-relative path (dev tooling).
2268// trace:exempt reason=internal-detail
2269pub fn locate_fixtures_dir() -> Option<PathBuf> {
2270    let mut dir = std::env::current_dir().ok()?;
2271    loop {
2272        if dir.join("fixtures").join("http-service-python").is_dir() {
2273            return Some(dir.join("fixtures"));
2274        }
2275        if !dir.pop() {
2276            break;
2277        }
2278    }
2279    let manifest: PathBuf = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
2280    let candidate = manifest
2281        .parent()
2282        .and_then(|p| p.parent())
2283        .map(|p| p.join("fixtures"))
2284        .filter(|p| p.join("http-service-python").is_dir());
2285    candidate
2286}
2287
2288/// Copy a tree, skipping `.scc` state dirs (mirrors golden::copy_tree).
2289// trace:exempt reason=internal-detail
2290fn copy_tree_skip_scc(src: &Path, dst: &Path) {
2291    std::fs::create_dir_all(dst).unwrap();
2292    for entry in std::fs::read_dir(src).unwrap() {
2293        let entry = entry.unwrap();
2294        let name = entry.file_name();
2295        if name == ".scc" {
2296            continue;
2297        }
2298        let from = entry.path();
2299        let to = dst.join(&name);
2300        if from.is_dir() {
2301            std::fs::create_dir_all(&to).unwrap();
2302            copy_tree_skip_scc(&from, &to);
2303        } else {
2304            std::fs::copy(&from, &to).unwrap();
2305        }
2306    }
2307}
2308
2309// ---------------------------------------------------------------------------
2310// Gap diagnosis (--diagnose)
2311// ---------------------------------------------------------------------------
2312
2313/// Normalized presence haystack of one entity: id + name + attribute JSON.
2314// trace:exempt reason=internal-detail
2315fn entity_norm(e: &Entity) -> String {
2316    let attrs = serde_json::to_string(&e.attributes).unwrap_or_default();
2317    norm(&format!("{} {} {}", e.id, e.name, attrs))
2318}
2319
2320/// Precompute the store-presence candidates for one repo (entity haystacks +
2321/// derived route strings), so gap diagnosis does not re-serialize every
2322/// entity per missed item.
2323// trace:exempt reason=internal-detail
2324fn store_candidates(comp: &Compiler<'_>) -> Vec<String> {
2325    let mut cands: Vec<String> = Vec::new();
2326    for e in comp.graph.entities.values() {
2327        cands.push(entity_norm(e));
2328    }
2329    for e in comp.graph.entities_of_kind(scc_core::kinds::ROUTE) {
2330        let method = e
2331            .attributes
2332            .get("method")
2333            .and_then(|v| v.as_str())
2334            .unwrap_or("");
2335        let path = e
2336            .attributes
2337            .get("path")
2338            .and_then(|v| v.as_str())
2339            .unwrap_or("");
2340        if !path.is_empty() {
2341            cands.push(norm(&format!("{method} {path}")));
2342        }
2343    }
2344    cands
2345}
2346
2347/// Classify a missed ground-truth item by where it disappeared, via the
2348/// deterministic ladder of the v2 spec: store presence -> flows ->
2349/// components -> rendered atlas text.
2350///
2351/// - nothing in the store: file-level parseability decides PARSER (disabled
2352///   language / no extractor for the format) vs EXTRACTOR (parsed but no
2353///   semantic fact emitted). WRITER is not observable from the store side
2354///   (it needs extractor output); parsed-but-absent facts map to EXTRACTOR.
2355/// - in the store as an isolated symbol (zero graph relationships):
2356///   RESOLUTION (resolution never connected it to a flow or component).
2357/// - in the store but never compiled into a component or flow: COMPILER.
2358/// - compiled into a flow but no component: COMPILER (flow-level only).
2359/// - compiled into a component but absent from the rendered atlas:
2360///   PROJECTION (dropped by budget/policy/rendering).
2361/// - present in the rendered atlas text only under a spelling the
2362///   structured layers do not carry (README prose, format variants):
2363///   ALIAS (the aliases did not reconcile it).
2364#[allow(clippy::too_many_arguments)]
2365// trace:exempt reason=internal-detail
2366fn classify_gap(
2367    section: &str,
2368    item: &str,
2369    repo_dir: &Path,
2370    store: &Store,
2371    config: &scc_indexer::Config,
2372    comp: &Compiler<'_>,
2373    layers: &AtlasLayers,
2374    store_cands: &[String],
2375) -> GapFinding {
2376    let n = norm(item);
2377    let section = section.to_string();
2378    let item = item.to_string();
2379
2380    // 1. Store presence.
2381    if !store_cands.iter().any(|c| c.contains(&n)) {
2382        // 2. Not in the store at all: file-level parseability.
2383        return match file_language(item.as_str(), repo_dir, store) {
2384            Some(lang) if config.language_enabled(lang) => GapFinding {
2385                section,
2386                item,
2387                kind: GapKind::Extractor,
2388                detail: format!(
2389                    "file exists and {} is enabled, but no semantic fact reached the store",
2390                    lang.as_str()
2391                ),
2392            },
2393            Some(lang) => GapFinding {
2394                section,
2395                item,
2396                kind: GapKind::Parser,
2397                detail: format!(
2398                    "file is {} but the extractor is disabled/ignored",
2399                    lang.as_str()
2400                ),
2401            },
2402            None => GapFinding {
2403                section,
2404                item,
2405                kind: GapKind::Extractor,
2406                detail: "nothing in the store and no extractor emits this fact".into(),
2407            },
2408        };
2409    }
2410
2411    // 3. In the store: a symbol with zero graph relationships was never
2412    // wired by resolution/compilation.
2413    let sym_matches: Vec<&Entity> = comp
2414        .graph
2415        .entities
2416        .values()
2417        .filter(|e| e.kind == scc_core::kinds::SYMBOL && entity_norm(e).contains(&n))
2418        .collect();
2419    let isolated = !sym_matches.is_empty()
2420        && sym_matches.iter().all(|e| {
2421            comp.graph
2422                .out
2423                .get(&e.id)
2424                .map(|r| r.is_empty())
2425                .unwrap_or(true)
2426                && comp
2427                    .graph
2428                    .inn
2429                    .get(&e.id)
2430                    .map(|r| r.is_empty())
2431                    .unwrap_or(true)
2432        });
2433    if isolated {
2434        return GapFinding {
2435            section,
2436            item,
2437            kind: GapKind::Resolution,
2438            detail: "symbol exists in the store but has zero graph relationships; resolution never connected it to a flow or component".into(),
2439        };
2440    }
2441
2442    // 4. Compiled into a component? Then only rendering could drop it.
2443    if layers.components.contains(&n) {
2444        return if layers.text.contains(&n) {
2445            GapFinding {
2446                section,
2447                item,
2448                kind: GapKind::Alias,
2449                detail: "compiled into a component and present in the rendered atlas; the structured layer's spelling/aliases did not match".into(),
2450            }
2451        } else {
2452            GapFinding {
2453                section,
2454                item,
2455                kind: GapKind::Projection,
2456                detail: "compiled into a component but dropped from the rendered atlas by budget/policy/rendering".into(),
2457            }
2458        };
2459    }
2460
2461    // 5. Reached a flow but never a component, or neither.
2462    if layers.flows.contains(&n) {
2463        return GapFinding {
2464            section,
2465            item,
2466            kind: GapKind::Compiler,
2467            detail: "reached a flow but was never compiled into a component".into(),
2468        };
2469    }
2470    if layers.text.contains(&n) {
2471        GapFinding {
2472            section,
2473            item,
2474            kind: GapKind::Alias,
2475            detail: "present in the rendered atlas text (e.g. README purpose or trust boundaries) under a spelling the structured layers do not carry".into(),
2476        }
2477    } else {
2478        GapFinding {
2479            section,
2480            item,
2481            kind: GapKind::Compiler,
2482            detail: "present in the store but not compiled into components or flows".into(),
2483        }
2484    }
2485}
2486
2487/// If `item` names a repo file, return the language it would be scanned
2488/// with (language from the file registry when scanned; otherwise inferred
2489/// from the extension when the file exists on disk).
2490// trace:exempt reason=internal-detail
2491fn file_language(item: &str, repo_dir: &Path, store: &Store) -> Option<Language> {
2492    if !item.contains('/') {
2493        return None;
2494    }
2495    let rel = item.trim_start_matches("./");
2496    if let Ok(files) = store.all_files() {
2497        for (path, _hash, lang, _kind, _size) in files {
2498            if path == rel {
2499                return lang_from_str(&lang);
2500            }
2501        }
2502    }
2503    let full = repo_dir.join(rel);
2504    if full.is_file() {
2505        lang_from_ext(rel)
2506    } else {
2507        None
2508    }
2509}
2510
2511// trace:exempt reason=internal-detail
2512fn lang_from_str(s: &str) -> Option<Language> {
2513    match s {
2514        "python" => Some(Language::Python),
2515        "typescript" | "javascript" => Some(Language::TypeScript),
2516        "go" => Some(Language::Go),
2517        "rust" => Some(Language::Rust),
2518        "java" => Some(Language::Java),
2519        _ => None,
2520    }
2521}
2522
2523// trace:exempt reason=internal-detail
2524fn lang_from_ext(path: &str) -> Option<Language> {
2525    let ext = path
2526        .rsplit('.')
2527        .next()
2528        .unwrap_or("")
2529        .to_ascii_lowercase();
2530    match ext.as_str() {
2531        "py" | "pyi" => Some(Language::Python),
2532        "ts" | "tsx" | "mts" | "cts" => Some(Language::TypeScript),
2533        "js" | "jsx" | "mjs" | "cjs" => Some(Language::TypeScript),
2534        "go" => Some(Language::Go),
2535        "rs" => Some(Language::Rust),
2536        "java" => Some(Language::Java),
2537        _ => None,
2538    }
2539}
2540
2541// trace:exempt reason=internal-detail
2542pub fn print_report(r: &AtlasRecallReport, diagnose: bool) {
2543    println!("scc bench atlas — startup-atlas recall vs independent ground truth (Wave 8 §57, v3)");
2544    println!("  mode: {}", r.mode);
2545    println!(
2546        "  gate: overall mean recall (architecture+entrypoints+behavior+state_authority+contracts) >= {ATLAS_GATE}"
2547    );
2548    println!(
2549        "  {:<24} {:>6} {:>6} {:>6} {:>6} {:>6} {:>6} {:>6} {:>8} {:>6} {:>6} {:>6} {:>5}  note",
2550        "repo", "arch", "entry", "behav", "state", "contr", "landm", "tests", "overall",
2551        "prec", "f2", "f/1k", "toks"
2552    );
2553    for repo in &r.repos {
2554        match &repo.skipped_reason {
2555            Some(reason) => println!(
2556                "  {:<24} {:>6} {:>6} {:>6} {:>6} {:>6} {:>6} {:>6} {:>8} {:>6} {:>6} {:>6} {:>5}  skipped: {reason}",
2557                repo.repo, "-", "-", "-", "-", "-", "-", "-", "-", "-", "-", "-", "-"
2558            ),
2559            None => {
2560                let note = if repo.resolved_calls > 0 {
2561                    format!("resolved:{}", repo.resolved_calls)
2562                } else {
2563                    String::new()
2564                };
2565                println!(
2566                    "  {:<24} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>8.3} {:>6.3} {:>6.3} {:>6.2} {:>5}  {}",
2567                    repo.repo,
2568                    repo.architecture,
2569                    repo.entrypoints,
2570                    repo.behavior,
2571                    repo.state_authority,
2572                    repo.contracts,
2573                    repo.landmarks,
2574                    repo.tests,
2575                    repo.overall,
2576                    repo.precision,
2577                    repo.f2,
2578                    repo.density,
2579                    repo.atlas_tokens,
2580                    note
2581                );
2582                for m in &repo.missed {
2583                    println!("      missed: {m}");
2584                }
2585            }
2586        }
2587    }
2588    println!(
2589        "  {:<24} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>6.3} {:>8.3} {:>6.3} {:>6.3} {:>6.2} {:>5}",
2590        "mean",
2591        r.mean_architecture,
2592        r.mean_entrypoints,
2593        r.mean_behavior,
2594        r.mean_state_authority,
2595        r.mean_contracts,
2596        r.mean_landmarks,
2597        r.mean_tests,
2598        r.mean_overall,
2599        r.mean_precision,
2600        r.mean_f2,
2601        r.mean_density,
2602        r.mean_atlas_tokens.round() as usize
2603    );
2604    println!("  scored: {}   skipped: {}", r.scored, r.skipped);
2605    let resolved_total: usize = r.repos.iter().map(|repo| repo.resolved_calls).sum();
2606    if resolved_total > 0 {
2607        println!("  resolved calls (semantic backends upgraded): {resolved_total}");
2608    }
2609    println!(
2610        "  gate: {} (overall mean recall {:.3} >= {ATLAS_GATE})",
2611        if r.gate_passed { "PASS" } else { "FAIL" },
2612        r.mean_overall
2613    );
2614
2615    if !diagnose {
2616        return;
2617    }
2618    println!("\ngap-kind histogram (all sections):");
2619    if r.gap_histogram.is_empty() {
2620        println!("  (none — no missed items)");
2621    }
2622    for (kind, count) in &r.gap_histogram {
2623        println!("  {kind:<12} {count}");
2624    }
2625    println!("\nPer-repo gap lines (regenerated into benchmarks/results/ground-truth-gaps.md):");
2626    for repo in &r.repos {
2627        if repo.gaps.is_empty() {
2628            continue;
2629        }
2630        println!("\n## {}", repo.repo);
2631        for g in &repo.gaps {
2632            println!("- `{}:{}` — {} GAP: {}", g.section, g.item, g.kind.as_str(), g.detail);
2633        }
2634    }
2635}
2636
2637#[cfg(test)]
2638// trace:exempt reason=internal-detail
2639mod tests {
2640    use super::*;
2641
2642    // trace:exempt reason=internal-detail
2643    const GT_MD: &str = r#"# synth repo
2644> synthetic | python | service
2645
2646## architecture
2647- root — the app root component
2648- services — business logic
2649
2650## entrypoints
2651- GET /api/items — fetch items
2652
2653## behavior
2654- `handle_items` — entry handler
2655
2656## state_authority
2657- db.items — owned by services
2658
2659## contracts
2660- POST /api/items
2661
2662## landmarks
2663- `ItemStore` — one zoom level deeper
2664
2665## tests
2666- test_create_item — creation test
2667"#;
2668
2669    #[test]
2670    // trace:exempt reason=internal-detail
2671    fn parses_ground_truth_sections() {
2672        let doc = parse_ground_truth(GT_MD);
2673        assert_eq!(doc.architecture, ["root", "services"]);
2674        assert_eq!(doc.entrypoints, ["GET /api/items"]);
2675        assert_eq!(doc.behavior, ["handle_items"], "inline-code backticks stripped");
2676        assert_eq!(doc.state_authority, ["db.items"]);
2677        assert_eq!(doc.contracts, ["POST /api/items"]);
2678        assert_eq!(doc.landmarks, ["ItemStore"]);
2679        assert_eq!(doc.tests, ["test_create_item"]);
2680    }
2681
2682    #[test]
2683    // trace:exempt reason=internal-detail
2684    fn parse_ground_truth_accepts_legacy_section_names() {
2685        let md = "## components\n- root\n## flows\n- handle\n## ownership\n- db.x\n## entrypoints\n- e\n## contracts\n- c\n## tests\n- t\n";
2686        let doc = parse_ground_truth(md);
2687        assert_eq!(doc.architecture, ["root"], "components -> architecture");
2688        assert_eq!(doc.behavior, ["handle"], "flows -> behavior");
2689        assert_eq!(doc.state_authority, ["db.x"], "ownership -> state_authority");
2690        assert_eq!(doc.entrypoints, ["e"]);
2691        assert_eq!(doc.contracts, ["c"]);
2692        assert_eq!(doc.tests, ["t"]);
2693        assert!(doc.landmarks.is_empty());
2694    }
2695
2696    #[test]
2697    // trace:exempt reason=internal-detail
2698    fn parse_ground_truth_dedupes_keys_per_section() {
2699        let md = "## contracts\n- GET /api/items\n- GET /api/items — duplicate bullet\n- POST /api/items\n";
2700        let doc = parse_ground_truth(md);
2701        assert_eq!(doc.contracts, ["GET /api/items", "POST /api/items"]);
2702        // same key in two sections is not a duplicate
2703        let md2 = "## entrypoints\n- GET /api/items\n## contracts\n- GET /api/items\n";
2704        let doc2 = parse_ground_truth(md2);
2705        assert_eq!(doc2.entrypoints, ["GET /api/items"]);
2706        assert_eq!(doc2.contracts, ["GET /api/items"]);
2707    }
2708
2709    #[test]
2710    // trace:exempt reason=internal-detail
2711    fn norm_applies_documented_aliases() {
2712        assert_eq!(norm("Controller::run"), "controller.run");
2713        // trace:exempt reason=internal-detail
2714        assert_eq!(norm("fn main"), "main");
2715        assert_eq!(norm("./src/index-client.js"), "src/index-client.js");
2716        assert_eq!(norm("ArgMatches"), "argmatches");
2717        assert_eq!(norm("GET /api/items"), "get /api/items");
2718    }
2719
2720    #[test]
2721    // trace:exempt reason=internal-detail
2722    fn recall_counts_all_hit_partial_and_zero() {
2723        // haystack already normalized: layer_recall applies norm() to items
2724        let hay = "root\nservices\nget /api/items\nhandle_items\ndb.items\npost /api/items\nitemstore\ntest_create_item";
2725        let doc = parse_ground_truth(GT_MD);
2726        let (a, _, _) = layer_recall(&doc.architecture, hay);
2727        assert_eq!(a, 1.0);
2728        let (e, _, _) = layer_recall(&doc.entrypoints, hay);
2729        assert_eq!(e, 1.0);
2730        let (b, _, _) = layer_recall(&doc.behavior, hay);
2731        assert_eq!(b, 1.0);
2732        let (s, _, _) = layer_recall(&doc.state_authority, hay);
2733        assert_eq!(s, 1.0);
2734        let (c, _, _) = layer_recall(&doc.contracts, hay);
2735        assert_eq!(c, 1.0);
2736        let (l, _, _) = layer_recall(&doc.landmarks, hay);
2737        assert_eq!(l, 1.0);
2738        let (t, _, _) = layer_recall(&doc.tests, hay);
2739        assert_eq!(t, 1.0);
2740
2741        // zero: no ground-truth item is a substring of empty haystack
2742        let (z, _, _) = layer_recall(&doc.architecture, "");
2743        assert_eq!(z, 0.0);
2744
2745        // partial: "root" hits, "services" does not
2746        let (p, hit, total) = layer_recall(&doc.architecture, "root only");
2747        assert_eq!((p, hit, total), (0.5, 1, 2));
2748
2749        // empty ground truth scores 1.0 (nothing to miss)
2750        let empty = GroundTruthDoc::default();
2751        let (x, hit0, total0) = layer_recall(&empty.architecture, "anything");
2752        assert_eq!((x, hit0, total0), (1.0, 0, 0));
2753    }
2754
2755    #[test]
2756    // trace:exempt reason=internal-detail
2757    fn layer_precision_counts_atlas_entries_matching_ground_truth() {
2758        // One haystack entry per line; an entry matches when a ground-truth
2759        // item is contained in it. 2 of 3 entries match -> 2/3.
2760        let items = vec!["services".to_string(), "db.items".to_string()];
2761        let hay = "services\ndb.items\nzzz_extra_fact";
2762        let p = layer_precision(&items, hay);
2763        assert!((p - 2.0 / 3.0).abs() < 1e-9, "2/3 entries match: {p}");
2764
2765        // non-matching entries drag precision down
2766        let p2 = layer_precision(&items, "services\nzzz_extra_fact\nzzz_extra_fact2");
2767        assert!((p2 - 1.0 / 3.0).abs() < 1e-9, "1/3 entries match: {p2}");
2768
2769        // empty layer haystack -> 1.0 (nothing spurious)
2770        assert_eq!(layer_precision(&items, ""), 1.0);
2771        // empty ground truth -> no entry can match -> 0.0
2772        assert_eq!(layer_precision(&[], hay), 0.0);
2773        // normalization applies to both sides
2774        let p3 = layer_precision(&["Controller::run".to_string()], "controller.run\nother");
2775        assert!((p3 - 0.5).abs() < 1e-9, ":: alias applied: {p3}");
2776    }
2777
2778    #[test]
2779    // trace:exempt reason=internal-detail
2780    fn f2_score_weights_recall_over_precision() {
2781        // F2 = 5PR/(4P+R): recall-favoring harmonic mean.
2782        let a = f2_score(0.9, 0.5);
2783        let b = f2_score(0.5, 0.9);
2784        assert!(b > a, "recall-weighted: {a} vs {b}");
2785        // exact values: P=0.5 R=0.5 -> 5*0.25/(2+0.5) = 1.25/2.5 = 0.5
2786        assert!((f2_score(0.5, 0.5) - 0.5).abs() < 1e-9);
2787        // zero when P+R==0
2788        assert_eq!(f2_score(0.0, 0.0), 0.0);
2789        assert_eq!(f2_score(0.0, 0.5), 0.0);
2790        assert_eq!(f2_score(0.5, 0.0), 0.0);
2791        // perfect P and R -> 1.0
2792        assert!((f2_score(1.0, 1.0) - 1.0).abs() < 1e-9);
2793    }
2794
2795    #[test]
2796    // trace:exempt reason=internal-detail
2797    fn aggregates_only_strips_per_repo_detail() {
2798        let r = AtlasRecallReport {
2799            mean_overall: 0.42,
2800            repos: vec![RepoRecall {
2801                repo: "secret-repo".into(),
2802                missed: vec!["architecture:secret_item".into()],
2803                ..Default::default()
2804            }],
2805            ..Default::default()
2806        };
2807        let a = r.aggregates_only();
2808        assert_eq!(a.repos.len(), 0);
2809        assert!((a.mean_overall - 0.42).abs() < 1e-9, "aggregates survive");
2810    }
2811
2812    #[test]
2813    // trace:exempt reason=internal-detail
2814    fn run_atlas_blind_refuses_diagnose() {
2815        let err = run_atlas_blind(true, false).unwrap_err();
2816        assert!(err.contains("blind corpus is not diagnosable"), "{err}");
2817    }
2818
2819    #[test]
2820    // trace:exempt reason=internal-detail
2821    fn blind_results_text_is_aggregates_only_and_deterministic() {
2822        let mut validation = AtlasRecallReport {
2823            mean_architecture: 0.3,
2824            mean_overall: 0.27,
2825            mean_precision: 0.6,
2826            mean_f2: 0.4,
2827            scored: 20,
2828            ..Default::default()
2829        };
2830        validation.mean_entrypoints = 0.2;
2831        validation.mean_behavior = 0.3;
2832        validation.mean_state_authority = 0.3;
2833        validation.mean_contracts = 0.2;
2834        validation.mean_landmarks = 0.1;
2835        validation.mean_tests = 0.1;
2836        validation.mean_density = 0.5;
2837        validation.mean_atlas_tokens = 46657.0;
2838        let mut blind = validation.clone();
2839        blind.mean_overall = 0.31;
2840        blind.mean_architecture = 0.34;
2841        blind.mean_contracts = 0.25;
2842
2843        let c = BlindComparison {
2844            gap_architecture: HoldoutComparison::layer_gap(validation.mean_architecture, blind.mean_architecture),
2845            gap_entrypoints: HoldoutComparison::layer_gap(validation.mean_entrypoints, blind.mean_entrypoints),
2846            gap_behavior: HoldoutComparison::layer_gap(validation.mean_behavior, blind.mean_behavior),
2847            gap_state_authority: HoldoutComparison::layer_gap(validation.mean_state_authority, blind.mean_state_authority),
2848            gap_contracts: HoldoutComparison::layer_gap(validation.mean_contracts, blind.mean_contracts),
2849            gap_landmarks: HoldoutComparison::layer_gap(validation.mean_landmarks, blind.mean_landmarks),
2850            gap_tests: HoldoutComparison::layer_gap(validation.mean_tests, blind.mean_tests),
2851            gap_overall: HoldoutComparison::layer_gap(validation.mean_overall, blind.mean_overall),
2852            manifest: BlindManifest {
2853                sha256: "0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef".into(),
2854                ground_truth_files: 20,
2855                repos: 20,
2856                text: String::new(),
2857            },
2858            results_file: "benchmarks/results/blind-v1.txt".to_string(),
2859            validation,
2860            blind,
2861        };
2862        let text = c.to_blind_text();
2863        let text2 = c.to_blind_text();
2864        assert_eq!(text, text2, "deterministic output");
2865        assert!(text.contains("aggregates only"), "{text}");
2866        assert!(text.contains("never shown to tuning agents"), "{text}");
2867        assert!(text.contains("overall (gate)"), "{text}");
2868        assert!(text.contains("generalization gap"), "{text}");
2869        assert!(text.contains("+0.040"), "gap +0.04 rendered: {text}");
2870        // Wave 11: manifest header + blind transfer ratios are printed
2871        assert!(
2872            text.contains("blind manifest sha256: 0123456789abcdef0123456789abcdef0123456789abcdef0123456789abcdef"),
2873            "manifest hash in header: {text}"
2874        );
2875        assert!(text.contains("blind transfer ratio"), "{text}");
2876        assert!(
2877            text.contains("overall"),
2878            "overall row present: {text}"
2879        );
2880        assert!(
2881            text.contains("     1.148"),
2882            "overall transfer ratio 0.31/0.27 = 1.148 rendered: {text}"
2883        );
2884        assert!(
2885            !text.contains("missed:"),
2886            "no per-repo miss lines in blind output: {text}"
2887        );
2888    }
2889
2890    #[test]
2891    // trace:exempt reason=internal-detail
2892    fn token_density_guards_zero_tokens() {
2893        assert_eq!(token_density(5, 0), 0.0);
2894        assert!((token_density(5, 5000) - 1.0).abs() < 1e-9, "5 facts / 5k tokens = 1 per 1k");
2895    }
2896
2897    #[test]
2898    // trace:exempt reason=internal-detail
2899    fn run_atlas_recall_skips_missing_dirs_without_panicking() {
2900        let tmp = tempfile::TempDir::new().unwrap();
2901        let corpus = tmp.path().join("corpus");
2902        let gt = tmp.path().join("ground-truth");
2903        std::fs::create_dir_all(&corpus).unwrap();
2904        std::fs::create_dir_all(&gt).unwrap();
2905        std::fs::create_dir_all(corpus.join("repo-a")).unwrap();
2906
2907        let names = ["repo-b".to_string(), "repo-a".to_string()];
2908        let report = run_atlas_recall(&corpus, &gt, &names, false, false).unwrap();
2909        assert_eq!(report.scored, 0);
2910        assert_eq!(report.skipped, 2);
2911        assert!(!report.gate_passed);
2912        assert_eq!(report.mean_overall, 0.0);
2913        // deterministic order regardless of input order
2914        assert_eq!(report.repos[0].repo, "repo-a");
2915        assert!(report.repos[0]
2916            .skipped_reason
2917            .as_deref()
2918            .unwrap()
2919            .contains("ground truth missing"));
2920        assert_eq!(report.repos[1].repo, "repo-b");
2921        assert!(report.repos[1]
2922            .skipped_reason
2923            .as_deref()
2924            .unwrap()
2925            .contains("corpus dir missing"));
2926    }
2927
2928    // trace:exempt reason=internal-detail
2929    fn synth_repo(tmp: &tempfile::TempDir, gt_md: &str) {
2930        let corpus = tmp.path().join("corpus");
2931        let gt = tmp.path().join("ground-truth");
2932        std::fs::create_dir_all(&corpus).unwrap();
2933        std::fs::create_dir_all(&gt).unwrap();
2934        let repo_dir = corpus.join("synth");
2935        std::fs::create_dir_all(repo_dir.join("services")).unwrap();
2936        std::fs::write(
2937            repo_dir.join("main.py"),
2938            "from fastapi import FastAPI\nfrom services.items import ItemRepository\n\napp = FastAPI()\n\n\n@app.get(\"/api/items\")\ndef handle_items() -> list:\n    \"\"\"List all items.\"\"\"\n    repo = ItemRepository()\n    return repo.find_all()\n",
2939        )
2940        .unwrap();
2941        std::fs::write(
2942            repo_dir.join("services/items.py"),
2943            "class ItemRepository:\n    def find_all(self):\n        return []\n",
2944        )
2945        .unwrap();
2946        std::fs::write(gt.join("synth.md"), gt_md).unwrap();
2947    }
2948
2949    #[test]
2950    // trace:exempt reason=internal-detail
2951    fn run_atlas_recall_scores_synthetic_repo_structurally() {
2952        let tmp = tempfile::TempDir::new().unwrap();
2953        synth_repo(
2954            &tmp,
2955            "## architecture\n- root\n- services\n## entrypoints\n- handle_items\n## behavior\n- ItemRepository\n## state_authority\n- zzz_nonexistent_store\n## contracts\n- GET /api/items\n- GET /api/zzz_nonexistent\n## landmarks\n- ItemRepository\n## tests\n- test_nonexistent\n",
2956        );
2957
2958        let report =
2959            run_atlas_recall(&tmp.path().join("corpus"), &tmp.path().join("ground-truth"), &["synth".to_string()], false, false)
2960                .unwrap();
2961        assert_eq!(report.scored, 1);
2962        assert_eq!(report.skipped, 0);
2963        assert!(report.gate_passed == (report.mean_overall >= ATLAS_GATE));
2964        let r = &report.repos[0];
2965        assert!(r.skipped_reason.is_none());
2966        // without the semantic pass the repo reports zero resolved calls
2967        assert_eq!(r.resolved_calls, 0);
2968        // overall is the equal-weighted mean of the five startup layers
2969        let expect = (r.architecture + r.entrypoints + r.behavior + r.state_authority + r.contracts) / 5.0;
2970        assert!((r.overall - expect).abs() < 1e-9);
2971        // the deliberately nonexistent items must be missed
2972        assert_eq!(r.state_authority, 0.0);
2973        assert_eq!(r.contracts, 0.5, "GET /api/items hits, zzz misses");
2974        assert!(r.missed.contains(&"state_authority:zzz_nonexistent_store".to_string()));
2975        assert!(r.missed.contains(&"contracts:GET /api/zzz_nonexistent".to_string()));
2976        assert!(!r.missed.contains(&"entrypoints:handle_items".to_string()));
2977        // v3 metrics are finite and in range
2978        assert!((0.0..=1.0).contains(&r.precision), "startup precision: {}", r.precision);
2979        assert!((0.0..=1.0).contains(&r.f2), "F2: {}", r.f2);
2980        assert_eq!(r.layer_precision.len(), 5, "five startup layers");
2981        assert_eq!(r.layer_f2.len(), 5);
2982        assert!(r.density >= 0.0);
2983        assert!(r.atlas_tokens > 0, "rendered atlas has tokens");
2984        assert_eq!(r.landmark_items, 1);
2985    }
2986
2987    #[test]
2988    // trace:exempt reason=internal-detail
2989    fn diagnose_classifies_missed_items_deterministically() {
2990        let tmp = tempfile::TempDir::new().unwrap();
2991        synth_repo(
2992            &tmp,
2993            "## architecture\n- root\n## state_authority\n- zzz_nonexistent_store\n## contracts\n- GET /api/zzz_nonexistent\n## tests\n- test_nonexistent\n",
2994        );
2995
2996        let report =
2997            run_atlas_recall(&tmp.path().join("corpus"), &tmp.path().join("ground-truth"), &["synth".to_string()], true, false)
2998                .unwrap();
2999        let r = &report.repos[0];
3000        assert!(!r.gaps.is_empty(), "diagnose produced gap findings");
3001        // nothing in the store, not a repo file -> EXTRACTOR
3002        for g in &r.gaps {
3003            assert_eq!(g.kind, GapKind::Extractor, "{:?}", g);
3004            assert!(!g.detail.is_empty());
3005        }
3006        // histogram covers exactly the diagnosed items
3007        let total: usize = report.gap_histogram.values().sum();
3008        assert_eq!(total, r.gaps.len());
3009    }
3010
3011    #[test]
3012    // trace:exempt reason=internal-detail
3013    fn fallback_ground_truth_maps_tasks_and_dedupes() {
3014        let task = serde_json::from_str::<BenchTask>(
3015            r#"{"id":"t1","repo":"r","goal":"g",
3016               "ground_truth":{"files":["a.py"],"symbols":["s1","s2"],"components":["root"],
3017                               "data":["db.x"],"routes":["GET /api/x"],"stores":["redis"],
3018                               "tests":["test_a"]},
3019               "hallucinations":[]}"#,
3020        )
3021        .unwrap();
3022        let doc = ground_truth_from_tasks(&[task.clone(), task], "r");
3023        assert_eq!(doc.architecture, ["root"]);
3024        assert_eq!(doc.entrypoints, ["GET /api/x"]);
3025        assert_eq!(doc.contracts, ["GET /api/x"]);
3026        assert_eq!(doc.behavior, ["s1", "s2"]);
3027        assert_eq!(doc.state_authority, ["redis", "db.x"]);
3028        assert_eq!(doc.tests, ["test_a"]);
3029    }
3030
3031    #[test]
3032    // trace:exempt reason=internal-detail
3033    fn holdout_verdict_matches_gap_bands() {
3034        // holdout >= dev -> NO OVERFIT
3035        assert_eq!(holdout_verdict(0.20, 0.25), HoldoutVerdict::NoOverfit);
3036        assert_eq!(holdout_verdict(0.20, 0.20), HoldoutVerdict::NoOverfit);
3037        // lag inside the tolerance band -> BORDERLINE
3038        assert_eq!(holdout_verdict(0.20, 0.17), HoldoutVerdict::Borderline);
3039        assert_eq!(holdout_verdict(0.20, 0.20 - 0.05 + 1e-9), HoldoutVerdict::Borderline);
3040        // lag beyond the tolerance band -> OVERFIT
3041        assert_eq!(holdout_verdict(0.20, 0.14), HoldoutVerdict::Overfit);
3042        assert_eq!(holdout_verdict(0.20, 0.0), HoldoutVerdict::Overfit);
3043        assert_eq!(holdout_verdict(0.20, 0.20 - 0.05 - 1e-9), HoldoutVerdict::Overfit);
3044    }
3045
3046    #[test]
3047    // trace:exempt reason=internal-detail
3048    fn layer_gap_is_clamped_and_signed() {
3049        assert!((HoldoutComparison::layer_gap(0.1, 0.3) - 0.2).abs() < 1e-9);
3050        assert!((HoldoutComparison::layer_gap(0.3, 0.1) - (-0.2)).abs() < 1e-9);
3051        assert_eq!(HoldoutComparison::layer_gap(0.0, 2.0), 1.0);
3052        assert_eq!(HoldoutComparison::layer_gap(2.0, 0.0), -1.0);
3053        assert_eq!(HoldoutComparison::layer_gap(0.5, 0.5), 0.0);
3054    }
3055
3056    #[test]
3057    // trace:exempt reason=internal-detail
3058    fn holdout_results_text_is_deterministic_and_contains_gap() {
3059        let dev = AtlasRecallReport {
3060            mean_architecture: 0.3,
3061            ..Default::default()
3062        };
3063        let mut dev = dev;
3064        dev.mean_entrypoints = 0.2;
3065        dev.mean_behavior = 0.4;
3066        dev.mean_state_authority = 0.1;
3067        dev.mean_contracts = 0.5;
3068        dev.mean_overall = 0.3;
3069        dev.scored = 20;
3070        let mut holdout = dev.clone();
3071        holdout.mean_overall = 0.26; // lag 0.04 -> BORDERLINE (inside 0.05 band)
3072        holdout.mean_contracts = 0.45;
3073
3074        let c = HoldoutComparison {
3075            gap_architecture: HoldoutComparison::layer_gap(dev.mean_architecture, holdout.mean_architecture),
3076            gap_entrypoints: HoldoutComparison::layer_gap(dev.mean_entrypoints, holdout.mean_entrypoints),
3077            gap_behavior: HoldoutComparison::layer_gap(dev.mean_behavior, holdout.mean_behavior),
3078            gap_state_authority: HoldoutComparison::layer_gap(
3079                dev.mean_state_authority,
3080                holdout.mean_state_authority,
3081            ),
3082            gap_contracts: HoldoutComparison::layer_gap(dev.mean_contracts, holdout.mean_contracts),
3083            gap_overall: HoldoutComparison::layer_gap(dev.mean_overall, holdout.mean_overall),
3084            verdict: holdout_verdict(dev.mean_overall, holdout.mean_overall),
3085            results_file: "benchmarks/results/holdout-v3.txt".to_string(),
3086            dev,
3087            holdout,
3088        };
3089        let text = c.to_results_text();
3090        let text2 = c.to_results_text();
3091        assert_eq!(text, text2, "deterministic output");
3092        assert!(text.contains("development corpus vs validation corpus"), "{text}");
3093        assert!(text.contains("overall (gate)"));
3094        assert!(text.contains("per-layer precision + F2"), "{text}");
3095        assert!(text.contains("verdict: BORDERLINE"));
3096        assert!(text.contains("-0.040"), "gap -0.04 rendered: {text}");
3097    }
3098    #[test]
3099    // trace:exempt reason=internal-detail
3100    fn behavior_chains_match_in_order() {
3101        // P1: ground-truth chains (A -> B -> C) match the per-step haystack
3102        // as an in-order subsequence, not as a substring.
3103        let hay = "src: Command.main\n-> src: Command.parse_args\n-> src: Command.invoke";
3104        assert!(chain_matches("Command.main -> Command.parse_args -> Command.invoke", hay));
3105        // gaps allowed (other steps between), order required
3106        let hay2 = "a: X\nb: main\nc: mid\nd: parse_args\ne: end\nf: invoke";
3107        assert!(chain_matches("main -> parse_args -> invoke", hay2));
3108        // wrong order must NOT match
3109        assert!(!chain_matches("invoke -> main", hay));
3110        assert!(!chain_matches("main -> invoke -> parse_args", hay2));
3111        // plain items still substring-match
3112        assert!(item_matches("parse_args", hay));
3113    }
3114
3115    // ---- Wave 11: generalization gates + blind manifest ----
3116
3117    #[test]
3118    // trace:exempt reason=internal-detail
3119    fn sha256_matches_standard_test_vectors() {
3120        assert_eq!(
3121            sha256::hex(b""),
3122            "e3b0c44298fc1c149afbf4c8996fb92427ae41e4649b934ca495991b7852b855"
3123        );
3124        assert_eq!(
3125            sha256::hex(b"abc"),
3126            "ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad"
3127        );
3128        assert_eq!(
3129            sha256::hex(b"abcdbcdecdefdefgefghfghighijhijkijkljklmklmnlmnomnopnopq"),
3130            "248d6a61d20638b8e5c026930c3e6039a33ce45964ff2167f6ecedd419db06c1"
3131        );
3132        let million: Vec<u8> = vec![b'a'; 1_000_000];
3133        assert_eq!(
3134            sha256::hex(&million),
3135            "cdc76e5c9914fb9281a1c7e284d73e67f1809a48a497200e046d39ccc7112cd0"
3136        );
3137    }
3138
3139    #[test]
3140    // trace:exempt reason=internal-detail
3141    fn generalization_efficiency_gate_fails_negative_passes_positive() {
3142        // dev improved, validation regressed -> negative GE -> the default
3143        // gate (MIN = 0.0) fails: the semantic wave overfit
3144        let ge = generalization_efficiency(0.10, -0.05);
3145        assert!((ge - (-0.5)).abs() < 1e-9);
3146        assert!(ge <= 0.0, "negative GE must fail the default gate");
3147        // both moved up -> positive GE -> passes
3148        let ge2 = generalization_efficiency(0.10, 0.04);
3149        assert!((ge2 - 0.4).abs() < 1e-9);
3150        assert!(ge2 > 0.0, "positive GE passes the default gate");
3151        // zero dev delta: validation-only improvement is pure generalization
3152        // (1.0); anything else is 0.0 — never NaN/inf
3153        assert_eq!(generalization_efficiency(0.0, 0.05), 1.0);
3154        assert_eq!(generalization_efficiency(0.0, 0.0), 0.0);
3155        assert_eq!(generalization_efficiency(0.0, -0.05), 0.0);
3156        for d in [0.0, -0.0, 1e-9, -1e-9] {
3157            assert!(generalization_efficiency(0.0, d).is_finite());
3158        }
3159    }
3160
3161    // trace:exempt reason=internal-detail
3162    fn holdout_with(
3163        dev_overall: f64,
3164        dev_sections: [f64; 5],
3165        val_overall: f64,
3166        val_sections: [f64; 5],
3167    ) -> HoldoutComparison {
3168        let dev = AtlasRecallReport {
3169            mean_overall: dev_overall,
3170            mean_architecture: dev_sections[0],
3171            mean_entrypoints: dev_sections[1],
3172            mean_behavior: dev_sections[2],
3173            mean_state_authority: dev_sections[3],
3174            mean_contracts: dev_sections[4],
3175            ..Default::default()
3176        };
3177        let holdout = AtlasRecallReport {
3178            mean_overall: val_overall,
3179            mean_architecture: val_sections[0],
3180            mean_entrypoints: val_sections[1],
3181            mean_behavior: val_sections[2],
3182            mean_state_authority: val_sections[3],
3183            mean_contracts: val_sections[4],
3184            ..Default::default()
3185        };
3186        HoldoutComparison {
3187            gap_architecture: 0.0,
3188            gap_entrypoints: 0.0,
3189            gap_behavior: 0.0,
3190            gap_state_authority: 0.0,
3191            gap_contracts: 0.0,
3192            gap_overall: 0.0,
3193            verdict: HoldoutVerdict::NoOverfit,
3194            results_file: String::new(),
3195            dev,
3196            holdout,
3197        }
3198    }
3199
3200    #[test]
3201    // trace:exempt reason=internal-detail
3202    fn compare_runs_ge_gate_and_section_guard() {
3203        let old = holdout_with(0.50, [0.5; 5], 0.50, [0.5; 5]);
3204        // new run: dev improved everywhere; validation improved overall but
3205        // contracts regressed 0.50 -> 0.30 (beyond the 0.05 guard)
3206        let mut new = holdout_with(0.55, [0.55; 5], 0.51, [0.55, 0.55, 0.55, 0.55, 0.30]);
3207        new.dev.mean_contracts = 0.56;
3208        let r = compare_runs(&old, &new, 0.0, 0.05);
3209        // GE = (0.51 - 0.50) / (0.55 - 0.50) = 0.2 > 0.0 -> passes
3210        assert!((r.generalization_efficiency - 0.2).abs() < 1e-9);
3211        assert!(r.ge_passed, "{:?}", r.failures);
3212        // contracts validation delta = 0.30 - 0.50 = -0.20 -> guard fails
3213        assert!((r.max_section_regression - 0.20).abs() < 1e-9);
3214        assert!(!r.guard_passed);
3215        assert!(!r.passed());
3216        assert!(r.failures.iter().any(|f| f.contains("contracts")));
3217        assert!((r.validation_deltas["contracts"] - (-0.20)).abs() < 1e-9);
3218
3219        // everything improved -> both gates pass
3220        let good = holdout_with(0.56, [0.56; 5], 0.53, [0.53; 5]);
3221        let r2 = compare_runs(&old, &good, 0.0, 0.05);
3222        assert!((r2.generalization_efficiency - 0.5).abs() < 1e-9);
3223        assert!(r2.ge_passed, "{:?}", r2.failures);
3224        assert!(r2.guard_passed, "{:?}", r2.failures);
3225        assert!(r2.passed());
3226        assert!(r2.failures.is_empty());
3227        assert_eq!(r2.max_section_regression, 0.0);
3228
3229        // a tight GE floor: ge 0.5 <= MIN 0.6 -> fails
3230        let r3 = compare_runs(&old, &good, 0.6, 0.05);
3231        assert!(!r3.ge_passed);
3232        assert!(r3.failures.iter().any(|f| f.contains("generalization efficiency")));
3233    }
3234
3235// trace:exempt reason=internal-detail
3236
3237    #[test]
3238// trace:exempt reason=internal-detail
3239    fn blind_manifest_hash_roundtrip_detects_change() {
3240        let tmp = tempfile::TempDir::new().unwrap();
3241        let root = tmp.path();
3242        let gt = root.join("benchmarks/blind-test-ground-truth");
3243        let blind = root.join("benchmarks/blind-test");
3244        std::fs::create_dir_all(&gt).unwrap();
3245        std::fs::create_dir_all(&blind).unwrap();
3246        std::fs::write(gt.join("axum.md"), "## architecture\n- axum\n").unwrap();
3247        std::fs::write(gt.join("echo.md"), "## architecture\n- echo\n").unwrap();
3248        std::fs::write(blind.join("README.md"), "# manifest\n| axum | url |\n").unwrap();
3249        std::fs::create_dir_all(blind.join("axum")).unwrap();
3250        std::fs::create_dir_all(blind.join("echo")).unwrap();
3251        std::fs::write(
3252            root.join("benchmarks/blind-lock.json"),
3253            r#"{"blind-test": {"axum": {"url": "https://github.com/tokio-rs/axum", "commit": "aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"}, "echo": {"url": "https://github.com/labstack/echo", "commit": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"}}}"#,
3254        )
3255        .unwrap();
3256
3257        let m1 = blind_manifest(root).unwrap();
3258        assert_eq!(m1.ground_truth_files, 2);
3259        assert_eq!(m1.repos, 2);
3260        assert_eq!(m1.sha256.len(), 64);
3261        assert!(m1.text.contains("ground-truth axum.md"), "{}", m1.text);
3262        assert!(m1.text.contains("clone axum"), "{}", m1.text);
3263        // Wave 13: the manifest carries the pinned commits from the lock
3264        assert!(
3265            m1.text.contains("lock axum aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa"),
3266            "{}",
3267            m1.text
3268        );
3269        assert!(
3270            m1.text.contains("lock echo bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"),
3271            "{}",
3272            m1.text
3273        );
3274        // deterministic roundtrip
3275        let m2 = blind_manifest(root).unwrap();
3276        assert_eq!(m1.sha256, m2.sha256);
3277        // tamper with a ground-truth key -> hash changes
3278        std::fs::write(gt.join("axum.md"), "## architecture\n- axum-changed\n").unwrap();
3279        assert_ne!(m1.sha256, blind_manifest(root).unwrap().sha256);
3280        // restore, then tamper with the clone list (drop a repo dir)
3281        std::fs::write(gt.join("axum.md"), "## architecture\n- axum\n").unwrap();
3282        std::fs::remove_dir_all(blind.join("echo")).unwrap();
3283        let m4 = blind_manifest(root).unwrap();
3284        assert_ne!(m1.sha256, m4.sha256);
3285        assert_eq!(m4.repos, 1);
3286        // re-pin one commit -> the digest covers the pinned commits
3287        std::fs::write(
3288            root.join("benchmarks/blind-lock.json"),
3289            r#"{"blind-test": {"axum": {"url": "https://github.com/tokio-rs/axum", "commit": "cccccccccccccccccccccccccccccccccccccccc"}, "echo": {"url": "https://github.com/labstack/echo", "commit": "bbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbbb"}}}"#,
3290        )
3291        .unwrap();
3292        assert_ne!(m1.sha256, blind_manifest(root).unwrap().sha256);
3293        // header roundtrip: the recorded hash is exactly what verification reads
3294        let header = format!("blind manifest sha256: {}\n", m1.sha256);
3295        assert_eq!(manifest_hash_from_header(&header).as_deref(), Some(m1.sha256.as_str()));
3296        assert_eq!(manifest_hash_from_header("no hash here"), None);
3297        // missing ground-truth dir is an error, not a silent empty manifest
3298        std::fs::remove_dir_all(&gt).unwrap();
3299        assert!(blind_manifest(root).is_err());
3300        // a missing lock is an error too — skipping it would silently
3301        // disable the commit-pin guard
3302        std::fs::create_dir_all(&gt).unwrap();
3303        std::fs::write(gt.join("axum.md"), "## architecture\n- axum\n").unwrap();
3304        std::fs::remove_file(root.join("benchmarks/blind-lock.json")).unwrap();
3305        let err = blind_manifest(root).unwrap_err();
3306        assert!(err.contains("blind-lock.json"), "{err}");
3307    }
3308
3309// trace:exempt reason=internal-detail
3310
3311    #[test]
3312// trace:exempt reason=internal-detail
3313    fn blind_head_lock_verifies_pinned_commit() {
3314        // Pure HEAD-vs-lock check (no git needed): a matching commit passes,
3315        // a mismatch is a hard error naming the repo, the actual commit, and
3316        // the pin.
3317        assert!(verify_head("axum", "abc123", "abc123").is_ok());
3318        let err = verify_head("axum", "abc123", "def456").unwrap_err();
3319        assert!(
3320            err.contains("blind-test repo axum at abc123, lock pins def456"),
3321            "{err}"
3322        );
3323    }
3324
3325    #[test]
3326    // trace:exempt reason=internal-detail
3327    fn blind_transfer_ratio_guards_zero_denominator() {
3328        assert_eq!(safe_ratio(1.0, 2.0), 0.5);
3329        assert_eq!(safe_ratio(0.0, 0.0), 0.0);
3330        assert_eq!(safe_ratio(5.0, 0.0), 0.0);
3331    }
3332}
3333
3334
3335