Skip to main content

scc_cli/
benchagent.rs

1//! Agent-run recorder and baseline harness (SCC-002, docs/TEST_PLAN.md §9).
2//!
3//! Runs each ground-truth task through an external agent command and records
4//! outcome metrics. When the agent emits a JSON event stream (`codex exec
5//! --json` emits JSONL: `{type:item.completed, item:{type:command_execution|
6//! mcp_tool_call|...}}`), the recorder additionally extracts tool-level
7//! exploration metrics: files opened, search/read tool calls, wrong-first
8//! locations opened before the first ground-truth file, and the wall time
9//! until the first ground-truth file is touched. When no JSON events are
10//! present the recorder falls back to the portable layer: wall time, exit
11//! status, output size, and per-task pass/fail localization.
12
13use std::collections::BTreeSet;
14use std::io::{BufRead, BufReader, Read};
15use std::path::{Path, PathBuf};
16use std::process::{Command, Stdio};
17use std::sync::OnceLock;
18use std::time::Instant;
19
20use scc_core::estimate_tokens;
21use serde_json::Value;
22
23use crate::benchctx::{locate_fixtures_dir, BenchmarkCorpus};
24
25#[derive(Debug, Clone, Default, serde::Serialize)]
26// trace:exempt reason=internal-detail  # agent-run recorder data, traced as part of impl.scc.bench.agent
27pub struct AgentTaskResult {
28    pub id: String,
29    pub exit_ok: bool,
30    pub duration_ms: u64,
31    pub output_bytes: usize,
32    /// ground-truth files mentioned in the agent's output (localization)
33    pub files_surfaced: usize,
34    pub files_total: usize,
35    // --- tool-level exploration metrics; 0 / None when the agent emits no
36    // JSON event stream (old-style output) ---
37    /// unique repo-relative file paths mentioned by tool events
38    pub files_opened: usize,
39    /// grep/glob/rg/find-style tool calls
40    pub search_tool_calls: usize,
41    /// read/cat/sed-style tool calls (including file reads)
42    pub read_tool_calls: usize,
43    /// all tool-like events (command executions + mcp tool calls + tool_use)
44    pub total_tool_calls: usize,
45    /// unique non-ground-truth files opened before the first ground-truth file was touched
46    pub wrong_first_locations: usize,
47    /// wall time until the first ground-truth file appears in a tool event or output
48    pub first_correct_ms: Option<u64>,
49    /// the agent's first plan (output before the first tool event) names a
50    /// ground-truth key (file or symbol)
51    pub first_plan_correct: bool,
52    /// knowledge-graph query tool calls (MCP tools named graph/gitnexus)
53    pub graph_tool_calls: usize,
54}
55
56#[derive(Debug, Clone, Default, serde::Serialize)]
57// trace:exempt reason=internal-detail  # agent-run recorder aggregate, traced as part of impl.scc.bench.agent
58pub struct AgentBenchSummary {
59    pub tasks: usize,
60    pub passed: usize,
61    pub mean_duration_ms: f64,
62    pub mean_localization: f64,
63    pub mean_files_opened: f64,
64    pub mean_search_tool_calls: f64,
65    pub mean_read_tool_calls: f64,
66    pub mean_wrong_first_locations: f64,
67    /// mean over tasks that had a first-correct observation (None if none did)
68    pub mean_first_correct_ms: Option<f64>,
69    /// knowledge-graph query tool calls per task (MCP tools named graph/gitnexus)
70    pub mean_graph_tool_calls: f64,
71    /// fraction of tasks whose first plan names a ground-truth key
72    pub mean_first_plan_correct: f64,
73    /// variant name for `scc bench external` runs (empty for `bench agent`)
74    #[serde(default)]
75    pub variant: String,
76    pub results: Vec<AgentTaskResult>,
77}
78
79/// A-vs-E agent-behavior gate result: does the atlas variant (E) reduce
80/// exploration vs the baseline (A)?
81// trace:v1 id=impl.crates-scc-cli-src-benchagent.agent-gate-result work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
82#[derive(Debug, Clone, serde::Serialize)]
83pub struct AgentGateResult {
84    pub baseline: AgentBenchSummary,
85    pub atlas: AgentBenchSummary,
86    /// E.mean_search_tool_calls < A.mean_search_tool_calls.
87    pub search_reduced: bool,
88    /// E.mean_files_opened <= A.mean_files_opened + 1.
89    pub files_bounded: bool,
90    /// E.mean_first_correct_ms <= A.mean_first_correct_ms. Fails closed:
91    /// when either side has no first-correct mean (no JSON event stream)
92    /// the clause cannot be verified and is false.
93    pub first_correct_bounded: bool,
94    pub passed: bool,
95}
96
97/// Evaluate the gate clauses from two summaries (A = baseline, E = atlas
98/// variant). The gate FAILS when the atlas variant does not reduce
99/// exploration: requires E.search_tool_calls < A.search_tool_calls AND
100/// E.files_opened <= A.files_opened + 1 AND E.first_correct_ms <=
101/// A.first_correct_ms (means).
102// trace:v1 id=impl.crates-scc-cli-src-benchagent.evaluate-gate work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
103pub fn evaluate_gate(a: &AgentBenchSummary, e: &AgentBenchSummary) -> AgentGateResult {
104    let search_reduced = e.mean_search_tool_calls < a.mean_search_tool_calls;
105    let files_bounded = e.mean_files_opened <= a.mean_files_opened + 1.0;
106    let first_correct_bounded = match (a.mean_first_correct_ms, e.mean_first_correct_ms) {
107        (Some(av), Some(ev)) => ev <= av,
108        _ => false, // fail closed: cannot verify
109    };
110    let passed = search_reduced && files_bounded && first_correct_bounded;
111    AgentGateResult {
112        baseline: a.clone(),
113        atlas: e.clone(),
114        search_reduced,
115        files_bounded,
116        first_correct_bounded,
117        passed,
118    }
119}
120
121/// Run the agent-behavior release gate: score the baseline (A) command and
122/// the atlas variant (E) command over the same corpus with the same
123/// harness, then evaluate the exploration-reduction clauses (see
124/// [`evaluate_gate`]). `min_files` applies to BOTH runs.
125// trace:v1 id=impl.crates-scc-cli-src-benchagent.run-agent-gate work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
126pub fn run_agent_gate(
127    baseline_cmd: &str,
128    atlas_cmd: &str,
129    min_files: f64,
130) -> Result<AgentGateResult, String> {
131    let baseline = run_agent_benchmark(baseline_cmd, min_files)?;
132    let atlas = run_agent_benchmark(atlas_cmd, min_files)?;
133    Ok(evaluate_gate(&baseline, &atlas))
134}
135
136// trace:v1 id=impl.crates-scc-cli-src-benchagent.print-agent-gate work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
137pub fn print_agent_gate(g: &AgentGateResult) {
138    println!("scc bench agent --gate — A (baseline) vs E (atlas variant)");
139    println!("\n--- baseline (A) ---");
140    print_agent_summary(&g.baseline);
141    println!("\n--- atlas variant (E) ---");
142    print_agent_summary(&g.atlas);
143    let a = &g.baseline;
144    let e = &g.atlas;
145    println!("\n=== exploration clauses (means) ===");
146    println!(
147        "  searches:      E {:.3} < A {:.3}              -> {}",
148        e.mean_search_tool_calls,
149        a.mean_search_tool_calls,
150        if g.search_reduced { "PASS" } else { "FAIL" }
151    );
152    println!(
153        "  files opened:  E {:.3} <= A {:.3} + 1        -> {}",
154        e.mean_files_opened,
155        a.mean_files_opened,
156        if g.files_bounded { "PASS" } else { "FAIL" }
157    );
158    let first = match (a.mean_first_correct_ms, e.mean_first_correct_ms) {
159        (Some(av), Some(ev)) => format!("E {ev:.0} ms <= A {av:.0} ms"),
160        _ => "not verifiable (missing first-correct mean)".to_string(),
161    };
162    println!(
163        "  first-correct: {first} -> {}",
164        if g.first_correct_bounded { "PASS" } else { "FAIL" }
165    );
166    println!(
167        "  gate: {} (atlas variant must reduce exploration: all three clauses)",
168        if g.passed { "PASS" } else { "FAIL" }
169    );
170}
171
172/// One ground-truth task definition used by the variant runners
173/// ([`run_variant_tasks`]). The `files` are the localization ground truth
174/// (the benchagent protocol compares tool events against them); `plan_keys`
175/// are the first-plan correctness keys (files + symbols).
176#[derive(Debug, Clone)]
177// trace:exempt reason=internal-detail  # data container of the variant runner (impl.scc.bench.variant)
178pub struct VariantTask {
179    pub id: String,
180    pub repo: String,
181    pub goal: String,
182    pub files: Vec<String>,
183    pub plan_keys: Vec<String>,
184}
185
186/// Load the `benchmarks/tasks.json` corpus as [`VariantTask`]s (plan keys =
187/// ground-truth files, matching the original `bench agent` protocol).
188// trace:exempt reason=internal-detail  # corpus loader of the variant runner (impl.scc.bench.variant)
189fn load_corpus_tasks() -> Result<Vec<VariantTask>, String> {
190    let fixtures = locate_fixtures_dir().ok_or("cannot locate fixtures/ directory")?;
191    let corpus_path = fixtures
192        .parent()
193        .map(|p| p.join("benchmarks/tasks.json"))
194        .or_else(|| {
195            PathBuf::from(env!("CARGO_MANIFEST_DIR"))
196                .parent()
197                .and_then(|p| p.parent())
198                .map(|p| p.join("benchmarks/tasks.json"))
199        })
200        .ok_or("cannot locate benchmarks/tasks.json")?;
201    let text = std::fs::read_to_string(&corpus_path).map_err(|e| e.to_string())?;
202    let corpus: BenchmarkCorpus = serde_json::from_str(&text).map_err(|e| e.to_string())?;
203    Ok(corpus
204        .tasks
205        .iter()
206        .map(|t| VariantTask {
207            id: t.id.clone(),
208            repo: t.repo.clone(),
209            goal: t.goal.clone(),
210            files: t.ground_truth.files.clone(),
211            plan_keys: t.ground_truth.files.clone(),
212        })
213        .collect())
214}
215
216/// Paid-model safety interlock: paid benchmark entrypoints refuse execution
217/// by default and spend no model quota unless the operator opted in with
218/// `SCC_ALLOW_PAID_BENCHMARKS=1`. The pure predicate is unit-testable
219/// without touching the process environment.
220// trace:v1 id=impl.scc.bench.paid-gate work=WORK-SCC-002 verifies=REQ-SCC-TEST
221pub fn paid_opt_in_allowed(env_val: Option<&str>) -> bool {
222    env_val == Some("1")
223}
224
225/// Read the process environment for the paid-benchmark opt-in.
226// trace:v1 id=impl.scc.bench.paid-gate-require work=WORK-SCC-002 verifies=REQ-SCC-TEST
227pub fn require_paid_opt_in() -> Result<(), String> {
228    if paid_opt_in_allowed(std::env::var("SCC_ALLOW_PAID_BENCHMARKS").ok().as_deref()) {
229        Ok(())
230    } else {
231        Err("refusing: this benchmark launches paid coding agents (codex/claude); paid model benchmarks are disabled by default and spend real API quota. Set SCC_ALLOW_PAID_BENCHMARKS=1 to opt in explicitly.".to_string())
232    }
233}
234#[cfg(test)]
235/// Serializes tests that mutate `SCC_ALLOW_PAID_BENCHMARKS` (Rust runs
236/// tests in parallel threads sharing one process environment).
237pub(crate) static PAID_TEST_ENV_LOCK: parking_lot::Mutex<()> = parking_lot::Mutex::new(());
238
239#[cfg(test)]
240/// Run `f` with the paid-benchmark opt-in set, restoring the previous
241/// value afterwards. Mock-agent tests use this: mocks never spend quota,
242/// but they exercise the same gated entrypoints as paid runs.
243// trace:v1 id=impl.crates-scc-cli-src-benchagent.with-paid-opt-in work=WORK-SCC-002
244pub(crate) fn with_paid_opt_in<T>(f: impl FnOnce() -> T) -> T {
245    let _guard = PAID_TEST_ENV_LOCK.lock();
246    let prev = std::env::var("SCC_ALLOW_PAID_BENCHMARKS").ok();
247    std::env::set_var("SCC_ALLOW_PAID_BENCHMARKS", "1");
248    let out = f();
249    match prev {
250        Some(v) => std::env::set_var("SCC_ALLOW_PAID_BENCHMARKS", v),
251        None => std::env::remove_var("SCC_ALLOW_PAID_BENCHMARKS"),
252    }
253    out
254}
255
256/// `scc bench agent --cmd "<command>"` — the command receives the task goal
257/// via the `SCC_GOAL` env var and the repo path as its working directory
258/// (like `claude -p "$SCC_GOAL"` or `codex exec -- "$SCC_GOAL"`).
259// trace:v1 id=impl.scc.bench.agent work=WORK-SCC-002 verifies=REQ-SCC-TEST
260pub fn run_agent_benchmark(cmd: &str, min_files: f64) -> Result<AgentBenchSummary, String> {
261    require_paid_opt_in()?;
262    let tasks = load_corpus_tasks()?;
263    run_variant_tasks("", &tasks, |_, _| Ok(cmd.to_string()), min_files)
264}
265
266/// Wave-15 variant benchmark: same protocol as [`run_agent_benchmark`] over
267/// the full `benchmarks/tasks.json` corpus, recording the variant name in
268/// the summary. `cmd` is a shell command; the variant context artifact is
269/// expected to be generated inside it (see benchmarks/run_agent_bench.sh).
270// trace:v1 id=impl.scc.bench.variant work=WORK-SCC-002 verifies=REQ-SCC-TEST
271pub fn run_variant_benchmark(
272    variant: &str,
273    cmd: &str,
274    min_files: f64,
275) -> Result<AgentBenchSummary, String> {
276    require_paid_opt_in()?;
277    let tasks = load_corpus_tasks()?;
278    run_variant_tasks(variant, &tasks, |_, _| Ok(cmd.to_string()), min_files)
279}
280
281/// The shared variant runner: for each task, copy the fixture repo, index
282/// it, ask `cmd_for` for the per-task shell command (this is where a
283/// variant generates its context artifact into the freshly indexed repo and
284/// returns the agent command that consumes it), then run the benchagent
285/// protocol. Aggregates the same metrics as `bench agent` plus
286/// first-plan accuracy and graph-query counts, and records `variant`.
287// trace:exempt reason=internal-detail  # shared runner internals; variant entry point is impl.scc.bench.variant
288pub fn run_variant_tasks<F>(
289    variant: &str,
290    tasks: &[VariantTask],
291    mut cmd_for: F,
292    min_files: f64,
293) -> Result<AgentBenchSummary, String>
294where
295    F: FnMut(&VariantTask, &Path) -> Result<String, String>,
296{
297    let fixtures = locate_fixtures_dir().ok_or("cannot locate fixtures/ directory")?;
298
299    let mut summary = AgentBenchSummary {
300        variant: variant.to_string(),
301        tasks: tasks.len(),
302        ..Default::default()
303    };
304    for task in tasks {
305        let repo_dir = fixtures.join(&task.repo);
306        let tmp = tempfile::TempDir::new().map_err(|e| e.to_string())?;
307        let root = tmp.path().join("repo");
308        copy_fixture_tree(&repo_dir, &root);
309        // index first so the agent starts warm (matches the SCC baseline flow)
310        crate::commands::cmd_index(&root, true).map_err(|e| e.to_string())?;
311
312        let cmd = cmd_for(task, &root)?;
313        let res = run_task(&cmd, &root, &task.id, &task.goal, &task.files, &task.plan_keys)?;
314        summary.mean_localization +=
315            res.files_surfaced as f64 / task.files.len().max(1) as f64;
316        summary.mean_duration_ms += res.duration_ms as f64;
317        summary.mean_files_opened += res.files_opened as f64;
318        summary.mean_search_tool_calls += res.search_tool_calls as f64;
319        summary.mean_read_tool_calls += res.read_tool_calls as f64;
320        summary.mean_wrong_first_locations += res.wrong_first_locations as f64;
321        summary.mean_graph_tool_calls += res.graph_tool_calls as f64;
322        summary.mean_first_plan_correct += res.first_plan_correct as usize as f64;
323        if res.exit_ok {
324            summary.passed += 1;
325        }
326        summary.results.push(res);
327    }
328    let n = tasks.len() as f64;
329    summary.mean_duration_ms /= n;
330    summary.mean_localization /= n;
331    summary.mean_files_opened /= n;
332    summary.mean_search_tool_calls /= n;
333    summary.mean_read_tool_calls /= n;
334    summary.mean_wrong_first_locations /= n;
335    summary.mean_graph_tool_calls /= n;
336    summary.mean_first_plan_correct /= n;
337    let with_first: Vec<u64> = summary
338        .results
339        .iter()
340        .filter_map(|r| r.first_correct_ms)
341        .collect();
342    summary.mean_first_correct_ms = if with_first.is_empty() {
343        None
344    } else {
345        Some(with_first.iter().sum::<u64>() as f64 / with_first.len() as f64)
346    };
347
348    if summary.mean_localization < min_files {
349        return Err(format!(
350            "agent benchmark gate failed: mean localization {:.3} < {min_files}",
351            summary.mean_localization
352        ));
353    }
354    Ok(summary)
355}
356
357/// Run one task through `sh -c <cmd>` while streaming stdout line by line:
358/// each line's arrival time is recorded (for first-correct timing) and lines
359/// that look like JSONL events are parsed for tool-level metrics.
360// trace:exempt reason=internal-detail  # agent-run recorder internals (impl.scc.bench.agent)
361fn run_task(
362    cmd: &str,
363    root: &Path,
364    id: &str,
365    goal: &str,
366    gt_files: &[String],
367    plan_keys: &[String],
368) -> Result<AgentTaskResult, String> {
369    let started = Instant::now();
370    let mut child = Command::new("sh")
371        .arg("-c")
372        .arg(cmd)
373        .env("SCC_GOAL", goal)
374        .current_dir(root)
375        .stdout(Stdio::piped())
376        .stderr(Stdio::piped())
377        .spawn()
378        .map_err(|e| format!("spawn agent command: {e}"))?;
379    let stdout = child.stdout.take().expect("stdout piped");
380    let stderr = child.stderr.take().expect("stderr piped");
381    let err_thread = std::thread::spawn(move || {
382        let mut buf = Vec::new();
383        let mut r = BufReader::new(stderr);
384        let _ = r.read_to_end(&mut buf);
385        buf
386    });
387
388    let mut out_buf = Vec::new();
389    let mut events: Vec<AgentEvent> = Vec::new();
390    let mut first_correct_ms: Option<u64> = None;
391    let mut gt_seen = false;
392    let mut wrong_seen: BTreeSet<String> = BTreeSet::new();
393    // The agent's first plan = the output stream before the first tool
394    // event; first-plan correctness compares that text against the
395    // ground-truth keys (files + symbols).
396    let mut plan_buf = String::new();
397    let mut plan_done = false;
398
399    let reader = BufReader::new(stdout);
400    for line in reader.split(b'\n') {
401        let line = line.map_err(|e| format!("read agent stdout: {e}"))?;
402        let line = String::from_utf8_lossy(&line);
403        let elapsed_ms = started.elapsed().as_millis() as u64;
404        out_buf.extend_from_slice(line.as_bytes());
405        out_buf.push(b'\n');
406        // first ground-truth file touched — in a tool event or any output
407        if !gt_seen && gt_files.iter().any(|f| line.contains(f.as_str())) {
408            gt_seen = true;
409            if first_correct_ms.is_none() {
410                first_correct_ms = Some(elapsed_ms);
411            }
412        }
413        if !plan_done {
414            plan_buf.push_str(&line);
415            plan_buf.push('\n');
416        }
417        if let Some(ev) = parse_event_line(&line, root) {
418            plan_done = true;
419            // wrong-first: non-ground-truth files opened before the first
420            // ground-truth file was touched (unique, in stream order)
421            if !gt_seen {
422                for p in &ev.paths {
423                    if !gt_files.iter().any(|f| f == p) {
424                        wrong_seen.insert(p.clone());
425                    }
426                }
427            }
428            events.push(ev);
429        }
430    }
431    let status = child
432        .wait()
433        .map_err(|e| format!("wait agent command: {e}"))?;
434    let duration_ms = started.elapsed().as_millis() as u64;
435    let stderr_bytes = err_thread
436        .join()
437        .map_err(|_| "join stderr thread".to_string())?;
438    out_buf.extend_from_slice(&stderr_bytes);
439
440    let output = String::from_utf8_lossy(&out_buf);
441    let files_surfaced = gt_files
442        .iter()
443        .filter(|f| output.contains(f.as_str()))
444        .count();
445    let first_plan_correct = plan_keys.iter().any(|k| plan_buf.contains(k.as_str()));
446    let mut opened: BTreeSet<String> = BTreeSet::new();
447    for ev in &events {
448        opened.extend(ev.paths.iter().cloned());
449    }
450    Ok(AgentTaskResult {
451        id: id.to_string(),
452        exit_ok: status.success(),
453        duration_ms,
454        output_bytes: output.len(),
455        files_surfaced,
456        files_total: gt_files.len(),
457        files_opened: opened.len(),
458        search_tool_calls: events.iter().filter(|e| e.kind == ToolKind::Search).count(),
459        read_tool_calls: events.iter().filter(|e| e.kind == ToolKind::Read).count(),
460        total_tool_calls: events.len(),
461        wrong_first_locations: wrong_seen.len(),
462        first_correct_ms,
463        first_plan_correct,
464        graph_tool_calls: events.iter().filter(|e| e.graph).count(),
465    })
466}
467
468/// Score an in-process JSONL tool stream the same way [`run_task`] scores an
469/// agent. Event index is used as `first_correct_ms` so deterministic
470/// pack-consumers have a stable ordinal (not wall-clock noise).
471// trace:v1 id=impl.scc.bench.jsonl-metrics work=WORK-phase-7-of-scc-x-ripwire-lessons-1-one-hop-type-narrowing-from-unique satisfies=REQ-implement-phase-7-of-scc-x-ripwire-lessons-1-one-hop-type-narrowing
472pub fn metrics_from_jsonl(
473    jsonl: &str,
474    root: &Path,
475    id: &str,
476    gt_files: &[String],
477) -> AgentTaskResult {
478    let mut events: Vec<AgentEvent> = Vec::new();
479    let mut first_correct_ms: Option<u64> = None;
480    let mut gt_seen = false;
481    let mut wrong_seen: BTreeSet<String> = BTreeSet::new();
482    let mut plan_buf = String::new();
483    let mut plan_done = false;
484    for (i, line) in jsonl.lines().enumerate() {
485        let elapsed_ms = i as u64;
486        if !gt_seen && gt_files.iter().any(|f| line.contains(f.as_str())) {
487            gt_seen = true;
488            if first_correct_ms.is_none() {
489                first_correct_ms = Some(elapsed_ms);
490            }
491        }
492        if !plan_done {
493            plan_buf.push_str(line);
494            plan_buf.push('\n');
495        }
496        if let Some(ev) = parse_event_line(line, root) {
497            plan_done = true;
498            if !gt_seen {
499                for p in &ev.paths {
500                    if !gt_files.iter().any(|f| f == p) {
501                        wrong_seen.insert(p.clone());
502                    }
503                }
504            }
505            events.push(ev);
506        }
507    }
508    let files_surfaced = gt_files
509        .iter()
510        .filter(|f| jsonl.contains(f.as_str()))
511        .count();
512    let mut opened: BTreeSet<String> = BTreeSet::new();
513    for ev in &events {
514        opened.extend(ev.paths.iter().cloned());
515    }
516    AgentTaskResult {
517        id: id.to_string(),
518        exit_ok: true,
519        duration_ms: jsonl.lines().count() as u64,
520        output_bytes: jsonl.len(),
521        files_surfaced,
522        files_total: gt_files.len(),
523        files_opened: opened.len(),
524        search_tool_calls: events.iter().filter(|e| e.kind == ToolKind::Search).count(),
525        read_tool_calls: events.iter().filter(|e| e.kind == ToolKind::Read).count(),
526        total_tool_calls: events.len(),
527        wrong_first_locations: wrong_seen.len(),
528        first_correct_ms,
529        first_plan_correct: gt_files.iter().any(|k| plan_buf.contains(k.as_str())),
530        graph_tool_calls: events.iter().filter(|e| e.graph).count(),
531    }
532}
533
534#[derive(Debug, Clone, Copy, PartialEq)]
535// trace:v1 id=impl.crates-scc-cli-src-benchagent.tool-kind work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
536enum ToolKind {
537    Search,
538    Read,
539    Other,
540}
541
542#[derive(Debug, Clone)]
543// trace:exempt reason=internal-detail  # event-model internals (impl.scc.bench.agent)
544struct AgentEvent {
545    paths: Vec<String>,
546    kind: ToolKind,
547    /// knowledge-graph query tool (mcp tool / tool_use named graph/gitnexus)
548    graph: bool,
549}
550
551/// Tolerant JSONL event parser. Recognizes the codex `--json` shape
552/// (`{type:"item.completed", item:{type:"command_execution"|"mcp_tool_call"}}`)
553/// and the generic `{type:"tool_use"|"tool_result"}` shapes. Returns None for
554/// non-JSON lines, in-progress events, and non-tool events (agent messages,
555/// errors, turn markers).
556// trace:exempt reason=internal-detail  # event parser internals (impl.scc.bench.agent)
557fn parse_event_line(line: &str, root: &Path) -> Option<AgentEvent> {
558    let trimmed = line.trim();
559    if !trimmed.starts_with('{') {
560        return None;
561    }
562    let value: Value = serde_json::from_str(trimmed).ok()?;
563    let item = match value.get("type").and_then(|v| v.as_str()) {
564        Some("item.completed") => value.get("item")?,
565        Some("item.started") => return None,
566        _ => &value,
567    };
568    let obj = item.as_object()?;
569    match obj.get("type").and_then(|v| v.as_str())? {
570        "command_execution" => {
571            let command = obj.get("command").and_then(|v| v.as_str()).unwrap_or("");
572            let tool = tool_from_command(command);
573            Some(AgentEvent {
574                paths: paths_from_command(command, root),
575                kind: kind_of(tool),
576                graph: false,
577            })
578        }
579        "mcp_tool_call" => {
580            let tool = obj.get("tool").and_then(|v| v.as_str()).unwrap_or("mcp_tool_call");
581            let paths = obj
582                .get("arguments")
583                .and_then(Value::as_object)
584                .map(|a| paths_from_args(a, root))
585                .unwrap_or_default();
586            Some(AgentEvent {
587                paths,
588                kind: kind_of(tool),
589                graph: is_graph_tool(tool),
590            })
591        }
592        "tool_use" => {
593            let tool = obj.get("name").and_then(|v| v.as_str()).unwrap_or("tool_use");
594            let paths = obj
595                .get("input")
596                .and_then(Value::as_object)
597                .map(|a| paths_from_args(a, root))
598                .unwrap_or_default();
599            Some(AgentEvent {
600                paths,
601                kind: kind_of(tool),
602                graph: is_graph_tool(tool),
603            })
604        }
605        // tool_result is a response, not a call; messages/errors carry no
606        // tool intent
607        _ => None,
608    }
609}
610
611/// First meaningful token of a shell command: unwraps `/bin/zsh -lc "tool …"`
612/// style wrappers and flag prefixes.
613// trace:v1 id=impl.crates-scc-cli-src-benchagent.tool-from-command work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
614fn tool_from_command(command: &str) -> &str {
615    let mut toks = command.split_whitespace();
616    let first = toks.next().unwrap_or("");
617    let first_trim = first.trim_matches(['\'', '"']);
618    let is_shell = (first_trim.starts_with('/') && first_trim.ends_with("sh"))
619        || first_trim == "zsh"
620        || first_trim == "bash"
621        || first_trim == "sh";
622    if is_shell {
623        for t in toks {
624            let t = t.trim_matches(['\'', '"']);
625            if !t.is_empty() && !t.starts_with('-') {
626                return t;
627            }
628        }
629        return "shell";
630    }
631    first_trim
632}
633
634/// File paths mentioned in a shell command: whitespace tokens that resolve
635/// inside the repo or carry a source-file extension. Flags, globs, env refs,
636/// and out-of-repo absolute paths are dropped.
637// trace:v1 id=impl.crates-scc-cli-src-benchagent.paths-from-command work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
638fn paths_from_command(command: &str, root: &Path) -> Vec<String> {
639    let mut out = Vec::new();
640    for raw in command.split_whitespace() {
641        let tok = raw.trim_matches(['\'', '"']);
642        if tok.is_empty()
643            || tok == "."
644            || tok == ".."
645            || tok.starts_with('-')
646            || tok.contains(['$', '|', '&', ';', '<', '>', '*', '?', '`', '='])
647        {
648            continue;
649        }
650        if let Some(p) = normalize_path(tok, root) {
651            out.push(p);
652        }
653    }
654    out
655}
656
657/// File paths from an MCP tool_use `arguments` / `input` object.
658// trace:v1 id=impl.crates-scc-cli-src-benchagent.paths-from-args work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
659fn paths_from_args(args: &serde_json::Map<String, Value>, root: &Path) -> Vec<String> {
660    let mut out = Vec::new();
661    for key in ["file_path", "path", "file", "filename"] {
662        if let Some(Value::String(s)) = args.get(key) {
663            if let Some(p) = normalize_path(s, root) {
664                out.push(p);
665            }
666        }
667    }
668    if let Some(Value::Array(arr)) = args.get("paths") {
669        for v in arr {
670            if let Value::String(s) = v {
671                if let Some(p) = normalize_path(s, root) {
672                    out.push(p);
673                }
674            }
675        }
676    }
677    out
678}
679
680/// Resolve a token to a repo-relative path when it exists under the repo
681/// root or looks like a source file. Absolute paths outside the repo are
682/// dropped (skill docs, system files are not repo locations).
683// trace:v1 id=impl.crates-scc-cli-src-benchagent.normalize-path work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
684fn normalize_path(tok: &str, root: &Path) -> Option<String> {
685    let tok = tok.trim_end_matches('/');
686    if tok.is_empty() {
687        return None;
688    }
689    let abs = if tok.starts_with('/') {
690        PathBuf::from(tok)
691    } else {
692        root.join(tok)
693    };
694    let rel = abs.strip_prefix(root).ok()?;
695    let rel_str = rel.to_string_lossy().replace('\\', "/");
696    if rel_str.is_empty() {
697        return None;
698    }
699    let exists = abs.exists();
700    let looks_like_file = SOURCE_EXTS.iter().any(|e| rel_str.ends_with(e));
701    if exists || looks_like_file {
702        Some(rel_str.trim_start_matches("./").to_string())
703    } else {
704        None
705    }
706}
707
708const SOURCE_EXTS: [&str; 40] = [
709    ".py", ".pyi", ".ts", ".tsx", ".js", ".jsx", ".mjs", ".cjs", ".rs", ".go", ".java", ".kt",
710    ".kts", ".rb", ".php", ".c", ".h", ".cpp", ".hpp", ".cc", ".cs", ".dart", ".proto", ".txt",
711    ".json", ".toml", ".yaml", ".yml", ".md", ".sh", ".sql", ".html", ".css", ".vue", ".svelte",
712    ".swift", ".lua", ".xml", ".gradle", ".dockerfile",
713];
714
715// trace:v1 id=impl.crates-scc-cli-src-benchagent.kind-of work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
716fn kind_of(tool: &str) -> ToolKind {
717    let t = tool.to_ascii_lowercase();
718    if t == "rg" || t == "ag" || t == "ack" || t == "fd" || t == "find" || t.contains("grep")
719        || t.contains("glob") || t.contains("search")
720    {
721        ToolKind::Search
722    } else if t == "cat" || t == "sed" || t == "less" || t == "more" || t == "head" || t == "tail"
723        || t == "wc" || t == "nl" || t == "open" || t.contains("read") || t.contains("view")
724    {
725        ToolKind::Read
726    } else {
727        ToolKind::Other
728    }
729}
730
731/// A knowledge-graph query tool (MCP graph servers, gitnexus, codegraph).
732// trace:exempt reason=internal-detail  # graph-query classifier feeding variant metrics (impl.scc.bench.variant)
733fn is_graph_tool(tool: &str) -> bool {
734    let t = tool.to_ascii_lowercase();
735    t.contains("graph") || t.contains("gitnexus")
736}
737
738// trace:v1 id=impl.crates-scc-cli-src-benchagent.copy-fixture-tree work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
739fn copy_fixture_tree(src: &Path, dst: &Path) {
740    std::fs::create_dir_all(dst).unwrap();
741    for entry in std::fs::read_dir(src).unwrap() {
742        let entry = entry.unwrap();
743        let name = entry.file_name();
744        if name == ".scc" {
745            continue;
746        }
747        let from = entry.path();
748        let to = dst.join(&name);
749        if from.is_dir() {
750            std::fs::create_dir_all(&to).unwrap();
751            copy_fixture_tree(&from, &to);
752        } else {
753            std::fs::copy(&from, &to).unwrap();
754        }
755    }
756}
757
758// trace:exempt reason=internal-detail  # summary printer internals (impl.scc.bench.agent)
759pub fn print_agent_summary(s: &AgentBenchSummary) {
760    if s.variant.is_empty() {
761        println!("scc bench agent — ground-truth corpus through an external agent command");
762    } else {
763        println!("scc bench external — variant {variant}", variant = s.variant);
764    }
765    println!(
766        "  tasks: {}   exit-ok: {}/{}   mean duration: {:.0} ms   mean localization: {:.3}",
767        s.tasks, s.passed, s.tasks, s.mean_duration_ms, s.mean_localization
768    );
769    println!(
770        "  exploration means: files opened {:.1}   searches {:.1}   reads {:.1}   graph {:.1}   wrong-first {:.1}   first-correct {}",
771        s.mean_files_opened,
772        s.mean_search_tool_calls,
773        s.mean_read_tool_calls,
774        s.mean_graph_tool_calls,
775        s.mean_wrong_first_locations,
776        match s.mean_first_correct_ms {
777            Some(v) => format!("{v:.0} ms"),
778            None => "— (no JSON event stream)".to_string(),
779        }
780    );
781    println!(
782        "  first-plan accuracy: {:.3}",
783        s.mean_first_plan_correct
784    );
785    for r in &s.results {
786        let first = match r.first_correct_ms {
787            Some(v) => format!("{v:>8} ms"),
788            None => "       —".to_string(),
789        };
790        println!(
791            "  {:<42} {} {:>8} ms {:>8} B  files {}/{}  opened {:>2}  srh {:>2}  read {:>2}  tot {:>3}  wrng {:>2}  1st-correct {}",
792            r.id,
793            if r.exit_ok { "ok  " } else { "FAIL" },
794            r.duration_ms,
795            r.output_bytes,
796            r.files_surfaced,
797            r.files_total,
798            r.files_opened,
799            r.search_tool_calls,
800            r.read_tool_calls,
801            r.total_tool_calls,
802            r.wrong_first_locations,
803            first,
804        );
805    }
806    if s.mean_first_correct_ms.is_none() {
807        println!(
808            "  (no JSONL events detected — agent did not emit a --json event stream; tool columns are 0)"
809        );
810    }
811}
812
813// ---------------------------------------------------------------------------
814// Wave-15 PPR ablation matrix + leakage-free structural selection
815// ---------------------------------------------------------------------------
816
817/// Harness-level surface ranking modes for the PPR ablation matrix. Every
818/// mode runs the PRODUCTION surface pipeline ([`render_ablation_surface`]
819/// → `build_surface_staged`) with exactly one stage toggled — never a
820/// harness reimplementation of ranking. `lexical` switches every stage
821/// off (lexical-only); `global-ppr` disables task PPR; `task-ppr` is the
822/// full task pipeline; `ppr-mmr`/`ppr-quotas`/`ppr-optimizer` disable the
823/// MMR/quotas/optimizer tail stages respectively. The ablation rows are
824/// the production rows with one stage removed. See
825/// benchmarks/external/README.md.
826#[derive(Debug, Clone, Copy, PartialEq, Eq)]
827// trace:exempt reason=internal-detail  # harness-level ablation mode flags (impl.scc.bench.ablation)
828pub enum SurfaceAblation {
829    /// Lexical goal match only — every pipeline stage off.
830    Lexical,
831    /// Global PPR + lexical (no task seeds): task-PPR stage off.
832    GlobalPpr,
833    /// The production task surface pipeline (every stage on).
834    TaskPpr,
835    /// Production pipeline with the MMR diversity stage off.
836    PprMmr,
837    /// Production pipeline with the token-aware quota stage off.
838    PprQuotas,
839    /// Production pipeline with the value/token budget optimizer off.
840    PprOptimizer,
841}
842
843// trace:exempt reason=internal-detail  # harness-level ablation mode flags (impl.scc.bench.ablation)
844impl SurfaceAblation {
845    /// Parse a variant name; `None` for non-ablation variants.
846    // trace:exempt reason=internal-detail  # harness-level ablation mode flags (impl.scc.bench.ablation)
847    pub fn parse(variant: &str) -> Option<SurfaceAblation> {
848        match variant {
849            "lexical" => Some(SurfaceAblation::Lexical),
850            "global-ppr" => Some(SurfaceAblation::GlobalPpr),
851            "task-ppr" => Some(SurfaceAblation::TaskPpr),
852            "ppr-mmr" => Some(SurfaceAblation::PprMmr),
853            "ppr-quotas" => Some(SurfaceAblation::PprQuotas),
854            "ppr-optimizer" => Some(SurfaceAblation::PprOptimizer),
855            _ => None,
856        }
857    }
858
859    // trace:exempt reason=internal-detail  # harness-level ablation mode flags (impl.scc.bench.ablation)
860    pub fn as_str(&self) -> &'static str {
861        match self {
862            SurfaceAblation::Lexical => "lexical",
863            SurfaceAblation::GlobalPpr => "global-ppr",
864            SurfaceAblation::TaskPpr => "task-ppr",
865            SurfaceAblation::PprMmr => "ppr-mmr",
866            SurfaceAblation::PprQuotas => "ppr-quotas",
867            SurfaceAblation::PprOptimizer => "ppr-optimizer",
868        }
869    }
870
871    /// The pipeline-stage toggle for this ablation: each mode disables
872    /// exactly one stage of the production pipeline (`lexical` disables
873    /// everything — lexical-only); every other stage runs exactly as
874    /// production runs it.
875    // trace:exempt reason=internal-detail  # harness-level ablation mode flags (impl.scc.bench.ablation)
876    pub fn stages(&self) -> scc_context::surface::SurfacePipelineStages {
877        use scc_context::surface::SurfacePipelineStages;
878        let all_on = SurfacePipelineStages {
879            lexical: true,
880            global_ppr: true,
881            task_ppr: true,
882            mmr: true,
883            quotas: true,
884            optimizer: true,
885        };
886        match self {
887            SurfaceAblation::Lexical => SurfacePipelineStages {
888                lexical: false,
889                global_ppr: false,
890                task_ppr: false,
891                mmr: false,
892                quotas: false,
893                optimizer: false,
894            },
895            SurfaceAblation::GlobalPpr => {
896                let mut s = all_on;
897                s.task_ppr = false;
898                s
899            }
900            SurfaceAblation::TaskPpr => all_on,
901            SurfaceAblation::PprMmr => {
902                let mut s = all_on;
903                s.mmr = false;
904                s
905            }
906            SurfaceAblation::PprQuotas => {
907                let mut s = all_on;
908                s.quotas = false;
909                s
910            }
911            SurfaceAblation::PprOptimizer => {
912                let mut s = all_on;
913                s.optimizer = false;
914                s
915            }
916        }
917    }
918}
919
920/// Render one ablation-mode task surface for `goal` under `budget_tokens`
921/// (the chars/4 estimate) through the PRODUCTION surface pipeline
922/// (`build_surface_staged`) with the mode's stage toggle — the ablation
923/// rows are the production rows with one stage removed, never a harness
924/// reimplementation of ranking. A one-line mode label prefixes the
925/// production render so artifacts stay identifiable; the label's token
926/// cost is subtracted from the budget (equal-token discipline: the final
927/// artifact never exceeds the cap).
928// trace:v1 id=impl.scc.bench.ablation work=WORK-SCC-014 satisfies=REQ-SCC-IR
929pub fn render_ablation_surface(
930    root: &Path,
931    mode: SurfaceAblation,
932    goal: &str,
933    budget_tokens: usize,
934) -> Result<String, String> {
935    let store = crate::open_store(root).map_err(|e| e.to_string())?;
936    let config = crate::load_config(root).map_err(|e| e.to_string())?;
937    let stale = crate::stale_paths(&store).map_err(|e| e.to_string())?;
938    let comp = crate::compiler(&store, &config, stale).map_err(|e| e.to_string())?;
939    let ctx = comp.ctx();
940    let label = format!("# SYSTEM SURFACE MAP (ablation {}: {})\n\n", mode.as_str(), goal);
941    let label_tokens = estimate_tokens(&label);
942    let budget = budget_tokens.saturating_sub(label_tokens);
943    let request = scc_context::surface::SurfaceRequest {
944        mode: scc_context::surface::SurfaceMode::Task {
945            goal,
946            visible: None,
947        },
948        budget,
949        explain: false,
950        // Ablation policy: the same quotas/MMR/coverage knobs as
951        // production, but a strict hard max = the budget — the ablation
952        // artifact never exceeds the cap (equal-token discipline; the
953        // required-coverage headroom production grants is not part of the
954        // ablation contract).
955        policy: scc_context::surface::SurfacePolicy {
956            quotas: true,
957            mmr: true,
958            coverage: true,
959            hard_max: budget,
960        },
961        // Ablations never use the semantic scorer (equivalence across
962        // inference on/off is the point); the 10% share is redistributed.
963        semantic: None,
964    };
965    let result = scc_context::surface::build_surface_staged(&ctx, request, &mode.stages());
966    let mut out = label;
967    out.push_str(&result.text);
968    Ok(out)
969}
970
971/// Parse repo-relative file paths from a rendered surface — the
972/// PRODUCTION format: groups headed by an uppercased component line, a
973/// blank line, then the group's path line (`<path>` or `<path>  [<sub>]`),
974/// then blank-line-separated `  <kind> <name>` entry blocks. Order
975/// preserved, deduped. This is the harness's file-selection oracle for
976/// the scc-full structural section: selection comes from the rendered
977/// surface (task PPR + lexical), NEVER from ground-truth `task.files`.
978// trace:exempt reason=internal-detail  # leakage-free selection oracle (impl.scc.bench.structural-select)
979pub fn parse_surface_paths(surface_text: &str) -> Vec<String> {
980    let mut out: Vec<String> = Vec::new();
981    let mut seen: BTreeSet<String> = BTreeSet::new();
982    let lines: Vec<&str> = surface_text.lines().collect();
983    // A group path line sits between two blank lines with a non-blank
984    // component line above the first blank: `<COMPONENT>\n\n<path>\n\n`.
985    // Path lines are left-aligned (entry lines are indented `  <kind>
986    // <name>`); component headers are uppercased by the renderer and
987    // carry no path punctuation; paths carry '/' or '.', so the filters
988    // keep component lines and entries out of the selection.
989    for i in 0..lines.len() {
990        let path = lines[i].trim();
991        if path.is_empty() || lines[i].starts_with(' ') {
992            continue; // entries and signatures are indented, never paths
993        }
994        let prev_blank = i
995            .checked_sub(1)
996            .map(|j| lines[j].trim().is_empty())
997            .unwrap_or(false);
998        let next_blank = lines
999            .get(i + 1)
1000            .map(|l| l.trim().is_empty())
1001            .unwrap_or(false);
1002        let above_component = i
1003            .checked_sub(2)
1004            .map(|j| !lines[j].trim().is_empty())
1005            .unwrap_or(false);
1006        if !(prev_blank && next_blank && above_component) {
1007            continue;
1008        }
1009        if !(path.contains('/') || path.contains('.')) {
1010            continue;
1011        }
1012        if path.chars().all(|c| !c.is_lowercase()) {
1013            continue; // uppercased component headers never name a path
1014        }
1015        let path = path.split('[').next().unwrap_or(path).trim().to_string();
1016        if seen.insert(path.clone()) {
1017            out.push(path);
1018        }
1019    }
1020    out
1021}
1022
1023/// Whether the running binary supports `scc context structural` (Wave-15
1024/// Item 7 wiring). Probed once per process via `--help`.
1025// trace:exempt reason=internal-detail  # structural CLI probe (impl.scc.bench.structural-select)
1026fn structural_cli_supported() -> bool {
1027    static SUPPORTED: OnceLock<bool> = OnceLock::new();
1028    *SUPPORTED.get_or_init(|| {
1029        std::env::current_exe()
1030            .ok()
1031            .map(|exe| {
1032                std::process::Command::new(&exe)
1033                    .args(["context", "structural", "--help"])
1034                    .output()
1035                    .map(|o| o.status.success())
1036                    .unwrap_or(false)
1037            })
1038            .unwrap_or(false)
1039    })
1040}
1041
1042/// Run the current executable with args in `root` and capture stdout.
1043// trace:exempt reason=internal-detail  # subprocess helper (impl.scc.bench.structural-select)
1044fn run_capture(exe: &Path, root: &Path, args: &[&str]) -> Result<String, String> {
1045    let out = std::process::Command::new(exe)
1046        .args(args)
1047        .current_dir(root)
1048        .output()
1049        .map_err(|e| format!("spawn {}: {e}", exe.display()))?;
1050    if !out.status.success() {
1051        return Err(format!(
1052            "`{} {}` exited {}: {}",
1053            exe.display(),
1054            args.join(" "),
1055            out.status,
1056            String::from_utf8_lossy(&out.stderr).trim()
1057        ));
1058    }
1059    Ok(String::from_utf8_lossy(&out.stdout).to_string())
1060}
1061
1062/// Render structural units in order, keeping complete units while the
1063/// chars/4 token estimate fits `budget_tokens` (equal-token discipline: a
1064/// file is never truncated mid-file).
1065// trace:exempt reason=internal-detail  # budgeted structural render (impl.scc.bench.structural-select)
1066fn render_units_budgeted(units: &[scc_core::StructuralSourceUnit], budget_tokens: usize) -> String {
1067    let mut out = String::new();
1068    let mut spent = 0usize;
1069    for u in units {
1070        let piece = scc_context::structural_source::render_structural(std::slice::from_ref(u));
1071        let cost = estimate_tokens(&piece);
1072        if !out.is_empty() && spent + cost > budget_tokens {
1073            break;
1074        }
1075        if !out.is_empty() {
1076            out.push('\n');
1077        }
1078        out.push_str(&piece);
1079        spent += cost;
1080    }
1081    out.trim_end().to_string()
1082}
1083
1084/// Build the structural-source section for a goal with NO ground truth:
1085/// file selection comes from the task-personalized surface (task PPR +
1086/// lexical), and per-file structural units are rendered while the token
1087/// estimate fits `budget_tokens`. When the running binary supports it, the
1088/// production `scc context structural --task "<goal>" --budget N` CLI
1089/// performs the same selection and render in one shot. `task.files` NEVER
1090/// enters context construction — ground truth is scoring-only.
1091// trace:v1 id=impl.scc.bench.structural-select work=WORK-SCC-014 satisfies=REQ-SCC-IR
1092pub fn structural_source_for_goal(
1093    root: &Path,
1094    goal: &str,
1095    budget_tokens: usize,
1096) -> Result<String, String> {
1097    if budget_tokens == 0 || goal.trim().is_empty() {
1098        return Ok(String::new());
1099    }
1100    let exe = std::env::current_exe().map_err(|e| format!("current exe: {e}"))?;
1101    let budget_s = budget_tokens.to_string();
1102    if structural_cli_supported() {
1103        return run_capture(
1104            &exe,
1105            root,
1106            &["context", "structural", "--task", goal, "--budget", &budget_s],
1107        );
1108    }
1109    // Fallback: same selection implemented in the harness — the rendered
1110    // task-personalized surface picks the files, then structural units are
1111    // rendered budget-capped (complete files only).
1112    let surface_text = run_capture(
1113        &exe,
1114        root,
1115        &["surface", "--task", goal, "--budget", &budget_s],
1116    )?;
1117    let files = parse_surface_paths(&surface_text);
1118    if files.is_empty() {
1119        return Ok(String::new());
1120    }
1121    let store = crate::open_store(root).map_err(|e| e.to_string())?;
1122    let config = crate::load_config(root).map_err(|e| e.to_string())?;
1123    let stale = crate::stale_paths(&store).map_err(|e| e.to_string())?;
1124    let comp = crate::compiler(&store, &config, stale).map_err(|e| e.to_string())?;
1125    let units = scc_context::structural_source::structural_source(&comp.ctx(), &files, usize::MAX);
1126    Ok(render_units_budgeted(&units, budget_tokens))
1127}
1128
1129#[cfg(test)]
1130mod tests {
1131    use super::*;
1132    use crate::benchctx::GroundTruth;
1133    use std::path::PathBuf;
1134
1135
1136    #[test]
1137    // trace:v1 id=impl.crates-scc-cli-src-benchagent.fake-agent-records-metrics work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
1138    // trace:v1 id=impl.crates-scc-cli-src-benchagent.fake-agent-records-metrics-2 work=WORK-SI-MMMJA4G6 satisfies=REQ-SI-503JSBGP
1139    fn fake_agent_records_metrics() {
1140        // the fake agent echoes the goal and lists the repo (shows files);
1141        // no JSON event stream → tool-level counters stay at their defaults
1142        let summary = with_paid_opt_in(|| run_agent_benchmark("echo \"$SCC_GOAL\" && ls -R .", 0.0).unwrap());
1143        assert_eq!(summary.tasks, 21);
1144        assert!(summary.results.iter().all(|r| r.exit_ok));
1145        assert!(summary.mean_duration_ms > 0.0);
1146        // no JSON event stream → tool-level counters stay at their defaults;
1147        // first_correct_ms may still fire because `ls -R` surfaces
1148        // ground-truth files in plain output (contract: "in a tool_use or output")
1149        assert!(summary.results.iter().all(|r| {
1150            r.files_opened == 0
1151                && r.search_tool_calls == 0
1152                && r.read_tool_calls == 0
1153                && r.total_tool_calls == 0
1154                && r.wrong_first_locations == 0
1155        }));
1156        let _ = PathBuf::new();
1157    }
1158
1159    #[test]
1160    // trace:v1 id=test.scc.bench.jsonl-metrics verifies=REQ-implement-phase-7-of-scc-x-ripwire-lessons-1-one-hop-type-narrowing exercises=impl.scc.bench.jsonl-metrics
1161    fn metrics_from_jsonl_counts_search_read_and_first_correct() {
1162        let root = PathBuf::from(".");
1163        let jsonl = r#"{"type":"tool_use","name":"grep","input":{"query":"orders"}}
1164{"type":"tool_use","name":"read","input":{"file_path":"wrong.py"}}
1165{"type":"tool_use","name":"read","input":{"file_path":"main.py"}}"#;
1166        let m = metrics_from_jsonl(jsonl, &root, "t", &["main.py".into()]);
1167        assert_eq!(m.search_tool_calls, 1);
1168        assert_eq!(m.read_tool_calls, 2);
1169        assert_eq!(m.files_opened, 2);
1170        assert_eq!(m.wrong_first_locations, 1);
1171        assert_eq!(m.first_correct_ms, Some(2));
1172        assert_eq!(m.files_surfaced, 1);
1173    }
1174
1175    #[test]
1176    // trace:v1 id=impl.crates-scc-cli-src-benchagent.jsonl-event-stream-metrics work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
1177    // trace:v1 id=impl.crates-scc-cli-src-benchagent.jsonl-event-stream-metrics-2 work=WORK-SI-MMMJA4G6 satisfies=REQ-SI-503JSBGP
1178    fn jsonl_event_stream_metrics() {
1179        // Synthetic codex --json stream (task 0 ground truth: main.py +
1180        // services/transcripts.py): a search, a wrong-file read, then the
1181        // correct file via an mcp read_file tool call.
1182        let cmd = r#"printf '%s\n' \
1183'{"type":"item.completed","item":{"id":"i1","type":"command_execution","command":"/bin/zsh -lc \"rg -n transcript .\"","aggregated_output":"","exit_code":0,"status":"completed"}}' \
1184'{"type":"item.completed","item":{"id":"i2","type":"command_execution","command":"/bin/zsh -lc \"sed -n 1,40p wrong_file.py\"","aggregated_output":"","exit_code":0,"status":"completed"}}' \
1185'{"type":"item.completed","item":{"id":"i3","type":"mcp_tool_call","server":"files","tool":"read_file","arguments":{"file_path":"main.py"},"result":{"content":"transcript"}}}' \
1186'{"type":"item.completed","item":{"id":"i4","type":"agent_message","text":"done"}}'"#;
1187        let summary = with_paid_opt_in(|| run_agent_benchmark(cmd, 0.0).unwrap());
1188        assert_eq!(summary.tasks, 21);
1189        assert!(summary.results.iter().all(|r| r.exit_ok));
1190        // task 0 = http-service.rename-transcript-field (GT: main.py, services/transcripts.py)
1191        let first = &summary.results[0];
1192        assert_eq!(first.files_opened, 2, "wrong_file.py + main.py");
1193        assert_eq!(first.search_tool_calls, 1, "rg");
1194        assert_eq!(first.read_tool_calls, 2, "sed + read_file");
1195        assert_eq!(first.total_tool_calls, 3, "agent_message is not a tool");
1196        assert_eq!(first.wrong_first_locations, 1, "wrong_file.py before main.py");
1197        assert!(
1198            first.first_correct_ms.is_some(),
1199            "main.py touched in the read_file event"
1200        );
1201        assert!(first.first_correct_ms.unwrap() <= first.duration_ms);
1202        assert!(summary.mean_first_correct_ms.is_some());
1203    }
1204
1205// trace:v1 id=impl.crates-scc-cli-src-benchagent.corpus-ground-truth-parses work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
1206    #[test]
1207    fn corpus_ground_truth_parses() {
1208        let fixtures = locate_fixtures_dir().unwrap();
1209        let path = fixtures.parent().unwrap().join("benchmarks/tasks.json");
1210        let corpus: BenchmarkCorpus =
1211            serde_json::from_str(&std::fs::read_to_string(path).unwrap()).unwrap();
1212        assert!(corpus.tasks.iter().all(|t| !t.goal.is_empty()));
1213        // GroundTruth fields must deserialize
1214        let gt: GroundTruth = serde_json::from_str(
1215            r#"{"files":["a.py"],"symbols":["f"],"components":["c"],"tests":["t"]}"#,
1216        )
1217        .unwrap();
1218        assert_eq!(gt.files, vec!["a.py"]);
1219    }
1220
1221    #[test]
1222    // trace:exempt reason=unit-test  # wave-15 variant runner test
1223fn variant_benchmark_records_variant_name() {
1224        // Same protocol as run_agent_benchmark, but the summary carries the
1225        // variant name (Wave 15 external suite).
1226        let summary = with_paid_opt_in(|| run_variant_benchmark("scc-atlas-surface", "echo \"$SCC_GOAL\"", 0.0).unwrap());
1227        assert_eq!(summary.variant, "scc-atlas-surface");
1228        assert_eq!(summary.tasks, 21);
1229        assert!(summary.results.iter().all(|r| r.exit_ok));
1230        // no tool events -> no graph queries; first plan = whole output
1231        assert!(summary.results.iter().all(|r| r.graph_tool_calls == 0));
1232    }
1233
1234    #[test]
1235    // trace:exempt reason=unit-test  # wave-15 variant runner test
1236fn run_variant_tasks_filters_and_first_plan() {
1237        // A repo-filtered task list: the closure builds the per-task shell
1238        // command (variant artifact injection point) and the runner records
1239        // first-plan correctness against the plan keys.
1240        let tasks = vec![VariantTask {
1241            id: "http-service.rename-transcript-field".into(),
1242            repo: "http-service-python".into(),
1243            goal: "rename the transcript field".into(),
1244            files: vec!["main.py".into(), "services/transcripts.py".into()],
1245            plan_keys: vec!["main.py".into(), "handle_transcripts".into()],
1246        }];
1247        let cmd_for = |_task: &VariantTask, _root: &Path| {
1248            // The fake agent's "first plan" names the correct file before
1249            // any tool event, then emits one search event.
1250            Ok(r#"printf '%s\n' \
1251'{"type":"item.completed","item":{"type":"agent_message","text":"Plan: main.py"}}' \
1252'{"type":"item.completed","item":{"type":"command_execution","command":"/bin/zsh -lc \"rg -n transcript .\"","exit_code":0}}'"#
1253                .to_string())
1254        };
1255        let summary = run_variant_tasks("scc-atlas", &tasks, cmd_for, 0.0).unwrap();
1256        assert_eq!(summary.tasks, 1);
1257        assert_eq!(summary.variant, "scc-atlas");
1258        let r = &summary.results[0];
1259        assert!(r.first_plan_correct, "plan mentions main.py before tools");
1260        assert_eq!(r.search_tool_calls, 1);
1261        assert_eq!(summary.mean_first_plan_correct, 1.0);
1262        assert_eq!(summary.mean_search_tool_calls, 1.0);
1263    }
1264
1265
1266// trace:v1 id=impl.crates-scc-cli-src-benchagent.summary-with work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
1267    fn summary_with(search: f64, files: f64, first: Option<f64>) -> AgentBenchSummary {
1268        AgentBenchSummary {
1269            tasks: 21,
1270            passed: 21,
1271            mean_search_tool_calls: search,
1272            mean_files_opened: files,
1273            mean_first_correct_ms: first,
1274            ..Default::default()
1275        }
1276    }
1277
1278// trace:v1 id=impl.crates-scc-cli-src-benchagent.agent-gate-evaluates-all-three-clauses work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
1279    #[test]
1280    fn agent_gate_evaluates_all_three_clauses() {
1281        let a = summary_with(2.0, 4.0, Some(1000.0));
1282        // E reduces searches AND files AND first-correct -> PASS
1283        let e = summary_with(1.0, 3.0, Some(900.0));
1284        let g = evaluate_gate(&a, &e);
1285        assert!(g.search_reduced && g.files_bounded && g.first_correct_bounded);
1286        assert!(g.passed);
1287        // files bounded is <= A + 1, so 5 vs 4 still passes
1288        let e2 = summary_with(1.0, 5.0, Some(900.0));
1289        let g2 = evaluate_gate(&a, &e2);
1290        assert!(g2.files_bounded && g2.passed);
1291        // slower first-correct -> FAIL
1292        let e3 = summary_with(1.0, 3.0, Some(1500.0));
1293        let g3 = evaluate_gate(&a, &e3);
1294        assert!(!g3.first_correct_bounded);
1295        assert!(!g3.passed);
1296        // search must be STRICTLY less (equal does not reduce)
1297        let e4 = summary_with(2.0, 3.0, Some(900.0));
1298        let g4 = evaluate_gate(&a, &e4);
1299        assert!(!g4.search_reduced);
1300        assert!(!g4.passed);
1301        // fails closed: missing first-correct mean on either side
1302        let e5 = summary_with(1.0, 3.0, None);
1303        let g5 = evaluate_gate(&a, &e5);
1304        assert!(!g5.first_correct_bounded);
1305        assert!(!g5.passed);
1306    }
1307
1308    #[test]
1309    // trace:v1 id=impl.crates-scc-cli-src-benchagent.agent-gate-fails-closed-without-json-streams work=WORK-wave-15-2-heterogeneous-hierarchy-edges-semantic-scoring-explain-rank-caching
1310    // trace:v1 id=impl.crates-scc-cli-src-benchagent.agent-gate-fails-closed-without-json-streams-2 work=WORK-SI-MMMJA4G6 satisfies=REQ-SI-503JSBGP
1311    fn agent_gate_fails_closed_without_json_streams() {
1312        // echo produces no JSON event stream -> no first-correct means ->
1313        // the gate cannot verify reduction and FAILS closed.
1314        let g = with_paid_opt_in(|| run_agent_gate("echo hi", "echo hi", 0.0).unwrap());
1315        assert!(!g.passed);
1316        assert!(!g.first_correct_bounded);
1317        assert_eq!(g.baseline.tasks, 21);
1318        assert_eq!(g.atlas.tasks, 21);
1319    }
1320
1321    // ---- Wave-15 ablation matrix + leakage-free selection ----------------
1322
1323    #[test]
1324    // trace:exempt reason=unit-test  # ablation matrix tests (impl.scc.bench.ablation)
1325    fn surface_paths_parse_from_production_render() {
1326        // The PRODUCTION surface format: `SCC SYSTEM SURFACE MAP` header,
1327        // then per-group `<COMPONENT>\n\n<path>[  [<sub>]]\n\n  <kind> <name>`
1328        // blocks. The path comes from the group header line, not the
1329        // entries. (concat! preserves the entry indentation that a `\`
1330        // line-continuation would strip.)
1331        let text = concat!(
1332            "SCC SYSTEM SURFACE MAP\n",
1333            "\n",
1334            "CLI\n",
1335            "\n",
1336            "cli.rs\n",
1337            "\n",
1338            "  method Cli.serve\n",
1339            "\n",
1340            "    def serve() -> None\n",
1341            "  Used by:\n",
1342            "    main\n",
1343            "\n",
1344            "ROUTER\n",
1345            "\n",
1346            "router.py  [routing]\n",
1347            "\n",
1348            "  function build_router\n",
1349            "\n",
1350            "    def build_router()\n",
1351            "\n",
1352            "DEPLOY\n",
1353            "\n",
1354            "deploy.go\n",
1355            "\n",
1356            "  method Deploy.deploy\n",
1357            "\n",
1358            "    func (d *Deploy) deploy()\n",
1359            "\n",
1360            "UTIL\n",
1361            "\n",
1362            "util.rs\n",
1363            "\n",
1364            "  function helper\n",
1365            "\n",
1366            "    fn helper()\n",
1367        );
1368        assert_eq!(
1369            parse_surface_paths(text),
1370            vec!["cli.rs", "router.py", "deploy.go", "util.rs"]
1371        );
1372    }
1373
1374    #[test]
1375    // trace:exempt reason=unit-test  # ablation matrix tests (impl.scc.bench.ablation)
1376    fn surface_paths_skip_headers_omitted_and_dedupe() {
1377        // Headers (including the uppercased component lines and the
1378        // OMITTED section), non-path lines, and duplicate groups carry no
1379        // paths; duplicate paths collapse. concat! preserves the entry
1380        // indentation a `\` line-continuation would strip.
1381        let text = concat!(
1382            "SCC SYSTEM SURFACE MAP\n",
1383            "\n",
1384            "CLI\n",
1385            "\n",
1386            "cli.rs\n",
1387            "\n",
1388            "  method Cli.serve\n",
1389            "\n",
1390            "    def serve() -> None\n",
1391            "\n",
1392            "CLI\n",
1393            "\n",
1394            "cli.rs\n",
1395            "\n",
1396            "  method Cli.serve\n",
1397            "\n",
1398            "    def serve() -> None\n",
1399            "OMITTED (token budget exceeded):\n",
1400            "  2 lower-ranked function definitions\n",
1401            "\n",
1402            "MY.SERVICE\n",
1403            "\n",
1404            "  method Thing.run\n",
1405            "\n",
1406            "    def run()\n",
1407            "\n",
1408            "DATA\n",
1409            "\n",
1410            "src/data/store.ts\n",
1411            "\n",
1412            "  class Store\n",
1413            "\n",
1414            "    export class Store {}\n",
1415        );
1416        assert_eq!(
1417            parse_surface_paths(text),
1418            vec!["cli.rs", "src/data/store.ts"]
1419        );
1420    }
1421
1422    #[test]
1423    // trace:exempt reason=unit-test  # ablation matrix tests (impl.scc.bench.ablation)
1424    fn ablation_modes_render_distinct_artifacts() {
1425        // Every mode renders a non-empty, mode-labeled artifact for the
1426        // cli-service fixture (indexed in a scratch copy).
1427        let fixtures = locate_fixtures_dir().unwrap();
1428        let src = fixtures.join("cli-service");
1429        assert!(src.is_dir(), "cli-service fixture exists");
1430        let tmp = tempfile::TempDir::new().unwrap();
1431        let root = tmp.path().join("repo");
1432        copy_fixture_tree(&src, &root);
1433        crate::commands::cmd_index(&root, true).unwrap();
1434        let goal = "add a --paging flag to the serve subcommand";
1435        for mode in [
1436            SurfaceAblation::Lexical,
1437            SurfaceAblation::GlobalPpr,
1438            SurfaceAblation::TaskPpr,
1439            SurfaceAblation::PprMmr,
1440            SurfaceAblation::PprQuotas,
1441            SurfaceAblation::PprOptimizer,
1442        ] {
1443            let out = render_ablation_surface(&root, mode, goal, 8000).unwrap();
1444            assert!(!out.is_empty(), "{mode:?} renders");
1445            assert!(
1446                out.contains(&format!("ablation {}", mode.as_str())),
1447                "{mode:?} header marks the mode: {out}"
1448            );
1449            // budget discipline: the chars/4 estimate never exceeds the cap
1450            assert!(
1451                estimate_tokens(&out) <= 8000,
1452                "{mode:?} fits the budget ({} tokens)",
1453                estimate_tokens(&out)
1454            );
1455        }
1456    }
1457
1458    #[test]
1459    // trace:exempt reason=unit-test  # ablation matrix tests (impl.scc.bench.ablation)
1460    fn lexical_ablation_ranks_goal_matches_first() {
1461        // Lexical mode must surface serve/paging symbols before unrelated
1462        // ones (pure lexical baseline).
1463        let fixtures = locate_fixtures_dir().unwrap();
1464        let tmp = tempfile::TempDir::new().unwrap();
1465        let root = tmp.path().join("repo");
1466        copy_fixture_tree(&fixtures.join("cli-service"), &root);
1467        crate::commands::cmd_index(&root, true).unwrap();
1468        let out = render_ablation_surface(
1469            &root,
1470            SurfaceAblation::Lexical,
1471            "add a --paging flag to the serve subcommand",
1472            8000,
1473        )
1474        .unwrap();
1475        let serve_lines: Vec<&str> = out
1476            .lines()
1477            .filter(|l| {
1478                // Entry lines are 2-space indented `  <kind> <name>`; the
1479                // label/headers/signatures are excluded.
1480                l.starts_with("  ") && !l.starts_with("    ") && l.contains("serve")
1481            })
1482            .collect();
1483        assert!(!serve_lines.is_empty(), "serve symbols surface lexically: {out}");
1484    }
1485    #[test]
1486    // trace:v1 id=test.scc.bench.paid-gate-predicate verifies=REQ-SCC-TEST exercises=impl.scc.bench.paid-gate
1487    fn paid_gate_predicate_allows_only_explicit_one() {
1488        assert!(paid_opt_in_allowed(Some("1")));
1489        assert!(!paid_opt_in_allowed(None));
1490        assert!(!paid_opt_in_allowed(Some("0")));
1491        assert!(!paid_opt_in_allowed(Some("")));
1492        assert!(!paid_opt_in_allowed(Some("true")));
1493    }
1494    #[test]
1495    // trace:v1 id=test.scc.bench.paid-gate-refuses verifies=REQ-SCC-TEST exercises=impl.scc.bench.paid-gate
1496    fn paid_gate_refuses_benchmark_without_opt_in() {
1497        let _guard = PAID_TEST_ENV_LOCK.lock();
1498        let prev = std::env::var("SCC_ALLOW_PAID_BENCHMARKS").ok();
1499        std::env::remove_var("SCC_ALLOW_PAID_BENCHMARKS");
1500        let err = run_agent_benchmark("echo hi", 0.0).unwrap_err();
1501        assert!(
1502            err.contains("SCC_ALLOW_PAID_BENCHMARKS"),
1503            "refusal must name the opt-in: {err}"
1504        );
1505        match prev {
1506            Some(v) => std::env::set_var("SCC_ALLOW_PAID_BENCHMARKS", v),
1507            None => std::env::remove_var("SCC_ALLOW_PAID_BENCHMARKS"),
1508        }
1509    }
1510
1511    #[test]
1512    // trace:v1 id=test.scc.bench.paid-gate-allows verifies=REQ-SCC-TEST exercises=impl.scc.bench.paid-gate
1513    fn paid_gate_opt_in_allows_mock_benchmark() {
1514        // A harmless mock command proves the gate opens with opt-in; no
1515        // model quota is spent by `echo`.
1516        let summary = with_paid_opt_in(|| run_agent_benchmark("echo hi", 0.0).unwrap());
1517        assert_eq!(summary.tasks, 21);
1518    }
1519
1520}