Skip to main content

supercode_harness/
audit.rs

1//! Corpus coverage audit.
2//!
3//! Walks a directory of session logs, parses every line through the typed
4//! [`crate::schema`], and reports — with real counts — exactly what we model,
5//! what we model-but-drop on normalization, and what we don't model at all.
6//! This is the machine that turns "what's missing?" into an enumerated answer
7//! rather than a guess.
8//!
9//! ```no_run
10//! use std::path::Path;
11//! use supercode_harness::audit::{audit_dir, Corpus};
12//!
13//! let report = audit_dir(Path::new("/home/me/.codex/sessions"), Corpus::Codex, None);
14//! report.print();
15//! ```
16
17use std::collections::BTreeMap;
18use std::path::{Path, PathBuf};
19
20use serde_json::Value;
21
22use crate::schema::{claude_code::*, codex::*, raw_block_tag, ContentBlock};
23use supercode_interchange::session::{
24    opencode_file_image_part, pi_content_has_unknown_image_shape,
25};
26
27/// Which corpus a directory holds.
28#[derive(Debug, Clone, Copy, PartialEq, Eq)]
29pub enum Corpus {
30    /// `~/.claude/projects`
31    ClaudeCode,
32    /// `~/.codex/sessions`
33    Codex,
34    /// `~/.pi/agent/sessions`
35    Pi,
36    /// `~/.local/share/opencode` (envelope-form fixtures/corpus — see
37    /// `docs/interop/opencode-pi-spec.md` §1.2/§4.1).
38    OpenCode,
39    /// `~/.grok/sessions` (`chat_history.jsonl` files only; companion
40    /// `updates.jsonl` streams are live protocol events, not transcripts).
41    Grok,
42    /// `~/.gemini/tmp/<project>/chats` Gemini CLI JSONL transcripts.
43    Gemini,
44    /// Goose's `sessions/sessions.db` native store.
45    Goose,
46}
47
48/// How a given discriminant is handled by the loader.
49#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
50pub enum Coverage {
51    /// Parsed and normalized into the canonical conversation.
52    Normalized,
53    /// PARITY-12/PARITY-13 (P012/P013): parsed and its content/provenance IS
54    /// captured by the loader — into `Session.meta` (Codex `session_meta`'s
55    /// id/cwd/model/base_instructions, `turn_context`'s model), replay
56    /// semantics (`thread_rolled_back` actually removes the rolled-back
57    /// turns, `exited_review_mode`'s `review_output.overall_explanation`
58    /// becomes a message with `review_output.findings` AND
59    /// `overall_correctness`/`overall_confidence_score` (N4) captured onto
60    /// that message's metadata (D4/N4), `thread_goal_updated`'s
61    /// `goal.objective` becomes a message with `goal.status`/
62    /// `goal.tokenBudget` captured onto that message's metadata too (D4),
63    /// `agent_message` can become a message when its text has no
64    /// `response_item` twin) — just not 1:1 into a `ChatMessage` the way
65    /// `Normalized` records are.
66    /// This is the "retained, not dropped" bucket the two audits were
67    /// missing: before this variant existed, every one of these landed in
68    /// `Dropped` indistinguishably from truly-inert UI noise (`token_count`,
69    /// `task_started`, …), which is exactly the false "silently dropped"
70    /// signal both items' dev/03 ACs flag.
71    ///
72    /// `response_item/reasoning` also belongs here (D5, PARITY-12): its
73    /// `summary` text, raw `content` chain-of-thought text when non-null
74    /// (N2 — previously dropped despite this very label claiming otherwise;
75    /// `content` is `null` on the vast majority of real turns, so this was
76    /// easy to miss until fixtures carried the key at all), and the
77    /// `encrypted_content` presence flag (N1: only a genuinely non-null
78    /// value counts — `serde_json` returns `Some(&Value::Null)` for a
79    /// present-but-null key, which is what EVERY real rollout's reasoning
80    /// item carries per upstream `codex-rs/protocol/src/models.rs:970-983`,
81    /// so a naive `.is_some()` false-flagged every reasoning item as
82    /// "encrypted" on real data) are captured by `Session::from_codex_str`
83    /// onto the *next* assistant `ChatMessage`'s metadata (`reasoning`/
84    /// `reasoning_content`/`reasoning_encrypted`) — see `audit_codex_item`'s
85    /// `Reasoning` arm. When no following assistant turn exists to attach to
86    /// (a non-assistant item interrupts, or the reasoning is dangling at
87    /// EOF — an aborted-turn shape, N3), it is flushed as its own synthesized
88    /// `[reasoning] (turn ended without a reply)` message instead of being
89    /// silently discarded, so this label stays honest for that shape too.
90    /// It isn't `Normalized` (no canonical "reasoning" `ChatMessage`), but it
91    /// is provably not a blind drop either.
92    ///
93    /// DISCLOSURE: every metadata key mentioned above (`review_findings`,
94    /// `review_overall_correctness`, `review_overall_confidence_score`,
95    /// `goal_status`, `goal_token_budget`, `reasoning`, `reasoning_content`,
96    /// `reasoning_encrypted`) is LOADER-CAPTURE ONLY. It survives the
97    /// verbatim Codex→Codex diagonal (raw bytes, untouched) and reads back
98    /// out of the native supercode format, but it does NOT survive a
99    /// cross-format writer or the `--session-id` re-serialized diagonal:
100    /// `ChatMessage.metadata` is never serialized (`message.rs:50-55`) and no
101    /// writer reads it back out. Don't misread `Retained` here as
102    /// cross-format-durable — it means "captured in-process", not "written
103    /// back out".
104    Retained,
105    /// Parsed and understood, but intentionally dropped (e.g. `token_count`,
106    /// `task_started`/`task_complete`, UI echoes of content already captured
107    /// elsewhere as `Normalized`/`Retained`). See `event_msg_coverage`'s doc
108    /// comment for the few real, currently-unrecovered exceptions (D1) —
109    /// e.g. `patch_apply_end`'s `changes[path].unified_diff` — where
110    /// `Dropped` means genuine, asserted content loss, not "duplicated
111    /// elsewhere".
112    Dropped,
113    /// Not modeled at all — falls into an `Unknown` typed bucket.
114    Unmodeled,
115}
116
117impl Coverage {
118    fn symbol(self) -> &'static str {
119        match self {
120            Coverage::Normalized => "✅ normalized",
121            Coverage::Retained => "◆ retained  ",
122            Coverage::Dropped => "➖ dropped   ",
123            Coverage::Unmodeled => "❌ UNMODELED ",
124        }
125    }
126}
127
128/// A tally for one discriminant value.
129#[derive(Debug, Clone, Default)]
130#[non_exhaustive]
131pub struct Tally {
132    /// How many times it occurred.
133    pub count: u64,
134    /// Field keys seen in `extra` (fields we didn't model), with counts.
135    pub unmodeled_fields: BTreeMap<String, u64>,
136}
137
138/// The full audit result.
139#[derive(Debug, Default)]
140#[non_exhaustive]
141pub struct Report {
142    /// Which corpus this is.
143    pub corpus: Option<&'static str>,
144    /// Files scanned.
145    pub files: u64,
146    /// Lines parsed.
147    pub lines: u64,
148    /// Lines that failed to deserialize even into the typed schema.
149    pub parse_errors: u64,
150    /// Per record/payload discriminant: (coverage, tally). Keyed by a readable
151    /// path like `response_item/custom_tool_call`.
152    pub records: BTreeMap<String, (Coverage, Tally)>,
153    /// Content block discriminants seen, with counts SPLIT by the coverage
154    /// each instance actually got (N1, Fable-5 review). Keyed by
155    /// `(tag, coverage)` rather than `tag` alone: D5 made `image` coverage
156    /// PER-INSTANCE (a `base64`/`url` source is `Normalized`, a Files-API/
157    /// `file` source is `Dropped`), so a single `tag -> (Coverage, count)`
158    /// entry — last-write-wins on `Coverage` — silently collapsed a mixed
159    /// corpus's genuinely-`Dropped` instances into whatever coverage the
160    /// LAST-seen instance of that tag happened to have, over- or
161    /// under-claiming fidelity depending on file order. Splitting the bucket
162    /// keeps every instance's actual coverage and never collapses counts.
163    pub blocks: BTreeMap<(String, Coverage), u64>,
164    /// Tool names seen, with counts.
165    pub tools: BTreeMap<String, u64>,
166    /// Structural notes discovered while scanning (e.g. sidechain lines).
167    pub notes: BTreeMap<String, u64>,
168}
169
170impl Report {
171    fn bump(&mut self, key: String, cov: Coverage, extra: &crate::schema::ExtraFields) {
172        let entry = self.records.entry(key).or_insert((cov, Tally::default()));
173        entry.0 = cov;
174        entry.1.count += 1;
175        for k in extra.keys() {
176            *entry.1.unmodeled_fields.entry(k.clone()).or_insert(0) += 1;
177        }
178    }
179
180    fn bump_block(&mut self, block: &ContentBlock, raw: &Value) {
181        let (tag, cov) = match block.tag() {
182            Some(t) => (t.to_string(), block_coverage(block)),
183            None => (
184                raw_block_tag(raw).unwrap_or_else(|| "<no-type>".into()),
185                Coverage::Unmodeled,
186            ),
187        };
188        // N1: bucket on (tag, coverage), not tag alone — see the `blocks`
189        // field doc. Each instance is counted under its OWN actual coverage
190        // instead of one shared, last-write-wins `Coverage` per tag.
191        *self.blocks.entry((tag, cov)).or_insert(0) += 1;
192    }
193
194    fn note(&mut self, key: &str) {
195        *self.notes.entry(key.to_string()).or_insert(0) += 1;
196    }
197
198    /// Serialize the report as structured JSON (for CI/dashboards).
199    pub fn to_json(&self) -> serde_json::Value {
200        let cov = |c: Coverage| match c {
201            Coverage::Normalized => "normalized",
202            Coverage::Retained => "retained",
203            Coverage::Dropped => "dropped",
204            Coverage::Unmodeled => "unmodeled",
205        };
206        let records: serde_json::Map<String, serde_json::Value> = self
207            .records
208            .iter()
209            .map(|(k, (c, t))| {
210                (
211                    k.clone(),
212                    serde_json::json!({
213                        "coverage": cov(*c),
214                        "count": t.count,
215                        "unmodeled_fields": t.unmodeled_fields.keys().collect::<Vec<_>>(),
216                    }),
217                )
218            })
219            .collect();
220        // N1: a tag can now have MULTIPLE coverage buckets (e.g. `image` ->
221        // Normalized:1, Dropped:1 on a mixed corpus), so each tag maps to a
222        // list of `{coverage, count}` entries rather than a single one.
223        let mut blocks_by_tag: BTreeMap<&str, Vec<serde_json::Value>> = BTreeMap::new();
224        for ((tag, c), n) in &self.blocks {
225            blocks_by_tag
226                .entry(tag.as_str())
227                .or_default()
228                .push(serde_json::json!({"coverage": cov(*c), "count": n}));
229        }
230        let blocks: serde_json::Map<String, serde_json::Value> = blocks_by_tag
231            .into_iter()
232            .map(|(k, v)| (k.to_string(), serde_json::Value::Array(v)))
233            .collect();
234        serde_json::json!({
235            "corpus": self.corpus,
236            "files": self.files,
237            "lines": self.lines,
238            "parse_errors": self.parse_errors,
239            "records": records,
240            "blocks": blocks,
241            "tools": self.tools,
242            "notes": self.notes,
243        })
244    }
245
246    /// Print a human-readable report to stdout.
247    pub fn print(&self) {
248        println!("# Coverage audit: {}", self.corpus.unwrap_or("?"));
249        println!(
250            "files={} lines={} parse_errors={}\n",
251            self.files, self.lines, self.parse_errors
252        );
253
254        println!("## Records (discriminant → coverage, count, unmodeled fields)");
255        for (key, (cov, tally)) in &self.records {
256            print!("  {}  {:<40} {:>9}", cov.symbol(), key, tally.count);
257            if !tally.unmodeled_fields.is_empty() {
258                let mut fields: Vec<_> = tally.unmodeled_fields.keys().cloned().collect();
259                fields.sort();
260                print!("   unmodeled fields: {}", fields.join(", "));
261            }
262            println!();
263        }
264
265        if !self.blocks.is_empty() {
266            println!("\n## Content blocks");
267            // N1: one row per (tag, coverage) bucket — a tag with mixed
268            // coverage (e.g. `image` seen both Normalized and Dropped) now
269            // prints as two distinct, honestly-counted rows instead of one
270            // row whose coverage was whichever instance was seen last.
271            for ((tag, cov), count) in &self.blocks {
272                println!("  {}  {:<28} {:>9}", cov.symbol(), tag, count);
273            }
274        }
275
276        if !self.notes.is_empty() {
277            println!("\n## Structural notes");
278            for (k, v) in &self.notes {
279                println!("  {k}: {v}");
280            }
281        }
282
283        if !self.tools.is_empty() {
284            println!("\n## Tools observed (top 30 by frequency)");
285            let mut tools: Vec<_> = self.tools.iter().collect();
286            tools.sort_by(|a, b| b.1.cmp(a.1));
287            for (name, count) in tools.into_iter().take(30) {
288                println!("  {count:>9}  {name}");
289            }
290        }
291
292        println!("\n## Summary of gaps (UNMODELED or dropped, non-UI)");
293        for (key, (cov, tally)) in &self.records {
294            if *cov == Coverage::Unmodeled {
295                println!("  ❌ {key} ({} occurrences) — not modeled", tally.count);
296            }
297        }
298        for ((tag, cov), count) in &self.blocks {
299            if *cov == Coverage::Unmodeled {
300                println!("  ❌ content block `{tag}` ({count}) — not modeled");
301            }
302        }
303    }
304}
305
306fn block_coverage(block: &ContentBlock) -> Coverage {
307    match block {
308        ContentBlock::Text { .. }
309        | ContentBlock::InputText { .. }
310        | ContentBlock::OutputText { .. }
311        | ContentBlock::ToolUse { .. }
312        | ContentBlock::ToolResult { .. } => Coverage::Normalized,
313        // D5 (Fable-5 review, confirmed): `image` used to be blanket-marked
314        // `Normalized` regardless of its `source` shape, but
315        // `claude_image_block_to_part` (session.rs) only actually converts
316        // `base64`/`url` sources into a replayable `content_parts` image —
317        // anything else (a Files-API `{"source":{"type":"file",...}}`
318        // reference, most commonly) is NOT carried through; the loader now
319        // emits a bracketed marker so the record survives (see
320        // `UNCONVERTIBLE_IMAGE_MARKER`), but the actual image content is
321        // still lost, so this must not claim full fidelity. Codex's
322        // `input_image` has no `source` sub-object (a bare, always-
323        // convertible `image_url` string via `codex_extract_images`) and
324        // `fallback` (folded into a text marker) are both still genuinely
325        // `Normalized`.
326        // N3 (Fable-5 review, ticket, fixed inline since it's the same
327        // `Image{source}` inspection N1 already touches): a well-typed but
328        // EMPTY `base64`/`url` source — e.g. `{"type":"base64","data":""}`
329        // — used to blanket-audit as `Normalized` just like a genuinely
330        // convertible one, but `claude_image_block_to_part` (session.rs)
331        // treats it as UNCONVERTIBLE (its own non-empty `mime`/`data`/`url`
332        // check returns `None`, same `UNCONVERTIBLE_IMAGE_MARKER` fallback
333        // path as a Files-API reference) — audit and loader must agree.
334        ContentBlock::Image { source } => image_source_coverage(source),
335        ContentBlock::InputImage { .. } | ContentBlock::Fallback { .. } => Coverage::Normalized,
336        // Provider-private reasoning: retained verbatim in
337        // (skip-serialized) `ChatMessage` metadata (`push_claude_assistant`)
338        // so a same-model continuation can replay it, but it has no slot in
339        // the canonical replayable conversation itself — "understood, not
340        // silently lost" rather than "normalized into the conversation".
341        ContentBlock::Thinking { .. } | ContentBlock::RedactedThinking { .. } => Coverage::Dropped,
342        ContentBlock::Unknown => Coverage::Unmodeled, // future blocks
343    }
344}
345
346/// Score a Claude `image` block's `source` object — factored out of
347/// `block_coverage`'s `Image` arm (D5/N3 discipline: `base64`/`url` with a
348/// non-empty payload is `Normalized`, anything else is `Dropped`) so PARITY-11
349/// can reuse the EXACT same test for an `image` block nested inside a
350/// `tool_result`'s own `content` array, not just a top-level one.
351fn image_source_coverage(source: &Value) -> Coverage {
352    match source.get("type").and_then(Value::as_str) {
353        Some("base64") => {
354            let mime = source
355                .get("media_type")
356                .and_then(Value::as_str)
357                .unwrap_or("");
358            let data = source.get("data").and_then(Value::as_str).unwrap_or("");
359            if mime.is_empty() || data.is_empty() {
360                Coverage::Dropped
361            } else {
362                Coverage::Normalized
363            }
364        }
365        Some("url") => {
366            let url = source.get("url").and_then(Value::as_str).unwrap_or("");
367            if url.is_empty() {
368                Coverage::Dropped
369            } else {
370                Coverage::Normalized
371            }
372        }
373        _ => Coverage::Dropped,
374    }
375}
376
377/// PARITY-11 (nested images, skeptic-confirmed on a real session): a Claude
378/// `tool_result` block's OWN `content` array can carry `image` blocks — the
379/// everyday "Read a PNG / screenshot tool output" shape. `block_coverage`
380/// blanket-labels the enclosing `tool_result` `Normalized` (true for its text
381/// portion), which used to be the ONLY signal `audit` gave — so a session
382/// whose `tool_result` held nothing but a dropped image still reported zero
383/// `image` blocks and a clean `tool_result: Normalized` line, i.e. coverage
384/// said "retained" while the loader silently dropped the bytes. This censuses
385/// each nested `image` block individually, under its own `tool_result/image`
386/// discriminant, scored with the SAME [`image_source_coverage`] test
387/// `session.rs`'s `extract_tool_result_content` uses to decide whether it
388/// actually captures the block into `content_parts` — so a genuinely
389/// unconvertible nested image (Files-API reference, empty payload, …) shows
390/// up here as `Dropped`, not folded invisibly into the outer `Normalized`
391/// tally.
392fn audit_nested_tool_result_images(content: &Value, report: &mut Report) {
393    let Some(items) = content.as_array() else {
394        return;
395    };
396    for item in items {
397        if item.get("type").and_then(Value::as_str) != Some("image") {
398            continue;
399        }
400        let cov = image_source_coverage(item.get("source").unwrap_or(&Value::Null));
401        *report
402            .blocks
403            .entry(("tool_result/image".to_string(), cov))
404            .or_insert(0) += 1;
405    }
406}
407
408/// Audit a directory. `limit` caps the number of files scanned (None = all).
409///
410/// `Corpus::OpenCode` (PARITY-4) is special-cased: a real OpenCode data root
411/// (`~/.local/share/opencode`) holds no `.jsonl` files at all — sessions live
412/// in `opencode*.db` (current installs) or a JSON-file tree (legacy). When
413/// [`supercode_interchange::session::detect_opencode_storage_surface`] resolves `dir` to the
414/// SQLite surface, this routes through
415/// [`supercode_interchange::session::opencode_sqlite_corpus_envelope_text`] (up to `limit`
416/// SESSIONS, not files — `report.files` counts sessions scanned in that
417/// case) instead of the `jsonl_files` walk below, so a real store actually
418/// gets audited rather than silently reporting zero files/lines. A directory
419/// with no detected SQLite surface (e.g. a fixture dir of committed
420/// envelope-form `.jsonl` files, or a not-yet-implemented legacy JSON tree)
421/// falls back to the original file-walk unchanged.
422pub fn audit_dir(dir: &Path, corpus: Corpus, limit: Option<usize>) -> Report {
423    let mut report = Report {
424        corpus: Some(match corpus {
425            Corpus::ClaudeCode => "claude-code",
426            Corpus::Codex => "codex",
427            Corpus::Pi => "pi",
428            Corpus::OpenCode => "opencode",
429            Corpus::Grok => "grok",
430            Corpus::Gemini => "gemini",
431            Corpus::Goose => "goose",
432        }),
433        ..Default::default()
434    };
435
436    if corpus == Corpus::OpenCode {
437        if let Some((supercode_interchange::session::OpenCodeStorageSurface::Sqlite, db_path)) =
438            supercode_interchange::session::detect_opencode_storage_surface(dir)
439        {
440            return audit_opencode_sqlite(&db_path, limit, report);
441        }
442    }
443    if corpus == Corpus::Goose {
444        return audit_goose(dir, limit, report);
445    }
446
447    let mut files = jsonl_files(dir);
448    if corpus == Corpus::Grok {
449        files.retain(|path| {
450            path.file_name().and_then(|name| name.to_str()) == Some("chat_history.jsonl")
451        });
452    }
453    let files = match limit {
454        Some(n) => &files[..files.len().min(n)],
455        None => &files[..],
456    };
457
458    for path in files {
459        report.files += 1;
460        let Ok(text) = std::fs::read_to_string(path) else {
461            continue;
462        };
463        for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
464            report.lines += 1;
465            match corpus {
466                Corpus::Codex => audit_codex_line(line, &mut report),
467                Corpus::ClaudeCode => audit_claude_line(line, &mut report),
468                Corpus::Pi => audit_pi_line(line, &mut report),
469                Corpus::OpenCode => audit_opencode_line(line, &mut report),
470                Corpus::Grok => audit_grok_line(line, &mut report),
471                Corpus::Gemini => audit_gemini_line(line, &mut report),
472                Corpus::Goose => unreachable!("Goose is audited through its SQLite store"),
473            }
474        }
475    }
476    report
477}
478
479fn audit_goose(root: &Path, limit: Option<usize>, mut report: Report) -> Report {
480    let catalog = crate::HarnessCatalog::new();
481    let discovery = catalog.discover(&crate::DiscoveryQuery {
482        harnesses: vec![crate::HarnessId::from(crate::HarnessId::GOOSE)],
483        homes: crate::HarnessHomes {
484            goose: root.to_path_buf(),
485            ..crate::HarnessHomes::default()
486        },
487        limit,
488        ..crate::DiscoveryQuery::default()
489    });
490    let descriptors = match discovery {
491        Ok(descriptors) => descriptors,
492        Err(error) => {
493            report.note(&format!("Goose discovery failed: {error}"));
494            return report;
495        }
496    };
497    let extra = EMPTY_EXTRA.get_or_init(Default::default);
498    for descriptor in descriptors {
499        let session = match catalog.load(&descriptor.locator) {
500            Ok(session) => session,
501            Err(error) => {
502                report.note(&format!(
503                    "Goose session {} failed to load: {error}",
504                    descriptor.locator.session_id
505                ));
506                continue;
507            }
508        };
509        report.files += 1;
510        for message in session.messages {
511            report.lines += 1;
512            report.bump(
513                format!("message/{:?}", message.role).to_lowercase(),
514                Coverage::Normalized,
515                extra,
516            );
517            for call in message.tool_calls() {
518                *report.tools.entry(call.function.name.clone()).or_insert(0) += 1;
519            }
520        }
521    }
522    report
523}
524
525/// The `Corpus::OpenCode` + SQLite branch of [`audit_dir`] (PARITY-4): reads
526/// every session's `session`/`message`/`part`/`todo` records out of
527/// `db_path` as envelope lines
528/// ([`supercode_interchange::session::opencode_sqlite_corpus_envelope_text`]) and scores each
529/// one exactly like a line from a committed envelope-form fixture
530/// (`audit_opencode_line` — same classifier, same coverage buckets, so a
531/// SQLite corpus and a JSON-tree/fixture corpus are held to the identical
532/// bar). `report.files` counts SESSIONS scanned (the natural unit for a
533/// single-DB corpus), not `.jsonl` files. A store that fails to open (bad
534/// path, corrupt DB, wrong schema) does not panic or silently return an
535/// empty report — the failure is recorded in `report.notes` so it is visible
536/// in both the text and `--json` renderings.
537fn audit_opencode_sqlite(db_path: &Path, limit: Option<usize>, mut report: Report) -> Report {
538    match supercode_interchange::session::opencode_sqlite_corpus_envelope_text(db_path, limit) {
539        Ok(text) => {
540            for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
541                report.lines += 1;
542                // A `session` envelope line is one-per-scanned-session — use
543                // it to derive `report.files` (sessions, not `.jsonl` files)
544                // without a second SQL pass.
545                if let Ok(v) = serde_json::from_str::<Value>(line) {
546                    if v.get("key")
547                        .and_then(Value::as_array)
548                        .and_then(|k| k.first())
549                        .and_then(Value::as_str)
550                        == Some("session")
551                    {
552                        report.files += 1;
553                    }
554                }
555                audit_opencode_line(line, &mut report);
556            }
557        }
558        Err(e) => {
559            report.note(&format!("opencode_sqlite_error: {e}"));
560        }
561    }
562    report
563}
564
565fn audit_codex_line(line: &str, report: &mut Report) {
566    let raw: Value = match serde_json::from_str(line) {
567        Ok(v) => v,
568        Err(_) => {
569            report.parse_errors += 1;
570            return;
571        }
572    };
573    let parsed: Result<CodexLine, _> = serde_json::from_str(line);
574    let Ok(parsed) = parsed else {
575        report.parse_errors += 1;
576        return;
577    };
578
579    match &parsed.record {
580        CodexRecord::Unknown => {
581            let tag = raw
582                .get("type")
583                .and_then(Value::as_str)
584                .unwrap_or("<no-type>");
585            report.bump(
586                format!("<line>/{tag}"),
587                Coverage::Unmodeled,
588                &Default::default(),
589            );
590        }
591        CodexRecord::ResponseItem { payload } => audit_codex_item(payload, &raw, report),
592        CodexRecord::EventMsg { payload } => {
593            let sub = payload.kind.clone().unwrap_or_else(|| "?".into());
594            report.bump(
595                format!("event_msg/{sub}"),
596                event_msg_coverage(&sub),
597                &payload.extra,
598            );
599        }
600        // PARITY-13 (P013): `session_meta` isn't discarded — `from_codex_str`
601        // (via `capture_codex_session_meta`) threads its id/cwd/model/
602        // base_instructions/lineage fields onto `Session.meta`, and the whole
603        // record is kept verbatim in `meta.codex_headers` so a same-format
604        // re-export (`to_codex_jsonl`) replays it byte-for-byte. `Dropped`
605        // read as a silent, untracked loss; it's retained, just not folded
606        // into a `ChatMessage`.
607        CodexRecord::SessionMeta { payload } => {
608            report.bump("session_meta".into(), Coverage::Retained, &payload.extra);
609        }
610        // Same reasoning as `SessionMeta` above: `turn_context`'s `model` is
611        // threaded onto `Session.meta.model` (first occurrence) and the
612        // record itself is kept verbatim in `meta.codex_headers` (P013).
613        CodexRecord::TurnContext { payload } => {
614            report.bump("turn_context".into(), Coverage::Retained, &payload.extra);
615        }
616        CodexRecord::Compacted { .. } => {
617            // replacement_history now replaces prior turns on load.
618            report.bump(
619                "compacted".into(),
620                Coverage::Normalized,
621                &Default::default(),
622            );
623        }
624    }
625}
626
627/// PARITY-12/PARITY-13 (P012/P013): `event_msg` coverage, mirroring EXACTLY
628/// the subset of `payload.type` values `Session::from_codex_str` special-cases
629/// (see its `Some("event_msg") if payload.get("type") == Some(...)` arms) —
630/// keep these two lists in lockstep; a subtype added there without a match
631/// here regresses to a false `Dropped` again.
632///
633/// - `agent_message`: real assistant narration with no `response_item`
634///   counterpart becomes a message (the sole source of truth in some
635///   collaboration/multi-agent sessions); a duplicate of an already-normalized
636///   `response_item/message` is skipped as a no-op. Either way the loader
637///   parses and acts on it — not a blind, untracked drop.
638/// - `thread_rolled_back`: directly mutates the canonical conversation
639///   (removes the rolled-back turns) — collaboration/undo provenance that's
640///   applied, not discarded.
641/// - `thread_goal_updated`: its `goal.objective` becomes a synthesized
642///   system message when present; `goal.status`/`goal.tokenBudget` (when
643///   present) are captured onto that message's metadata too (D4) —
644///   `goal.tokensUsed`/`timeUsedSeconds`/timestamps are still real, minor
645///   residue, not claimed as retained.
646/// - `exited_review_mode`: its `review_output.overall_explanation` becomes a
647///   synthesized assistant message; `review_output.findings` (verbatim JSON:
648///   `title`/`body`/`confidence_score`/`priority`/`code_location`) AND the
649///   review verdict itself, `overall_correctness`/`overall_confidence_score`
650///   (N4 — previously neither captured nor disclosed, unlike the
651///   `thread_goal_updated` arm above which already disclosed its own
652///   residue), are captured onto that message's metadata too (D4/N4) — the
653///   only place review-mode findings/verdict live.
654///
655/// `token_count`, `task_started`/`task_complete`, `user_message` (a
656/// duplicate of the already-normalized `response_item/message[user]`),
657/// `entered_review_mode` has no canonical chat turn, but PARITY-13's portable
658/// Codex provenance envelope now retains its exact target/hint record across
659/// every foreign-format hop, so it is `Retained` rather than an untracked
660/// drop. Other UI-only echoes with no unique replayable content stay
661/// `Dropped` honestly.
662///
663/// D1 correction — this used to also claim `exec_command_begin`/`end` and
664/// `mcp_tool_call_begin`/`patch_apply_begin` were safe to drop because
665/// "already captured via the paired `response_item/function_call*`". That
666/// framing was FALSE for what those events would carry if they were ever
667/// actually present: verified against upstream `openai/codex`'s
668/// `codex-rs/rollout/src/policy.rs` `should_persist_event_msg`, all four of
669/// `EventMsg::ExecCommandBegin`, `EventMsg::ExecCommandEnd`,
670/// `EventMsg::McpToolCallBegin`, and `EventMsg::PatchApplyBegin` hit that
671/// function's `=> false` arm — **codex never writes these event kinds to a
672/// real rollout file at all.** So in a genuine `~/.codex/sessions` corpus
673/// this isn't "content safely captured elsewhere"; it's a branch that is
674/// simply never reached. `Dropped` below is defensive (a hand-edited or
675/// legacy-schema file could still carry one, and the typed schema should
676/// keep parsing it rather than falling into `Unmodeled`), not a claim that
677/// real sessions lose this content on every turn.
678///
679/// `patch_apply_end` and `mcp_tool_call_end`, by contrast, ARE persisted by
680/// real Codex (`should_persist_event_msg` `=> true` for both) — and here the
681/// old "already captured" framing is mostly right but not entirely: their
682/// short `stdout`/`result` text does duplicate the paired
683/// `response_item/function_call_output` or
684/// `response_item/custom_tool_call_output`. `patch_apply_end`'s
685/// `changes[path]`'s `unified_diff` is NOT duplicated there — the paired
686/// `function_call_output` only carries the apply summary text, never the
687/// diff body — and the loader does not capture it, so this is genuine,
688/// currently-real content loss on cross-format export. (N5: the diff body's
689/// raw hunk TEXT does have a counterpart — the paired `response_item/
690/// function_call.arguments` for the preceding `apply_patch` call carries the
691/// same added/removed lines in its own `*** Begin Patch` format, since
692/// that's literally what was applied. What's genuinely unique to
693/// `unified_diff` and absent from `function_call.arguments` is its
694/// standard-diff framing — the `--- a/<path>`/`+++ b/<path>`/`@@ …@@` header
695/// lines `apply_patch`'s custom patch format never emits. The dev/02 test
696/// below keys its residue assertion on those header lines specifically, not
697/// on the shared hunk body, so it proves the part that's actually
698/// unrecovered rather than merely re-finding text that was never at risk.)
699/// `Dropped` is the honest label for it, not "already captured" — and
700/// `parity12_cross_format_export_retains_tool_outputs_as_transcript_content`
701/// (dev/02, `crates/cli/tests/codex_fidelity_cli.rs`) now asserts this
702/// residue explicitly instead of staying silent about it.
703fn event_msg_coverage(sub: &str) -> Coverage {
704    match sub {
705        "agent_message"
706        | "thread_rolled_back"
707        | "thread_goal_updated"
708        | "entered_review_mode"
709        | "exited_review_mode" => Coverage::Retained,
710        _ => Coverage::Dropped,
711    }
712}
713
714fn audit_codex_item(item: &ResponseItem, raw: &Value, report: &mut Report) {
715    let raw_payload = raw.get("payload").cloned().unwrap_or(Value::Null);
716    let cov = if item.is_normalized() {
717        Coverage::Normalized
718    } else if matches!(item, ResponseItem::Reasoning { .. }) {
719        // D5/N1/N2/N3: `Session::from_codex_str` captures `summary` text,
720        // the raw `content` chain-of-thought text when genuinely present
721        // (N2), and a correctly-computed `encrypted_content` presence flag
722        // (N1: only a non-null value counts, not merely a present-but-null
723        // key) onto the NEXT assistant `ChatMessage`'s metadata
724        // (`reasoning`/`reasoning_content`/`reasoning_encrypted`) — or, when
725        // there is no following assistant turn to attach to, flushes it as
726        // its own synthesized message instead of discarding it (N3). Not a
727        // blind drop. The opaque `encrypted_content` blob itself isn't
728        // replayed cross-model, so this is an honest `Retained`, not
729        // `Normalized` (there's no 1:1 canonical "reasoning" `ChatMessage`).
730        // See `Coverage::Retained`'s doc comment for the LOADER-CAPTURE-ONLY
731        // disclosure that applies to all of this.
732        Coverage::Retained
733    } else {
734        // custom_tool_call, web_search_call, tool_search_*, image_generation,
735        // and any future Unknown — all not yet normalized.
736        Coverage::Unmodeled
737    };
738
739    let tag = item.tag().map(str::to_string).unwrap_or_else(|| {
740        raw_payload
741            .get("type")
742            .and_then(Value::as_str)
743            .unwrap_or("<no-type>")
744            .to_string()
745    });
746
747    match item {
748        ResponseItem::Message {
749            content,
750            extra,
751            role,
752        } => {
753            report.bump(format!("response_item/message[{role}]"), cov, extra);
754            audit_blocks(content, &raw_payload, report);
755        }
756        ResponseItem::FunctionCall { name, extra, .. } => {
757            *report.tools.entry(name.clone()).or_insert(0) += 1;
758            report.bump("response_item/function_call".into(), cov, extra);
759        }
760        ResponseItem::FunctionCallOutput { extra, .. } => {
761            report.bump("response_item/function_call_output".into(), cov, extra);
762        }
763        ResponseItem::CustomToolCall { name, extra, .. } => {
764            if let Some(n) = name {
765                *report.tools.entry(n.clone()).or_insert(0) += 1;
766            }
767            report.bump("response_item/custom_tool_call".into(), cov, extra);
768        }
769        other => {
770            let extra = item_extra(other);
771            report.bump(format!("response_item/{tag}"), cov, extra);
772        }
773    }
774}
775
776fn item_extra(item: &ResponseItem) -> &crate::schema::ExtraFields {
777    match item {
778        ResponseItem::CustomToolCallOutput { extra, .. }
779        | ResponseItem::Reasoning { extra }
780        | ResponseItem::WebSearchCall { extra }
781        | ResponseItem::ToolSearchCall { extra }
782        | ResponseItem::ToolSearchOutput { extra }
783        | ResponseItem::ImageGenerationCall { extra } => extra,
784        _ => EMPTY_EXTRA.get_or_init(Default::default),
785    }
786}
787
788static EMPTY_EXTRA: std::sync::OnceLock<crate::schema::ExtraFields> = std::sync::OnceLock::new();
789
790fn audit_claude_line(line: &str, report: &mut Report) {
791    let raw: Value = match serde_json::from_str(line) {
792        Ok(v) => v,
793        Err(_) => {
794            report.parse_errors += 1;
795            return;
796        }
797    };
798    let parsed: Result<ClaudeRecord, _> = serde_json::from_str(line);
799    let Ok(parsed) = parsed else {
800        report.parse_errors += 1;
801        return;
802    };
803
804    match &parsed {
805        ClaudeRecord::Unknown => {
806            let tag = raw
807                .get("type")
808                .and_then(Value::as_str)
809                .unwrap_or("<no-type>");
810            report.bump(
811                format!("<line>/{tag}"),
812                Coverage::Unmodeled,
813                &Default::default(),
814            );
815        }
816        ClaudeRecord::User { message, meta } | ClaudeRecord::Assistant { message, meta } => {
817            let role = match &parsed {
818                ClaudeRecord::Assistant { .. } => "assistant",
819                _ => "user",
820            };
821            report.bump(role.to_string(), Coverage::Normalized, &meta.extra);
822            if meta.is_sidechain {
823                report.note("sidechain (subagent) lines — flattened, not separated");
824            }
825            match &message.content {
826                MessageContent::Text(_) => {
827                    report.bump_block(
828                        &ContentBlock::Text {
829                            text: String::new(),
830                        },
831                        &Value::Null,
832                    );
833                }
834                MessageContent::Blocks(blocks) => {
835                    let raw_blocks = raw
836                        .get("message")
837                        .and_then(|m| m.get("content"))
838                        .cloned()
839                        .unwrap_or(Value::Null);
840                    audit_blocks(blocks, &Value::Null, report);
841                    let _ = raw_blocks;
842                    // capture tool names
843                    for b in blocks {
844                        if let ContentBlock::ToolUse { name, .. } = b {
845                            *report.tools.entry(name.clone()).or_insert(0) += 1;
846                        }
847                    }
848                }
849            }
850        }
851        ClaudeRecord::System { subtype, extra } => {
852            let sub = subtype.clone().unwrap_or_else(|| "?".into());
853            // Content-bearing system subtypes are now folded into the conversation.
854            let cov = match sub.as_str() {
855                "scheduled_task_fire" | "local_command" | "away_summary" => Coverage::Normalized,
856                _ => Coverage::Dropped,
857            };
858            report.bump(format!("system/{sub}"), cov, extra);
859        }
860        other => {
861            let tag = other.tag().unwrap_or("?");
862            let cov = match tag {
863                // Metadata/UI we deliberately skip.
864                "permission-mode" | "mode" | "last-prompt" | "queue-operation" | "ai-title"
865                | "pr-link" | "frame-link" | "agent-name" | "worktree-state" => Coverage::Dropped,
866                // Content-bearing attachment subtypes are now folded into the
867                // conversation (regenerable ones are still skipped).
868                "attachment" => Coverage::Normalized,
869                // PARITY-10: captured into `Session::meta.lineage` on load
870                // (`capture_claude_meta` in `session.rs`) and re-emitted
871                // verbatim by the Claude Code writer — no longer silently
872                // dropped, even though (like `attachment`) it has no slot in
873                // the OpenAI-shaped canonical message conversation itself.
874                "fork-context-ref" => Coverage::Normalized,
875                // These still carry real content/structure we don't yet use:
876                // file-history-snapshot / file-history-delta (undo state),
877                // started/result (subagent task lifecycle).
878                _ => Coverage::Unmodeled,
879            };
880            report.bump(tag.to_string(), cov, record_extra(other));
881        }
882    }
883}
884
885fn record_extra(rec: &ClaudeRecord) -> &crate::schema::ExtraFields {
886    match rec {
887        ClaudeRecord::Attachment { extra }
888        | ClaudeRecord::FileHistorySnapshot { extra }
889        | ClaudeRecord::FileHistoryDelta { extra }
890        | ClaudeRecord::AiTitle { extra }
891        | ClaudeRecord::PermissionMode { extra }
892        | ClaudeRecord::Mode { extra }
893        | ClaudeRecord::LastPrompt { extra }
894        | ClaudeRecord::QueueOperation { extra }
895        | ClaudeRecord::PrLink { extra }
896        | ClaudeRecord::FrameLink { extra }
897        | ClaudeRecord::AgentName { extra }
898        | ClaudeRecord::Started { extra }
899        | ClaudeRecord::Result { extra }
900        | ClaudeRecord::WorktreeState { extra }
901        | ClaudeRecord::ForkContextRef { extra } => extra,
902        _ => EMPTY_EXTRA.get_or_init(Default::default),
903    }
904}
905
906fn audit_blocks(blocks: &[ContentBlock], raw_payload: &Value, report: &mut Report) {
907    let raw_blocks = raw_payload.get("content").and_then(Value::as_array);
908    for (i, b) in blocks.iter().enumerate() {
909        let raw = raw_blocks
910            .and_then(|arr| arr.get(i))
911            .cloned()
912            .unwrap_or(Value::Null);
913        report.bump_block(b, &raw);
914        // PARITY-11: census any `image` block nested inside this
915        // `tool_result`'s own `content` array separately — see
916        // `audit_nested_tool_result_images`'s doc comment.
917        if let ContentBlock::ToolResult { content, .. } = b {
918            audit_nested_tool_result_images(content, report);
919        }
920    }
921}
922
923/// Audit one line of a pi session file (`docs/interop/opencode-pi-spec.md`
924/// §1.1/§4.1, `pi-fields.md`). Unlike the Claude/Codex auditors this walks
925/// raw [`Value`]s rather than a typed `crate::schema` module — Wave A scopes
926/// the typed-schema mirror to a later pass; the tally/Unknown-bucket
927/// machinery this function drives is the same [`Report`] used everywhere
928/// else, so the coverage guard test reads identically.
929///
930/// `message.role` is tallied as a **second-level discriminant** under its own
931/// `message/…` keys, with an `message/UnknownRole:<role>` bucket for any role
932/// outside pi's five modeled ones — pi's `message.role` is an OPEN,
933/// extension-mergeable union (§1.1 S6), so a role the loader doesn't
934/// recognize must surface here as a scored `Unmodeled` entry, not vanish.
935///
936/// A second, orthogonal second-level bucket — `message/UnknownImageShape` —
937/// covers FIX #2: `user`/`toolResult`/`custom` content can carry an
938/// `ImageContent` block whose `{mimeType, data}` shape is an unverified guess
939/// (`pi-fields.md` never enumerates `ImageContent`'s own fields). A block
940/// that doesn't match that shape must score `Unmodeled` here too, instead of
941/// letting the loader silently synthesize an empty/corrupt `image_url` part.
942fn audit_pi_line(line: &str, report: &mut Report) {
943    let raw: Value = match serde_json::from_str(line) {
944        Ok(v) => v,
945        Err(_) => {
946            report.parse_errors += 1;
947            return;
948        }
949    };
950    let extra = EMPTY_EXTRA.get_or_init(Default::default);
951    let Some(ty) = raw.get("type").and_then(Value::as_str) else {
952        report.bump("<line>/<no-type>".to_string(), Coverage::Unmodeled, extra);
953        return;
954    };
955    match ty {
956        "session" => report.bump("session".to_string(), Coverage::Normalized, extra),
957        "message" => {
958            let message = raw.get("message");
959            let role = message.and_then(|m| m.get("role")).and_then(Value::as_str);
960            // FIX #2: `user`/`toolResult`/`custom` all carry the shared
961            // `(TextContent|ImageContent)[]` content union (`pi-fields.md`
962            // §3a/§3c/§3e) — an `ImageContent` block that doesn't match the
963            // loader's assumed (and unverified) `{mimeType, data}` shape
964            // must score as `message/UnknownImageShape`, never silently
965            // `Normalized`, mirroring `UnknownRole`'s "surface it, don't
966            // vanish" rule exactly.
967            let content = message.and_then(|m| m.get("content"));
968            let unknown_image = matches!(role, Some("user") | Some("toolResult") | Some("custom"))
969                && pi_content_has_unknown_image_shape(content);
970            match role {
971                _ if unknown_image => report.bump(
972                    "message/UnknownImageShape".to_string(),
973                    Coverage::Unmodeled,
974                    extra,
975                ),
976                Some("user") => {
977                    report.bump("message/user".to_string(), Coverage::Normalized, extra)
978                }
979                Some("assistant") => {
980                    report.bump("message/assistant".to_string(), Coverage::Normalized, extra)
981                }
982                Some("toolResult") => report.bump(
983                    "message/toolResult".to_string(),
984                    Coverage::Normalized,
985                    extra,
986                ),
987                Some("bashExecution") => report.bump(
988                    "message/bashExecution".to_string(),
989                    Coverage::Normalized,
990                    extra,
991                ),
992                Some("custom") => {
993                    report.bump("message/custom".to_string(), Coverage::Normalized, extra)
994                }
995                Some(other) => report.bump(
996                    format!("message/UnknownRole:{other}"),
997                    Coverage::Unmodeled,
998                    extra,
999                ),
1000                None => report.bump(
1001                    "message/UnknownRole:<none>".to_string(),
1002                    Coverage::Unmodeled,
1003                    extra,
1004                ),
1005            }
1006        }
1007        "custom_message" => report.bump("custom_message".to_string(), Coverage::Normalized, extra),
1008        "compaction" => report.bump("compaction".to_string(), Coverage::Normalized, extra),
1009        "branch_summary" => report.bump("branch_summary".to_string(), Coverage::Normalized, extra),
1010        "thinking_level_change" => report.bump(
1011            "thinking_level_change".to_string(),
1012            Coverage::Dropped,
1013            extra,
1014        ),
1015        "model_change" => report.bump("model_change".to_string(), Coverage::Normalized, extra),
1016        "custom" => report.bump("custom".to_string(), Coverage::Dropped, extra),
1017        "label" => report.bump("label".to_string(), Coverage::Dropped, extra),
1018        "session_info" => report.bump("session_info".to_string(), Coverage::Normalized, extra),
1019        other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1020    }
1021}
1022
1023/// Score one Gemini CLI session record against the native loader.
1024fn audit_gemini_line(line: &str, report: &mut Report) {
1025    let raw: Value = match serde_json::from_str(line) {
1026        Ok(value) => value,
1027        Err(_) => {
1028            report.parse_errors += 1;
1029            return;
1030        }
1031    };
1032    let extra = EMPTY_EXTRA.get_or_init(Default::default);
1033    let kind = raw.get("type").and_then(Value::as_str);
1034    if kind.is_none() {
1035        let key = if raw.get("sessionId").is_some_and(Value::is_string) {
1036            "session_header"
1037        } else if raw.get("$set").is_some() {
1038            "metadata_update"
1039        } else {
1040            "<line>/<no-type>"
1041        };
1042        let coverage = if key == "<line>/<no-type>" {
1043            Coverage::Unmodeled
1044        } else {
1045            Coverage::Retained
1046        };
1047        report.bump(key.to_string(), coverage, extra);
1048        return;
1049    }
1050    match kind.unwrap_or_default() {
1051        "user" | "gemini" => {
1052            report.bump(
1053                kind.unwrap_or_default().to_string(),
1054                Coverage::Normalized,
1055                extra,
1056            );
1057            if let Some(parts) = raw.get("content").and_then(Value::as_array) {
1058                for part in parts {
1059                    if part.get("text").is_some() {
1060                        report.bump("content/text".into(), Coverage::Normalized, extra);
1061                    } else if part.get("inlineData").is_some() {
1062                        report.bump("content/inlineData".into(), Coverage::Normalized, extra);
1063                    } else if let Some(call) = part.get("functionCall") {
1064                        report.bump("content/functionCall".into(), Coverage::Normalized, extra);
1065                        if let Some(name) = call.get("name").and_then(Value::as_str) {
1066                            *report.tools.entry(name.to_string()).or_insert(0) += 1;
1067                        }
1068                    } else if let Some(response) = part.get("functionResponse") {
1069                        report.bump(
1070                            "content/functionResponse".into(),
1071                            Coverage::Normalized,
1072                            extra,
1073                        );
1074                        if let Some(name) = response.get("name").and_then(Value::as_str) {
1075                            *report.tools.entry(name.to_string()).or_insert(0) += 1;
1076                        }
1077                    } else {
1078                        report.bump("content/unknown".into(), Coverage::Unmodeled, extra);
1079                    }
1080                }
1081            } else if !raw
1082                .get("content")
1083                .is_some_and(|content| content.is_string() || content.is_null())
1084            {
1085                report.bump("content/nonstandard".into(), Coverage::Unmodeled, extra);
1086            }
1087            if raw.get("thoughts").is_some() {
1088                report.bump("thoughts".into(), Coverage::Retained, extra);
1089            }
1090        }
1091        "info" | "error" => report.bump(
1092            kind.unwrap_or_default().to_string(),
1093            Coverage::Retained,
1094            extra,
1095        ),
1096        other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1097    }
1098}
1099
1100/// Score one Grok `chat_history.jsonl` record against the same shapes the
1101/// native loader actually consumes. Companion `updates.jsonl` streams are
1102/// excluded by [`audit_dir`], because they are ACP/runtime evidence rather
1103/// than the resumable transcript.
1104fn audit_grok_line(line: &str, report: &mut Report) {
1105    let raw: Value = match serde_json::from_str(line) {
1106        Ok(value) => value,
1107        Err(_) => {
1108            report.parse_errors += 1;
1109            return;
1110        }
1111    };
1112    let extra = EMPTY_EXTRA.get_or_init(Default::default);
1113    let Some(kind) = raw.get("type").and_then(Value::as_str) else {
1114        report.bump("<line>/<no-type>".to_string(), Coverage::Unmodeled, extra);
1115        return;
1116    };
1117    match kind {
1118        "system" => {
1119            let coverage = if raw.get("content").is_some_and(Value::is_string) {
1120                Coverage::Retained
1121            } else {
1122                Coverage::Unmodeled
1123            };
1124            report.bump("system".to_string(), coverage, extra);
1125        }
1126        "user" => {
1127            let (key, coverage) = classify_grok_user(&raw);
1128            report.bump(key.clone(), coverage, extra);
1129            audit_grok_content(raw.get("content"), &key, coverage, report);
1130        }
1131        "assistant" => {
1132            let content_supported = raw
1133                .get("content")
1134                .is_none_or(|content| content.is_null() || content.is_string());
1135            report.bump(
1136                "assistant".to_string(),
1137                if content_supported {
1138                    Coverage::Normalized
1139                } else {
1140                    Coverage::Unmodeled
1141                },
1142                extra,
1143            );
1144            if let Some(calls) = raw.get("tool_calls").and_then(Value::as_array) {
1145                for call in calls {
1146                    let modeled = call.get("id").is_some_and(Value::is_string)
1147                        && call.get("name").is_some_and(Value::is_string);
1148                    report.bump(
1149                        if modeled {
1150                            "assistant/tool_call".to_string()
1151                        } else {
1152                            "assistant/tool_call:invalid".to_string()
1153                        },
1154                        if modeled {
1155                            Coverage::Normalized
1156                        } else {
1157                            Coverage::Unmodeled
1158                        },
1159                        extra,
1160                    );
1161                    if let Some(name) = call.get("name").and_then(Value::as_str) {
1162                        *report.tools.entry(name.to_string()).or_insert(0) += 1;
1163                    }
1164                }
1165            } else if raw.get("tool_calls").is_some() {
1166                report.bump(
1167                    "assistant/tool_calls:non-array".to_string(),
1168                    Coverage::Unmodeled,
1169                    extra,
1170                );
1171            }
1172        }
1173        "tool_result" => {
1174            let modeled = raw.get("tool_call_id").is_some_and(Value::is_string);
1175            report.bump(
1176                "tool_result".to_string(),
1177                if modeled {
1178                    Coverage::Normalized
1179                } else {
1180                    Coverage::Unmodeled
1181                },
1182                extra,
1183            );
1184            audit_grok_content(
1185                raw.get("content"),
1186                "tool_result",
1187                Coverage::Normalized,
1188                report,
1189            );
1190        }
1191        // These are understood native records but deliberately remain in
1192        // the byte-exact raw prefix instead of becoming replayable messages.
1193        "reasoning" | "backend_tool_call" => {
1194            report.bump(kind.to_string(), Coverage::Dropped, extra)
1195        }
1196        other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1197    }
1198}
1199
1200/// Classify a Grok `user` record using the exact replay boundary enforced by
1201/// `Session::from_grok_str`: generated context wrappers are native session
1202/// state, not human turns, and therefore remain raw-only. Separate record
1203/// keys are deliberate — [`Report::records`] stores one coverage value per
1204/// key, so mixing replayed and discarded users under a single `user` key
1205/// would make the last line scanned overwrite the truth for the whole corpus.
1206fn classify_grok_user(raw: &Value) -> (String, Coverage) {
1207    if raw.get("synthetic_reason").and_then(Value::as_str) == Some("supercode_system_event") {
1208        return ("user/system_event".to_string(), Coverage::Normalized);
1209    }
1210
1211    let content = grok_audit_text(raw.get("content"));
1212    let text = content.trim();
1213    if text.starts_with("<user_info>") {
1214        return (
1215            "user/injected_context:user_info".to_string(),
1216            Coverage::Dropped,
1217        );
1218    }
1219    if text.starts_with("<system-reminder>") {
1220        return (
1221            "user/injected_context:system-reminder".to_string(),
1222            Coverage::Dropped,
1223        );
1224    }
1225    if text.is_empty() {
1226        return ("user/empty".to_string(), Coverage::Dropped);
1227    }
1228    if text
1229        .strip_prefix("<user_query>")
1230        .and_then(|value| value.strip_suffix("</user_query>"))
1231        .is_some_and(|value| value.trim().is_empty())
1232    {
1233        return ("user/empty_query".to_string(), Coverage::Dropped);
1234    }
1235    ("user".to_string(), Coverage::Normalized)
1236}
1237
1238/// Mirror the loader's text extraction for the two organic Grok shapes:
1239/// direct string content and arrays containing either `{text: ...}` blocks
1240/// or string entries. Other scalar/object values stringify exactly as the
1241/// loader does, making the audit a behavioral classification rather than a
1242/// narrower invented schema.
1243fn grok_audit_text(content: Option<&Value>) -> String {
1244    match content {
1245        Some(Value::String(text)) => text.clone(),
1246        Some(Value::Array(items)) => items
1247            .iter()
1248            .filter_map(|item| {
1249                item.get("text")
1250                    .and_then(Value::as_str)
1251                    .or_else(|| item.as_str())
1252            })
1253            .collect::<Vec<_>>()
1254            .join("\n"),
1255        Some(other) => other.to_string(),
1256        None => String::new(),
1257    }
1258}
1259
1260fn audit_grok_content(
1261    content: Option<&Value>,
1262    prefix: &str,
1263    record_coverage: Coverage,
1264    report: &mut Report,
1265) {
1266    let extra = EMPTY_EXTRA.get_or_init(Default::default);
1267    match content {
1268        Some(Value::Array(items)) => {
1269            for item in items {
1270                let tag = item
1271                    .get("type")
1272                    .and_then(Value::as_str)
1273                    .unwrap_or("<no-type>");
1274                let modeled = item.is_string() || item.get("text").is_some_and(Value::is_string);
1275                let coverage = if record_coverage == Coverage::Dropped {
1276                    Coverage::Dropped
1277                } else if modeled {
1278                    Coverage::Normalized
1279                } else {
1280                    Coverage::Unmodeled
1281                };
1282                report.bump(format!("{prefix}/content/{tag}"), coverage, extra);
1283            }
1284        }
1285        Some(_) => report.bump(format!("{prefix}/content"), record_coverage, extra),
1286        None => report.bump(
1287            format!("{prefix}/content:<missing>"),
1288            if record_coverage == Coverage::Dropped {
1289                Coverage::Dropped
1290            } else {
1291                Coverage::Unmodeled
1292            },
1293            extra,
1294        ),
1295    }
1296}
1297
1298/// The frozen 12-part union discriminant values
1299/// (`docs/interop/research/opencode-fields.md` §3, `v1/session.ts:357-370`).
1300/// Anything outside this set is an UNKNOWN part type — never silently
1301/// dropped, always scored `Unmodeled` (§4.1's "no record/part discriminant
1302/// falls into an Unknown bucket" completeness guard).
1303const OPENCODE_KNOWN_PART_TYPES: &[&str] = &[
1304    "text",
1305    "reasoning",
1306    "tool",
1307    "file",
1308    "step-start",
1309    "step-finish",
1310    "snapshot",
1311    "patch",
1312    "agent",
1313    "subtask",
1314    "retry",
1315    "compaction",
1316];
1317
1318/// `ToolState`'s frozen discriminant values (`opencode-fields.md` §3.3,
1319/// `v1/session.ts:259-313`).
1320const OPENCODE_KNOWN_TOOL_STATUSES: &[&str] = &["pending", "running", "completed", "error"];
1321
1322/// Audit one envelope line of an OpenCode session
1323/// (`docs/interop/opencode-pi-spec.md` §1.2/§4.1): `{"key":[...],"value":...}`,
1324/// classified by the envelope `key`'s first component exactly like
1325/// [`supercode_interchange::session::Session::from_opencode_str`]. Two second-level
1326/// discriminants get their own `Unknown*` buckets, mirroring pi's
1327/// `UnknownRole`/`UnknownImageShape` discipline (S6): `message/UnknownRole:*`
1328/// for a `message` record whose `role` isn't `user`/`assistant`, and
1329/// `part/UnknownType:*` for a `part` record whose `type` isn't one of the
1330/// frozen 12 — plus a THIRD level for `tool` parts specifically,
1331/// `part/tool/UnknownStatus:*`, for a `state.status` outside the frozen
1332/// four. All three must be empty over the committed fixture + real corpus.
1333fn audit_opencode_line(line: &str, report: &mut Report) {
1334    let raw: Value = match serde_json::from_str(line) {
1335        Ok(v) => v,
1336        Err(_) => {
1337            report.parse_errors += 1;
1338            return;
1339        }
1340    };
1341    let extra = EMPTY_EXTRA.get_or_init(Default::default);
1342    let Some(key) = raw.get("key").and_then(Value::as_array) else {
1343        report.bump("<line>/<no-key>".to_string(), Coverage::Unmodeled, extra);
1344        return;
1345    };
1346    let value = raw.get("value").cloned().unwrap_or(Value::Null);
1347    let kind = key.first().and_then(Value::as_str).unwrap_or("<no-kind>");
1348    match kind {
1349        "session" => report.bump("session".to_string(), Coverage::Normalized, extra),
1350        "message" => match value.get("role").and_then(Value::as_str) {
1351            Some("user") => report.bump("message/user".to_string(), Coverage::Normalized, extra),
1352            Some("assistant") => {
1353                report.bump("message/assistant".to_string(), Coverage::Normalized, extra)
1354            }
1355            Some(other) => report.bump(
1356                format!("message/UnknownRole:{other}"),
1357                Coverage::Unmodeled,
1358                extra,
1359            ),
1360            None => report.bump(
1361                "message/UnknownRole:<none>".to_string(),
1362                Coverage::Unmodeled,
1363                extra,
1364            ),
1365        },
1366        "part" => match value.get("type").and_then(Value::as_str) {
1367            Some(t) if OPENCODE_KNOWN_PART_TYPES.contains(&t) => {
1368                if t == "tool" {
1369                    // D5: tally the tool NAME (`tool`, e.g. "bash"/"edit"),
1370                    // not just the call-status bucket — previously
1371                    // `report.tools` was always empty for opencode corpora.
1372                    if let Some(name) = value.get("tool").and_then(Value::as_str) {
1373                        *report.tools.entry(name.to_string()).or_insert(0) += 1;
1374                    }
1375                    match value
1376                        .get("state")
1377                        .and_then(|s| s.get("status"))
1378                        .and_then(Value::as_str)
1379                    {
1380                        Some(s) if OPENCODE_KNOWN_TOOL_STATUSES.contains(&s) => {
1381                            report.bump(format!("part/tool/{s}"), Coverage::Normalized, extra)
1382                        }
1383                        Some(other) => report.bump(
1384                            format!("part/tool/UnknownStatus:{other}"),
1385                            Coverage::Unmodeled,
1386                            extra,
1387                        ),
1388                        None => report.bump(
1389                            "part/tool/UnknownStatus:<none>".to_string(),
1390                            Coverage::Unmodeled,
1391                            extra,
1392                        ),
1393                    }
1394                } else if t == "text" {
1395                    // D5: an `ignored:true` text part is EXCLUDED from
1396                    // replay by design (§2.2: "must not be re-emitted to
1397                    // the model") — it is recognized and preserved in
1398                    // `raw`, but never lands in canonical `messages`, so it
1399                    // is Dropped, not Normalized. A separate discriminant
1400                    // key keeps the two counted (and displayed) apart
1401                    // rather than one overwriting the other's coverage.
1402                    let ignored = value.get("ignored").and_then(Value::as_bool) == Some(true);
1403                    if ignored {
1404                        report.bump("part/text:ignored".to_string(), Coverage::Dropped, extra);
1405                    } else {
1406                        report.bump("part/text".to_string(), Coverage::Normalized, extra);
1407                    }
1408                } else if t == "file" {
1409                    // D5: the loader only canonicalizes a `data:`-URI
1410                    // `image/*` file part into `content_parts` (the SAME
1411                    // test `opencode_file_image_part` uses, reused here so
1412                    // audit can never drift from what convert actually
1413                    // replays). An `https:` link, a bare path, a PDF, or
1414                    // any other non-image/non-data-URI file is raw-only
1415                    // residue — Dropped, not Normalized.
1416                    if opencode_file_image_part(&value).is_some() {
1417                        report.bump("part/file".to_string(), Coverage::Normalized, extra);
1418                    } else {
1419                        report.bump("part/file:residue".to_string(), Coverage::Dropped, extra);
1420                    }
1421                } else {
1422                    // compaction drives the `compacted_out` boundary —
1423                    // Normalized. reasoning feeds `metadata["thinking"]`
1424                    // (recognized, deliberately not canonical content —
1425                    // Dropped, same label Claude/Codex `thinking` blocks
1426                    // get). step-start/step-finish/snapshot/patch/agent/
1427                    // subtask/retry are recognized but have NO clean home
1428                    // at all (§2.3) — also Dropped. Only a truly
1429                    // unrecognized type is Unmodeled.
1430                    let cov = match t {
1431                        "compaction" => Coverage::Normalized,
1432                        _ => Coverage::Dropped,
1433                    };
1434                    report.bump(format!("part/{t}"), cov, extra);
1435                }
1436            }
1437            Some(other) => report.bump(
1438                format!("part/UnknownType:{other}"),
1439                Coverage::Unmodeled,
1440                extra,
1441            ),
1442            None => report.bump(
1443                "part/UnknownType:<none>".to_string(),
1444                Coverage::Unmodeled,
1445                extra,
1446            ),
1447        },
1448        "session_diff" => report.bump("session_diff".to_string(), Coverage::Normalized, extra),
1449        "todo" => report.bump("todo".to_string(), Coverage::Normalized, extra),
1450        other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1451    }
1452}
1453
1454fn jsonl_files(dir: &Path) -> Vec<PathBuf> {
1455    let mut out = Vec::new();
1456    let walker = ignore::WalkBuilder::new(dir)
1457        .standard_filters(false)
1458        .build();
1459    for entry in walker.flatten() {
1460        let p = entry.into_path();
1461        if p.extension().and_then(|e| e.to_str()) == Some("jsonl") {
1462            out.push(p);
1463        }
1464    }
1465    out.sort();
1466    out
1467}