supercode_harness/audit.rs
1//! Corpus coverage audit.
2//!
3//! Walks a directory of session logs, parses every line through the typed
4//! [`crate::schema`], and reports — with real counts — exactly what we model,
5//! what we model-but-drop on normalization, and what we don't model at all.
6//! This is the machine that turns "what's missing?" into an enumerated answer
7//! rather than a guess.
8//!
9//! ```no_run
10//! use std::path::Path;
11//! use supercode_harness::audit::{audit_dir, Corpus};
12//!
13//! let report = audit_dir(Path::new("/home/me/.codex/sessions"), Corpus::Codex, None);
14//! report.print();
15//! ```
16
17use std::collections::BTreeMap;
18use std::path::{Path, PathBuf};
19
20use serde_json::Value;
21
22use crate::schema::{claude_code::*, codex::*, raw_block_tag, ContentBlock};
23use supercode_interchange::session::{
24 opencode_file_image_part, pi_content_has_unknown_image_shape,
25};
26
27/// Which corpus a directory holds.
28#[derive(Debug, Clone, Copy, PartialEq, Eq)]
29pub enum Corpus {
30 /// `~/.claude/projects`
31 ClaudeCode,
32 /// `~/.codex/sessions`
33 Codex,
34 /// `~/.pi/agent/sessions`
35 Pi,
36 /// `~/.local/share/opencode` (envelope-form fixtures/corpus — see
37 /// `docs/interop/opencode-pi-spec.md` §1.2/§4.1).
38 OpenCode,
39 /// `~/.grok/sessions` (`chat_history.jsonl` files only; companion
40 /// `updates.jsonl` streams are live protocol events, not transcripts).
41 Grok,
42 /// `~/.gemini/tmp/<project>/chats` Gemini CLI JSONL transcripts.
43 Gemini,
44 /// Goose's `sessions/sessions.db` native store.
45 Goose,
46}
47
48/// How a given discriminant is handled by the loader.
49#[derive(Debug, Clone, Copy, PartialEq, Eq, PartialOrd, Ord)]
50pub enum Coverage {
51 /// Parsed and normalized into the canonical conversation.
52 Normalized,
53 /// PARITY-12/PARITY-13 (P012/P013): parsed and its content/provenance IS
54 /// captured by the loader — into `Session.meta` (Codex `session_meta`'s
55 /// id/cwd/model/base_instructions, `turn_context`'s model), replay
56 /// semantics (`thread_rolled_back` actually removes the rolled-back
57 /// turns, `exited_review_mode`'s `review_output.overall_explanation`
58 /// becomes a message with `review_output.findings` AND
59 /// `overall_correctness`/`overall_confidence_score` (N4) captured onto
60 /// that message's metadata (D4/N4), `thread_goal_updated`'s
61 /// `goal.objective` becomes a message with `goal.status`/
62 /// `goal.tokenBudget` captured onto that message's metadata too (D4),
63 /// `agent_message` can become a message when its text has no
64 /// `response_item` twin) — just not 1:1 into a `ChatMessage` the way
65 /// `Normalized` records are.
66 /// This is the "retained, not dropped" bucket the two audits were
67 /// missing: before this variant existed, every one of these landed in
68 /// `Dropped` indistinguishably from truly-inert UI noise (`token_count`,
69 /// `task_started`, …), which is exactly the false "silently dropped"
70 /// signal both items' dev/03 ACs flag.
71 ///
72 /// `response_item/reasoning` also belongs here (D5, PARITY-12): its
73 /// `summary` text, raw `content` chain-of-thought text when non-null
74 /// (N2 — previously dropped despite this very label claiming otherwise;
75 /// `content` is `null` on the vast majority of real turns, so this was
76 /// easy to miss until fixtures carried the key at all), and the
77 /// `encrypted_content` presence flag (N1: only a genuinely non-null
78 /// value counts — `serde_json` returns `Some(&Value::Null)` for a
79 /// present-but-null key, which is what EVERY real rollout's reasoning
80 /// item carries per upstream `codex-rs/protocol/src/models.rs:970-983`,
81 /// so a naive `.is_some()` false-flagged every reasoning item as
82 /// "encrypted" on real data) are captured by `Session::from_codex_str`
83 /// onto the *next* assistant `ChatMessage`'s metadata (`reasoning`/
84 /// `reasoning_content`/`reasoning_encrypted`) — see `audit_codex_item`'s
85 /// `Reasoning` arm. When no following assistant turn exists to attach to
86 /// (a non-assistant item interrupts, or the reasoning is dangling at
87 /// EOF — an aborted-turn shape, N3), it is flushed as its own synthesized
88 /// `[reasoning] (turn ended without a reply)` message instead of being
89 /// silently discarded, so this label stays honest for that shape too.
90 /// It isn't `Normalized` (no canonical "reasoning" `ChatMessage`), but it
91 /// is provably not a blind drop either.
92 ///
93 /// DISCLOSURE: every metadata key mentioned above (`review_findings`,
94 /// `review_overall_correctness`, `review_overall_confidence_score`,
95 /// `goal_status`, `goal_token_budget`, `reasoning`, `reasoning_content`,
96 /// `reasoning_encrypted`) is LOADER-CAPTURE ONLY. It survives the
97 /// verbatim Codex→Codex diagonal (raw bytes, untouched) and reads back
98 /// out of the native supercode format, but it does NOT survive a
99 /// cross-format writer or the `--session-id` re-serialized diagonal:
100 /// `ChatMessage.metadata` is never serialized (`message.rs:50-55`) and no
101 /// writer reads it back out. Don't misread `Retained` here as
102 /// cross-format-durable — it means "captured in-process", not "written
103 /// back out".
104 Retained,
105 /// Parsed and understood, but intentionally dropped (e.g. `token_count`,
106 /// `task_started`/`task_complete`, UI echoes of content already captured
107 /// elsewhere as `Normalized`/`Retained`). See `event_msg_coverage`'s doc
108 /// comment for the few real, currently-unrecovered exceptions (D1) —
109 /// e.g. `patch_apply_end`'s `changes[path].unified_diff` — where
110 /// `Dropped` means genuine, asserted content loss, not "duplicated
111 /// elsewhere".
112 Dropped,
113 /// Not modeled at all — falls into an `Unknown` typed bucket.
114 Unmodeled,
115}
116
117impl Coverage {
118 fn symbol(self) -> &'static str {
119 match self {
120 Coverage::Normalized => "✅ normalized",
121 Coverage::Retained => "◆ retained ",
122 Coverage::Dropped => "➖ dropped ",
123 Coverage::Unmodeled => "❌ UNMODELED ",
124 }
125 }
126}
127
128/// A tally for one discriminant value.
129#[derive(Debug, Clone, Default)]
130#[non_exhaustive]
131pub struct Tally {
132 /// How many times it occurred.
133 pub count: u64,
134 /// Field keys seen in `extra` (fields we didn't model), with counts.
135 pub unmodeled_fields: BTreeMap<String, u64>,
136}
137
138/// The full audit result.
139#[derive(Debug, Default)]
140#[non_exhaustive]
141pub struct Report {
142 /// Which corpus this is.
143 pub corpus: Option<&'static str>,
144 /// Files scanned.
145 pub files: u64,
146 /// Lines parsed.
147 pub lines: u64,
148 /// Lines that failed to deserialize even into the typed schema.
149 pub parse_errors: u64,
150 /// Per record/payload discriminant: (coverage, tally). Keyed by a readable
151 /// path like `response_item/custom_tool_call`.
152 pub records: BTreeMap<String, (Coverage, Tally)>,
153 /// Content block discriminants seen, with counts SPLIT by the coverage
154 /// each instance actually got (N1, Fable-5 review). Keyed by
155 /// `(tag, coverage)` rather than `tag` alone: D5 made `image` coverage
156 /// PER-INSTANCE (a `base64`/`url` source is `Normalized`, a Files-API/
157 /// `file` source is `Dropped`), so a single `tag -> (Coverage, count)`
158 /// entry — last-write-wins on `Coverage` — silently collapsed a mixed
159 /// corpus's genuinely-`Dropped` instances into whatever coverage the
160 /// LAST-seen instance of that tag happened to have, over- or
161 /// under-claiming fidelity depending on file order. Splitting the bucket
162 /// keeps every instance's actual coverage and never collapses counts.
163 pub blocks: BTreeMap<(String, Coverage), u64>,
164 /// Tool names seen, with counts.
165 pub tools: BTreeMap<String, u64>,
166 /// Structural notes discovered while scanning (e.g. sidechain lines).
167 pub notes: BTreeMap<String, u64>,
168}
169
170impl Report {
171 fn bump(&mut self, key: String, cov: Coverage, extra: &crate::schema::ExtraFields) {
172 let entry = self.records.entry(key).or_insert((cov, Tally::default()));
173 entry.0 = cov;
174 entry.1.count += 1;
175 for k in extra.keys() {
176 *entry.1.unmodeled_fields.entry(k.clone()).or_insert(0) += 1;
177 }
178 }
179
180 fn bump_block(&mut self, block: &ContentBlock, raw: &Value) {
181 let (tag, cov) = match block.tag() {
182 Some(t) => (t.to_string(), block_coverage(block)),
183 None => (
184 raw_block_tag(raw).unwrap_or_else(|| "<no-type>".into()),
185 Coverage::Unmodeled,
186 ),
187 };
188 // N1: bucket on (tag, coverage), not tag alone — see the `blocks`
189 // field doc. Each instance is counted under its OWN actual coverage
190 // instead of one shared, last-write-wins `Coverage` per tag.
191 *self.blocks.entry((tag, cov)).or_insert(0) += 1;
192 }
193
194 fn note(&mut self, key: &str) {
195 *self.notes.entry(key.to_string()).or_insert(0) += 1;
196 }
197
198 /// Serialize the report as structured JSON (for CI/dashboards).
199 pub fn to_json(&self) -> serde_json::Value {
200 let cov = |c: Coverage| match c {
201 Coverage::Normalized => "normalized",
202 Coverage::Retained => "retained",
203 Coverage::Dropped => "dropped",
204 Coverage::Unmodeled => "unmodeled",
205 };
206 let records: serde_json::Map<String, serde_json::Value> = self
207 .records
208 .iter()
209 .map(|(k, (c, t))| {
210 (
211 k.clone(),
212 serde_json::json!({
213 "coverage": cov(*c),
214 "count": t.count,
215 "unmodeled_fields": t.unmodeled_fields.keys().collect::<Vec<_>>(),
216 }),
217 )
218 })
219 .collect();
220 // N1: a tag can now have MULTIPLE coverage buckets (e.g. `image` ->
221 // Normalized:1, Dropped:1 on a mixed corpus), so each tag maps to a
222 // list of `{coverage, count}` entries rather than a single one.
223 let mut blocks_by_tag: BTreeMap<&str, Vec<serde_json::Value>> = BTreeMap::new();
224 for ((tag, c), n) in &self.blocks {
225 blocks_by_tag
226 .entry(tag.as_str())
227 .or_default()
228 .push(serde_json::json!({"coverage": cov(*c), "count": n}));
229 }
230 let blocks: serde_json::Map<String, serde_json::Value> = blocks_by_tag
231 .into_iter()
232 .map(|(k, v)| (k.to_string(), serde_json::Value::Array(v)))
233 .collect();
234 serde_json::json!({
235 "corpus": self.corpus,
236 "files": self.files,
237 "lines": self.lines,
238 "parse_errors": self.parse_errors,
239 "records": records,
240 "blocks": blocks,
241 "tools": self.tools,
242 "notes": self.notes,
243 })
244 }
245
246 /// Print a human-readable report to stdout.
247 pub fn print(&self) {
248 println!("# Coverage audit: {}", self.corpus.unwrap_or("?"));
249 println!(
250 "files={} lines={} parse_errors={}\n",
251 self.files, self.lines, self.parse_errors
252 );
253
254 println!("## Records (discriminant → coverage, count, unmodeled fields)");
255 for (key, (cov, tally)) in &self.records {
256 print!(" {} {:<40} {:>9}", cov.symbol(), key, tally.count);
257 if !tally.unmodeled_fields.is_empty() {
258 let mut fields: Vec<_> = tally.unmodeled_fields.keys().cloned().collect();
259 fields.sort();
260 print!(" unmodeled fields: {}", fields.join(", "));
261 }
262 println!();
263 }
264
265 if !self.blocks.is_empty() {
266 println!("\n## Content blocks");
267 // N1: one row per (tag, coverage) bucket — a tag with mixed
268 // coverage (e.g. `image` seen both Normalized and Dropped) now
269 // prints as two distinct, honestly-counted rows instead of one
270 // row whose coverage was whichever instance was seen last.
271 for ((tag, cov), count) in &self.blocks {
272 println!(" {} {:<28} {:>9}", cov.symbol(), tag, count);
273 }
274 }
275
276 if !self.notes.is_empty() {
277 println!("\n## Structural notes");
278 for (k, v) in &self.notes {
279 println!(" {k}: {v}");
280 }
281 }
282
283 if !self.tools.is_empty() {
284 println!("\n## Tools observed (top 30 by frequency)");
285 let mut tools: Vec<_> = self.tools.iter().collect();
286 tools.sort_by(|a, b| b.1.cmp(a.1));
287 for (name, count) in tools.into_iter().take(30) {
288 println!(" {count:>9} {name}");
289 }
290 }
291
292 println!("\n## Summary of gaps (UNMODELED or dropped, non-UI)");
293 for (key, (cov, tally)) in &self.records {
294 if *cov == Coverage::Unmodeled {
295 println!(" ❌ {key} ({} occurrences) — not modeled", tally.count);
296 }
297 }
298 for ((tag, cov), count) in &self.blocks {
299 if *cov == Coverage::Unmodeled {
300 println!(" ❌ content block `{tag}` ({count}) — not modeled");
301 }
302 }
303 }
304}
305
306fn block_coverage(block: &ContentBlock) -> Coverage {
307 match block {
308 ContentBlock::Text { .. }
309 | ContentBlock::InputText { .. }
310 | ContentBlock::OutputText { .. }
311 | ContentBlock::ToolUse { .. }
312 | ContentBlock::ToolResult { .. } => Coverage::Normalized,
313 // D5 (Fable-5 review, confirmed): `image` used to be blanket-marked
314 // `Normalized` regardless of its `source` shape, but
315 // `claude_image_block_to_part` (session.rs) only actually converts
316 // `base64`/`url` sources into a replayable `content_parts` image —
317 // anything else (a Files-API `{"source":{"type":"file",...}}`
318 // reference, most commonly) is NOT carried through; the loader now
319 // emits a bracketed marker so the record survives (see
320 // `UNCONVERTIBLE_IMAGE_MARKER`), but the actual image content is
321 // still lost, so this must not claim full fidelity. Codex's
322 // `input_image` has no `source` sub-object (a bare, always-
323 // convertible `image_url` string via `codex_extract_images`) and
324 // `fallback` (folded into a text marker) are both still genuinely
325 // `Normalized`.
326 // N3 (Fable-5 review, ticket, fixed inline since it's the same
327 // `Image{source}` inspection N1 already touches): a well-typed but
328 // EMPTY `base64`/`url` source — e.g. `{"type":"base64","data":""}`
329 // — used to blanket-audit as `Normalized` just like a genuinely
330 // convertible one, but `claude_image_block_to_part` (session.rs)
331 // treats it as UNCONVERTIBLE (its own non-empty `mime`/`data`/`url`
332 // check returns `None`, same `UNCONVERTIBLE_IMAGE_MARKER` fallback
333 // path as a Files-API reference) — audit and loader must agree.
334 ContentBlock::Image { source } => image_source_coverage(source),
335 ContentBlock::InputImage { .. } | ContentBlock::Fallback { .. } => Coverage::Normalized,
336 // Provider-private reasoning: retained verbatim in
337 // (skip-serialized) `ChatMessage` metadata (`push_claude_assistant`)
338 // so a same-model continuation can replay it, but it has no slot in
339 // the canonical replayable conversation itself — "understood, not
340 // silently lost" rather than "normalized into the conversation".
341 ContentBlock::Thinking { .. } | ContentBlock::RedactedThinking { .. } => Coverage::Dropped,
342 ContentBlock::Unknown => Coverage::Unmodeled, // future blocks
343 }
344}
345
346/// Score a Claude `image` block's `source` object — factored out of
347/// `block_coverage`'s `Image` arm (D5/N3 discipline: `base64`/`url` with a
348/// non-empty payload is `Normalized`, anything else is `Dropped`) so PARITY-11
349/// can reuse the EXACT same test for an `image` block nested inside a
350/// `tool_result`'s own `content` array, not just a top-level one.
351fn image_source_coverage(source: &Value) -> Coverage {
352 match source.get("type").and_then(Value::as_str) {
353 Some("base64") => {
354 let mime = source
355 .get("media_type")
356 .and_then(Value::as_str)
357 .unwrap_or("");
358 let data = source.get("data").and_then(Value::as_str).unwrap_or("");
359 if mime.is_empty() || data.is_empty() {
360 Coverage::Dropped
361 } else {
362 Coverage::Normalized
363 }
364 }
365 Some("url") => {
366 let url = source.get("url").and_then(Value::as_str).unwrap_or("");
367 if url.is_empty() {
368 Coverage::Dropped
369 } else {
370 Coverage::Normalized
371 }
372 }
373 _ => Coverage::Dropped,
374 }
375}
376
377/// PARITY-11 (nested images, skeptic-confirmed on a real session): a Claude
378/// `tool_result` block's OWN `content` array can carry `image` blocks — the
379/// everyday "Read a PNG / screenshot tool output" shape. `block_coverage`
380/// blanket-labels the enclosing `tool_result` `Normalized` (true for its text
381/// portion), which used to be the ONLY signal `audit` gave — so a session
382/// whose `tool_result` held nothing but a dropped image still reported zero
383/// `image` blocks and a clean `tool_result: Normalized` line, i.e. coverage
384/// said "retained" while the loader silently dropped the bytes. This censuses
385/// each nested `image` block individually, under its own `tool_result/image`
386/// discriminant, scored with the SAME [`image_source_coverage`] test
387/// `session.rs`'s `extract_tool_result_content` uses to decide whether it
388/// actually captures the block into `content_parts` — so a genuinely
389/// unconvertible nested image (Files-API reference, empty payload, …) shows
390/// up here as `Dropped`, not folded invisibly into the outer `Normalized`
391/// tally.
392fn audit_nested_tool_result_images(content: &Value, report: &mut Report) {
393 let Some(items) = content.as_array() else {
394 return;
395 };
396 for item in items {
397 if item.get("type").and_then(Value::as_str) != Some("image") {
398 continue;
399 }
400 let cov = image_source_coverage(item.get("source").unwrap_or(&Value::Null));
401 *report
402 .blocks
403 .entry(("tool_result/image".to_string(), cov))
404 .or_insert(0) += 1;
405 }
406}
407
408/// Audit a directory. `limit` caps the number of files scanned (None = all).
409///
410/// `Corpus::OpenCode` (PARITY-4) is special-cased: a real OpenCode data root
411/// (`~/.local/share/opencode`) holds no `.jsonl` files at all — sessions live
412/// in `opencode*.db` (current installs) or a JSON-file tree (legacy). When
413/// [`supercode_interchange::session::detect_opencode_storage_surface`] resolves `dir` to the
414/// SQLite surface, this routes through
415/// [`supercode_interchange::session::opencode_sqlite_corpus_envelope_text`] (up to `limit`
416/// SESSIONS, not files — `report.files` counts sessions scanned in that
417/// case) instead of the `jsonl_files` walk below, so a real store actually
418/// gets audited rather than silently reporting zero files/lines. A directory
419/// with no detected SQLite surface (e.g. a fixture dir of committed
420/// envelope-form `.jsonl` files, or a not-yet-implemented legacy JSON tree)
421/// falls back to the original file-walk unchanged.
422pub fn audit_dir(dir: &Path, corpus: Corpus, limit: Option<usize>) -> Report {
423 let mut report = Report {
424 corpus: Some(match corpus {
425 Corpus::ClaudeCode => "claude-code",
426 Corpus::Codex => "codex",
427 Corpus::Pi => "pi",
428 Corpus::OpenCode => "opencode",
429 Corpus::Grok => "grok",
430 Corpus::Gemini => "gemini",
431 Corpus::Goose => "goose",
432 }),
433 ..Default::default()
434 };
435
436 if corpus == Corpus::OpenCode {
437 if let Some((supercode_interchange::session::OpenCodeStorageSurface::Sqlite, db_path)) =
438 supercode_interchange::session::detect_opencode_storage_surface(dir)
439 {
440 return audit_opencode_sqlite(&db_path, limit, report);
441 }
442 }
443 if corpus == Corpus::Goose {
444 return audit_goose(dir, limit, report);
445 }
446
447 let mut files = jsonl_files(dir);
448 if corpus == Corpus::Grok {
449 files.retain(|path| {
450 path.file_name().and_then(|name| name.to_str()) == Some("chat_history.jsonl")
451 });
452 }
453 let files = match limit {
454 Some(n) => &files[..files.len().min(n)],
455 None => &files[..],
456 };
457
458 for path in files {
459 report.files += 1;
460 let Ok(text) = std::fs::read_to_string(path) else {
461 continue;
462 };
463 for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
464 report.lines += 1;
465 match corpus {
466 Corpus::Codex => audit_codex_line(line, &mut report),
467 Corpus::ClaudeCode => audit_claude_line(line, &mut report),
468 Corpus::Pi => audit_pi_line(line, &mut report),
469 Corpus::OpenCode => audit_opencode_line(line, &mut report),
470 Corpus::Grok => audit_grok_line(line, &mut report),
471 Corpus::Gemini => audit_gemini_line(line, &mut report),
472 Corpus::Goose => unreachable!("Goose is audited through its SQLite store"),
473 }
474 }
475 }
476 report
477}
478
479fn audit_goose(root: &Path, limit: Option<usize>, mut report: Report) -> Report {
480 let catalog = crate::HarnessCatalog::new();
481 let discovery = catalog.discover(&crate::DiscoveryQuery {
482 harnesses: vec![crate::HarnessId::from(crate::HarnessId::GOOSE)],
483 homes: crate::HarnessHomes {
484 goose: root.to_path_buf(),
485 ..crate::HarnessHomes::default()
486 },
487 limit,
488 ..crate::DiscoveryQuery::default()
489 });
490 let descriptors = match discovery {
491 Ok(descriptors) => descriptors,
492 Err(error) => {
493 report.note(&format!("Goose discovery failed: {error}"));
494 return report;
495 }
496 };
497 let extra = EMPTY_EXTRA.get_or_init(Default::default);
498 for descriptor in descriptors {
499 let session = match catalog.load(&descriptor.locator) {
500 Ok(session) => session,
501 Err(error) => {
502 report.note(&format!(
503 "Goose session {} failed to load: {error}",
504 descriptor.locator.session_id
505 ));
506 continue;
507 }
508 };
509 report.files += 1;
510 for message in session.messages {
511 report.lines += 1;
512 report.bump(
513 format!("message/{:?}", message.role).to_lowercase(),
514 Coverage::Normalized,
515 extra,
516 );
517 for call in message.tool_calls() {
518 *report.tools.entry(call.function.name.clone()).or_insert(0) += 1;
519 }
520 }
521 }
522 report
523}
524
525/// The `Corpus::OpenCode` + SQLite branch of [`audit_dir`] (PARITY-4): reads
526/// every session's `session`/`message`/`part`/`todo` records out of
527/// `db_path` as envelope lines
528/// ([`supercode_interchange::session::opencode_sqlite_corpus_envelope_text`]) and scores each
529/// one exactly like a line from a committed envelope-form fixture
530/// (`audit_opencode_line` — same classifier, same coverage buckets, so a
531/// SQLite corpus and a JSON-tree/fixture corpus are held to the identical
532/// bar). `report.files` counts SESSIONS scanned (the natural unit for a
533/// single-DB corpus), not `.jsonl` files. A store that fails to open (bad
534/// path, corrupt DB, wrong schema) does not panic or silently return an
535/// empty report — the failure is recorded in `report.notes` so it is visible
536/// in both the text and `--json` renderings.
537fn audit_opencode_sqlite(db_path: &Path, limit: Option<usize>, mut report: Report) -> Report {
538 match supercode_interchange::session::opencode_sqlite_corpus_envelope_text(db_path, limit) {
539 Ok(text) => {
540 for line in text.lines().map(str::trim).filter(|l| !l.is_empty()) {
541 report.lines += 1;
542 // A `session` envelope line is one-per-scanned-session — use
543 // it to derive `report.files` (sessions, not `.jsonl` files)
544 // without a second SQL pass.
545 if let Ok(v) = serde_json::from_str::<Value>(line) {
546 if v.get("key")
547 .and_then(Value::as_array)
548 .and_then(|k| k.first())
549 .and_then(Value::as_str)
550 == Some("session")
551 {
552 report.files += 1;
553 }
554 }
555 audit_opencode_line(line, &mut report);
556 }
557 }
558 Err(e) => {
559 report.note(&format!("opencode_sqlite_error: {e}"));
560 }
561 }
562 report
563}
564
565fn audit_codex_line(line: &str, report: &mut Report) {
566 let raw: Value = match serde_json::from_str(line) {
567 Ok(v) => v,
568 Err(_) => {
569 report.parse_errors += 1;
570 return;
571 }
572 };
573 let parsed: Result<CodexLine, _> = serde_json::from_str(line);
574 let Ok(parsed) = parsed else {
575 report.parse_errors += 1;
576 return;
577 };
578
579 match &parsed.record {
580 CodexRecord::Unknown => {
581 let tag = raw
582 .get("type")
583 .and_then(Value::as_str)
584 .unwrap_or("<no-type>");
585 report.bump(
586 format!("<line>/{tag}"),
587 Coverage::Unmodeled,
588 &Default::default(),
589 );
590 }
591 CodexRecord::ResponseItem { payload } => audit_codex_item(payload, &raw, report),
592 CodexRecord::EventMsg { payload } => {
593 let sub = payload.kind.clone().unwrap_or_else(|| "?".into());
594 report.bump(
595 format!("event_msg/{sub}"),
596 event_msg_coverage(&sub),
597 &payload.extra,
598 );
599 }
600 // PARITY-13 (P013): `session_meta` isn't discarded — `from_codex_str`
601 // (via `capture_codex_session_meta`) threads its id/cwd/model/
602 // base_instructions/lineage fields onto `Session.meta`, and the whole
603 // record is kept verbatim in `meta.codex_headers` so a same-format
604 // re-export (`to_codex_jsonl`) replays it byte-for-byte. `Dropped`
605 // read as a silent, untracked loss; it's retained, just not folded
606 // into a `ChatMessage`.
607 CodexRecord::SessionMeta { payload } => {
608 report.bump("session_meta".into(), Coverage::Retained, &payload.extra);
609 }
610 // Same reasoning as `SessionMeta` above: `turn_context`'s `model` is
611 // threaded onto `Session.meta.model` (first occurrence) and the
612 // record itself is kept verbatim in `meta.codex_headers` (P013).
613 CodexRecord::TurnContext { payload } => {
614 report.bump("turn_context".into(), Coverage::Retained, &payload.extra);
615 }
616 CodexRecord::Compacted { .. } => {
617 // replacement_history now replaces prior turns on load.
618 report.bump(
619 "compacted".into(),
620 Coverage::Normalized,
621 &Default::default(),
622 );
623 }
624 }
625}
626
627/// PARITY-12/PARITY-13 (P012/P013): `event_msg` coverage, mirroring EXACTLY
628/// the subset of `payload.type` values `Session::from_codex_str` special-cases
629/// (see its `Some("event_msg") if payload.get("type") == Some(...)` arms) —
630/// keep these two lists in lockstep; a subtype added there without a match
631/// here regresses to a false `Dropped` again.
632///
633/// - `agent_message`: real assistant narration with no `response_item`
634/// counterpart becomes a message (the sole source of truth in some
635/// collaboration/multi-agent sessions); a duplicate of an already-normalized
636/// `response_item/message` is skipped as a no-op. Either way the loader
637/// parses and acts on it — not a blind, untracked drop.
638/// - `thread_rolled_back`: directly mutates the canonical conversation
639/// (removes the rolled-back turns) — collaboration/undo provenance that's
640/// applied, not discarded.
641/// - `thread_goal_updated`: its `goal.objective` becomes a synthesized
642/// system message when present; `goal.status`/`goal.tokenBudget` (when
643/// present) are captured onto that message's metadata too (D4) —
644/// `goal.tokensUsed`/`timeUsedSeconds`/timestamps are still real, minor
645/// residue, not claimed as retained.
646/// - `exited_review_mode`: its `review_output.overall_explanation` becomes a
647/// synthesized assistant message; `review_output.findings` (verbatim JSON:
648/// `title`/`body`/`confidence_score`/`priority`/`code_location`) AND the
649/// review verdict itself, `overall_correctness`/`overall_confidence_score`
650/// (N4 — previously neither captured nor disclosed, unlike the
651/// `thread_goal_updated` arm above which already disclosed its own
652/// residue), are captured onto that message's metadata too (D4/N4) — the
653/// only place review-mode findings/verdict live.
654///
655/// `token_count`, `task_started`/`task_complete`, `user_message` (a
656/// duplicate of the already-normalized `response_item/message[user]`),
657/// `entered_review_mode` has no canonical chat turn, but PARITY-13's portable
658/// Codex provenance envelope now retains its exact target/hint record across
659/// every foreign-format hop, so it is `Retained` rather than an untracked
660/// drop. Other UI-only echoes with no unique replayable content stay
661/// `Dropped` honestly.
662///
663/// D1 correction — this used to also claim `exec_command_begin`/`end` and
664/// `mcp_tool_call_begin`/`patch_apply_begin` were safe to drop because
665/// "already captured via the paired `response_item/function_call*`". That
666/// framing was FALSE for what those events would carry if they were ever
667/// actually present: verified against upstream `openai/codex`'s
668/// `codex-rs/rollout/src/policy.rs` `should_persist_event_msg`, all four of
669/// `EventMsg::ExecCommandBegin`, `EventMsg::ExecCommandEnd`,
670/// `EventMsg::McpToolCallBegin`, and `EventMsg::PatchApplyBegin` hit that
671/// function's `=> false` arm — **codex never writes these event kinds to a
672/// real rollout file at all.** So in a genuine `~/.codex/sessions` corpus
673/// this isn't "content safely captured elsewhere"; it's a branch that is
674/// simply never reached. `Dropped` below is defensive (a hand-edited or
675/// legacy-schema file could still carry one, and the typed schema should
676/// keep parsing it rather than falling into `Unmodeled`), not a claim that
677/// real sessions lose this content on every turn.
678///
679/// `patch_apply_end` and `mcp_tool_call_end`, by contrast, ARE persisted by
680/// real Codex (`should_persist_event_msg` `=> true` for both) — and here the
681/// old "already captured" framing is mostly right but not entirely: their
682/// short `stdout`/`result` text does duplicate the paired
683/// `response_item/function_call_output` or
684/// `response_item/custom_tool_call_output`. `patch_apply_end`'s
685/// `changes[path]`'s `unified_diff` is NOT duplicated there — the paired
686/// `function_call_output` only carries the apply summary text, never the
687/// diff body — and the loader does not capture it, so this is genuine,
688/// currently-real content loss on cross-format export. (N5: the diff body's
689/// raw hunk TEXT does have a counterpart — the paired `response_item/
690/// function_call.arguments` for the preceding `apply_patch` call carries the
691/// same added/removed lines in its own `*** Begin Patch` format, since
692/// that's literally what was applied. What's genuinely unique to
693/// `unified_diff` and absent from `function_call.arguments` is its
694/// standard-diff framing — the `--- a/<path>`/`+++ b/<path>`/`@@ …@@` header
695/// lines `apply_patch`'s custom patch format never emits. The dev/02 test
696/// below keys its residue assertion on those header lines specifically, not
697/// on the shared hunk body, so it proves the part that's actually
698/// unrecovered rather than merely re-finding text that was never at risk.)
699/// `Dropped` is the honest label for it, not "already captured" — and
700/// `parity12_cross_format_export_retains_tool_outputs_as_transcript_content`
701/// (dev/02, `crates/cli/tests/codex_fidelity_cli.rs`) now asserts this
702/// residue explicitly instead of staying silent about it.
703fn event_msg_coverage(sub: &str) -> Coverage {
704 match sub {
705 "agent_message"
706 | "thread_rolled_back"
707 | "thread_goal_updated"
708 | "entered_review_mode"
709 | "exited_review_mode" => Coverage::Retained,
710 _ => Coverage::Dropped,
711 }
712}
713
714fn audit_codex_item(item: &ResponseItem, raw: &Value, report: &mut Report) {
715 let raw_payload = raw.get("payload").cloned().unwrap_or(Value::Null);
716 let cov = if item.is_normalized() {
717 Coverage::Normalized
718 } else if matches!(item, ResponseItem::Reasoning { .. }) {
719 // D5/N1/N2/N3: `Session::from_codex_str` captures `summary` text,
720 // the raw `content` chain-of-thought text when genuinely present
721 // (N2), and a correctly-computed `encrypted_content` presence flag
722 // (N1: only a non-null value counts, not merely a present-but-null
723 // key) onto the NEXT assistant `ChatMessage`'s metadata
724 // (`reasoning`/`reasoning_content`/`reasoning_encrypted`) — or, when
725 // there is no following assistant turn to attach to, flushes it as
726 // its own synthesized message instead of discarding it (N3). Not a
727 // blind drop. The opaque `encrypted_content` blob itself isn't
728 // replayed cross-model, so this is an honest `Retained`, not
729 // `Normalized` (there's no 1:1 canonical "reasoning" `ChatMessage`).
730 // See `Coverage::Retained`'s doc comment for the LOADER-CAPTURE-ONLY
731 // disclosure that applies to all of this.
732 Coverage::Retained
733 } else {
734 // custom_tool_call, web_search_call, tool_search_*, image_generation,
735 // and any future Unknown — all not yet normalized.
736 Coverage::Unmodeled
737 };
738
739 let tag = item.tag().map(str::to_string).unwrap_or_else(|| {
740 raw_payload
741 .get("type")
742 .and_then(Value::as_str)
743 .unwrap_or("<no-type>")
744 .to_string()
745 });
746
747 match item {
748 ResponseItem::Message {
749 content,
750 extra,
751 role,
752 } => {
753 report.bump(format!("response_item/message[{role}]"), cov, extra);
754 audit_blocks(content, &raw_payload, report);
755 }
756 ResponseItem::FunctionCall { name, extra, .. } => {
757 *report.tools.entry(name.clone()).or_insert(0) += 1;
758 report.bump("response_item/function_call".into(), cov, extra);
759 }
760 ResponseItem::FunctionCallOutput { extra, .. } => {
761 report.bump("response_item/function_call_output".into(), cov, extra);
762 }
763 ResponseItem::CustomToolCall { name, extra, .. } => {
764 if let Some(n) = name {
765 *report.tools.entry(n.clone()).or_insert(0) += 1;
766 }
767 report.bump("response_item/custom_tool_call".into(), cov, extra);
768 }
769 other => {
770 let extra = item_extra(other);
771 report.bump(format!("response_item/{tag}"), cov, extra);
772 }
773 }
774}
775
776fn item_extra(item: &ResponseItem) -> &crate::schema::ExtraFields {
777 match item {
778 ResponseItem::CustomToolCallOutput { extra, .. }
779 | ResponseItem::Reasoning { extra }
780 | ResponseItem::WebSearchCall { extra }
781 | ResponseItem::ToolSearchCall { extra }
782 | ResponseItem::ToolSearchOutput { extra }
783 | ResponseItem::ImageGenerationCall { extra } => extra,
784 _ => EMPTY_EXTRA.get_or_init(Default::default),
785 }
786}
787
788static EMPTY_EXTRA: std::sync::OnceLock<crate::schema::ExtraFields> = std::sync::OnceLock::new();
789
790fn audit_claude_line(line: &str, report: &mut Report) {
791 let raw: Value = match serde_json::from_str(line) {
792 Ok(v) => v,
793 Err(_) => {
794 report.parse_errors += 1;
795 return;
796 }
797 };
798 let parsed: Result<ClaudeRecord, _> = serde_json::from_str(line);
799 let Ok(parsed) = parsed else {
800 report.parse_errors += 1;
801 return;
802 };
803
804 match &parsed {
805 ClaudeRecord::Unknown => {
806 let tag = raw
807 .get("type")
808 .and_then(Value::as_str)
809 .unwrap_or("<no-type>");
810 report.bump(
811 format!("<line>/{tag}"),
812 Coverage::Unmodeled,
813 &Default::default(),
814 );
815 }
816 ClaudeRecord::User { message, meta } | ClaudeRecord::Assistant { message, meta } => {
817 let role = match &parsed {
818 ClaudeRecord::Assistant { .. } => "assistant",
819 _ => "user",
820 };
821 report.bump(role.to_string(), Coverage::Normalized, &meta.extra);
822 if meta.is_sidechain {
823 report.note("sidechain (subagent) lines — flattened, not separated");
824 }
825 match &message.content {
826 MessageContent::Text(_) => {
827 report.bump_block(
828 &ContentBlock::Text {
829 text: String::new(),
830 },
831 &Value::Null,
832 );
833 }
834 MessageContent::Blocks(blocks) => {
835 let raw_blocks = raw
836 .get("message")
837 .and_then(|m| m.get("content"))
838 .cloned()
839 .unwrap_or(Value::Null);
840 audit_blocks(blocks, &Value::Null, report);
841 let _ = raw_blocks;
842 // capture tool names
843 for b in blocks {
844 if let ContentBlock::ToolUse { name, .. } = b {
845 *report.tools.entry(name.clone()).or_insert(0) += 1;
846 }
847 }
848 }
849 }
850 }
851 ClaudeRecord::System { subtype, extra } => {
852 let sub = subtype.clone().unwrap_or_else(|| "?".into());
853 // Content-bearing system subtypes are now folded into the conversation.
854 let cov = match sub.as_str() {
855 "scheduled_task_fire" | "local_command" | "away_summary" => Coverage::Normalized,
856 _ => Coverage::Dropped,
857 };
858 report.bump(format!("system/{sub}"), cov, extra);
859 }
860 other => {
861 let tag = other.tag().unwrap_or("?");
862 let cov = match tag {
863 // Metadata/UI we deliberately skip.
864 "permission-mode" | "mode" | "last-prompt" | "queue-operation" | "ai-title"
865 | "pr-link" | "frame-link" | "agent-name" | "worktree-state" => Coverage::Dropped,
866 // Content-bearing attachment subtypes are now folded into the
867 // conversation (regenerable ones are still skipped).
868 "attachment" => Coverage::Normalized,
869 // PARITY-10: captured into `Session::meta.lineage` on load
870 // (`capture_claude_meta` in `session.rs`) and re-emitted
871 // verbatim by the Claude Code writer — no longer silently
872 // dropped, even though (like `attachment`) it has no slot in
873 // the OpenAI-shaped canonical message conversation itself.
874 "fork-context-ref" => Coverage::Normalized,
875 // These still carry real content/structure we don't yet use:
876 // file-history-snapshot / file-history-delta (undo state),
877 // started/result (subagent task lifecycle).
878 _ => Coverage::Unmodeled,
879 };
880 report.bump(tag.to_string(), cov, record_extra(other));
881 }
882 }
883}
884
885fn record_extra(rec: &ClaudeRecord) -> &crate::schema::ExtraFields {
886 match rec {
887 ClaudeRecord::Attachment { extra }
888 | ClaudeRecord::FileHistorySnapshot { extra }
889 | ClaudeRecord::FileHistoryDelta { extra }
890 | ClaudeRecord::AiTitle { extra }
891 | ClaudeRecord::PermissionMode { extra }
892 | ClaudeRecord::Mode { extra }
893 | ClaudeRecord::LastPrompt { extra }
894 | ClaudeRecord::QueueOperation { extra }
895 | ClaudeRecord::PrLink { extra }
896 | ClaudeRecord::FrameLink { extra }
897 | ClaudeRecord::AgentName { extra }
898 | ClaudeRecord::Started { extra }
899 | ClaudeRecord::Result { extra }
900 | ClaudeRecord::WorktreeState { extra }
901 | ClaudeRecord::ForkContextRef { extra } => extra,
902 _ => EMPTY_EXTRA.get_or_init(Default::default),
903 }
904}
905
906fn audit_blocks(blocks: &[ContentBlock], raw_payload: &Value, report: &mut Report) {
907 let raw_blocks = raw_payload.get("content").and_then(Value::as_array);
908 for (i, b) in blocks.iter().enumerate() {
909 let raw = raw_blocks
910 .and_then(|arr| arr.get(i))
911 .cloned()
912 .unwrap_or(Value::Null);
913 report.bump_block(b, &raw);
914 // PARITY-11: census any `image` block nested inside this
915 // `tool_result`'s own `content` array separately — see
916 // `audit_nested_tool_result_images`'s doc comment.
917 if let ContentBlock::ToolResult { content, .. } = b {
918 audit_nested_tool_result_images(content, report);
919 }
920 }
921}
922
923/// Audit one line of a pi session file (`docs/interop/opencode-pi-spec.md`
924/// §1.1/§4.1, `pi-fields.md`). Unlike the Claude/Codex auditors this walks
925/// raw [`Value`]s rather than a typed `crate::schema` module — Wave A scopes
926/// the typed-schema mirror to a later pass; the tally/Unknown-bucket
927/// machinery this function drives is the same [`Report`] used everywhere
928/// else, so the coverage guard test reads identically.
929///
930/// `message.role` is tallied as a **second-level discriminant** under its own
931/// `message/…` keys, with an `message/UnknownRole:<role>` bucket for any role
932/// outside pi's five modeled ones — pi's `message.role` is an OPEN,
933/// extension-mergeable union (§1.1 S6), so a role the loader doesn't
934/// recognize must surface here as a scored `Unmodeled` entry, not vanish.
935///
936/// A second, orthogonal second-level bucket — `message/UnknownImageShape` —
937/// covers FIX #2: `user`/`toolResult`/`custom` content can carry an
938/// `ImageContent` block whose `{mimeType, data}` shape is an unverified guess
939/// (`pi-fields.md` never enumerates `ImageContent`'s own fields). A block
940/// that doesn't match that shape must score `Unmodeled` here too, instead of
941/// letting the loader silently synthesize an empty/corrupt `image_url` part.
942fn audit_pi_line(line: &str, report: &mut Report) {
943 let raw: Value = match serde_json::from_str(line) {
944 Ok(v) => v,
945 Err(_) => {
946 report.parse_errors += 1;
947 return;
948 }
949 };
950 let extra = EMPTY_EXTRA.get_or_init(Default::default);
951 let Some(ty) = raw.get("type").and_then(Value::as_str) else {
952 report.bump("<line>/<no-type>".to_string(), Coverage::Unmodeled, extra);
953 return;
954 };
955 match ty {
956 "session" => report.bump("session".to_string(), Coverage::Normalized, extra),
957 "message" => {
958 let message = raw.get("message");
959 let role = message.and_then(|m| m.get("role")).and_then(Value::as_str);
960 // FIX #2: `user`/`toolResult`/`custom` all carry the shared
961 // `(TextContent|ImageContent)[]` content union (`pi-fields.md`
962 // §3a/§3c/§3e) — an `ImageContent` block that doesn't match the
963 // loader's assumed (and unverified) `{mimeType, data}` shape
964 // must score as `message/UnknownImageShape`, never silently
965 // `Normalized`, mirroring `UnknownRole`'s "surface it, don't
966 // vanish" rule exactly.
967 let content = message.and_then(|m| m.get("content"));
968 let unknown_image = matches!(role, Some("user") | Some("toolResult") | Some("custom"))
969 && pi_content_has_unknown_image_shape(content);
970 match role {
971 _ if unknown_image => report.bump(
972 "message/UnknownImageShape".to_string(),
973 Coverage::Unmodeled,
974 extra,
975 ),
976 Some("user") => {
977 report.bump("message/user".to_string(), Coverage::Normalized, extra)
978 }
979 Some("assistant") => {
980 report.bump("message/assistant".to_string(), Coverage::Normalized, extra)
981 }
982 Some("toolResult") => report.bump(
983 "message/toolResult".to_string(),
984 Coverage::Normalized,
985 extra,
986 ),
987 Some("bashExecution") => report.bump(
988 "message/bashExecution".to_string(),
989 Coverage::Normalized,
990 extra,
991 ),
992 Some("custom") => {
993 report.bump("message/custom".to_string(), Coverage::Normalized, extra)
994 }
995 Some(other) => report.bump(
996 format!("message/UnknownRole:{other}"),
997 Coverage::Unmodeled,
998 extra,
999 ),
1000 None => report.bump(
1001 "message/UnknownRole:<none>".to_string(),
1002 Coverage::Unmodeled,
1003 extra,
1004 ),
1005 }
1006 }
1007 "custom_message" => report.bump("custom_message".to_string(), Coverage::Normalized, extra),
1008 "compaction" => report.bump("compaction".to_string(), Coverage::Normalized, extra),
1009 "branch_summary" => report.bump("branch_summary".to_string(), Coverage::Normalized, extra),
1010 "thinking_level_change" => report.bump(
1011 "thinking_level_change".to_string(),
1012 Coverage::Dropped,
1013 extra,
1014 ),
1015 "model_change" => report.bump("model_change".to_string(), Coverage::Normalized, extra),
1016 "custom" => report.bump("custom".to_string(), Coverage::Dropped, extra),
1017 "label" => report.bump("label".to_string(), Coverage::Dropped, extra),
1018 "session_info" => report.bump("session_info".to_string(), Coverage::Normalized, extra),
1019 other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1020 }
1021}
1022
1023/// Score one Gemini CLI session record against the native loader.
1024fn audit_gemini_line(line: &str, report: &mut Report) {
1025 let raw: Value = match serde_json::from_str(line) {
1026 Ok(value) => value,
1027 Err(_) => {
1028 report.parse_errors += 1;
1029 return;
1030 }
1031 };
1032 let extra = EMPTY_EXTRA.get_or_init(Default::default);
1033 let kind = raw.get("type").and_then(Value::as_str);
1034 if kind.is_none() {
1035 let key = if raw.get("sessionId").is_some_and(Value::is_string) {
1036 "session_header"
1037 } else if raw.get("$set").is_some() {
1038 "metadata_update"
1039 } else {
1040 "<line>/<no-type>"
1041 };
1042 let coverage = if key == "<line>/<no-type>" {
1043 Coverage::Unmodeled
1044 } else {
1045 Coverage::Retained
1046 };
1047 report.bump(key.to_string(), coverage, extra);
1048 return;
1049 }
1050 match kind.unwrap_or_default() {
1051 "user" | "gemini" => {
1052 report.bump(
1053 kind.unwrap_or_default().to_string(),
1054 Coverage::Normalized,
1055 extra,
1056 );
1057 if let Some(parts) = raw.get("content").and_then(Value::as_array) {
1058 for part in parts {
1059 if part.get("text").is_some() {
1060 report.bump("content/text".into(), Coverage::Normalized, extra);
1061 } else if part.get("inlineData").is_some() {
1062 report.bump("content/inlineData".into(), Coverage::Normalized, extra);
1063 } else if let Some(call) = part.get("functionCall") {
1064 report.bump("content/functionCall".into(), Coverage::Normalized, extra);
1065 if let Some(name) = call.get("name").and_then(Value::as_str) {
1066 *report.tools.entry(name.to_string()).or_insert(0) += 1;
1067 }
1068 } else if let Some(response) = part.get("functionResponse") {
1069 report.bump(
1070 "content/functionResponse".into(),
1071 Coverage::Normalized,
1072 extra,
1073 );
1074 if let Some(name) = response.get("name").and_then(Value::as_str) {
1075 *report.tools.entry(name.to_string()).or_insert(0) += 1;
1076 }
1077 } else {
1078 report.bump("content/unknown".into(), Coverage::Unmodeled, extra);
1079 }
1080 }
1081 } else if !raw
1082 .get("content")
1083 .is_some_and(|content| content.is_string() || content.is_null())
1084 {
1085 report.bump("content/nonstandard".into(), Coverage::Unmodeled, extra);
1086 }
1087 if raw.get("thoughts").is_some() {
1088 report.bump("thoughts".into(), Coverage::Retained, extra);
1089 }
1090 }
1091 "info" | "error" => report.bump(
1092 kind.unwrap_or_default().to_string(),
1093 Coverage::Retained,
1094 extra,
1095 ),
1096 other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1097 }
1098}
1099
1100/// Score one Grok `chat_history.jsonl` record against the same shapes the
1101/// native loader actually consumes. Companion `updates.jsonl` streams are
1102/// excluded by [`audit_dir`], because they are ACP/runtime evidence rather
1103/// than the resumable transcript.
1104fn audit_grok_line(line: &str, report: &mut Report) {
1105 let raw: Value = match serde_json::from_str(line) {
1106 Ok(value) => value,
1107 Err(_) => {
1108 report.parse_errors += 1;
1109 return;
1110 }
1111 };
1112 let extra = EMPTY_EXTRA.get_or_init(Default::default);
1113 let Some(kind) = raw.get("type").and_then(Value::as_str) else {
1114 report.bump("<line>/<no-type>".to_string(), Coverage::Unmodeled, extra);
1115 return;
1116 };
1117 match kind {
1118 "system" => {
1119 let coverage = if raw.get("content").is_some_and(Value::is_string) {
1120 Coverage::Retained
1121 } else {
1122 Coverage::Unmodeled
1123 };
1124 report.bump("system".to_string(), coverage, extra);
1125 }
1126 "user" => {
1127 let (key, coverage) = classify_grok_user(&raw);
1128 report.bump(key.clone(), coverage, extra);
1129 audit_grok_content(raw.get("content"), &key, coverage, report);
1130 }
1131 "assistant" => {
1132 let content_supported = raw
1133 .get("content")
1134 .is_none_or(|content| content.is_null() || content.is_string());
1135 report.bump(
1136 "assistant".to_string(),
1137 if content_supported {
1138 Coverage::Normalized
1139 } else {
1140 Coverage::Unmodeled
1141 },
1142 extra,
1143 );
1144 if let Some(calls) = raw.get("tool_calls").and_then(Value::as_array) {
1145 for call in calls {
1146 let modeled = call.get("id").is_some_and(Value::is_string)
1147 && call.get("name").is_some_and(Value::is_string);
1148 report.bump(
1149 if modeled {
1150 "assistant/tool_call".to_string()
1151 } else {
1152 "assistant/tool_call:invalid".to_string()
1153 },
1154 if modeled {
1155 Coverage::Normalized
1156 } else {
1157 Coverage::Unmodeled
1158 },
1159 extra,
1160 );
1161 if let Some(name) = call.get("name").and_then(Value::as_str) {
1162 *report.tools.entry(name.to_string()).or_insert(0) += 1;
1163 }
1164 }
1165 } else if raw.get("tool_calls").is_some() {
1166 report.bump(
1167 "assistant/tool_calls:non-array".to_string(),
1168 Coverage::Unmodeled,
1169 extra,
1170 );
1171 }
1172 }
1173 "tool_result" => {
1174 let modeled = raw.get("tool_call_id").is_some_and(Value::is_string);
1175 report.bump(
1176 "tool_result".to_string(),
1177 if modeled {
1178 Coverage::Normalized
1179 } else {
1180 Coverage::Unmodeled
1181 },
1182 extra,
1183 );
1184 audit_grok_content(
1185 raw.get("content"),
1186 "tool_result",
1187 Coverage::Normalized,
1188 report,
1189 );
1190 }
1191 // These are understood native records but deliberately remain in
1192 // the byte-exact raw prefix instead of becoming replayable messages.
1193 "reasoning" | "backend_tool_call" => {
1194 report.bump(kind.to_string(), Coverage::Dropped, extra)
1195 }
1196 other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1197 }
1198}
1199
1200/// Classify a Grok `user` record using the exact replay boundary enforced by
1201/// `Session::from_grok_str`: generated context wrappers are native session
1202/// state, not human turns, and therefore remain raw-only. Separate record
1203/// keys are deliberate — [`Report::records`] stores one coverage value per
1204/// key, so mixing replayed and discarded users under a single `user` key
1205/// would make the last line scanned overwrite the truth for the whole corpus.
1206fn classify_grok_user(raw: &Value) -> (String, Coverage) {
1207 if raw.get("synthetic_reason").and_then(Value::as_str) == Some("supercode_system_event") {
1208 return ("user/system_event".to_string(), Coverage::Normalized);
1209 }
1210
1211 let content = grok_audit_text(raw.get("content"));
1212 let text = content.trim();
1213 if text.starts_with("<user_info>") {
1214 return (
1215 "user/injected_context:user_info".to_string(),
1216 Coverage::Dropped,
1217 );
1218 }
1219 if text.starts_with("<system-reminder>") {
1220 return (
1221 "user/injected_context:system-reminder".to_string(),
1222 Coverage::Dropped,
1223 );
1224 }
1225 if text.is_empty() {
1226 return ("user/empty".to_string(), Coverage::Dropped);
1227 }
1228 if text
1229 .strip_prefix("<user_query>")
1230 .and_then(|value| value.strip_suffix("</user_query>"))
1231 .is_some_and(|value| value.trim().is_empty())
1232 {
1233 return ("user/empty_query".to_string(), Coverage::Dropped);
1234 }
1235 ("user".to_string(), Coverage::Normalized)
1236}
1237
1238/// Mirror the loader's text extraction for the two organic Grok shapes:
1239/// direct string content and arrays containing either `{text: ...}` blocks
1240/// or string entries. Other scalar/object values stringify exactly as the
1241/// loader does, making the audit a behavioral classification rather than a
1242/// narrower invented schema.
1243fn grok_audit_text(content: Option<&Value>) -> String {
1244 match content {
1245 Some(Value::String(text)) => text.clone(),
1246 Some(Value::Array(items)) => items
1247 .iter()
1248 .filter_map(|item| {
1249 item.get("text")
1250 .and_then(Value::as_str)
1251 .or_else(|| item.as_str())
1252 })
1253 .collect::<Vec<_>>()
1254 .join("\n"),
1255 Some(other) => other.to_string(),
1256 None => String::new(),
1257 }
1258}
1259
1260fn audit_grok_content(
1261 content: Option<&Value>,
1262 prefix: &str,
1263 record_coverage: Coverage,
1264 report: &mut Report,
1265) {
1266 let extra = EMPTY_EXTRA.get_or_init(Default::default);
1267 match content {
1268 Some(Value::Array(items)) => {
1269 for item in items {
1270 let tag = item
1271 .get("type")
1272 .and_then(Value::as_str)
1273 .unwrap_or("<no-type>");
1274 let modeled = item.is_string() || item.get("text").is_some_and(Value::is_string);
1275 let coverage = if record_coverage == Coverage::Dropped {
1276 Coverage::Dropped
1277 } else if modeled {
1278 Coverage::Normalized
1279 } else {
1280 Coverage::Unmodeled
1281 };
1282 report.bump(format!("{prefix}/content/{tag}"), coverage, extra);
1283 }
1284 }
1285 Some(_) => report.bump(format!("{prefix}/content"), record_coverage, extra),
1286 None => report.bump(
1287 format!("{prefix}/content:<missing>"),
1288 if record_coverage == Coverage::Dropped {
1289 Coverage::Dropped
1290 } else {
1291 Coverage::Unmodeled
1292 },
1293 extra,
1294 ),
1295 }
1296}
1297
1298/// The frozen 12-part union discriminant values
1299/// (`docs/interop/research/opencode-fields.md` §3, `v1/session.ts:357-370`).
1300/// Anything outside this set is an UNKNOWN part type — never silently
1301/// dropped, always scored `Unmodeled` (§4.1's "no record/part discriminant
1302/// falls into an Unknown bucket" completeness guard).
1303const OPENCODE_KNOWN_PART_TYPES: &[&str] = &[
1304 "text",
1305 "reasoning",
1306 "tool",
1307 "file",
1308 "step-start",
1309 "step-finish",
1310 "snapshot",
1311 "patch",
1312 "agent",
1313 "subtask",
1314 "retry",
1315 "compaction",
1316];
1317
1318/// `ToolState`'s frozen discriminant values (`opencode-fields.md` §3.3,
1319/// `v1/session.ts:259-313`).
1320const OPENCODE_KNOWN_TOOL_STATUSES: &[&str] = &["pending", "running", "completed", "error"];
1321
1322/// Audit one envelope line of an OpenCode session
1323/// (`docs/interop/opencode-pi-spec.md` §1.2/§4.1): `{"key":[...],"value":...}`,
1324/// classified by the envelope `key`'s first component exactly like
1325/// [`supercode_interchange::session::Session::from_opencode_str`]. Two second-level
1326/// discriminants get their own `Unknown*` buckets, mirroring pi's
1327/// `UnknownRole`/`UnknownImageShape` discipline (S6): `message/UnknownRole:*`
1328/// for a `message` record whose `role` isn't `user`/`assistant`, and
1329/// `part/UnknownType:*` for a `part` record whose `type` isn't one of the
1330/// frozen 12 — plus a THIRD level for `tool` parts specifically,
1331/// `part/tool/UnknownStatus:*`, for a `state.status` outside the frozen
1332/// four. All three must be empty over the committed fixture + real corpus.
1333fn audit_opencode_line(line: &str, report: &mut Report) {
1334 let raw: Value = match serde_json::from_str(line) {
1335 Ok(v) => v,
1336 Err(_) => {
1337 report.parse_errors += 1;
1338 return;
1339 }
1340 };
1341 let extra = EMPTY_EXTRA.get_or_init(Default::default);
1342 let Some(key) = raw.get("key").and_then(Value::as_array) else {
1343 report.bump("<line>/<no-key>".to_string(), Coverage::Unmodeled, extra);
1344 return;
1345 };
1346 let value = raw.get("value").cloned().unwrap_or(Value::Null);
1347 let kind = key.first().and_then(Value::as_str).unwrap_or("<no-kind>");
1348 match kind {
1349 "session" => report.bump("session".to_string(), Coverage::Normalized, extra),
1350 "message" => match value.get("role").and_then(Value::as_str) {
1351 Some("user") => report.bump("message/user".to_string(), Coverage::Normalized, extra),
1352 Some("assistant") => {
1353 report.bump("message/assistant".to_string(), Coverage::Normalized, extra)
1354 }
1355 Some(other) => report.bump(
1356 format!("message/UnknownRole:{other}"),
1357 Coverage::Unmodeled,
1358 extra,
1359 ),
1360 None => report.bump(
1361 "message/UnknownRole:<none>".to_string(),
1362 Coverage::Unmodeled,
1363 extra,
1364 ),
1365 },
1366 "part" => match value.get("type").and_then(Value::as_str) {
1367 Some(t) if OPENCODE_KNOWN_PART_TYPES.contains(&t) => {
1368 if t == "tool" {
1369 // D5: tally the tool NAME (`tool`, e.g. "bash"/"edit"),
1370 // not just the call-status bucket — previously
1371 // `report.tools` was always empty for opencode corpora.
1372 if let Some(name) = value.get("tool").and_then(Value::as_str) {
1373 *report.tools.entry(name.to_string()).or_insert(0) += 1;
1374 }
1375 match value
1376 .get("state")
1377 .and_then(|s| s.get("status"))
1378 .and_then(Value::as_str)
1379 {
1380 Some(s) if OPENCODE_KNOWN_TOOL_STATUSES.contains(&s) => {
1381 report.bump(format!("part/tool/{s}"), Coverage::Normalized, extra)
1382 }
1383 Some(other) => report.bump(
1384 format!("part/tool/UnknownStatus:{other}"),
1385 Coverage::Unmodeled,
1386 extra,
1387 ),
1388 None => report.bump(
1389 "part/tool/UnknownStatus:<none>".to_string(),
1390 Coverage::Unmodeled,
1391 extra,
1392 ),
1393 }
1394 } else if t == "text" {
1395 // D5: an `ignored:true` text part is EXCLUDED from
1396 // replay by design (§2.2: "must not be re-emitted to
1397 // the model") — it is recognized and preserved in
1398 // `raw`, but never lands in canonical `messages`, so it
1399 // is Dropped, not Normalized. A separate discriminant
1400 // key keeps the two counted (and displayed) apart
1401 // rather than one overwriting the other's coverage.
1402 let ignored = value.get("ignored").and_then(Value::as_bool) == Some(true);
1403 if ignored {
1404 report.bump("part/text:ignored".to_string(), Coverage::Dropped, extra);
1405 } else {
1406 report.bump("part/text".to_string(), Coverage::Normalized, extra);
1407 }
1408 } else if t == "file" {
1409 // D5: the loader only canonicalizes a `data:`-URI
1410 // `image/*` file part into `content_parts` (the SAME
1411 // test `opencode_file_image_part` uses, reused here so
1412 // audit can never drift from what convert actually
1413 // replays). An `https:` link, a bare path, a PDF, or
1414 // any other non-image/non-data-URI file is raw-only
1415 // residue — Dropped, not Normalized.
1416 if opencode_file_image_part(&value).is_some() {
1417 report.bump("part/file".to_string(), Coverage::Normalized, extra);
1418 } else {
1419 report.bump("part/file:residue".to_string(), Coverage::Dropped, extra);
1420 }
1421 } else {
1422 // compaction drives the `compacted_out` boundary —
1423 // Normalized. reasoning feeds `metadata["thinking"]`
1424 // (recognized, deliberately not canonical content —
1425 // Dropped, same label Claude/Codex `thinking` blocks
1426 // get). step-start/step-finish/snapshot/patch/agent/
1427 // subtask/retry are recognized but have NO clean home
1428 // at all (§2.3) — also Dropped. Only a truly
1429 // unrecognized type is Unmodeled.
1430 let cov = match t {
1431 "compaction" => Coverage::Normalized,
1432 _ => Coverage::Dropped,
1433 };
1434 report.bump(format!("part/{t}"), cov, extra);
1435 }
1436 }
1437 Some(other) => report.bump(
1438 format!("part/UnknownType:{other}"),
1439 Coverage::Unmodeled,
1440 extra,
1441 ),
1442 None => report.bump(
1443 "part/UnknownType:<none>".to_string(),
1444 Coverage::Unmodeled,
1445 extra,
1446 ),
1447 },
1448 "session_diff" => report.bump("session_diff".to_string(), Coverage::Normalized, extra),
1449 "todo" => report.bump("todo".to_string(), Coverage::Normalized, extra),
1450 other => report.bump(format!("<line>/{other}"), Coverage::Unmodeled, extra),
1451 }
1452}
1453
1454fn jsonl_files(dir: &Path) -> Vec<PathBuf> {
1455 let mut out = Vec::new();
1456 let walker = ignore::WalkBuilder::new(dir)
1457 .standard_filters(false)
1458 .build();
1459 for entry in walker.flatten() {
1460 let p = entry.into_path();
1461 if p.extension().and_then(|e| e.to_str()) == Some("jsonl") {
1462 out.push(p);
1463 }
1464 }
1465 out.sort();
1466 out
1467}