use std::path::{Path, PathBuf};
use crate::core::db::Database;
use crate::core::errors::TgaError;
use super::attest::{attest, DIFF_API_SITES, DIFF_TEXT_CONSUMERS, NO_CONTENT_CLAIM};
use super::schema::{snapshot, ObjectKind};
use super::text_columns::{classify, TextClass, CONSTRAINED, EMBEDDED_PAYLOAD, FREE_TEXT};
use super::{open_read_only, render};
struct TempDir {
path: PathBuf,
}
impl Drop for TempDir {
fn drop(&mut self) {
let _ = std::fs::remove_dir_all(&self.path);
}
}
fn temp_dir(tag: &str) -> TempDir {
let mut path = std::env::temp_dir();
path.push(format!(
"tga-inspect-{tag}-{}-{}",
std::process::id(),
std::time::SystemTime::now()
.duration_since(std::time::UNIX_EPOCH)
.map_or(0, |d| d.as_nanos())
));
std::fs::create_dir_all(&path).expect("mkdir");
TempDir { path }
}
fn migrated_db(dir: &TempDir) -> PathBuf {
let path = dir.path.join("tga.db");
Database::open(&path).expect("open and migrate");
path
}
#[test]
fn open_read_only_names_a_missing_database() {
let dir = temp_dir("missing");
let missing = dir.path.join("absent.db");
let err = open_read_only(&missing).expect_err("a missing database must not open");
assert!(
matches!(err, TgaError::NotFound(_)),
"expected NotFound, got {err:?}"
);
assert!(
err.to_string().contains("absent.db"),
"the error must name the path it could not find: {err}"
);
assert!(
!missing.exists(),
"inspecting a missing database must not create one"
);
}
#[test]
fn open_read_only_names_a_directory() {
let dir = temp_dir("isdir");
let err = open_read_only(&dir.path).expect_err("a directory must not open");
assert!(
matches!(err, TgaError::ValidationError(_)),
"expected ValidationError, got {err:?}"
);
assert!(
err.to_string().contains("is a directory"),
"the error must name the cause: {err}"
);
}
#[test]
fn open_read_only_names_a_non_sqlite_file() {
let dir = temp_dir("notdb");
let path = dir.path.join("notes.txt");
std::fs::write(&path, b"this is not a database").expect("write");
let err = open_read_only(&path).expect_err("a text file must not open as a database");
assert!(
matches!(err, TgaError::ValidationError(_)),
"expected ValidationError, got {err:?}"
);
assert!(
err.to_string().contains("not a SQLite database"),
"the error must name the cause: {err}"
);
}
#[test]
fn open_read_only_does_not_migrate() {
let dir = temp_dir("nomigrate");
let path = dir.path.join("empty.db");
rusqlite::Connection::open(&path)
.expect("create")
.execute_batch("CREATE TABLE marker (x INTEGER);")
.expect("seed");
let conn = open_read_only(&path).expect("open the valid, un-migrated file");
let snap = snapshot(&conn).expect("snapshot");
assert_eq!(
snap.schema_version, None,
"no schema_migrations table exists"
);
let names: Vec<&str> = snap.objects.iter().map(|o| o.name.as_str()).collect();
assert_eq!(names, vec!["marker"], "inspection must not create tables");
let write = conn.execute_batch("CREATE TABLE sneaky (y INTEGER);");
assert!(write.is_err(), "the connection must be read-only");
}
#[test]
fn snapshot_reads_every_table_and_column() {
let dir = temp_dir("snapshot");
let path = migrated_db(&dir);
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let names: Vec<&str> = snap.objects.iter().map(|o| o.name.as_str()).collect();
for expected in [
"commits",
"files",
"work_items",
"fact_pm_effort",
"schema_migrations",
"v_lead_time",
] {
assert!(
names.contains(&expected),
"{expected} missing from {names:?}"
);
}
let commits = snap
.objects
.iter()
.find(|o| o.name == "commits")
.expect("commits table");
assert_eq!(commits.kind, ObjectKind::Table);
let cols: Vec<&str> = commits.columns.iter().map(|c| c.name.as_str()).collect();
assert!(cols.contains(&"message"), "column from 0001 missing");
assert!(cols.contains(&"agentic_mode"), "column from 0021 missing");
let view = snap
.objects
.iter()
.find(|o| o.name == "v_lead_time")
.expect("view");
assert_eq!(view.kind, ObjectKind::View);
assert_eq!(view.row_count, None, "views carry no row count");
}
#[test]
fn snapshot_reports_row_counts() {
let dir = temp_dir("counts");
let path = migrated_db(&dir);
{
let db = Database::open(&path).expect("reopen");
db.connection()
.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository) \
VALUES ('abc123', 'Ada', 'ada@example.com', '2026-01-01T00:00:00Z', 'feat: x', 'r')",
[],
)
.expect("insert");
}
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let commits = snap
.objects
.iter()
.find(|o| o.name == "commits")
.expect("commits");
assert_eq!(commits.row_count, Some(1));
}
#[test]
fn every_text_column_is_classified() {
let dir = temp_dir("classified");
let path = migrated_db(&dir);
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let unclassified: Vec<String> = snap
.text_columns()
.into_iter()
.filter(|(_, c)| c.text_class == Some(TextClass::Unclassified))
.map(|(t, c)| format!("{}.{}", t.name, c.name))
.collect();
assert!(
unclassified.is_empty(),
"these TEXT columns are in the schema but in none of FREE_TEXT / \
EMBEDDED_PAYLOAD / CONSTRAINED — classify them in \
core::inspect::text_columns: {unclassified:?}"
);
let live: Vec<String> = snap
.text_columns()
.into_iter()
.map(|(t, c)| format!("{}.{}", t.name, c.name))
.collect();
let mut stale: Vec<&str> = Vec::new();
for key in FREE_TEXT.iter().chain(EMBEDDED_PAYLOAD).chain(CONSTRAINED) {
if !live.iter().any(|l| l == key) {
stale.push(key);
}
}
assert!(
stale.is_empty(),
"these inventory entries name columns the schema no longer has: {stale:?}"
);
}
#[test]
fn text_class_lists_are_disjoint() {
for key in FREE_TEXT {
assert!(!EMBEDDED_PAYLOAD.contains(key), "{key} in two lists");
assert!(!CONSTRAINED.contains(key), "{key} in two lists");
}
for key in EMBEDDED_PAYLOAD {
assert!(!CONSTRAINED.contains(key), "{key} in two lists");
}
assert_eq!(
classify("commits", "message"),
TextClass::FreeText,
"the highest-exposure column must classify as free text"
);
assert_eq!(
classify("work_items", "raw_json"),
TextClass::EmbeddedPayload
);
assert_eq!(classify("commits", "sha"), TextClass::Constrained);
assert_eq!(classify("nope", "nope"), TextClass::Unclassified);
}
#[test]
fn claim_never_says_no_code() {
assert!(
!NO_CONTENT_CLAIM.to_lowercase().contains("no code"),
"the claim must never assert the database contains no code: {NO_CONTENT_CLAIM}"
);
for term in ["file content", "diffs", "patches", "hunks", "blobs"] {
assert!(
NO_CONTENT_CLAIM.contains(term),
"the claim must name {term}: {NO_CONTENT_CLAIM}"
);
}
}
#[test]
fn attest_on_a_fresh_database_is_consistent() {
let dir = temp_dir("attest-clean");
let path = migrated_db(&dir);
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let report = attest(&conn, &snap).expect("attest");
assert!(
report.content_columns.is_empty(),
"tga's schema must hold no content-bearing column: {:?}",
report.content_columns
);
assert_eq!(report.verdict, super::attest::Verdict::Consistent);
assert!(
report
.scanned_columns
.iter()
.any(|s| s.table == "commits" && s.column == "message"),
"commits.message must be scanned"
);
assert!(
report
.scanned_columns
.iter()
.any(|s| s.table == "work_items" && s.column == "raw_json"),
"work_items.raw_json must be scanned at runtime, not read off the migration"
);
assert!(
report
.scanned_columns
.iter()
.all(|s| !matches!(s.class, TextClass::Constrained)),
"constrained columns must not be scanned"
);
}
#[test]
fn attest_flags_a_diff_pasted_into_a_commit_message() {
let dir = temp_dir("attest-dirty");
let path = migrated_db(&dir);
let payload = serde_json::to_string(&serde_json::json!({
"description": "--- a/x\n+++ b/x\n@@ -1 +1 @@\n-a\n+b\n",
}))
.expect("serialize");
assert!(
!payload.contains('\n'),
"the fixture must carry the JSON escape, not a newline byte: {payload}"
);
{
let db = Database::open(&path).expect("reopen");
db.connection()
.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository) \
VALUES ('deadbee', 'Ada', 'ada@example.com', '2026-01-01T00:00:00Z', ?1, 'r')",
["fix: paste\n\ndiff --git a/src/lib.rs b/src/lib.rs\n@@ -1 +1 @@\n-old\n+new\n"],
)
.expect("insert commit");
db.connection()
.execute(
"INSERT INTO work_items (id, source, title, status, item_type, raw_json) \
VALUES ('W-1', 'jira', 'T', 'Done', 'Bug', ?1)",
[&payload],
)
.expect("insert work item");
}
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let report = attest(&conn, &snap).expect("attest");
let message = report
.scanned_columns
.iter()
.find(|s| s.table == "commits" && s.column == "message")
.expect("commits.message scan");
assert_eq!(
message.diff_shaped_rows, 1,
"a pasted unified diff must be counted"
);
assert_eq!(message.populated, 1);
assert!(message.max_len > 0);
let raw_json = report
.scanned_columns
.iter()
.find(|s| s.table == "work_items" && s.column == "raw_json")
.expect("work_items.raw_json scan");
assert_eq!(raw_json.populated, 1);
assert_eq!(
raw_json.diff_shaped_rows, 1,
"a diff serialized into JSON must be counted — its newlines are the \
two-character escape, so the scan has to normalise before matching"
);
assert_eq!(report.verdict, super::attest::Verdict::Findings);
}
#[test]
fn diff_probe_normalises_json_escaped_newlines() {
let cases: &[(&str, &str, i64)] = &[
("raw", "fix\n\ndiff --git a/x b/x\n@@ -1 +1 @@\n-a\n+b\n", 1),
(
"json_lf",
r#"{"description":"--- a/x\n+++ b/x\n@@ -1 +1 @@"}"#,
1,
),
(
"json_crlf",
r#"{"description":"--- a/x\r\n+++ b/x\r\n@@ -1 +1 @@"}"#,
1,
),
(
"prose",
"Reviewed the @@ hunk headers and the --- a/ prefix in passing",
0,
),
];
for (label, message, expected) in cases {
let dir = temp_dir(&format!("probe-{label}"));
let path = migrated_db(&dir);
{
let db = Database::open(&path).expect("reopen");
db.connection()
.execute(
"INSERT INTO commits (sha, author_name, author_email, timestamp, message, repository) \
VALUES ('s1', 'Ada', 'ada@example.com', '2026-01-01T00:00:00Z', ?1, 'r')",
[message],
)
.expect("insert");
}
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let report = attest(&conn, &snap).expect("attest");
let scan = report
.scanned_columns
.iter()
.find(|s| s.table == "commits" && s.column == "message")
.expect("commits.message scan");
assert_eq!(
scan.diff_shaped_rows, *expected,
"case {label}: expected {expected} diff-shaped row(s) for {message:?}"
);
}
}
#[test]
fn diff_for_commit_callers_match_the_attestation() {
let root = PathBuf::from(env!("CARGO_MANIFEST_DIR"));
let mut found: Vec<String> = Vec::new();
collect_callers(&root.join("src"), &root, &mut found);
found.sort();
found.dedup();
let mut attested: Vec<String> = DIFF_TEXT_CONSUMERS
.iter()
.map(|c| c.source_path.to_string())
.collect();
attested.sort();
assert_eq!(
found, attested,
"the non-test callers of diff_for_commit changed. Update DIFF_TEXT_CONSUMERS in \
core::inspect::attest, and confirm the new caller does not write diff text to the \
database — see #5218."
);
}
fn collect_callers(dir: &Path, root: &Path, out: &mut Vec<String>) {
let entries = std::fs::read_dir(dir).expect("read src dir");
for entry in entries {
let entry = entry.expect("dir entry");
let path = entry.path();
if path.is_dir() {
collect_callers(&path, root, out);
continue;
}
if path.extension().is_none_or(|e| e != "rs") {
continue;
}
let rel = path
.strip_prefix(root)
.expect("path under crate root")
.to_string_lossy()
.replace('\\', "/");
if DIFF_API_SITES.contains(&rel.as_str()) || is_test_file(&rel) {
continue;
}
let body = std::fs::read_to_string(&path).expect("read source file");
if body.lines().any(is_caller_line) {
out.push(rel);
}
}
}
fn is_test_file(rel: &str) -> bool {
rel.contains("/tests/")
|| rel.ends_with("/tests.rs")
|| rel.ends_with("_test.rs")
|| rel.ends_with("_tests.rs")
}
fn is_caller_line(line: &str) -> bool {
let trimmed = line.trim_start();
if trimmed.starts_with("//") {
return false;
}
strip_string_literals(line).contains("diff_for_commit")
}
fn strip_string_literals(line: &str) -> String {
let mut out = String::with_capacity(line.len());
let mut in_string = false;
for ch in line.chars() {
if ch == '"' {
in_string = !in_string;
continue;
}
if !in_string {
out.push(ch);
}
}
out
}
#[test]
fn schema_report_marks_free_text_columns() {
let dir = temp_dir("render-schema");
let path = migrated_db(&dir);
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let text = render::schema_report(&snap);
assert!(text.contains("TABLE commits"), "tables must be listed");
assert!(text.contains("VIEW v_lead_time"), "views must be listed");
assert!(
text.contains("message") && text.contains("← FREE TEXT"),
"free-text columns must be marked"
);
assert!(
text.contains("← EMBEDDED PAYLOAD"),
"work_items.raw_json must be marked"
);
assert!(
text.contains("Free-text and payload columns"),
"the trailing summary must be present"
);
}
#[test]
fn attestation_report_states_the_claim_and_the_caveat() {
let dir = temp_dir("render-attest");
let path = migrated_db(&dir);
let conn = open_read_only(&path).expect("open");
let snap = snapshot(&conn).expect("snapshot");
let report = attest(&conn, &snap).expect("attest");
let text = render::attestation_report(&report);
assert!(text.contains(NO_CONTENT_CLAIM));
assert!(text.contains("not a claim that the database contains no code"));
assert!(
text.lines()
.next()
.is_some_and(|first| first == NO_CONTENT_CLAIM),
"the claim must lead the report, so a reader quoting the first line quotes it"
);
assert!(text.contains("commits.message"));
assert!(text.contains("src/profile/diff_sampler/sampler.rs"));
assert!(text.contains("VERDICT: consistent"));
}