use super::*;
#[derive(Debug, Clone, Args, Default)]
#[command(
about = "Regex-search sessions, returning complete request/response round-trip exchanges",
long_about = "Regex-search transcripts, returning the COMPLETE round-trip exchange \
containing each hit; never a bare fragment. A turn is delimited by a genuine \
user message; the emitted exchange is the whole turn (the opening user record \
plus every assistant/thinking/tool_use/tool_result record chained under it \
until the next genuine user). So a matched tool_use comes WITH its \
tool_result, a matched user turn WITH the agent's response, etc.\n\n\
The PATTERN is ripgrep-like and defaults to SMART-CASE: case-insensitive \
unless it contains an uppercase letter; `-i` forces case-insensitive and \
always wins. `--multiline` lets `.` cross newlines. An EMPTY pattern is a \
pure filter; it matches every label-eligible record, so combine it with \
`--label` / `--since` / `--turn` (a bare empty pattern with no other \
filter warns that it will emit a lot).\n\n\
CATEGORIES (`-t`, repeatable): a dotted `role.class.sub` SELECTOR. A selector matches a \
record label iff it is a dot-SEGMENT prefix of the label's path, so `-t agent` covers the \
whole agent role while `-t agent.tool` covers use+result. The leaf labels: \
user.message | user.answer | user.rejection | agent.message | agent.thinking | \
agent.tool.use | agent.tool.result | agent.communication.{inbox,sent,signal} | \
harness.notification.{workflow,monitor,subagent,background-command,task} | \
harness.compaction.{summary,boundary} | harness.command.{invocation,stdout} | \
harness.interrupt.{user,tool} | harness.schedule.{wakeup,continuation} | \
harness.meta.{hook,loop}. With none given, EVERY label is eligible. `-T`/`--label-not` \
EXCLUDES with the same selector grammar (the rg -t/-T duality): the effective set is \
(-t selectors, or ALL) minus (-T selectors); a combination that excludes everything it \
includes is a hard error. The human turn is \
`user.message`; an AskUserQuestion answer is `user.answer` (the full Q+options+answer \
unit); a plan-rejection-with-message is `user.rejection` (+ a [plan: …] pointer). An \
inbound peer/teammate message is `agent.communication.inbox` (NOT `user`); a \
`<task-notification>` automation pulse is `harness.notification.*` (NOT `user`). A \
tool named after harness machinery still classifies by ROLE: a `ScheduleWakeup` CALL \
(arming a timer) is `agent.tool.use` like any other tool: `harness.schedule.wakeup` \
is only the FIRED tick, the harness-injected marker-carrying wakeup prompt (a \
custom-prompt tick lands as an isMeta record, excluded like all isMeta).\n\n\
AUTOMATION TRIGGERS: a `<task-notification>` (a background-command / workflow / \
spawned-agent / monitor COMPLETION pulse Claude Code injects as a `type:\"user\"` record) \
OPENS a turn like a human message but classifies under `harness.notification.<kind>` \
(kind = background-command | workflow | subagent | monitor | task, read from the \
summary). It renders as the parsed `[<kind> <task-id> <status>] <summary>` attribution \
label; never the raw `<task-id>`/`<output-file>` XML. Match it like any other text (e.g. \
`search 'background-command' -t harness.notification.background-command`).\n\n\
WINDOWING: `--turn` takes the shared range grammar: `N` (one turn) · `A..B` \
(closed) · `N..` (turn N → the end) · `..N` (start → N) · `-k` = k-th FROM THE END \
(`-3..` = the last 3 turns), 0-based on turn-boundary order, and INTERSECTS with \
`--since`/`--until` (both filters AND). Time bounds accept ISO8601 (`2026-06-01` = local \
midnight · `2026-06-01T05:00:00` BARE = that LOCAL wall-clock time · \
`2026-06-01T05:00:00Z`/`…+10:00` = explicit zone) or a relative form (`2h`, `3d`, `90m`, `45s`, `1w`) meaning \
\"that long ago\" in the system local timezone.\n\n\
ZERO MATCHES IS A DEFINITIVE ANSWER, NOT A FAILURE: a no-match search prints a stderr \
diagnosis: \"DEFINITIVE absence (exit 0), NOT an error\", the active filters, and (when a \
`-t`/`-T` filter was on) an active probe that NAMES the label(s) the pattern DOES occur \
under (e.g. it was excluded by your `-t user.message` but occurs under `agent.tool.use`). \
Read the diagnosis and adjust the filter; do NOT assume a syntax error. To SEE a scope's \
record-types before you filter, run `--count-by label` (a per-leaf census; with an empty \
pattern it censuses the whole scope).\n\n\
`--max-count` caps emitted exchanges but reports the dropped count (default: \
unlimited; no cap); there is NO silent truncation anywhere.",
after_help = "EXAMPLES\n \
csift search \"carry\" # all projects, smart-case\n \
csift search \"carry\" . # this project (positional PATH, like every sibling)\n \
csift search -i \"askuserquestion\" -t agent.tool.use # tool_use blocks naming AUQ\n \
csift search \"\" -t user --since 2h . # user turns, last 2h, this project\n \
csift search \"tail.read\" --multiline @0a1b2c3d-4e5f-4a6b-8c7d-9e0f1a2b3c4d\n \
csift search \"panic\" -t agent.message -t agent.thinking --turn 10..20 --max-count 50\n \
csift search \"persisted-output\" --resolve-persisted --format json\n \
csift search \"refactor\" -c # COUNT matches only (ripgrep -c idiom)\n \
csift search \"refactor\" -l # WHICH sessions matched, one id per line (rg -l idiom)\n \
csift search \"refactor\" -l | csift files --sessions-from - # …then scope the NEXT command to them\n \
csift search \"\" @<uuid> -t agent -T agent.thinking # the agent role MINUS its thinking (-T excludes)\n \
csift search \"\" @<uuid> -t agent.message --raw | jq -r '.message.model' # raw lines: any unrendered field\n \
csift search \"let's chat\" -t user --siblings # the match WITH the turn's other side\n \
csift search \"let's chat\" -t user --siblings --no-truncate # …and READ the reply end-to-end\n \
csift search \"X\" --max-count 1 # when did X FIRST happen? (earliest exchange)\n \
csift search \"X\" --max-count -1 # most recent occurrence of X\n \
csift show @<tok> --line <n> # follow up a hit: paste its header token + L<n>\n\n\
OUTPUT GEOMETRY (text mode)\n \
Exchanges emit oldest-first (stable chronological across every transcript in \
scope; undated exchanges last). Each exchange header opens with a STABLE id-prefix \
token: the first 8 chars of the owning transcript id (a within-output collision \
lengthens the colliding group to 12, then the full id; a teammate id renders whole), \
directly usable as an `@` target, identical across invocations. A subagent \
exchange carries `(parent <first-8>)` on every header. The head carries scope + \
match totals + direction; the tail repeats the totals and adds integrity notes and \
refetch guidance; each over-long fragment marks its own truncation inline \
(`(+N chars)`). To limit output, prefer `--max-count N` (earliest N) or \
`--max-count -N` (latest N) over piping into `head`/`tail`: a capped run keeps \
every note; a pipe amputates one end of the ledger.\n\n\
SIBLINGS (`--siblings`)\n \
A match renders only the records that MATCHED. `--siblings` additionally renders \
the OTHER records of the same turn (the back-and-forth around the hit) under a `·` marker, \
so a matched user question surfaces WITH the agent's reply. Fixed policy: message units \
always render (user.*, agent.message, agent.communication.*); chattier machinery is \
capped per leaf (thinking ≤2, tool.use ≤3, tool.result ≤3, harness ≤2); the capped-away \
remainder surfaces as an explicit `(+N more · csift show @<id> --line A..B)` pointer. A \
record that itself matched is never duplicated as a sibling.\n\n\
COUNT (`-c` / `--count-only`)\n \
`-c`/`--count-only` prints just the integer EXCHANGE total: matched round-trips, \
the ripgrep `-c` idiom, honoring every filter (per-RECORD counts are `--count-by`). \
That total is ALSO always in the normal output's footer (alongside the \
distinct-session total); `--count-only` just isolates that ONE integer for a pipe. To \
list WHICH sessions matched, use `-l` (one owning uuid per line; it pipes straight \
into `--sessions-from -`).\n\n\
REGEX DIALECT: linear-time (RE2-class)\n \
The pattern is the Rust `regex` crate (regex::bytes), which GUARANTEES \
linear-time matching in the input length: NO catastrophic backtracking, ever.\n \
Supported: literals; character classes [...] / [^...] / \\d \\w \\s and \
Unicode classes \\p{...}; alternation |; groups (...) and non-capturing \
(?:...); quantifiers * + ? {m,n} (greedy + lazy *?); anchors ^ $ \\b \\B; \
dot . (use --multiline to let it cross newlines); inline flags (?i)(?m)(?s)(?x); \
Unicode-aware by default.\n \
NOT supported (these need non-linear engines): backreferences \\1; \
lookahead/lookbehind (?=) (?!) (?<=) (?<!); atomic groups / possessive \
quantifiers (?>...) / a*+. A pattern using these fails to COMPILE with a clear \
error (by design, not a bug).\n \
Case: smart-case by default (insensitive unless the pattern has an uppercase \
letter); -i forces insensitive. --multiline lives in the SAME dialect (it sets \
the (?s)(?m) flags). CAVEAT: tool_use.input is matched RE-SERIALIZED: every \
tool_use's matchable text is its name + the re-serialized JSON input (not just \
AskUserQuestion's), so a real newline inside e.g. a Bash `input.command` is \
already the two-character sequence \\n by match time; match the literal `\\\\n`; \
--multiline is correctly irrelevant there (it helps only where the RENDERED text \
keeps real newlines: message text, thinking, tool_result bodies).\n\n\
AUTOMATION TRIGGERS (`harness.notification.*`)\n \
A machine `<task-notification>` (a background-command / workflow / spawned-agent / \
monitor-tick COMPLETION pulse) OPENS a turn but classifies under \
`harness.notification.<kind>` (NOT `user`). It renders as a PARSED attribution label \
`[<kind> <task-id> <status>] <summary>` (kind = background-command | workflow | subagent | \
monitor | task, read from the summary); never the raw XML. Match it like any text, e.g. \
`csift search 'background-command' -t harness.notification`. The `<kind>` prefix \
distinguishes a machine opener from a genuine human message.\n\n\
EMPTY RESULTS ARE AN ANSWER, NOT A FAILURE\n \
With NO `-t`/`--label`, EVERY label is searched. A ZERO-match result is a DEFINITIVE \
absence (exit 0), never an error, and it SELF-DIAGNOSES on stderr: it echoes the active \
filters and, when a `-t`/`-T` was on, an active probe NAMES the label(s) the pattern DOES \
occur under (so an empty `-t user.message` that hid tool-name hits under `agent.tool.use` \
tells you exactly that). Read the diagnosis and adjust the filter; do NOT assume a syntax \
error or fall back to hand-parsing jsonl. To SEE a scope's record-types BEFORE you guess a \
filter, run `--count-by label` (a per-leaf census; empty pattern = whole-scope census; a \
leaf's count is exactly how many records `-t <leaf>` would surface; JSON `census` \
rows).\n\n\
THE LABEL TAXONOMY (-t / -T select by dot-segment prefix): 3 roles, 25 leaves\n \
user .message genuine human prose (a slash command with typed\n \
prose renders as `/name args`)\n \
.answer an answered AskUserQuestion: question, options and\n \
the picked answer as one unit\n \
.rejection a plan/tool rejection carrying the user's typed\n \
instruction (+ a `[plan: …]` pointer when resolvable)\n \
agent .message · .thinking assistant prose · reasoning (a redacted block\n \
renders \"[redacted thinking]\")\n \
.tool.use · .tool.result tool traffic, paired by tool_use_id (the `▹` join)\n \
.communication.{inbox,sent,signal} peer messages, rendered `from ⇨ to`\n \
harness .notification.{workflow,monitor,subagent,background-command,task}\n \
.compaction.{summary,boundary} · .command.{invocation,stdout}\n \
.interrupt.{user,tool} · .schedule.{wakeup,continuation} · .meta.{hook,loop}\n \
`-t agent` selects the whole role, `-t agent.tool` both tool leaves, a full path\n \
just that leaf; `-T` excludes with the same grammar (a combination that excludes\n \
everything it includes is a parse error, as is a selector typo, with suggestions).\n \
A record carrying several labels prints ONCE, under its richest view (an AUQ answer\n \
is `user.answer`, not `agent.tool.result`). Glyphs: ◂ user · ▸ agent · ⚙ harness ·\n \
▹ tool use↔result pairing · ⇨ message direction · · sibling.\n\n\
JSON SCHEMA (per --format json)\n \
One ENVELOPE object PER matched exchange (NOT one bare record per line): \
{session_id, is_subagent, parent_session_id, turn_index, ts_utc, ts_local, \
record_uuids:[…], hits:[{session_id, is_subagent, parent_session_id, label, \
labels:[…], line, uuid, excerpt, tool_name, pairing, \
from, to, ts_utc, ts_local, refetch}, …]}: `label` is the matched dotted path, `labels` \
the record's full label set, `pairing` the tool_use↔tool_result join state \
(paired | pending | orphan; null off the tool axis), `from`/`to` the comm direction \
when the hit is `agent.communication.*`, and `refetch` is the ready-to-run `csift show` \
command addressed at the RIGHT id (run it verbatim). With `--count-by <axis>` the rows are `census` \
objects instead. The \
id trio rides EVERY hit object too (so bare `.hits[]` flattening keeps real ids); \
`refetch` stays the preferred single-record path. With `--siblings`, the \
envelope also carries a `siblings:[…]` array (same per-hit shape) for the turn's \
non-matched records. Envelopes stream in \
a COMBINED STABLE CHRONOLOGICAL order (subagent exchanges interleaved with top-level \
by `ts_utc`, the turn-opening timestamp; timestamp-less exchanges sort last); the \
per-hit `ts_utc` may be later than the envelope's for a deep tool_use match. \
`session_id` is the transcript's own id: a re-feedable top-level uuid, OR a bare \
SUBAGENT hex when `is_subagent` is true (that hex is NOT a re-feedable `@<uuid>` target; \
re-feed `parent_session_id`, which is always the owning top-level uuid). \
`record_uuids` lists every record stitched into the round-trip (§6.4 completeness \
evidence). A trailing footer object {matched, sessions, transcript_ids, dropped_by_cap, \
skipped_lines, with_elicitation_sidecar, excerpts_truncated} closes the stream, plus \
{definitive_absence, active_filters, excluded_by_label} on a ZERO-match run. \
(`transcript_ids` is the per-TRANSCRIPT matching-id set, named apart from `-l`'s \
owning-session ids.) (Whole-document `json.load` fails; parse line-by-line as JSONL: N \
envelopes then the footer.)"
)]
pub struct SearchArgs {
#[arg(value_name = "PATTERN", default_value = "")]
pub pattern: String,
#[arg(
value_name = "PATH",
allow_hyphen_values = true,
value_parser = parse_project_target
)]
pub paths: Vec<PathBuf>,
#[arg(long = "sessions-from", value_name = "FILE|-")]
pub sessions_from: Option<std::path::PathBuf>,
#[arg(long = "no-subagents")]
pub no_subagents: bool,
#[arg(long = "subagents", conflicts_with = "no_subagents")]
pub subagents: bool,
#[arg(
short = 't',
long = "label",
value_name = "SELECTOR",
value_parser = parse_label_selector
)]
pub labels: Vec<String>,
#[arg(
short = 'T',
long = "label-not",
value_name = "SELECTOR",
value_parser = parse_label_selector
)]
pub labels_not: Vec<String>,
#[arg(short = 'i', long)]
pub ignore_case: bool,
#[arg(long)]
pub multiline: bool,
#[arg(
long = "turn",
value_name = "N|A..B|N..|-k",
allow_hyphen_values = true
)]
pub turn_range: Option<String>,
#[arg(long, value_name = "WHEN")]
pub since: Option<String>,
#[arg(long, value_name = "WHEN")]
pub until: Option<String>,
#[arg(long, value_name = "N", allow_negative_numbers = true)]
pub max_count: Option<i64>,
#[arg(long = "count-only", short = 'c')]
pub count_only: bool,
#[arg(
short = 'l',
long = "sessions-with-matches",
conflicts_with_all = ["count_only", "siblings"]
)]
pub sessions_with_matches: bool,
#[arg(
long = "count-by",
value_enum,
value_name = "AXIS",
conflicts_with_all = ["count_only", "sessions_with_matches", "siblings", "raw"]
)]
pub count_by: Option<CountAxis>,
#[arg(long)]
pub siblings: bool,
#[arg(
long,
conflicts_with_all = ["siblings", "count_only", "sessions_with_matches", "no_truncate"]
)]
pub raw: bool,
#[arg(long)]
pub no_truncate: bool,
#[arg(long)]
pub resolve_persisted: bool,
#[arg(long)]
pub additional_context: bool,
#[arg(long, value_enum, default_value_t = OutputFormat::Text)]
pub format: OutputFormat,
}
impl SearchArgs {
#[must_use]
pub fn want_subagents(&self) -> bool {
self.subagents || !self.no_subagents
}
#[must_use]
pub fn targets(&self) -> Vec<PathBuf> {
self.paths.clone()
}
#[must_use]
pub fn label_filter(&self) -> LabelFilter<'_> {
LabelFilter::new(&self.labels, &self.labels_not)
}
}