yog 0.0.1

yog: a balls-oriented session manager for lernie loops (egui frontend)
Documentation
//! Transcript view-model (DESIGN §5.1 #12, §11 Altitude-2 Transcript tab).
//!
//! An agent's conversation is a directory of message files at
//! `<workspace>/agents/<agent-id>/messages/`. Each file is
//! `NNN-<origin>.<ext>` — **order lives in the filename** (the `NNN`
//! counter), and **origin lives in the filename token**:
//!
//! | ext  | origin token | entry                                        |
//! |------|--------------|----------------------------------------------|
//! | `md` | `<sender>`   | delivered message, body verbatim             |
//! | `json` | `tool`     | `tool_result` (content / is_error / tool_use_id) |
//! | `json` | `<model-id>` | model output — canonical content blocks    |
//! | other / unparseable name / unparseable bytes | — | Raw bucket (never dropped) |
//!
//! Everything is a pure function of the on-disk bytes (§3.5 stateless
//! re-read): no field caches a fact the files already carry. "Tool in
//! progress" is a *query* over the entries (a committed `tool_use` with no
//! committed `tool_result`), never a stored flag (PRINCIPLES: single source
//! of truth). When the agent's latest step is in flight, [`build`] appends
//! the live streaming tail as a virtual trailing entry, folded through the
//! one shared JSONL parser ([`crate::git_tree::streaming_text_from_disk`]).

use std::path::{Path, PathBuf};

mod render;
pub use render::render;

/// Directory under the workspace holding the per-agent worktrees (ARCH §2.3).
const AGENTS_DIR: &str = "agents";
/// The committed-transcript directory inside an agent's worktree.
const MESSAGES_DIR: &str = "messages";
/// The one reserved `.json` origin token: a `tool_result` payload.
const TOOL_ORIGIN: &str = "tool";
const MD_EXT: &str = "md";
const JSON_EXT: &str = "json";
/// Compact-JSON cap for a `tool_use` input summary chip; longer inputs are
/// truncated with an ellipsis (the full bytes remain under the Raw toggle).
const INPUT_SUMMARY_CAP: usize = 200;
/// Synthetic filename for the virtual live-streaming entry.
const STREAMING_NAME: &str = "«live»";

/// A parsed agent transcript: the ordered `messages/` entries, plus — when
/// the latest step is in flight — the live streaming tail as a virtual
/// trailing entry.
#[derive(Debug, Clone, PartialEq, Eq, Default)]
pub struct Transcript {
    pub entries: Vec<Entry>,
}

/// One transcript row. `raw` is the verbatim backing bytes surfaced by the
/// Raw toggle for *any* entry (§11 "every tab has a Raw toggle showing
/// verbatim bytes"); `kind` is the parsed projection.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct Entry {
    /// Source filename (`003-claude-opus.json`), or [`STREAMING_NAME`] for
    /// the virtual streaming entry.
    pub name: String,
    /// Verbatim backing bytes (the file's contents; the folded text for the
    /// streaming entry).
    pub raw: Vec<u8>,
    pub kind: EntryKind,
}

/// The origin classification of a transcript entry (see the module table).
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum EntryKind {
    /// `.md` — a delivered message; `sender` is the filename origin token.
    Delivered { sender: String, body: String },
    /// `NNN-<model>.json` — model output as canonical content blocks.
    Model {
        model_id: String,
        blocks: Vec<Block>,
    },
    /// `NNN-tool.json` — a `tool_result`.
    ToolResult {
        tool_use_id: String,
        content: String,
        is_error: bool,
    },
    /// The live streaming tail folded from the open `response.json`.
    Streaming { text: String },
    /// Unparseable filename or unparseable bytes — surfaced verbatim rather
    /// than dropped (§15 Y12: "surface them in a Raw bucket").
    Raw,
}

/// One content block of a model message (§4.4 canonical blocks).
#[derive(Debug, Clone, PartialEq, Eq)]
pub enum Block {
    Text(String),
    Thinking(String),
    /// A tool call: rendered as a chip with id / name / input summary.
    ToolUse {
        id: String,
        name: String,
        input_summary: String,
    },
}

impl Transcript {
    /// Is `tool_use_id` a committed `tool_use` still in progress — i.e. no
    /// committed `tool_result` anywhere in the sequence names it? Derived on
    /// demand, never stored (DESIGN §5.1 #12, §11).
    pub fn tool_in_progress(&self, tool_use_id: &str) -> bool {
        !self.entries.iter().any(|e| {
            matches!(&e.kind, EntryKind::ToolResult { tool_use_id: id, .. } if id == tool_use_id)
        })
    }
}

/// Build the transcript for `agent_id` in `workspace`. When `in_flight`
/// (the caller's `AgentState::InFlight` — the latest step's `response.json`
/// fd is still open, §3.5), the folded live tail is appended as a virtual
/// trailing entry.
pub fn build(workspace: &Path, agent_id: &str, in_flight: bool) -> Transcript {
    let dir = workspace.join(AGENTS_DIR).join(agent_id).join(MESSAGES_DIR);
    let mut entries = read_messages(&dir);
    if in_flight && let Some(text) = crate::git_tree::streaming_text_from_disk(workspace, agent_id)
    {
        entries.push(Entry {
            name: STREAMING_NAME.to_string(),
            raw: text.clone().into_bytes(),
            kind: EntryKind::Streaming { text },
        });
    }
    Transcript { entries }
}

/// Enumerate `messages/` as entries in filename order. The zero-padded `NNN`
/// counter makes lexicographic filename order the true message order, so a
/// plain string sort suffices. Non-files (a stray subdir) are skipped; an
/// absent directory yields no entries.
fn read_messages(dir: &Path) -> Vec<Entry> {
    let Ok(read) = std::fs::read_dir(dir) else {
        return Vec::new();
    };
    let mut files: Vec<(String, PathBuf)> = read
        .flatten()
        .filter_map(|e| {
            let path = e.path();
            path.is_file()
                .then(|| (e.file_name().to_string_lossy().into_owned(), path))
        })
        .collect();
    files.sort_by(|a, b| a.0.cmp(&b.0));
    files
        .into_iter()
        .map(|(name, path)| {
            let raw = std::fs::read(&path).unwrap_or_default();
            let kind = classify(&name, &raw);
            Entry { name, raw, kind }
        })
        .collect()
}

/// Classify one entry by its filename origin token and bytes.
fn classify(name: &str, raw: &[u8]) -> EntryKind {
    let Some((origin, ext)) = parse_name(name) else {
        return EntryKind::Raw;
    };
    match ext {
        MD_EXT => EntryKind::Delivered {
            sender: origin.to_string(),
            body: String::from_utf8_lossy(raw).into_owned(),
        },
        JSON_EXT if origin == TOOL_ORIGIN => parse_tool_result(raw).unwrap_or(EntryKind::Raw),
        JSON_EXT => parse_model(origin, raw),
        _ => EntryKind::Raw,
    }
}

/// Split `NNN-<origin>.<ext>` into `(origin, ext)`. The `NNN` is validated
/// (leading digit run) but not returned — filename order already carries it.
/// Anything not matching the shape is `None` (→ Raw bucket).
fn parse_name(name: &str) -> Option<(&str, &str)> {
    let (stem, ext) = name.rsplit_once('.')?;
    let (num, origin) = stem.split_once('-')?;
    if num.is_empty() || !num.bytes().all(|b| b.is_ascii_digit()) || origin.is_empty() {
        return None;
    }
    Some((origin, ext))
}

/// Parse a model `.json` into content blocks. Valid JSON always yields a
/// `Model` (possibly with no recognized blocks); only unparseable bytes fall
/// to the Raw bucket.
fn parse_model(model_id: &str, raw: &[u8]) -> EntryKind {
    match serde_json::from_slice::<serde_json::Value>(raw) {
        Ok(value) => EntryKind::Model {
            model_id: model_id.to_string(),
            blocks: blocks_from_value(&value),
        },
        Err(_) => EntryKind::Raw,
    }
}

/// Content blocks from a model payload — forgiving over the two canonical
/// shapes: a bare block array, or a message object with a `content` array or
/// string. Anything else yields no blocks (the Raw toggle still shows bytes).
fn blocks_from_value(value: &serde_json::Value) -> Vec<Block> {
    if let Some(arr) = value.as_array() {
        return arr.iter().filter_map(block_from_value).collect();
    }
    match value.get("content") {
        Some(serde_json::Value::Array(arr)) => arr.iter().filter_map(block_from_value).collect(),
        Some(serde_json::Value::String(s)) => vec![Block::Text(s.clone())],
        _ => Vec::new(),
    }
}

/// One canonical content block, or `None` for an untyped / unknown-typed
/// element (skipped — the whole entry stays inspectable via Raw).
fn block_from_value(v: &serde_json::Value) -> Option<Block> {
    match v.get("type").and_then(|t| t.as_str())? {
        "text" => Some(Block::Text(str_field(v, "text"))),
        "thinking" => Some(Block::Thinking(str_field(v, "thinking"))),
        "tool_use" => Some(Block::ToolUse {
            id: str_field(v, "id"),
            name: str_field(v, "name"),
            input_summary: summarize_input(v.get("input")),
        }),
        _ => None,
    }
}

/// A compact one-line JSON summary of a `tool_use` input, capped.
fn summarize_input(input: Option<&serde_json::Value>) -> String {
    let Some(v) = input else {
        return String::new();
    };
    let compact = v.to_string();
    if compact.chars().count() > INPUT_SUMMARY_CAP {
        let head: String = compact.chars().take(INPUT_SUMMARY_CAP).collect();
        format!("{head}…")
    } else {
        compact
    }
}

/// Parse a `tool` `.json` into a `ToolResult`. `None` (→ Raw) when the bytes
/// don't parse or carry no `tool_use_id`-bearing result block.
fn parse_tool_result(raw: &[u8]) -> Option<EntryKind> {
    let value: serde_json::Value = serde_json::from_slice(raw).ok()?;
    let block = find_tool_result(&value)?;
    Some(EntryKind::ToolResult {
        tool_use_id: str_field(block, "tool_use_id"),
        content: tool_result_content(block.get("content")),
        is_error: block
            .get("is_error")
            .and_then(serde_json::Value::as_bool)
            .unwrap_or(false),
    })
}

/// Locate the result block — the value itself when it carries a
/// `tool_use_id`, else the first such element of a wrapping `content` array.
fn find_tool_result(value: &serde_json::Value) -> Option<&serde_json::Value> {
    if value.get("tool_use_id").is_some() {
        return Some(value);
    }
    value
        .get("content")?
        .as_array()?
        .iter()
        .find(|b| b.get("tool_use_id").is_some())
}

/// Flatten a `tool_result` `content` field to text: a bare string verbatim,
/// an array of blocks concatenated (text blocks' text; else compact JSON),
/// anything else compact JSON, absent/null empty.
fn tool_result_content(content: Option<&serde_json::Value>) -> String {
    match content {
        None | Some(serde_json::Value::Null) => String::new(),
        Some(serde_json::Value::String(s)) => s.clone(),
        Some(serde_json::Value::Array(arr)) => arr.iter().map(content_piece).collect(),
        Some(other) => other.to_string(),
    }
}

/// One element of a `tool_result` content array: a text block's text, else
/// its compact JSON.
fn content_piece(v: &serde_json::Value) -> String {
    if v.get("type").and_then(|t| t.as_str()) == Some("text") {
        str_field(v, "text")
    } else {
        v.to_string()
    }
}

/// A string field of a JSON object, or empty when absent / non-string.
fn str_field(v: &serde_json::Value, key: &str) -> String {
    v.get(key)
        .and_then(|x| x.as_str())
        .unwrap_or_default()
        .to_string()
}

#[cfg(test)]
mod tests;