tga 9.0.0

Developer productivity analytics — git commit collection, classification, and reporting
Documentation
//! PM work meaningfulness classification — the WORK tier of the
//! Activity / Work / Effort model for tickets (issue #3916, epic #3914).
//!
//! Why: a raw ticket count is an ACTIVITY number. It counts a one-word
//! sub-task ("Final", "Phase 3") identically to a substantive story, so a PM
//! who decomposes work into many stubs outscores one who does not. This
//! module filters that activity down to the meaningful subset, and the
//! verdict is persisted to `fact_pm_work` (`core::db::pm_work`).
//!
//! What: [`classify`] is a deterministic function of one ticket's title,
//! description, reporter, and whether a human ever moved it. There is no LLM
//! tier in v1 — every input the classifier reads is already in the database.
//!
//! Test: `tests` in `tests.rs`.
//!
//! # Versioning
//!
//! Every persisted row records [`FORMULA_VERSION`]. The thresholds in
//! [`thresholds`] are v1 constants: a retune ships as a NEW version string,
//! never as an edit of the values below, so a stored verdict always names the
//! rule set that produced it.

pub mod extract;

use std::fmt;

/// Threshold set version baked into every persisted `fact_pm_work` row.
///
/// Why: a later retune of the constants in [`thresholds`] must be
/// distinguishable from v1 in already-stored rows.
/// What: the string `"pm-work-1"`, written to `fact_pm_work.formula_version`.
/// Test: `classifier_reports_the_v1_formula_version`.
pub const FORMULA_VERSION: &str = "pm-work-1";

/// v1 meaningfulness thresholds (`formula_version = "pm-work-1"`).
///
/// Why: the exclusion rules in issue #3916 are heuristics calibrated against
/// one JIRA instance, so they will be retuned. Keeping every tunable number
/// in one named block means a retune is a reviewable diff of this module and
/// a bump of [`FORMULA_VERSION`] — never a scattered edit that silently
/// changes what already-stored rows meant.
/// What: the word-count bounds and bot-account markers the v1 rules read.
/// Nothing outside this block is tunable.
/// Test: `terse_title_boundary_is_the_documented_threshold`.
pub mod thresholds {
    /// A title of at most this many words is "terse" (issue #3916: ≤5 words).
    pub const TERSE_TITLE_MAX_WORDS: usize = 5;

    /// A description of at least this many words rescues a terse title.
    /// One encodes issue #3916's "AND no description body" clause: any real
    /// prose at all makes the ticket meaningful. Raising it is how a retune
    /// would demand a substantive body rather than merely a non-empty one.
    pub const SUBSTANTIVE_BODY_MIN_WORDS: usize = 1;

    /// Substrings that identify an automation account when found anywhere in
    /// a lowercased reporter name. These are unambiguous — no human display
    /// name contains them.
    pub const BOT_NAME_MARKERS: &[&str] = &[
        "[bot]",
        "dependabot",
        "renovate",
        "github-actions",
        "jenkins",
        "automation",
        "service-account",
        "serviceaccount",
        "svc-",
        "noreply",
        "codecov",
        "snyk",
        "sonarqube",
    ];

    /// Whole words that identify an automation account. Matched against
    /// alphanumeric tokens of the reporter name rather than as substrings,
    /// because each is a substring of ordinary names — "bot" of "Abbott",
    /// "ci" of "Lucia" and "Marie-Cindy". Token matching is what lets `ci`
    /// catch the generic `release-ci` / `deploy-ci` / `ci-runner` accounts
    /// without those false positives.
    pub const BOT_NAME_TOKENS: &[&str] = &["bot", "bots", "robot", "system", "automation", "ci"];
}

/// Why a ticket was excluded from the meaningful-work subset.
///
/// Why: the exclusion list is issue #3916's, verbatim — the consumer
/// (cto-reports) groups by this value to report which filter removed what.
/// What: a four-valued enum stored as text in
/// `fact_pm_work.exclusion_reason`; [`ExclusionReason::None`] is stored as
/// `'NONE'` rather than SQL NULL so every row carries a groupable verdict.
/// Test: `exclusion_reason_round_trips_through_its_wire_string`.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Hash)]
pub enum ExclusionReason {
    /// Not excluded — the ticket is meaningful PM work.
    None,
    /// Short title with no description body: a decomposition stub such as
    /// "Final" or "Phase 3", not a unit of management labor.
    TerseTitle,
    /// Filed by an automation account and never touched by a human — a
    /// system-driven ticket, not PM intent.
    AutoGenerated,
    /// Filed by an automation account but later transitioned by a human —
    /// engineering automation a person triaged, still not PM authorship.
    BotFiled,
}

impl ExclusionReason {
    /// The value stored in `fact_pm_work.exclusion_reason`.
    #[must_use]
    pub const fn as_wire_str(self) -> &'static str {
        match self {
            Self::None => "NONE",
            Self::TerseTitle => "TERSE_TITLE",
            Self::AutoGenerated => "AUTO_GENERATED",
            Self::BotFiled => "BOT_FILED",
        }
    }

    /// Parse a stored `fact_pm_work.exclusion_reason` value.
    ///
    /// Returns `None` for an unrecognised string — a row written by a newer
    /// `formula_version` than this binary knows about — so callers decide
    /// whether to skip or fail rather than silently reading it as `NONE`.
    #[must_use]
    pub fn from_wire_str(s: &str) -> Option<Self> {
        match s {
            "NONE" => Some(Self::None),
            "TERSE_TITLE" => Some(Self::TerseTitle),
            "AUTO_GENERATED" => Some(Self::AutoGenerated),
            "BOT_FILED" => Some(Self::BotFiled),
            _ => None,
        }
    }
}

impl fmt::Display for ExclusionReason {
    fn fmt(&self, f: &mut fmt::Formatter<'_>) -> fmt::Result {
        f.write_str(self.as_wire_str())
    }
}

/// Everything the v1 classifier reads about one ticket.
///
/// Borrowed rather than owned: the caller already holds these as columns of a
/// `work_items` row and the classifier never keeps them.
#[derive(Debug, Clone, Copy, Default)]
pub struct PmWorkInput<'a> {
    /// `work_items.title`.
    pub title: &'a str,
    /// Plain-text description extracted from `work_items.raw_json`; see
    /// [`extract::extract_fields`]. `None` when the payload carried none.
    pub description: Option<&'a str>,
    /// Reporter display name from the same payload. `None` when absent — an
    /// unknown reporter is treated as human, since excluding tickets for a
    /// missing field would filter on payload completeness, not on labor.
    pub reporter: Option<&'a str>,
    /// Whether any `fact_ticket_transitions` row for this ticket names a
    /// non-bot author. Distinguishes [`ExclusionReason::BotFiled`] from
    /// [`ExclusionReason::AutoGenerated`].
    pub human_transitioned: bool,
}

/// One ticket's meaningfulness verdict.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
pub struct PmWorkVerdict {
    /// `true` when the ticket counts toward meaningful PM work.
    pub is_meaningful: bool,
    /// Which rule excluded it; [`ExclusionReason::None`] when meaningful.
    pub exclusion_reason: ExclusionReason,
    /// Words in the title, recorded so a retune can be evaluated against
    /// stored rows without re-reading the source payloads.
    pub title_word_count: usize,
    /// Words in the description body, recorded for the same reason.
    pub body_word_count: usize,
}

/// Whether `name` is an automation account rather than a person.
///
/// Why: two of the three exclusion rules key off reporter identity, and a
/// naive `contains("bot")` misfires on surnames like "Abbott".
/// What: lowercases `name`, tests it against
/// [`thresholds::BOT_NAME_MARKERS`] as substrings, then splits it into
/// alphanumeric tokens and tests those against
/// [`thresholds::BOT_NAME_TOKENS`] as whole words.
/// Test: `bot_detection_matches_automation_accounts`,
/// `bot_detection_does_not_misfire_on_human_names`.
#[must_use]
pub fn is_bot_account(name: &str) -> bool {
    let lower = name.to_lowercase();
    if thresholds::BOT_NAME_MARKERS
        .iter()
        .any(|m| lower.contains(m))
    {
        return true;
    }
    lower
        .split(|c: char| !c.is_alphanumeric())
        .any(|token| thresholds::BOT_NAME_TOKENS.contains(&token))
}

/// Count whitespace-separated words.
///
/// Deliberately naive: the input is already plain text (HTML tags and
/// Atlassian-Document-Format wrappers are stripped by
/// [`extract::extract_fields`]), and a locale-aware tokenizer would change
/// the meaning of the stored counts without changing any verdict near the
/// v1 thresholds.
#[must_use]
pub fn word_count(text: &str) -> usize {
    text.split_whitespace().count()
}

/// Classify one ticket's meaningfulness under `formula_version = "pm-work-1"`.
///
/// Why: see the module docs — this is the WORK tier of #3914, filtering the
/// raw ticket count down to tickets that represent management labor.
/// What: applies issue #3916's three exclusion rules in a fixed precedence,
/// because a ticket can satisfy more than one and the stored reason must be
/// deterministic:
///
/// 1. The reporter is an automation account ([`is_bot_account`]). A human
///    later transitioned it → [`ExclusionReason::BotFiled`]; nobody did →
///    [`ExclusionReason::AutoGenerated`]. Provenance is checked first: a
///    bot-filed stub is excluded for being bot-filed, which is the more
///    specific fact about it.
/// 2. The title is at most [`thresholds::TERSE_TITLE_MAX_WORDS`] words AND
///    the body is under [`thresholds::SUBSTANTIVE_BODY_MIN_WORDS`] →
///    [`ExclusionReason::TerseTitle`].
/// 3. Otherwise the ticket is meaningful.
///
/// Test: `terse_title_with_no_body_is_excluded`,
/// `bot_reporter_with_no_human_transition_is_auto_generated`,
/// `bot_reporter_with_human_transition_is_bot_filed`,
/// `substantive_story_is_meaningful`.
#[must_use]
pub fn classify(input: &PmWorkInput<'_>) -> PmWorkVerdict {
    let title_word_count = word_count(input.title);
    let body_word_count = input.description.map_or(0, word_count);

    let reason = if input.reporter.is_some_and(is_bot_account) {
        if input.human_transitioned {
            ExclusionReason::BotFiled
        } else {
            ExclusionReason::AutoGenerated
        }
    } else if title_word_count <= thresholds::TERSE_TITLE_MAX_WORDS
        && body_word_count < thresholds::SUBSTANTIVE_BODY_MIN_WORDS
    {
        ExclusionReason::TerseTitle
    } else {
        ExclusionReason::None
    };

    PmWorkVerdict {
        is_meaningful: reason == ExclusionReason::None,
        exclusion_reason: reason,
        title_word_count,
        body_word_count,
    }
}

#[cfg(test)]
#[path = "tests.rs"]
mod tests;