tga 4.0.2

Developer productivity analytics — git commit collection, classification, and reporting
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
//! Domain models corresponding to the v1 database schema.
//!
//! These structs are the in-memory representation of rows in the core
//! tables. They are intentionally serialization-friendly via `serde` so
//! that they can be emitted as JSON in reports without an intermediate
//! DTO layer.

use chrono::{DateTime, Utc};
use serde::{Deserialize, Serialize};

/// A single commit observed in a repository.
///
/// Why: rows in the `commits` SQLite table need a typed in-memory
/// counterpart that both extractors and aggregators can share.
/// What: maps 1:1 onto the v1 `commits` schema. `Serialize`/`Deserialize`
/// derives let report formatters emit it as JSON without a DTO layer.
/// Test: covered indirectly by every test that inserts into the
/// `commits` table (see `core::tests::database_opens_with_wal_and_migrations_apply`).
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Commit {
    /// Primary key (database-assigned).
    pub id: i64,

    /// Full git OID (hex).
    pub sha: String,

    /// Foreign key into [`Author`].
    pub author_id: Option<i64>,

    /// Author display name as recorded in the commit.
    pub author_name: String,

    /// Author email as recorded in the commit.
    pub author_email: String,

    /// Author timestamp (UTC).
    pub timestamp: DateTime<Utc>,

    /// Commit message body (raw).
    pub message: String,

    /// Repository identifier (path or canonical name).
    pub repository: String,

    /// Number of files changed.
    pub files_changed: u32,

    /// Lines added.
    pub insertions: u32,

    /// Lines deleted.
    pub deletions: u32,

    /// Foreign key into [`Classification`], if classified.
    pub classification_id: Option<i64>,

    /// Confidence assigned by the classifier (0.0–1.0).
    pub confidence: Option<f64>,

    /// True for merge commits (parents > 1).
    pub is_merge: bool,

    /// True if the commit message references a known ticket system
    /// (JIRA/Linear-style `PROJ-123`, GitHub `fixes #123`, or bare `#123`).
    ///
    /// Computed at extraction time by [`crate::collect::ticket::is_ticketed`]
    /// and persisted on the `commits` row.
    pub ticketed: bool,
}

/// A canonical author / developer identity.
///
/// Why: the same physical developer often commits under multiple
/// `(name, email)` pairs; the `authors` table holds one row per resolved
/// identity so reports collapse them.
/// What: maps to the `authors` v1 schema with the alias list stored as a
/// JSON-encoded string.
/// Test: covered by `collect::identity::resolver` tests that exercise the
/// upsert path.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Author {
    /// Primary key (database-assigned).
    pub id: i64,

    /// Canonical display name.
    pub canonical_name: String,

    /// Canonical email address.
    pub canonical_email: String,

    /// JSON-encoded array of alias strings (names or emails).
    pub aliases: String,
}

/// A classification verdict produced by the cascade.
///
/// Why: classifications are stored once per unique outcome and referenced
/// by `commits.classification_id`, so the same `(category, subcategory,
/// method)` triple is not duplicated per commit.
/// What: maps to the `classifications` v1 schema; `method` records which
/// cascade tier produced the verdict.
/// Test: covered by `classify::pipeline` tests that exercise full-cascade
/// runs against an in-memory DB.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct Classification {
    /// Primary key (database-assigned).
    pub id: i64,

    /// Top-level category (e.g. `feature`, `bugfix`, `chore`).
    pub category: String,

    /// Optional finer-grained label.
    pub subcategory: Option<String>,

    /// Associated ticket identifier (e.g. `API-123`), if any.
    pub ticket_id: Option<String>,

    /// Confidence in this verdict (0.0–1.0).
    pub confidence: f64,

    /// Which tier of the cascade produced this verdict.
    pub method: ClassificationMethod,
}

/// File-level change record attached to a commit.
///
/// Why: per-file change data feeds the "files churned" and
/// "complexity" metrics; the per-file granularity must survive
/// round-tripping through SQLite.
/// What: maps to the `files` v1 schema with a typed `change_type`
/// (added / modified / deleted / renamed).
/// Test: covered indirectly by the git-extractor tests
/// (`collect::git::extractor`).
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct FileChange {
    /// Primary key (database-assigned).
    pub id: i64,

    /// Foreign key into [`Commit`].
    pub commit_id: i64,

    /// Relative path within the repository.
    pub path: String,

    /// Type of change.
    pub change_type: ChangeType,

    /// Lines added in this file.
    pub insertions: u32,

    /// Lines deleted in this file.
    pub deletions: u32,
}

/// A pull request record (typically GitHub).
///
/// Why: PR data drives the velocity / DORA lead-time / cycle-time
/// metrics; storing the full PR row lets us recompute those metrics
/// without re-fetching from the provider.
/// What: maps to the `pull_requests` v1 schema. Provider-specific PR
/// numbering means the `(provider, pr_number, repository)` triple is
/// the persistence-level unique identity.
/// Test: covered by `collect::github::client` and
/// `collect::bitbucket::client` tests that round-trip PR data through
/// the DB.
#[derive(Debug, Clone, Serialize, Deserialize)]
pub struct PullRequest {
    /// Primary key (database-assigned).
    pub id: i64,

    /// PR number within its repository.
    pub pr_number: u64,

    /// Repository this PR belongs to. Together with `provider` and
    /// `pr_number` this forms the persistence-level unique identity of a
    /// PR. GitHub assigns `pr_number` per-repository (so #1 in repo A is
    /// not the same PR as #1 in repo B); without this field the
    /// `(provider, pr_number)` unique index from migration v10 silently
    /// dropped cross-repo collisions during multi-repo collection (#88).
    ///
    /// Format is provider-specific:
    /// - GitHub: `"owner/repo"` (e.g. `"acme/widgets"`)
    /// - Bitbucket: `"workspace/repo_slug"`
    /// - Azure DevOps: `"project"` (PRs are project-scoped)
    pub repository: String,

    /// PR title.
    pub title: String,

    /// Author login.
    pub author: String,

    /// Lifecycle state.
    pub state: PrState,

    /// PR creation timestamp (UTC).
    pub created_at: DateTime<Utc>,

    /// Merge timestamp, if merged.
    pub merged_at: Option<DateTime<Utc>>,

    /// JSON-encoded array of commit SHAs in the PR.
    pub commit_shas: String,

    /// Client-set ingestion/snapshot timestamp (RFC3339, UTC).
    ///
    /// Why: issue #821, advisory follow-up to #752/#808. The upsert's
    /// `ON CONFLICT … DO UPDATE` unconditionally overwrote `state` /
    /// `merged_at` / `commit_shas` on every re-ingest; if a background job
    /// re-ingests an OLDER snapshot, a PR already recorded as `merged`
    /// could be overwritten back to `open`. `created_at` is the PR's
    /// immutable creation date — identical across re-ingests — so it
    /// cannot detect staleness. `fetched_at` changes on every fetch, so
    /// `store_pull_requests` guards the update with
    /// `WHERE excluded.fetched_at > pull_requests.fetched_at`.
    /// What: set once per fetch (`Utc::now().to_rfc3339()`) by the GitHub
    /// and Bitbucket collectors when the PR row is constructed; persisted
    /// via migration v22 (`0022_pull_requests_fetched_at.sql`).
    /// Test: `store_pull_requests_stale_write_guard_*` in
    /// `collect::github::client_tests` and `collect::bitbucket::tests`.
    pub fetched_at: String,

    /// Raw source-branch name this PR was opened from, when the provider
    /// supplies one.
    ///
    /// Why: #5734 — this is the ONLY place a commit's authoring branch
    /// survives. A commit object carries no branch, and `git branch --contains`
    /// answers a different question: under a squash-merge workflow it returns
    /// `main` for every merged commit and the authoring branch for none. Joined
    /// through [`Self::commit_shas`], this field is the per-commit branch
    /// reference that #5734 asks for.
    ///
    /// What: `None` and `Some("")` mean different things, and the distinction
    /// is what keeps a missing branch from being silently harvested as nothing.
    /// `None` is "this provider makes no claim" — Bitbucket and Azure DevOps
    /// PRs fetched before #5734 wired their source branch through. `Some("")`
    /// is "the provider answered and the ref was empty", an anomaly the
    /// collector records as a [`crate::collect::CollectionFault`] with
    /// `skip_item` severity rather than passing over.
    ///
    /// GitHub reports a bare name (`feature/PROJ-123-thing`); Azure DevOps
    /// reports a full ref (`refs/heads/...`).
    /// [`crate::collect::ticket::branch_ticket_key`] normalizes both.
    ///
    /// Test: `collect::correlate::tests::branch_name_supplies_a_key_the_subject_lacks`.
    #[serde(default)]
    pub head_ref: Option<String>,

    /// Issue reference extracted from this PR's body at fetch time.
    ///
    /// Why: #5734 — a PR body routinely declares `Closes #N` for work the
    /// commit subject never names; this recovers a key for 51 merged commits in
    /// this repository. The EXTRACTED key is stored rather than the body
    /// because the body is 9.4 MB across 2433 PRs here and an unrestricted scan
    /// of it yields 2501 false JIRA-shaped keys — see
    /// [`crate::collect::ticket::pr_body_ticket_key`] for the measurement and
    /// `0024_pull_requests_head_ref_and_body_ticket.sql` for the trade.
    ///
    /// What: `Some("#N")` when the body declared an issue with an action
    /// keyword, `None` otherwise. Because the key is derived at fetch time,
    /// sharpening the extraction rule needs a re-collect to take effect on
    /// existing rows — that is the cost of not keeping the prose.
    ///
    /// Test: `collect::correlate::tests::pr_body_supplies_a_key_the_subject_lacks`.
    #[serde(default)]
    pub body_ticket_id: Option<String>,
}

/// Cascade tier that produced a classification.
///
/// Why: knowing which tier of the four-tier cascade produced a verdict
/// lets analytics tools surface low-confidence verdicts (e.g. the
/// catch-all routes through `LlmFallback` when LLM is enabled).
/// What: enum tagged with snake_case string values for DB persistence.
/// Test: covered by `classify::tests::engine_classify_batch_does_not_panic`
/// which asserts the cascade reports the correct tier.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ClassificationMethod {
    /// Matched a deterministic exact rule.
    ExactRule,
    /// Matched a regex rule.
    RegexRule,
    /// Matched via fuzzy similarity.
    FuzzyMatch,
    /// Assigned by an LLM fallback.
    LlmFallback,
    /// Set manually by a user override.
    Manual,
    /// Derived from an external ticket source (JIRA issue type or GitHub
    /// Issues label). Added for issue #260.
    ExternalSource,
    /// Composed from multiple weak signals via a weighted-sum model.
    ///
    /// Tier 2.5 — sits between the regex tier (Tier 2) and the fuzzy tier
    /// (Tier 3). Blends keyword-density, ticket-prefix presence, message-length
    /// bucket, merge indicator, and file-path signals into per-category scores.
    /// Added for issue #270.
    WeightedSum,
    /// Derived from the catch-all rule (lowest-priority, confidence 0.3).
    ///
    /// Why: distinguishes the explicit catch-all from other fuzzy verdicts so
    /// downstream consumers can filter "true unknowns" separately.
    /// Added for issue #445 batch C.
    CatchAll,
    /// Applied by the `repo_categories` fallback tier (Tier 5, #445 batch C).
    ///
    /// Why: lets callers distinguish a confident classification from a
    /// repo-default assignment, enabling metric-level filtering.
    RepoCategoryFallback,
}

impl ClassificationMethod {
    /// Stable string representation used for DB storage.
    ///
    /// Why: rusqlite needs a `&str` to bind to the `method` column; the
    /// values must stay stable across releases so existing rows continue
    /// to round-trip correctly.
    /// What: returns the lowercase snake_case label for each variant.
    /// Test: covered indirectly by every classification test that reads
    /// or writes the `classifications` table.
    pub fn as_str(&self) -> &'static str {
        match self {
            ClassificationMethod::ExactRule => "exact_rule",
            ClassificationMethod::RegexRule => "regex_rule",
            ClassificationMethod::FuzzyMatch => "fuzzy_match",
            ClassificationMethod::LlmFallback => "llm_fallback",
            ClassificationMethod::Manual => "manual",
            ClassificationMethod::ExternalSource => "external_source",
            ClassificationMethod::WeightedSum => "weighted_sum",
            ClassificationMethod::CatchAll => "catch_all",
            ClassificationMethod::RepoCategoryFallback => "repo_category_fallback",
        }
    }
}

/// File change kind for [`FileChange`].
///
/// Why: distinguishing add / modify / delete / rename lets reports
/// separate "new code" from "code shuffled around" without re-parsing
/// the git diff.
/// What: 4-variant enum with snake_case strings for DB persistence.
/// Test: covered by `collect::git::diff` extractor tests.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum ChangeType {
    /// File was added.
    Added,
    /// File contents were modified.
    Modified,
    /// File was deleted.
    Deleted,
    /// File was renamed (and possibly modified).
    Renamed,
}

impl ChangeType {
    /// Stable string representation used for DB storage.
    ///
    /// Why: see [`ClassificationMethod::as_str`] — same persistence
    /// invariant applies.
    /// What: returns the snake_case label for each variant.
    /// Test: covered by `collect::git::diff` tests.
    pub fn as_str(&self) -> &'static str {
        match self {
            ChangeType::Added => "added",
            ChangeType::Modified => "modified",
            ChangeType::Deleted => "deleted",
            ChangeType::Renamed => "renamed",
        }
    }
}

/// Lifecycle state of a [`PullRequest`].
///
/// Why: cycle-time and DORA lead-time only apply to merged PRs;
/// surfacing the state lets reports filter without joining extra tables.
/// What: open / closed / merged tri-state with snake_case persistence.
/// Test: covered by `collect::github::client` round-trip tests.
#[derive(Debug, Clone, Copy, PartialEq, Eq, Serialize, Deserialize)]
#[serde(rename_all = "snake_case")]
pub enum PrState {
    /// PR is open.
    Open,
    /// PR was closed without merging.
    Closed,
    /// PR was merged.
    Merged,
}

impl PrState {
    /// Stable string representation used for DB storage.
    ///
    /// Why: see [`ClassificationMethod::as_str`] — same persistence
    /// invariant applies.
    /// What: returns the snake_case label for each variant.
    /// Test: covered by `collect::github::client` round-trip tests.
    pub fn as_str(&self) -> &'static str {
        match self {
            PrState::Open => "open",
            PrState::Closed => "closed",
            PrState::Merged => "merged",
        }
    }
}