tga 10.3.0

Developer productivity analytics — git commit collection, classification, and reporting
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
//! `tga eval` — the classifier precision harness (#111).
//!
//! Why: measure how often the classification cascade is right, per rule and
//! per method, from a human-labelled stratified sample.
//! What: `sample` draws the sample from a read-only database copy; `subsample`
//! draws a proportional subset of a sample; `repredict` re-derives a sample's
//! predictions under another config (#111); `score`
//! turns rater labels into a precision report. The library side lives in
//! [`tga::eval`]; this module only parses flags and prints summaries.
//! Test: `tests/eval_harness.rs`.

use std::path::{Path, PathBuf};

use anyhow::{bail, Context, Result};
use clap::{Args, Subcommand};

use tga::core::config::Config;
use tga::eval::{self, RepredictParams, SampleParams, ScoreParams, Stratum, SubsampleParams};

const PRIVACY: &str = "PRIVACY: every file this command writes contains commit text \
(subjects, bodies, paths, PR titles). Store the output directory privately, outside any \
repository, and delete it when the evaluation is done. Nothing is written unless you \
name the directory with --out.";

/// Arguments for `tga eval`.
#[derive(Args, Debug)]
#[command(after_help = PRIVACY)]
pub struct EvalArgs {
    /// Harness step.
    #[command(subcommand)]
    pub step: EvalSubcommand,
}

/// The harness steps.
#[derive(Subcommand, Debug)]
// #111: non_exhaustive so a new step is an additive change, not a SemVer break.
#[non_exhaustive]
pub enum EvalSubcommand {
    /// Draw a stratified, capped, seeded sample of classified commits for labelling.
    #[command(after_help = PRIVACY)]
    Sample(SampleArgs),
    /// Draw a seeded subset of an existing sample, proportional per stratum.
    ///
    /// Writes sample.jsonl, a blind labels.csv and strata.json for the subset.
    /// No label file is read. Score a rater of the subset alongside a rater of
    /// the full sample with `score --sample <source sample.jsonl>`.
    #[command(after_help = PRIVACY)]
    Subsample(SubsampleArgs),
    /// Re-derive an existing sample's predictions under the rules of --config.
    ///
    /// Keeps every row, its order and its stratum; replaces predicted_category,
    /// method, rule_id and confidence with what the config's rules give for
    /// that commit in --db. Writes <out>.jsonl and <out>.provenance.json (config
    /// and rules-file BLAKE3 hashes, tga version). A row whose commit is not in
    /// --db is an error. Requires an explicit --config; --rules replaces its
    /// rules file for this run.
    #[command(after_help = PRIVACY)]
    Repredict(RepredictArgs),
    /// Score rater labels against a sample: precision per rule, method and stratum.
    ///
    /// Valid labels are the categories recorded in strata.json, or the taxonomy
    /// and rule categories of the config when --config or --rules is passed
    /// explicitly, plus every predicted category in the sample, `unclear`, `mixed` and
    /// `release_merge`. Those three are reported per label but score as no
    /// answer. Merge commits (2+ parents) are excluded with their labels; a
    /// sample written without merge flags needs --db to resolve them.
    #[command(after_help = PRIVACY)]
    Score(ScoreArgs),
}

/// Flags for `tga eval sample`. The rules come from the global `--config`.
#[derive(Args, Debug)]
pub struct SampleArgs {
    /// A COPY of the tga database; opened read-only, refused if writable.
    /// The global --database flag is not used by this command.
    #[arg(long)]
    pub db: PathBuf,
    /// Window length in weeks, ending at the newest commit in the database.
    #[arg(long, default_value_t = 26)]
    pub weeks: u32,
    /// Total sample size, split equally across non-empty strata.
    #[arg(long, default_value_t = 400)]
    pub size: usize,
    /// RNG seed; the same seed on the same database draws the same sample.
    #[arg(long)]
    pub seed: u64,
    /// Maximum commits per repository and per author within each stratum.
    #[arg(long, default_value_t = 5)]
    pub cap: usize,
    /// Salt for author hashes; generated and saved as salt.txt when omitted.
    #[arg(long)]
    pub salt: Option<String>,
    /// Output directory (private, outside any repository). Required.
    #[arg(long)]
    pub out: PathBuf,
}

/// Flags for `tga eval subsample`.
#[derive(Args, Debug)]
pub struct SubsampleArgs {
    /// sample.jsonl written by `tga eval sample`.
    #[arg(long)]
    pub from: PathBuf,
    /// strata.json of that sample; defaults to the one next to it.
    #[arg(long)]
    pub strata: Option<PathBuf>,
    /// Rows in the subset (at most the source's rows).
    #[arg(long)]
    pub size: usize,
    /// RNG seed; the same seed on the same sample draws the same subset.
    #[arg(long)]
    pub seed: u64,
    /// A COPY of the tga database the sample was drawn from; opened read-only.
    /// Resolves the merge flag of rows that lack one so merges are left out
    /// of the subset (#111). Without it such a sample is refused.
    #[arg(long)]
    pub db: Option<PathBuf>,
    /// Output directory (private, outside any repository); must not already
    /// hold sample.jsonl, labels.csv or strata.json. Required.
    #[arg(long)]
    pub out: PathBuf,
}

/// Flags for `tga eval repredict`. The rules come from the global `--config`.
#[derive(Args, Debug)]
pub struct RepredictArgs {
    /// sample.jsonl whose predictions are re-derived.
    #[arg(long)]
    pub sample: PathBuf,
    /// A COPY of the tga database the sample was drawn from; opened read-only.
    #[arg(long)]
    pub db: PathBuf,
    /// Output .jsonl (private, outside any repository); it and its
    /// .provenance.json must not exist. Required.
    #[arg(long)]
    pub out: PathBuf,
    /// Rules file for this run, in place of `classification.rules_file`
    /// (as `tga rules --rules`). Its categories and `buckets:` map apply.
    #[arg(long)]
    pub rules: Option<PathBuf>,
}

/// #111: `--rules` stands in for `classification.rules_file`, as in
/// `tga rules --rules`.
fn with_rules(mut config: Config, rules: Option<PathBuf>) -> Config {
    if let Some(path) = rules {
        config
            .classification
            .get_or_insert_with(Default::default)
            .rules_files = vec![path];
    }
    config
}

/// Flags for `tga eval score`.
#[derive(Args, Debug)]
pub struct ScoreArgs {
    /// sample.jsonl written by `tga eval sample`.
    #[arg(long)]
    pub sample: PathBuf,
    /// Rater label file (labels.csv with the `label` column filled; a blank
    /// label is unlabelled). The first file is the scored rater: precision
    /// and coverage use its labels over its rows. Repeat once for a second
    /// rater, whose rows may differ; Cohen's kappa is then reported over the
    /// SHAs both labelled.
    #[arg(long, required = true, num_args = 1)]
    pub labels: Vec<PathBuf>,
    /// Adjudicated labels (same format); they replace the first rater's label
    /// on the rows they name, which that rater must have labelled.
    #[arg(long)]
    pub adjudicated: Option<PathBuf>,
    /// strata.json; defaults to the one next to the sample.
    #[arg(long)]
    pub strata: Option<PathBuf>,
    /// A COPY of the tga database the sample was drawn from; opened read-only.
    /// Resolves the merge flag of rows that lack one (samples written before
    /// merges were excluded) so merges can be dropped. Without it such a
    /// sample is refused rather than scored with merges in it.
    #[arg(long)]
    pub db: Option<PathBuf>,
    /// Output directory for report.md and report.json. Required.
    #[arg(long)]
    pub out: PathBuf,
    /// Rules file for this run, in place of `classification.rules_file`
    /// (as `tga rules --rules`). Its categories are valid labels and its
    /// `buckets:` map applies when the config has none.
    #[arg(long)]
    pub rules: Option<PathBuf>,
}

fn warn_if_in_repo(out: &Path) {
    let probe = out
        .ancestors()
        .find(|p| p.exists())
        .unwrap_or(Path::new("."));
    if let Ok(repo) = git2::Repository::discover(probe) {
        let root = repo.workdir().unwrap_or(repo.path());
        eprintln!(
            "warning: {} is inside the git work tree {}; these files contain commit text — \
             keep them out of version control",
            out.display(),
            root.display()
        );
    }
}

/// Run `tga eval`.
///
/// `config` is the loaded global config; `config_path` says which file it came
/// from, and `config_explicit` whether the user passed `--config`.
pub fn run(
    args: EvalArgs,
    config: Config,
    config_path: &Path,
    config_explicit: bool,
) -> Result<()> {
    match args.step {
        EvalSubcommand::Sample(a) => run_sample(a, config, config_path),
        EvalSubcommand::Subsample(a) => run_subsample(a),
        EvalSubcommand::Repredict(a) => run_repredict(a, config, config_path, config_explicit),
        EvalSubcommand::Score(a) => run_score(a, config, config_explicit),
    }
}

fn run_sample(a: SampleArgs, config: Config, config_path: &Path) -> Result<()> {
    if !config_path.exists() {
        bail!(
            "config file {} not found — `tga eval sample` evaluates a config's rules, pass it with --config",
            config_path.display()
        );
    }
    warn_if_in_repo(&a.out);
    let summary = eval::run_sample(&SampleParams {
        db: a.db,
        config,
        weeks: a.weeks,
        size: a.size,
        seed: a.seed,
        cap: a.cap,
        salt: a.salt,
        out: a.out.clone(),
    })
    .context("tga eval sample")?;

    let s = &summary.strata;
    println!(
        "Window {} → {} ({} weeks): {} commits, seed {}, cap {}",
        s.window_start, s.window_end, s.weeks, s.population, s.seed, s.cap
    );
    // #111: merges (2+ parents) never enter the eval.
    println!(
        "Excluded {} merge commits (2+ parents) from the window.",
        s.merges_excluded
    );
    println!(
        "{:<14} {:>10} {:>8} {:>10}",
        "stratum", "population", "sampled", "weight"
    );
    let mut total = 0;
    for stratum in Stratum::ALL {
        let c = s.strata.get(stratum.as_str()).cloned().unwrap_or_default();
        total += c.sampled;
        let weight = if c.sampled > 0 {
            format!("{:.2}", c.population as f64 / c.sampled as f64)
        } else {
            "—".into()
        };
        println!(
            "{:<14} {:>10} {:>8} {:>10}",
            stratum.as_str(),
            c.population,
            c.sampled,
            weight
        );
    }
    println!("Sampled {total} of {} requested.", s.requested_size);
    if total < s.requested_size && total < s.population {
        println!(
            "Note: the per-repo and per-author --cap limited the sample; with few repositories or authors, raise --cap."
        );
    }
    if summary.drifted > 0 {
        println!(
            "Note: {} stored verdicts differ from the current rules; the sample measures the current rules.",
            summary.drifted
        );
    }
    if summary.bad_timestamps > 0 {
        println!(
            "Note: {} commits skipped for an unparseable timestamp.",
            summary.bad_timestamps
        );
    }
    for f in &summary.files {
        println!("wrote {}", f.display());
    }
    if let Some(salt) = &summary.salt_file {
        println!(
            "Generated salt saved to {}; pass it with --salt to redraw identical author hashes.",
            salt.display()
        );
    }
    println!(
        "Valid labels: {}, {}",
        s.categories.join(", "),
        eval::score::NO_ANSWER_LABELS.join(", ")
    );
    println!("{PRIVACY}");
    Ok(())
}

fn run_subsample(a: SubsampleArgs) -> Result<()> {
    warn_if_in_repo(&a.out);
    let summary = eval::run_subsample(&SubsampleParams {
        from: a.from,
        strata: a.strata,
        size: a.size,
        seed: a.seed,
        db: a.db,
        out: a.out,
    })
    .context("tga eval subsample")?;
    let s = &summary.strata;
    let origin = s.subsample.clone().unwrap_or_default();
    println!(
        "Subset of {} rows from {}, seed {}",
        origin.size, origin.source_size, origin.seed
    );
    println!("{:<14} {:>10} {:>8}", "stratum", "population", "subset");
    for stratum in Stratum::ALL {
        let c = s.strata.get(stratum.as_str()).cloned().unwrap_or_default();
        if c.sampled > 0 {
            println!(
                "{:<14} {:>10} {:>8}",
                stratum.as_str(),
                c.population,
                c.sampled
            );
        }
    }
    for f in &summary.files {
        println!("wrote {}", f.display());
    }
    println!("{PRIVACY}");
    Ok(())
}

/// #111: re-predict a sample under an explicitly named config.
fn run_repredict(
    a: RepredictArgs,
    config: Config,
    config_path: &Path,
    config_explicit: bool,
) -> Result<()> {
    if !config_explicit || !config_path.exists() {
        bail!(
            "`tga eval repredict` applies a config's rules; name an existing config file with --config"
        );
    }
    warn_if_in_repo(&a.out);
    let summary = eval::run_repredict(&RepredictParams {
        sample: a.sample,
        db: a.db,
        config: with_rules(config, a.rules),
        config_path: config_path.to_path_buf(),
        out: a.out,
    })
    .context("tga eval repredict")?;
    let p = &summary.provenance;
    println!(
        "Re-predicted {} rows under {} (blake3 {}): {} changed, {} abstain, {} carried from the \
         database {:?}, {} stored verdicts superseded",
        p.rows,
        p.config.path,
        &p.config.blake3[..16],
        p.changed,
        p.abstentions,
        p.carried.values().sum::<u64>(),
        p.carried,
        p.superseded
    );
    for f in &summary.files {
        println!("wrote {}", f.display());
    }
    println!(
        "Score it with `tga eval score --sample {} --strata <the source strata.json> --config <config>`.",
        summary.files[0].display()
    );
    println!("{PRIVACY}");
    Ok(())
}

fn run_score(a: ScoreArgs, config: Config, config_explicit: bool) -> Result<()> {
    if a.labels.len() > 2 {
        bail!("pass at most two --labels files");
    }
    warn_if_in_repo(&a.out);
    let explicit = config_explicit || a.rules.is_some();
    let config = with_rules(config, a.rules);
    let categories = if explicit {
        Some(eval::config_categories(&config).context("loading categories from --config")?)
    } else {
        None
    };
    // #111: `classification.buckets`, else the rules file's map, else the
    // built-in fallback.
    let (buckets, source) = tga::classify::ClassificationPipeline::new(config)
        .effective_bucket_map()
        .context("loading the bucket map")?;
    println!("Bucket map from {}", source.describe());
    let report = eval::run_score_with_buckets(
        &ScoreParams {
            sample: a.sample,
            strata: a.strata,
            labels: a.labels,
            adjudicated: a.adjudicated,
            categories,
            db: a.db,
            out: a.out.clone(),
        },
        &buckets,
    )
    .context("tga eval score")?;

    println!(
        "Sample {} · labelled {} · scored {} · unclear {} · mixed {} · release_merge {} · unresolved {}",
        report.sample_size,
        report.labelled,
        report.scored,
        report.unclear,
        report.mixed,
        report.release_merge,
        report.unresolved_disagreements
    );
    // #111: merges are excluded from the eval; say how many.
    println!("{} rows excluded as merges", report.merges_excluded);
    if report.window_merges_estimated > 0 || report.window_merges_exact.is_some() {
        println!(
            "Window merges: {} estimated from the sample's strata, {} exact from --db",
            report.window_merges_estimated,
            report
                .window_merges_exact
                .map_or_else(|| "—".to_string(), |n| n.to_string())
        );
    }
    if let Some(w) = &report.weighted_accuracy {
        println!(
            "Stratum-weighted accuracy {:.1}% [{:.1}%, {:.1}%]",
            w.estimate * 100.0,
            w.ci_low * 100.0,
            w.ci_high * 100.0
        );
    }
    for (what, w) in [
        ("Primary (bucket)", &report.buckets.primary),
        ("Secondary (fine within bucket)", &report.buckets.secondary),
    ] {
        if let Some(w) = w {
            println!(
                "{what} stratum-weighted accuracy {:.1}% [{:.1}%, {:.1}%]",
                w.estimate * 100.0,
                w.ci_low * 100.0,
                w.ci_high * 100.0
            );
        }
    }
    println!("Abstention share {:.1}%", report.abstention.share * 100.0);
    if let Some(k) = report.kappa.as_ref().and_then(|k| k.kappa) {
        println!("Cohen's kappa {k:.3}");
    }
    println!("wrote {}", a.out.join("report.md").display());
    println!("wrote {}", a.out.join("report.json").display());
    Ok(())
}