trusty-common 0.52.3

Shared utilities and provider-agnostic streaming chat (ChatProvider, OllamaProvider, OpenRouter, tool-use) for trusty-* projects
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
//! Upload trusty-* log files to object storage (#6533, epic phase 1+2).
//!
//! Why: every trusty daemon writes logs to a local path that nothing prunes and
//! nothing collects. Diagnosing a failure on someone else's machine means
//! asking them to find and send a file. This module is the piece that gets
//! those bytes somewhere durable, scrubbed of credentials, without re-uploading
//! what has not changed.
//!
//! What: the drain CORE — a [`LogDestination`](crate::log_drain::LogDestination)
//! trait with `s3://` and `file://` adapters, a
//! [`DestinationUri`](crate::log_drain::DestinationUri) parser, the key layout,
//! an idempotency [`DrainManifest`](crate::log_drain::DrainManifest), the
//! [`collect`](crate::log_drain::collect) pipeline, and
//! [`run_once`](crate::log_drain::run_once). Phase 1+2 is
//! the core ONLY: there is no scheduler, no config plumbing, and no consumer
//! wired up. Phase 3 adds those, plus project-identity resolution — this module
//! deliberately does not resolve an identity, it demands one.
//!
//! Test: `tests` (sibling module). Run it with
//! `cargo test -p trusty-common --features log-drain --no-fail-fast`.
//!
//! # Key layout
//!
//! ```text
//! <destination prefix>/<owner>/<project>/<crate>/<relative file>
//! ```
//!
//! The destination prefix comes from the URI (`s3://bucket/PREFIX`); everything
//! after it is built by
//! [`DrainTarget::key_prefix`](crate::log_drain::DrainTarget::key_prefix).
//! #6657 replaced the earlier `<github_id>/<session_id>/logs/…` layout;
//! objects already uploaded under it stay where they are, because the manifest
//! is keyed by destination and key. Per-target reference
//! documentation: `docs/reference/log-drain.md`.
//!
//! # What the caller owns
//!
//! - **Identity.** [`DrainTarget`](crate::log_drain::DrainTarget) is supplied,
//!   never resolved here. An empty `owner` or `project` is refused with
//!   [`DrainError::MissingIdentity`](crate::log_drain::DrainError::MissingIdentity)
//!   rather than defaulted.
//! - **Single-flight.** [`run_once`](crate::log_drain::run_once) is exactly what
//!   its name says: one pass,
//!   no locking. Two concurrent runs against one target will both upload and
//!   both rewrite the manifest, and the loser's entries are lost. The scheduler
//!   Phase 3 adds owns that mutual exclusion.
//! - **The secret list.** [`crate::credentials::scrub_secrets`] removes
//!   values it is GIVEN; it does not detect secret-shaped strings. A caller that
//!   passes an empty list gets no scrubbing.

mod collector;
mod destination;
mod error;
mod manifest;
mod pipeline;
mod single_source;
mod uri;

#[cfg(test)]
mod tests;

use std::path::PathBuf;

pub use collector::{
    CollectLimits, Collected, CollectedFile, DEFAULT_MAX_FILE_BYTES, DEFAULT_MAX_WIRE_BYTES, Level,
    LogSource, OversizeFile, collect,
};
pub use destination::{LIST_LIMIT, LogDestination, ObjectMeta, ObjectStoreDestination, PutMeta};
pub use error::DrainError;
pub use manifest::{
    DrainManifest, MANIFEST_FILENAME, MANIFEST_VERSION, ManifestEntry, ManifestOrigin, SkipReason,
    SkipRecord, StatDecision,
};
pub use single_source::{
    DEFAULT_INTERVAL_SECS, DrainOutcome, LogDrainSetting, ResolvedLogDrain, SingleSourceError,
    SingleSourceSection, resolve_single_source, run_tick,
};
pub use uri::{DestinationScheme, DestinationUri};

/// Which project's logs a drain run is uploading.
///
/// Why: the two components are the whole key namespace below the destination
/// prefix, and getting either wrong files one project's logs under another's.
/// They are the caller's to supply because the core has no business shelling
/// out to git — the consumer resolves the identity and passes the answer in.
/// What: [`DrainTarget::validate`] refuses an empty component, so no code path
/// can produce a key containing `//` or an anonymous segment.
/// Test: `tests::run_once_refuses_an_empty_owner`,
/// `tests::run_once_refuses_an_empty_project`, `tests::key_layout_shape`.
///
/// #6657 replaced the previous `<github_id>/<session_id>` pair. A per-session
/// segment made every project on a host share one namespace keyed by whoever
/// ran the daemon; the owner and project of the repo the logs came from is what
/// an operator actually looks under.
#[derive(Debug, Clone, PartialEq, Eq)]
pub struct DrainTarget {
    /// Repository owner — the GitHub user or org, verbatim from the remote.
    pub owner: String,
    /// Repository name the logs belong to, verbatim from the remote.
    pub project: String,
}

impl DrainTarget {
    /// Refuse an empty identity component.
    ///
    /// Fail-closed by design: see [`DrainError::MissingIdentity`].
    ///
    /// # Errors
    /// [`DrainError::MissingIdentity`] naming the empty field.
    pub fn validate(&self) -> Result<(), DrainError> {
        if self.owner.trim().is_empty() {
            return Err(DrainError::MissingIdentity { field: "owner" });
        }
        if self.project.trim().is_empty() {
            return Err(DrainError::MissingIdentity { field: "project" });
        }
        Ok(())
    }

    /// The `<owner>/<project>` prefix every object for this target sits beneath.
    ///
    /// Destination-relative: the adapter joins the URI's own prefix on top.
    /// Test: `tests::key_layout_shape`.
    pub fn key_prefix(&self) -> String {
        format!("{}/{}", self.owner, self.project)
    }

    /// Full key for one collected file's `relative_key`.
    pub fn object_key(&self, relative_key: &str) -> String {
        format!("{}/{}", self.key_prefix(), relative_key)
    }

    /// Key of this target's manifest object.
    pub fn manifest_key(&self) -> String {
        format!("{}/{}", self.key_prefix(), MANIFEST_FILENAME)
    }

    /// Path segment identifying this target inside the local state directory.
    ///
    /// The TARGET half only. [`DrainManifest::cache_path`] puts the
    /// destination's namespace above it, so a record made for one destination
    /// is never read back for another (#6548).
    fn cache_key(&self) -> String {
        self.key_prefix()
    }
}

/// Everything a run needs that is not the destination, target, or sources.
///
/// `#[non_exhaustive]` so Phase 3 can add scheduling knobs without a breaking
/// change; construct it with [`DrainConfig::new`].
#[derive(Debug, Clone)]
#[non_exhaustive]
pub struct DrainConfig {
    /// Where the local manifest cache lives.
    pub state_dir: PathBuf,
    /// Values [`collect`] removes from every body before upload.
    pub secrets: Vec<String>,
    /// Plaintext source ceiling. Files over it are skipped, never truncated.
    pub max_file_bytes: u64,
    /// Compressed body ceiling (#6547). See [`DEFAULT_MAX_WIRE_BYTES`].
    pub max_wire_bytes: u64,
}

impl DrainConfig {
    /// A config with the default [`CollectLimits`] and no secrets.
    ///
    /// A caller that leaves `secrets` empty gets no scrubbing — see the module
    /// docs.
    pub fn new(state_dir: impl Into<PathBuf>) -> Self {
        let limits = CollectLimits::default();
        Self {
            state_dir: state_dir.into(),
            secrets: Vec::new(),
            max_file_bytes: limits.max_file_bytes,
            max_wire_bytes: limits.max_wire_bytes,
        }
    }

    /// Set the values to scrub from every uploaded body.
    #[must_use]
    pub fn with_secrets(mut self, secrets: Vec<String>) -> Self {
        self.secrets = secrets;
        self
    }

    /// Set the plaintext source ceiling.
    #[must_use]
    pub fn with_max_file_bytes(mut self, max: u64) -> Self {
        self.max_file_bytes = max;
        self
    }

    /// Set the compressed-body ceiling (#6547).
    #[must_use]
    pub fn with_max_wire_bytes(mut self, max: u64) -> Self {
        self.max_wire_bytes = max;
        self
    }

    /// The two bounds, as [`collect`] takes them.
    pub fn limits(&self) -> CollectLimits {
        CollectLimits::new(self.max_file_bytes, self.max_wire_bytes)
    }
}

/// What one [`run_once`] pass did.
///
/// Why: the scheduler Phase 3 adds decides whether to back off from these
/// counts, and an operator reading a doctor check needs to tell "nothing
/// changed" from "everything failed". Both are zero uploads.
/// What: four counters, two byte totals, and the per-file errors that did not
/// abort the batch.
/// Test: `tests::run_once_end_to_end`, `tests::run_once_collects_per_file_errors_without_aborting`.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
#[non_exhaustive]
pub struct DrainReport {
    /// Files uploaded this pass.
    pub uploaded: usize,
    /// Files skipped because the manifest already had them.
    pub skipped_unchanged: usize,
    /// Files skipped for exceeding one of the size bounds.
    pub skipped_too_large: usize,
    /// Of those, the ones whose decision was recorded for the FIRST time (#6547).
    ///
    /// A steady state where every oversize file has already been decided reads
    /// `skipped_too_large: 29, skips_recorded: 0` — which is the signal that no
    /// warning was logged this pass, and the difference between a backlog that
    /// is settled and one that keeps churning.
    pub skips_recorded: usize,
    /// Total plaintext bytes of everything uploaded.
    pub bytes_plain: u64,
    /// Total gzipped bytes actually written to the destination.
    pub bytes_wire: u64,
    /// Per-file failures, as `(key or path, message)`. Never aborts the batch.
    pub errors: Vec<(String, String)>,
    /// Sampled remote-manifest entries whose object was missing (#6548).
    ///
    /// 0 or 1 per run — one entry is sampled, not all of them. Non-zero means
    /// the destination's own manifest lists something the destination does not
    /// have, so every file it lists is being skipped rather than uploaded.
    /// Repair is documented in `docs/reference/log-drain.md`.
    pub manifest_spot_check_missing: usize,
}

/// Run the drain once: collect, compare against the manifest, upload what changed.
///
/// Why: one entry point means the ordering — validate identity, load manifest,
/// collect, filter by manifest, upload, rewrite manifest — is fixed rather than
/// reassembled per caller.
///
/// What: validates `target` FIRST, so a bad identity costs no filesystem walk
/// and can never reach a `put`. Loads the manifest (remote authoritative, local
/// cache as fallback, and the cache is scoped to the DESTINATION as well as the
/// target — see [`DrainManifest::cache_path`]). Collects every matching file.
/// A manifest that came from the destination gets one sampled
/// [`DrainManifest::spot_check`]. For each file: the stat-only
/// fast path skips unchanged files without comparing digests; a file whose stat
/// moved but whose SHA-256 matches is still skipped, with its manifest entry
/// refreshed so the next run takes the fast path. Everything else is uploaded.
/// A per-file failure is recorded in [`DrainReport::errors`] and the batch
/// CONTINUES — one unreadable file must not strand every other log on the
/// machine. The manifest is rewritten once at the end, reflecting only what
/// actually landed.
///
/// A file over one of the size bounds is decided ONCE and the decision is
/// written to the manifest as a [`SkipRecord`] (#6547); a later pass that sees
/// the same `(file, size, mtime)` counts it and says nothing. Any file the pass
/// CAN read has its skip record dropped, so raising a bound takes effect on the
/// next pass rather than needing the manifest cleared by hand.
///
/// Single-flight is the CALLER's responsibility; see the module docs.
///
/// Test: `tests::run_once_end_to_end`, `tests::run_once_is_idempotent`,
/// `tests::run_once_reuploads_a_mutated_file`, `tests::run_once_refuses_an_empty_owner`,
/// `tests::run_once_records_an_oversize_skip_once`,
/// `tests::run_once_re_evaluates_a_skip_when_the_file_changes`.
///
/// # Errors
/// - [`DrainError::MissingIdentity`] when `target` has an empty component.
/// - [`DrainError::Uri`] for a malformed include glob.
/// - [`DrainError::Transport`] when the manifest itself could not be read or
///   written. A per-FILE transport failure is collected, not raised.
pub async fn run_once(
    cfg: &DrainConfig,
    dest: &dyn LogDestination,
    target: &DrainTarget,
    sources: &[LogSource],
) -> Result<DrainReport, DrainError> {
    // #6533: fail closed before anything touches the filesystem or the network.
    target.validate()?;

    let manifest_key = target.manifest_key();
    let cache_key = target.cache_key();
    let (mut manifest, origin) =
        DrainManifest::load_with_origin(dest, &cfg.state_dir, &manifest_key, &cache_key).await?;

    let collected = collect(sources, &cfg.secrets, cfg.limits())?;

    let mut report = DrainReport {
        skipped_too_large: collected.oversize.len(),
        ..DrainReport::default()
    };
    for (path, message) in collected.errors {
        report.errors.push((path.display().to_string(), message));
    }

    // #6548: a manifest written before the cache-keying fix can list objects
    // this destination never received, and every one of them then skips
    // forever. One sampled `head` turns that into a warning an operator sees.
    if origin == ManifestOrigin::Remote
        && let Some(missing) = manifest.spot_check(dest, &target.key_prefix()).await
    {
        report.manifest_spot_check_missing += 1;
        tracing::warn!(
            key = %missing,
            manifest = %manifest_key,
            "log-drain manifest lists an object this destination does not have; \
             delete the manifest object to force a full re-upload (see #6548)"
        );
    }

    let now = chrono::Utc::now().to_rfc3339();
    let mut manifest_dirty = false;

    // #6547: decide each oversize file ONCE. A file that has already been
    // recorded at this exact size and mtime cannot have changed its answer, so
    // it is counted and passed over in silence rather than warned about again
    // every cycle.
    for over in &collected.oversize {
        if manifest.skip_recorded(&over.relative_key, over.size, over.mtime_unix) {
            tracing::debug!(
                path = %over.path.display(),
                size = over.size,
                "log-drain skip already recorded for this size and mtime"
            );
            continue;
        }
        tracing::warn!(
            path = %over.path.display(),
            size = over.size,
            limit = over.reason.limit_name(),
            "log-drain is not uploading this file; the decision is recorded in the \
             manifest and is not logged again until the file's size or mtime changes"
        );
        manifest.record_skip(SkipRecord {
            relative_file: over.relative_key.clone(),
            size: over.size,
            mtime_unix: over.mtime_unix,
            reason: over.reason,
            decided_at: now.clone(),
        });
        report.skips_recorded += 1;
        manifest_dirty = true;
    }

    for file in collected.files {
        // A file the pass CAN read has no business carrying a skip record —
        // raising a bound, or a rotation that shrank it, makes it drainable.
        if manifest.forget_skip(&file.relative_key) {
            manifest_dirty = true;
        }

        // Fast path: stat alone says nothing changed, so never compare digests.
        if manifest.decide(&file.relative_key, file.plaintext_len, file.mtime_unix)
            == StatDecision::SkipUnchanged
        {
            report.skipped_unchanged += 1;
            continue;
        }

        // Slow path: the stat moved. SHA-256 decides, and it wins over mtime —
        // identical bytes are never re-uploaded just because a timestamp moved.
        if manifest.digest_matches(&file.relative_key, &file.sha256_plaintext) {
            report.skipped_unchanged += 1;
            manifest.record(entry_for(&file, &now));
            manifest_dirty = true;
            continue;
        }

        let key = target.object_key(&file.relative_key);
        let wire_len = file.body.len() as u64;
        match dest
            .put(&key, file.body.clone(), PutMeta::gzipped_text())
            .await
        {
            Ok(()) => {
                report.uploaded += 1;
                report.bytes_plain += file.plaintext_len;
                report.bytes_wire += wire_len;
                manifest.record(entry_for(&file, &now));
                manifest_dirty = true;
            }
            Err(e) => report.errors.push((key, e.to_string())),
        }
    }

    if manifest_dirty {
        manifest
            .save(dest, &cfg.state_dir, &manifest_key, &cache_key)
            .await?;
    }

    Ok(report)
}

/// What one [`prune_confirmed`] pass did (#6536, `prune_after_upload`).
///
/// Why: a destructive pass and an upload pass need the same "nothing changed
/// vs. everything failed" distinction [`DrainReport`] gives uploads.
/// What: a count of objects actually removed, plus per-key failures that never
/// abort the batch — the same shape [`DrainReport::errors`] uses. `refused`
/// carries the [`PruneRefusal`] when the whole batch was declined by the
/// sanity cap (#7154); `pruned` is always `0` and `errors` always empty in
/// that case, because a refused batch deletes nothing.
/// Test: `tests::prune_confirmed_deletes_object_and_manifest_entry`.
#[derive(Debug, Clone, Default, PartialEq, Eq)]
#[non_exhaustive]
pub struct PruneOutcome {
    /// Objects deleted from the destination this pass, with their manifest
    /// entry dropped in the same save.
    pub pruned: usize,
    /// Per-key failures, as `(key, message)`. Never aborts the batch.
    pub errors: Vec<(String, String)>,
    /// Set instead of acting when the batch tripped the sanity cap.
    pub refused: Option<PruneRefusal>,
}

/// Ceiling on how many objects one [`prune_confirmed`] call may delete in a
/// single tick.
///
/// Why (#7154): the two-tick `RetentionDebounce` only proves a candidate was
/// STILL prunable on two consecutive ticks — it does not bound how MANY can
/// qualify at once. A forward clock jump (a bad NTP sync, a suspended host
/// waking up days later) ages every manifest entry past `prune_retention` in
/// one stroke, and both ticks agree, so the debounce confirms the whole
/// history in one pass. A destination's entire backlog vanishing in a single
/// tick is exactly the failure mode retention pruning must never cause.
/// What: an absolute count, independent of the fraction ceiling below —
/// either one refuses the batch. `500` is generous for the steady-state
/// workload (one object per source file per upload pass) while still well
/// short of "this destination's whole history".
/// Test: `tests::prune_batch_refusal_over_the_count_ceiling`.
pub const PRUNE_BATCH_COUNT_CEILING: usize = 500;

/// Ceiling on what FRACTION of the manifest's own entries one tick may
/// delete, once the manifest is large enough for a fraction to mean anything.
///
/// What: `0.5` — a batch may claim at most half of what the manifest
/// currently lists. Paired with [`PRUNE_FRACTION_FLOOR`] so a manifest with a
/// handful of entries (where "half" is one or two files) does not trip on
/// perfectly ordinary retention behaviour.
/// Test: `tests::prune_batch_refusal_over_the_fraction_ceiling`,
/// `tests::prune_batch_refusal_allows_a_small_manifest_at_any_fraction`.
pub const PRUNE_BATCH_FRACTION_CEILING: f64 = 0.5;

/// Manifests at or below this many entries are exempt from
/// [`PRUNE_BATCH_FRACTION_CEILING`] — the count ceiling alone still applies.
pub const PRUNE_FRACTION_FLOOR: usize = 20;

/// Why a confirmed prune batch was refused outright.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[non_exhaustive]
pub enum PruneRefusalReason {
    /// `batch_len` exceeded [`PRUNE_BATCH_COUNT_CEILING`].
    CountCeiling,
    /// `batch_len` exceeded [`PRUNE_BATCH_FRACTION_CEILING`] of `manifest_len`,
    /// with `manifest_len` over [`PRUNE_FRACTION_FLOOR`].
    FractionCeiling,
}

impl std::fmt::Display for PruneRefusalReason {
    fn fmt(&self, f: &mut std::fmt::Formatter<'_>) -> std::fmt::Result {
        match self {
            Self::CountCeiling => {
                write!(f, "batch exceeds the {PRUNE_BATCH_COUNT_CEILING}-entry cap")
            }
            Self::FractionCeiling => write!(
                f,
                "batch exceeds {:.0}% of the manifest",
                PRUNE_BATCH_FRACTION_CEILING * 100.0
            ),
        }
    }
}

/// A confirmed prune batch the sanity cap declined to act on (#7154).
///
/// Why: the destructive decision already survived the two-tick debounce by
/// the time [`prune_confirmed`] sees it — this is the LAST guard, and it has
/// to say what it saw, not just that it said no.
/// What: the batch size and the manifest size the refusal was judged against,
/// plus which ceiling tripped. Neither count implies the other: a huge batch
/// against a huge manifest can still trip [`PruneRefusalReason::CountCeiling`]
/// first.
#[derive(Debug, Clone, Copy, PartialEq, Eq)]
#[non_exhaustive]
pub struct PruneRefusal {
    /// How many keys the debounce confirmed this tick.
    pub batch_len: usize,
    /// How many entries the manifest held when the batch was judged.
    pub manifest_len: usize,
    /// Which ceiling refused the batch.
    pub reason: PruneRefusalReason,
}

/// Pure decision: would this batch trip the sanity cap?
///
/// Why: kept free of I/O so the two ceilings are testable directly against
/// plain integers rather than a real manifest and destination.
/// What: [`PRUNE_BATCH_COUNT_CEILING`] is checked first, then
/// [`PRUNE_BATCH_FRACTION_CEILING`] once `manifest_len` is over
/// [`PRUNE_FRACTION_FLOOR`]. `None` means the batch may proceed.
/// Test: `tests::prune_batch_refusal_over_the_count_ceiling`,
/// `tests::prune_batch_refusal_over_the_fraction_ceiling`,
/// `tests::prune_batch_refusal_allows_a_small_manifest_at_any_fraction`,
/// `tests::prune_batch_refusal_allows_a_batch_under_both_ceilings`.
fn prune_batch_refusal(batch_len: usize, manifest_len: usize) -> Option<PruneRefusal> {
    if batch_len > PRUNE_BATCH_COUNT_CEILING {
        return Some(PruneRefusal {
            batch_len,
            manifest_len,
            reason: PruneRefusalReason::CountCeiling,
        });
    }
    if manifest_len > PRUNE_FRACTION_FLOOR
        && (batch_len as f64) > (manifest_len as f64) * PRUNE_BATCH_FRACTION_CEILING
    {
        return Some(PruneRefusal {
            batch_len,
            manifest_len,
            reason: PruneRefusalReason::FractionCeiling,
        });
    }
    None
}

/// Manifest-confirmed object keys at least `retention` old, as of now.
///
/// Why: `prune_after_upload` (#6536) must never touch an object the manifest
/// does not itself vouch for — a failed or partial upload never gets an
/// entry, so it can never appear here. Read-only: nothing is deleted or
/// re-saved by this call. The caller (the trusty-mpm log-drain scheduler)
/// puts the result through its own two-tick `RetentionDebounce` before ever
/// calling [`prune_confirmed`] with it — the destructive decision is
/// deliberately made one layer up, in the crate that owns that debounce.
/// What: loads the manifest exactly as [`run_once`] does (remote
/// authoritative, local cache as fallback) and delegates the age decision to
/// [`DrainManifest::prunable`].
/// Test: `tests::prunable_candidates_never_names_a_file_that_failed_to_upload`.
///
/// # Errors
/// [`DrainError::Transport`] when the manifest itself could not be read.
pub async fn prunable_candidates(
    dest: &dyn LogDestination,
    target: &DrainTarget,
    state_dir: &std::path::Path,
    retention: std::time::Duration,
) -> Result<Vec<String>, DrainError> {
    let manifest =
        DrainManifest::load(dest, state_dir, &target.manifest_key(), &target.cache_key()).await?;
    Ok(manifest.prunable(chrono::Utc::now(), retention))
}

/// Delete the destination object and drop the manifest entry for every key in
/// `confirm`, then persist the manifest once (#6536, `prune_after_upload`).
///
/// Why: `confirm` is exactly what survived the caller's `RetentionDebounce` —
/// this function trusts it completely and re-derives nothing about the
/// retention window itself.
/// What: loads a fresh copy of the manifest, so a key another process already
/// pruned (or a fresh upload just re-confirmed) is re-checked immediately
/// before acting rather than trusted from whenever `confirm` was computed. A
/// key no longer listed is silently skipped — never an error, because "already
/// gone" is exactly the state pruning wants. A delete failure is recorded in
/// [`PruneOutcome::errors`] and its entry is KEPT, so the manifest never claims
/// an object is gone when the destination still has it; [`LogDestination::delete`]
/// itself treats "already gone" as success, so that case still drops the
/// entry. The manifest is saved once at the end, only if something changed.
///
/// Test: `tests::prune_confirmed_deletes_object_and_manifest_entry`,
/// `tests::prune_confirmed_ignores_a_key_the_manifest_no_longer_lists`,
/// `tests::prune_confirmed_refuses_a_batch_over_the_count_ceiling`,
/// `tests::prune_confirmed_refuses_a_batch_over_the_fraction_ceiling`,
/// `tests::prune_confirmed_prunes_a_batch_under_both_ceilings`.
///
/// # Errors
/// [`DrainError::Transport`] only when the manifest itself could not be
/// loaded or saved — a per-key delete failure is collected, not raised. A
/// batch that trips the sanity cap ([`PRUNE_BATCH_COUNT_CEILING`],
/// [`PRUNE_BATCH_FRACTION_CEILING`]) is reported in [`PruneOutcome::refused`]
/// rather than as an error — it is a deliberate refusal, not a failure.
pub async fn prune_confirmed(
    dest: &dyn LogDestination,
    target: &DrainTarget,
    state_dir: &std::path::Path,
    confirm: &[String],
) -> Result<PruneOutcome, DrainError> {
    let manifest_key = target.manifest_key();
    let cache_key = target.cache_key();
    let mut manifest = DrainManifest::load(dest, state_dir, &manifest_key, &cache_key).await?;

    // #7154: judged against the manifest as loaded, before anything this
    // batch would remove — the sanity cap asks "is it safe to trust this
    // batch", not "what would be left afterward". Delete nothing this tick
    // when it is not.
    if let Some(refusal) = prune_batch_refusal(confirm.len(), manifest.entries.len()) {
        return Ok(PruneOutcome {
            refused: Some(refusal),
            ..PruneOutcome::default()
        });
    }

    let mut outcome = PruneOutcome::default();
    let mut dirty = false;
    for relative_file in confirm {
        if !manifest
            .entries
            .iter()
            .any(|e| &e.relative_file == relative_file)
        {
            continue;
        }
        let key = target.object_key(relative_file);
        match dest.delete(&key).await {
            Ok(()) => {
                manifest.remove_entry(relative_file);
                outcome.pruned += 1;
                dirty = true;
            }
            Err(e) => outcome.errors.push((key, e.to_string())),
        }
    }

    if dirty {
        manifest
            .save(dest, state_dir, &manifest_key, &cache_key)
            .await?;
    }
    Ok(outcome)
}

/// Build the manifest entry recording one collected file.
fn entry_for(file: &CollectedFile, uploaded_at: &str) -> ManifestEntry {
    ManifestEntry {
        relative_file: file.relative_key.clone(),
        size: file.plaintext_len,
        mtime_unix: file.mtime_unix,
        sha256: file.sha256_plaintext.clone(),
        uploaded_at: uploaded_at.to_string(),
    }
}