onlyne-client 2.0.0

Onlyne client: role runtime, dispatch, intent, adapter socket
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
use super::*;

use super::outbound::store_ack;
use super::projection::stored_task_state;
use super::state::{
    DispatchInner, DispatchState, forget_tools_binding, has_attached_transport, session_exited,
    slot_key_serving_task,
};
use super::transport::{held_read_only, names_session};

/// Reason a session that stopped answering is closed with.
///
/// The wire word the ledger and `onlyne sessions` read for a death this client
/// judged: the reconnect sweep stamps it on the refusal ack that buries the
/// session's delivery, and an operator's own
/// `repair fail --reason session_dead` writes the same word.
pub const SESSION_DEAD: &str = "session_dead";

/// Retire one session. The stored tuple decides whether a live resource
/// remains to close, and the caller's reason reaches the backend unchanged, so an
/// operator cancel stops reporting itself as a completion.
pub fn on_recycled(
    state: &DispatchState,
    task_id: &str,
    reason: crate::backend::CloseReason,
) -> Result<()> {
    let mut inner = state.inner.lock();
    release_locked(&mut inner, task_id, Some(reason))
}

/// Why the task of one session ended, as the retirement reason the task table
/// records.
///
/// The task's own row is the only place a verdict lives: the session tuple says
/// nothing about how its work ended, and an open task — `pending`, or no row at
/// all, which is the same reading — has no reason to retire anything.
pub(super) fn stored_close_reason(
    inner: &DispatchInner,
    task_id: &str,
) -> Option<crate::backend::CloseReason> {
    match stored_task_state(inner, task_id) {
        TaskState::Pending => None,
        TaskState::Done => Some(crate::backend::CloseReason::Completed),
        // A blocked delivery leaves the work owed, which is what a fault reason
        // names here, exactly as it does for a failed one.
        TaskState::Failed | TaskState::Blocked => Some(crate::backend::CloseReason::Fault),
        TaskState::Cancelled => Some(crate::backend::CloseReason::Cancelled),
    }
}

/// The reason one session the reconnect grace retires is closed with, read from
/// what this client holds rather than from a settle that never came.
///
/// The id is the session's own (`slot.session.task_id`), never the task binding
/// beside it: the sweep is what feeds that session's row and closes the resource
/// its agent was holding, and the two ids part company exactly where a slot's
/// binding is not the session it was born for. A session still owing work closes
/// as that task's own record reads: a `done` task is a `Completed`, a
/// `cancelled` one is a `Cancelled`, and a `failed` task — like one that never
/// settled at all — is a `Fault`, because the work was still owed when the agent
/// left.
fn grace_close_reason(inner: &DispatchInner, task_id: &str) -> crate::backend::CloseReason {
    match stored_task_state(inner, task_id) {
        TaskState::Done => crate::backend::CloseReason::Completed,
        TaskState::Pending | TaskState::Failed | TaskState::Blocked => {
            crate::backend::CloseReason::Fault
        }
        TaskState::Cancelled => crate::backend::CloseReason::Cancelled,
    }
}

/// Whether one slot is due for the reconnect grace to take it.
///
/// The window as it always was: the connection that would have sent this
/// session's next heartbeat has ended, and the window runs from the moment it
/// left. A slot a live connection serves is not this arm's to end, because an
/// attached transport is the one thing that says the agent is still reachable.
fn dropped_past_window(
    inner: &DispatchInner,
    key: &str,
    slot: &SessionSlot,
    now: Instant,
    window: Duration,
) -> bool {
    slot.dropped_at.is_some_and(|dropped| {
        !has_attached_transport(inner, key, slot)
            && now
                .checked_duration_since(dropped)
                .is_some_and(|away| away >= window)
    })
}

/// Whether one session's own agent has gone quiet on a socket that is still up.
///
/// The window's other door, and it exists precisely because an attached
/// transport — the one thing the arm above trusts — can lie. A socket that is
/// still up proves the connection survived; it says nothing about the agent
/// behind it, and a plugin whose event loop is blocked holds its socket and
/// stops beating. Nothing the client reads before this could see that: no socket
/// ends, so no drop clock ever starts, and the session keeps its slot, its
/// projected row and its host resource for as long as the client runs.
///
/// So the reading is taken off the frames themselves. A session whose task is
/// still bound and unsettled and whose last accepted frame is older than the
/// protocol's heartbeat interval by [`HEARTBEAT_SILENCE_MARGIN`] has no agent
/// behind its socket.
///
/// Why no task bound is excluded: the plugin stops its heartbeat loop with the
/// last task it was given, so a task-free session that has gone quiet is an
/// agent waiting for work by design — the ordinary shape between deliveries.
/// Sweeping it would retire the very connection the next payload is staged onto.
/// The unsettled check beside the binding says the same thing about work that
/// already landed: a settled task has nothing left for its session to answer,
/// and `release_locked` has already given the binding back.
///
/// Why the drop clock's arm never reads this stamp: the two are mutually
/// exclusive by construction. A re-mount clears `dropped_at`, so a session that
/// came back is judged by its beat alone, and a session whose connection went
/// away is judged by the clock alone and never by a stamp its agent can no
/// longer refresh.
fn silent_past_window(inner: &DispatchInner, key: &str, slot: &SessionSlot, now: Instant) -> bool {
    if slot.dropped_at.is_some() || !has_attached_transport(inner, key, slot) {
        return false;
    }
    let Some(task_id) = slot.task_id.as_deref() else {
        return false;
    };
    if stored_task_state(inner, task_id) != TaskState::Pending {
        return false;
    }
    let quiet = HEARTBEAT_INTERVAL * HEARTBEAT_SILENCE_MARGIN;
    slot.last_beat.is_some_and(|beat| {
        now.checked_duration_since(beat)
            .is_some_and(|away| away >= quiet)
    })
}

/// Stop holding a session's connection as its transport.
///
/// The act a socket ending performs, run here on the client's own verdict
/// instead: a session judged dead by its silence still holds its socket, and the
/// connection is not going to end on its own while the agent behind it is
/// blocked. The binding goes now, so the retirement below runs as the ordinary
/// one, and a frame from that connection afterwards is refused the way every
/// frame from a connection no session answers for is — it is served no state,
/// which is the same door a stale reporter already comes to.
fn unbind_transports(inner: &mut DispatchInner, key: &str, slot: &SessionSlot) {
    inner
        .transports
        .retain(|served, _| !names_session(key, slot, served));
}

/// One backend close a retirement owed, carried off the dispatch lock.
///
/// A close is a host round trip — a pane kill, a terminal close, an agent reap —
/// and it can take seconds. The dispatch lock is the one lock every adapter frame,
/// every report, and every slot of this role queues behind, so a sweep that
/// retires more than one session at a time must not spend that lock on the hosts
/// it is calling. A retirement therefore records what it took down here and runs
/// the closes once it is off the lock.
pub(super) struct PendingClose {
    pub(super) backend: Arc<dyn SessionBackend>,
    pub(super) session: SessionRef,
    pub(super) reason: crate::backend::CloseReason,
}

/// Run the closes a retirement collected.
///
/// Four of the five callers hold the dispatch lock and can put it down first — the
/// two sweeps, the goodbye path, and the merged-handoff retire — and each does,
/// because a close that blocks must not hold every adapter frame of this role
/// behind it. The fifth is `release_locked`, which runs inside a caller that
/// already holds the lock for the store work above it and cannot leave; it takes
/// the close where it stands, which is the behavior that path has always had.
///
/// A failed close is reported and not propagated. The slot is gone from this
/// client's books either way and its row already reads closed, so the caller has
/// nothing left to undo; an error that travelled back would replace the answer
/// the caller is waiting for — which sessions left, and therefore must be
/// published — with a failure to say so.
pub(super) fn close_retired(pending: Vec<PendingClose>) {
    for close in pending {
        if let Err(error) = close.backend.close(&close.session, close.reason, false) {
            tracing::warn!(
                task = %close.session.task_id,
                backend = %close.session.backend,
                resource = %close.session.backend_ref,
                error = %error,
                "session resource retirement failed"
            );
        }
    }
}

/// Whether this role's scope keeps one session alive after its delivery settles.
///
/// `oneshot` is the rule the client has always had: the session served its one
/// delivery and the slot is done with it. A `task` or `role` session outlives
/// the delivery that opened it, which is the whole point of the scope, so its
/// slot stays serving nothing until the scope sends it the next delivery or
/// `idle_close` releases its process.
///
/// A read-only slot is never kept: it is a connection that came back for a task
/// a newer session already took, and it owns no delivery of this role.
pub(super) fn keeps_idle(inner: &DispatchInner, key: &str) -> bool {
    inner
        .sessions
        .get(key)
        .is_some_and(|slot| slot.keeps_idle && !slot.read_only)
}

/// Retire one task-free session after its transport set becomes empty.
///
/// This is the `oneshot` ending, and the ending of every session whose scope
/// does not keep it: its resource closes because the agent able to run another
/// task in it has left. [`keeps_idle`] answers for the scopes that keep a
/// session between deliveries, and those never reach here with work behind them.
///
/// The idle slot releases its backend resource because the agent able to run
/// another task in it has left. An attached transport keeps the resource because
/// that agent remains reachable. The dispatch lock serializes the final transport
/// check, reference refresh, lifecycle projection, and slot removal with adapter
/// binding. The backend close is deliberately NOT one of them: it is recorded in
/// `pending` for the caller to run once it is off the lock, so a sweep of sessions
/// cannot hold every frame of this role behind a host round trip. See
/// [`close_retired`] for the one path that cannot put the lock down.
pub(super) fn retire_idle_locked(
    inner: &mut DispatchInner,
    key: &str,
    reason: crate::backend::CloseReason,
    pending: &mut Vec<PendingClose>,
) -> bool {
    let Some(slot) = inner.sessions.get(key) else {
        return false;
    };
    if slot.task_id.is_some() || has_attached_transport(inner, key, slot) {
        return false;
    }

    let original = slot.session.clone();
    let task_id = original.task_id.clone();
    let resource = inner
        .store
        .get_session(&task_id)
        .ok()
        .flatten()
        .map(|row| row.resource_state)
        .unwrap_or_else(|| "detached".to_string());
    if resource != "detached" && resource != "closed" {
        let session = match inner.backend.attach(&original) {
            Ok(refreshed) => {
                if refreshed != original {
                    inner.bridge.track_live(refreshed.clone());
                    if let Some(slot) = inner.sessions.get_mut(key) {
                        slot.session = refreshed.clone();
                    }
                }
                refreshed
            }
            Err(_) => original,
        };
        tracing::info!(
            task = %task_id,
            backend = %session.backend,
            resource = %session.backend_ref,
            ?reason,
            "retiring idle session resource"
        );
        if let Err(error) = feed_resource_closed(&inner.bridge, &inner.store, &task_id) {
            tracing::warn!(
                task = %task_id,
                backend = %session.backend,
                resource = %session.backend_ref,
                error = %error,
                "session resource close projection failed"
            );
        }
        pending.push(PendingClose {
            backend: Arc::clone(&inner.backend),
            session,
            reason,
        });
    }
    if reason == crate::backend::CloseReason::Completed {
        if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, &task_id) {
            tracing::warn!(
                task = %task_id,
                error = %error,
                "agent-gone projection failed for a completed session"
            );
        }
    }
    inner.bridge.untrack_live(&task_id);
    forget_tools_binding(inner, key);
    inner.sessions.remove(key);
    true
}

/// Give one session's task slot back. Settled tasks enter idle retirement, and
/// explicit reasons drive the control-close path.
pub(super) fn release_locked(
    inner: &mut DispatchInner,
    task_id: &str,
    reason: Option<crate::backend::CloseReason>,
) -> Result<()> {
    let resource = inner
        .store
        .get_session(task_id)?
        .map(|row| row.resource_state)
        .unwrap_or_else(|| "detached".to_string());
    if let Some((key, slot)) = slot_key_serving_task(inner, task_id)
        .and_then(|key| inner.sessions.get(&key).map(|slot| (key, slot.clone())))
    {
        if let Some(reason) = reason {
            if resource != "detached" && resource != "closed" {
                feed_resource_closed(&inner.bridge, &inner.store, task_id)?;
                inner.backend.close(&slot.session, reason, false)?;
            }
            // The agent goes with the resource: this path closes a session whose work an
            // operator ended or whose backend faulted, and the slot below leaves the map in
            // the same breath. Without this feed the tuple keeps the agent phase its last
            // beat reported, and `project` answers `working` for a `cancelled` or `failed`
            // task whenever the agent is not `Gone` — so the mirrored row read `working` for
            // a session the client had already closed, which is what `onlyne sessions` and
            // the board showed beside a ledger row that had settled. The reconnect sweep
            // feeds the same event for the same ending.
            if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, task_id) {
                tracing::warn!(
                    task = %task_id,
                    error = %error,
                    "agent-gone projection failed for a closed session"
                );
            }
            inner.bridge.untrack_live(task_id);
            // The delivery row this session is still holding is answered here,
            // while the slot that holds its handle is still in the map. The close
            // takes the slot, and the handle goes with it: a row this client
            // never answered is a row the server still reads as owed, so the
            // pull that would have taken it passes and the release of a session
            // the server judges gone hands it back to the queue — which
            // dispatches the task a second time and runs it again, as the live
            // run showed for a task whose work the close had already ended.
            //
            // The handle is taken, so the row is answered once: the control
            // settle watchdog and the reconnect sweep refuse the same handle
            // the same way, and whichever of the three runs first spends it and
            // the others find nothing left to answer.
            let held = inner
                .sessions
                .get_mut(&key)
                .and_then(|slot| slot.msg_id.take());
            if let Some(msg_id) = held {
                store_ack(
                    inner,
                    AckArgs {
                        msg_id,
                        op_id: None,
                        accepted: false,
                        reason: Some(close_refusal(reason).to_string()),
                    },
                );
            }
            forget_tools_binding(inner, &key);
            inner.sessions.remove(&key);
        } else {
            let session_id = inner
                .sessions
                .get(&key)
                .map(|slot| slot.session.task_id.clone())
                .unwrap_or_else(|| task_id.to_string());
            let keeps = keeps_idle(inner, &key);
            let now = Instant::now();
            if let Some(session) = inner.sessions.get_mut(&key) {
                session.task_id = None;
                session.ready = false;
                session.idle_since = Some(now);
            }
            // The binding goes with the delivery. A session that serves several
            // deliveries in turn serves none between them, and both the hello
            // claim and the mirrored row read the open binding to say which
            // delivery a session is on.
            if let Err(error) = inner.store.release_binding(&session_id, task_id) {
                tracing::warn!(
                        session = %session_id,
                        task = %task_id,
                        error = %error,
                        "a settled session's binding was not handed back"
                );
            }
            if keeps {
                // A `task` or `role` session outlives the delivery that opened
                // it: that is what the scope is for. It stays idle, with its
                // process and its row, until the scope sends it the next
                // delivery or `idle_close` releases the process.
                tracing::info!(
                        session = %session_id,
                        task = %task_id,
                        "delivery settled; the session stays idle for its scope's next one"
                );
            } else {
                let mut pending = Vec::new();
                retire_idle_locked(
                    inner,
                    &key,
                    crate::backend::CloseReason::Completed,
                    &mut pending,
                );
                // This function answers to its caller's lock, which is already held
                // across the store work above, so the close it owes cannot wait for an
                // unlock this scope cannot perform. Running it here is the same
                // under-lock call the path has always made; the sweeps that CAN leave
                // the lock — `retire_dropped_ghosts`, `reclaim_exited_resources`,
                // `close_all` — are the ones that now defer it.
                close_retired(pending);
            }
        }
    }
    inner.stall.forget(task_id);
    if reason.is_some() && resource == "detached" {
        inner
            .store
            .note_alert(format!("session recycled {task_id}"));
    }
    Ok(())
}

/// The operator's word a control close stands for, as the refusal that answers
/// the row its session was still holding.
///
/// The words are the ones the settle fallback already writes for the same
/// commands ([`ControlWord::refusal`]), so a row reads the same thing whichever
/// door refused it, and each names the command that was given rather than the
/// verdict that command left behind.
fn close_refusal(reason: crate::backend::CloseReason) -> &'static str {
    match reason {
        // The close an operator's `cancel` runs: the `ControlOp::Cancel` arm of
        // `on_control`, which is also how the server asks for `repair close` and
        // `repair fail` to reach this client.
        crate::backend::CloseReason::Cancelled => ControlWord::Cancel.refusal(),
        // The close an operator's `recycle` runs: the `ControlOp::Recycle` arm
        // of `on_control`.
        crate::backend::CloseReason::Operator => ControlWord::Recycle.refusal(),
        // No control command reaches this branch with another reason: a
        // `completed`, `fault` or `replaced` close retires an idle slot, and a
        // `shutdown` close runs `close_all`, neither of which comes through
        // here. The word is the server's own for a close that named no command
        // — `repair close` without a `--reason` writes it on the task's row.
        _ => "operator close",
    }
}

/// Close every live session's resource with `reason` and forget the slots.
///
/// This is the shutdown path: a stopped client must not leave resources behind
/// that only it can address, and each backend's own record of the resource —
/// the Orca tab map included — ends with the session. `budget` bounds the whole
/// sweep, because an operator's SIGTERM must not turn into a hang while a slow
/// backend CLI exits; whatever the budget cuts off is reported and dropped
/// anyway.
///
/// The order is deliberate: every slot is taken off this client's books under the
/// dispatch lock — untracked and removed, so nothing still reads a live session —
/// and the closes it collected run once the lock is let go. A shutdown that
/// closed each pane while holding that lock spent the whole budget on the host,
/// with the lock no adapter frame could reach.
pub fn close_all(state: &DispatchState, reason: crate::backend::CloseReason, budget: Duration) {
    let started = Instant::now();
    let pending: Vec<PendingClose> = {
        let mut inner = state.inner.lock();
        let sessions: Vec<(String, SessionRef)> = inner
            .sessions
            .iter()
            .map(|(key, slot)| (key.clone(), slot.session.clone()))
            .collect();
        let mut pending = Vec::with_capacity(sessions.len());
        for (key, session) in sessions {
            inner.bridge.untrack_live(&session.task_id);
            forget_tools_binding(&mut inner, &key);
            inner.sessions.remove(&key);
            pending.push(PendingClose {
                backend: Arc::clone(&inner.backend),
                session,
                reason,
            });
        }
        pending
    };
    for close in pending {
        if started.elapsed() > budget {
            tracing::warn!(
                task = %close.session.task_id,
                "shutdown close budget reached; the resource is left behind"
            );
            continue;
        }
        if let Err(error) = close.backend.close(&close.session, close.reason, false) {
            tracing::warn!(
                task = %close.session.task_id,
                error = %error,
                "session close failed during shutdown"
            );
        }
    }
}

/// Which way a session's window closed.
///
/// Two arms reach the same verdict through different facts, and an operator reading one
/// aggregate log line cannot tell them apart: one means the connection ended and the agent
/// stayed away, the other means the connection is still up while nothing the client accepts
/// arrives over it. The words exist so the log can name which reading retired a session.
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
pub enum RetirementArm {
    /// The plugin connection ended and `[client] reconnect_grace_secs` expired.
    Dropped,
    /// The connection stayed up while the session went quiet past the heartbeat window.
    Silent,
}

impl RetirementArm {
    pub fn word(self) -> &'static str {
        match self {
            Self::Dropped => "reconnect_grace",
            Self::Silent => "heartbeat_silence",
        }
    }
}

/// One session the sweep retired, with what it read on the way.
pub struct Retired {
    /// The retired session's own id, which a client-held session shares with its task.
    pub session_id: String,
    /// The arm that decided it.
    pub arm: RetirementArm,
    /// Seconds since this session's last accepted frame.
    pub quiet_secs: u64,
    /// Seconds since its connection ended, when one did.
    pub away_secs: Option<u64>,
}

/// How long a session has gone without a frame this client accepted.
fn quiet_secs(slot: &SessionSlot, now: Instant) -> u64 {
    slot.last_beat
        .map(|beat| now.saturating_duration_since(beat).as_secs())
        .unwrap_or(u64::MAX)
}

/// How long a session's connection has been gone, for a session still waiting on one.
fn away_secs(slot: &SessionSlot, now: Instant) -> Option<u64> {
    slot.dropped_at
        .map(|left| now.saturating_duration_since(left).as_secs())
}

impl DispatchState {
    /// Retire tracked resources whose stored lifecycle has reached `Exited`.
    ///
    /// The periodic readiness tick calls this after completed work becomes an
    /// idle slot. Task-free sessions with an attached transport stay bound to
    /// their host resource, and task-free sessions whose agent has left release it.
    ///
    /// The session ids come back because the retirement wrote each one of those
    /// rows and the server mirrors only what this client reports: the resource
    /// close and, for a completed session, the agent's exit both moved the row
    /// this tick found, and a publish cannot run under this lock. The caller is
    /// handed what to publish, the same answer [`DispatchState::retire_dropped_ghosts`]
    /// gives the sweep above.
    pub fn reclaim_exited_resources(&self) -> Vec<String> {
        let mut inner = self.inner.lock();
        let candidates: Vec<(String, crate::backend::CloseReason)> = inner
            .sessions
            .iter()
            .filter(|(key, slot)| {
                // A suspended session is not an exited one. Its process is gone
                // *because* it was released, its conversation is what the
                // family's next delivery is for, and the row it published says
                // `idle` for exactly that reason — `binding_task_state` answers
                // `Pending` for a slot that serves nothing and whose scope keeps
                // it, so the two derivations of "is this over" disagree here.
                //
                // This sweep asks the other one, and it asks about the *work*:
                // a suspended session's delivery is finished, so it reads
                // `Exited` and the slot is retired within a tick. The family's
                // next delivery then finds nothing to resume and opens a second
                // conversation for one chain — which is the failure this filter
                // existed to prevent, reached through the sweep that was meant
                // to clean up after a session that was already gone.
                !slot.suspended
                    && slot.task_id.is_none()
                    && session_exited(&inner, &slot.session.task_id)
                    && !has_attached_transport(&inner, key, slot)
            })
            .filter_map(|(key, slot)| {
                stored_close_reason(&inner, &slot.session.task_id)
                    .map(|reason| (key.clone(), reason))
            })
            .collect();
        let mut retired: Vec<String> = Vec::new();
        let mut pending: Vec<PendingClose> = Vec::new();
        for (key, reason) in candidates {
            // The id that travels is the session's own, the one whose row the
            // retirement below is about to write.
            let Some(task_id) = inner
                .sessions
                .get(&key)
                .map(|slot| slot.session.task_id.clone())
            else {
                continue;
            };
            if retire_idle_locked(&mut inner, &key, reason, &mut pending) {
                retired.push(task_id);
            }
        }
        // The tick that found a batch of finished sessions owes a host close for
        // each, and this sweep runs every 250 ms: on the lock, one slow backend
        // would spend the whole window and every adapter frame queued behind it.
        drop(inner);
        close_retired(pending);
        retired
    }

    /// Retire the sessions whose plugin connection dropped and never came back, or
    /// whose connection stayed up while they went quiet, and answer which ones left
    /// and why.
    ///
    /// A connection that ends without a `detach` frame leaves its session tracked
    /// so an agent that restarts inside `[client] reconnect_grace_secs` finds the
    /// resource it was using. That promise has to expire: a process that is
    /// simply gone would otherwise hold a slot, a projected `idle` row, and a live
    /// host resource forever, and on a role with `max_sessions = 1` it stops every
    /// later delivery. The window answers for the agent itself, so a session still
    /// bound to a task goes with it: the plugin connection that would have
    /// reported the ending is the one that dropped. The agent-gone feed is what
    /// says the process left — the session's own tuple reaches `Exited` through
    /// `AgentPhase::Gone` rather than through a task result — and the reason the
    /// backend is handed is the one `grace_close_reason` reads off what the slot
    /// still owes.
    ///
    /// What the slot owed is settled too: the task a bound session was serving
    /// ends `failed` here, because the agent that would have reported its ending
    /// is the one that left. A task with no verdict stays open for the server to
    /// re-offer and for `open_tasks` to keep reading, and no later caller exists
    /// to write one.
    ///
    /// A slot this client holds read-only is not this sweep's to end, agent gone
    /// or not: the session id it would feed is the task id, so the ghost's death
    /// would take the live session's mirror and its delivery row down with it.
    /// That retirement belongs to `retire_revived`, which runs when the
    /// completion that answers the held connection merges.
    ///
    /// The window has a second way to open, and it is the one a socket cannot
    /// report: a plugin whose event loop is blocked keeps its connection and
    /// stops beating, so no socket ends and no clock this sweep could read
    /// before moved. What such a session leaves behind is a stamp going stale
    /// while its task stays bound and unsettled, and that is the reading this
    /// sweep takes now. It is the same window and the same verdict — one clock,
    /// one retirement, no second threshold beside `[client]
    /// reconnect_grace_secs` and no fault row of the kind `stall_report_secs`
    /// records and leaves behind.
    ///
    /// The sessions' own ids come back rather than a count, because a retirement
    /// still owes the server the session's own ending: it is the only writer left
    /// for that task, and a mirror nobody tells keeps that session's last reading —
    /// `working`, for one that had beaten — until the server's own observer records
    /// a fault about it. The publish is
    /// [`sync_session`](crate::session::dispatch::sync_session)'s, which is the
    /// report an ordinary ending travels on, and it cannot run under this lock —
    /// so the caller is handed what to publish instead of a second writer being
    /// invented here.
    pub fn retire_dropped_ghosts(&self, now: Instant, grace_secs: u64) -> Vec<Retired> {
        if grace_secs == 0 {
            return Vec::new();
        }
        let window = Duration::from_secs(grace_secs);
        let mut inner = self.inner.lock();
        // The arm is decided from the pair of readings here, while both still
        // describe the slot: which fact closed the window is what the operator
        // has to be able to tell apart once the retirement itself is a line in
        // the log.
        let due: Vec<(String, RetirementArm)> = inner
            .sessions
            .iter()
            .filter_map(|(key, slot)| {
                let arm = if dropped_past_window(&inner, key, slot, now, window) {
                    RetirementArm::Dropped
                } else if silent_past_window(&inner, key, slot, now) {
                    RetirementArm::Silent
                } else {
                    return None;
                };
                Some((key.clone(), arm))
            })
            .collect();
        let mut retired: Vec<Retired> = Vec::new();
        let mut pending: Vec<PendingClose> = Vec::new();
        for (key, arm) in due {
            let Some(slot) = inner.sessions.get(&key).cloned() else {
                continue;
            };
            // A slot this client holds read-only is not this sweep's to end, for
            // the reason the doc above gives. The demotion alone does not decide
            // it: the held connection that owns the slot can go without the task
            // ever completing, and a slot nothing owns any more is what this
            // window is for.
            if held_read_only(&inner, &key, &slot) {
                continue;
            }
            // A session judged dead on its silence is the one case where the
            // connection is still there: the socket has not ended and will not
            // while the agent behind it is blocked, so the client's own verdict
            // has to take the binding the way the death of the socket would have.
            // The retirement below refuses a slot an attached transport still
            // serves, and that refusal is what this unbinding answers: the
            // verdict has already been reached here, so the binding goes rather
            // than the death of the socket that would normally take it.
            if slot.dropped_at.is_none() {
                unbind_transports(&mut inner, &key, &slot);
            }
            let task_id = slot.session.task_id.clone();
            // Both the feed and the reason name the session's own task, so the id
            // that travels is the one the row and the resource are keyed by.
            let reason = grace_close_reason(&inner, &task_id);
            // The work this session still owed ends here, and this sweep is the
            // only writer left to say so: the plugin connection that would have
            // reported the ending is the one that dropped. A task nobody answers
            // stays `settled_at IS NULL` forever, so `open_tasks` keeps reading
            // it and the server keeps re-offering a delivery no client can take.
            // The binding is what the slot owed — a slot past its window with no
            // task bound owes nothing — and `failed` is the verdict the close
            // reason above already carries for it. A row an earlier verdict
            // settled keeps that one: `settle_task` updates only where
            // `settled_at IS NULL` and answers `false`.
            //
            // The write runs before the agent-gone feed and before the binding
            // hand-back, both of which end this slot's turn through the sweep:
            // the id is captured here, and the verdict is on disk before the
            // only handle on it goes away.
            if let Some(owed) = slot.task_id.clone() {
                // A verdict written here answers the task for good, so a
                // `control` command that is still waiting for its plugin's report
                // has nothing left to authorise: the note goes with the verdict
                // that outranks it.
                inner.control_settles.retain(|noted| noted.task_id != owed);
                if let Err(error) = inner.store.settle_task(&owed, TaskState::Failed) {
                    tracing::warn!(
                        task = %owed,
                        error = %error,
                        "the task of a retired ghost was not settled"
                    );
                }
                // The delivery handle this session was holding is spent as a
                // refusal that names the death, and it is the ledger half of the
                // verdict above. A row left `in_flight` is handed to a pull no
                // longer — `pull` passes by a row whose ticket is armed, and a
                // role-level pull's ticket carries no session id for the release
                // path to match — so nothing would answer for this task until the
                // link dropped, and an operator reading `onlyne ledger` would see
                // a session that has been buried as one still holding its
                // delivery. The reason is this client's own word for a death
                // (`SESSION_DEAD`), and a refusal is terminal: the work comes back
                // through `repair retry`, not by itself.
                let handle = inner
                    .sessions
                    .get_mut(&key)
                    .and_then(|slot| slot.msg_id.take());
                if let Some(msg_id) = handle {
                    store_ack(
                        &inner,
                        AckArgs {
                            msg_id,
                            op_id: None,
                            accepted: false,
                            reason: Some(SESSION_DEAD.to_string()),
                        },
                    );
                }
            }
            if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, &task_id) {
                tracing::warn!(
                    task = %task_id,
                    error = %error,
                    "agent-gone projection failed for a retired ghost"
                );
            }
            // The agent left, so the session owes no task any more: the binding
            // goes back before the idle retirement takes the slot.
            if let Some(slot) = inner.sessions.get_mut(&key) {
                slot.task_id = None;
            }
            if retire_idle_locked(&mut inner, &key, reason, &mut pending) {
                // The id that travels is the session's own, the one whose row was
                // just fed agent-gone and resource-closed: that row is what the
                // server mirrors, and its ending is what the caller publishes. The
                // ages are read off the slot as it stood before the retirement.
                retired.push(Retired {
                    session_id: task_id,
                    arm,
                    quiet_secs: quiet_secs(&slot, now),
                    away_secs: away_secs(&slot, now),
                });
            }
        }
        // Every session this sweep took is off the slot map and fully written down
        // by now; what remains is the host's own work, and the lock that answers an
        // agent's next frame is not to be held through it.
        drop(inner);
        close_retired(pending);
        retired
    }
}