Skip to main content

onlyne_client/session/dispatch/
retire.rs

1use super::*;
2
3use super::outbound::store_ack;
4use super::projection::stored_task_state;
5use super::state::{
6    DispatchInner, DispatchState, forget_tools_binding, has_attached_transport, session_exited,
7    slot_key_serving_task,
8};
9use super::transport::{held_read_only, names_session};
10
11/// Reason a session that stopped answering is closed with.
12///
13/// The wire word the ledger and `onlyne sessions` read for a death this client
14/// judged: the reconnect sweep stamps it on the refusal ack that buries the
15/// session's delivery, and an operator's own
16/// `repair fail --reason session_dead` writes the same word.
17pub const SESSION_DEAD: &str = "session_dead";
18
19/// Retire one session. The stored tuple decides whether a live resource
20/// remains to close, and the caller's reason reaches the backend unchanged, so an
21/// operator cancel stops reporting itself as a completion.
22pub fn on_recycled(
23    state: &DispatchState,
24    task_id: &str,
25    reason: crate::backend::CloseReason,
26) -> Result<()> {
27    let mut inner = state.inner.lock();
28    release_locked(&mut inner, task_id, Some(reason))
29}
30
31/// Why the task of one session ended, as the retirement reason the task table
32/// records.
33///
34/// The task's own row is the only place a verdict lives: the session tuple says
35/// nothing about how its work ended, and an open task — `pending`, or no row at
36/// all, which is the same reading — has no reason to retire anything.
37pub(super) fn stored_close_reason(
38    inner: &DispatchInner,
39    task_id: &str,
40) -> Option<crate::backend::CloseReason> {
41    match stored_task_state(inner, task_id) {
42        TaskState::Pending => None,
43        TaskState::Done => Some(crate::backend::CloseReason::Completed),
44        // A blocked delivery leaves the work owed, which is what a fault reason
45        // names here, exactly as it does for a failed one.
46        TaskState::Failed | TaskState::Blocked => Some(crate::backend::CloseReason::Fault),
47        TaskState::Cancelled => Some(crate::backend::CloseReason::Cancelled),
48    }
49}
50
51/// The reason one session the reconnect grace retires is closed with, read from
52/// what this client holds rather than from a settle that never came.
53///
54/// The id is the session's own (`slot.session.task_id`), never the task binding
55/// beside it: the sweep is what feeds that session's row and closes the resource
56/// its agent was holding, and the two ids part company exactly where a slot's
57/// binding is not the session it was born for. A session still owing work closes
58/// as that task's own record reads: a `done` task is a `Completed`, a
59/// `cancelled` one is a `Cancelled`, and a `failed` task — like one that never
60/// settled at all — is a `Fault`, because the work was still owed when the agent
61/// left.
62fn grace_close_reason(inner: &DispatchInner, task_id: &str) -> crate::backend::CloseReason {
63    match stored_task_state(inner, task_id) {
64        TaskState::Done => crate::backend::CloseReason::Completed,
65        TaskState::Pending | TaskState::Failed | TaskState::Blocked => {
66            crate::backend::CloseReason::Fault
67        }
68        TaskState::Cancelled => crate::backend::CloseReason::Cancelled,
69    }
70}
71
72/// Whether one slot is due for the reconnect grace to take it.
73///
74/// The window as it always was: the connection that would have sent this
75/// session's next heartbeat has ended, and the window runs from the moment it
76/// left. A slot a live connection serves is not this arm's to end, because an
77/// attached transport is the one thing that says the agent is still reachable.
78fn dropped_past_window(
79    inner: &DispatchInner,
80    key: &str,
81    slot: &SessionSlot,
82    now: Instant,
83    window: Duration,
84) -> bool {
85    slot.dropped_at.is_some_and(|dropped| {
86        !has_attached_transport(inner, key, slot)
87            && now
88                .checked_duration_since(dropped)
89                .is_some_and(|away| away >= window)
90    })
91}
92
93/// Whether one session's own agent has gone quiet on a socket that is still up.
94///
95/// The window's other door, and it exists precisely because an attached
96/// transport — the one thing the arm above trusts — can lie. A socket that is
97/// still up proves the connection survived; it says nothing about the agent
98/// behind it, and a plugin whose event loop is blocked holds its socket and
99/// stops beating. Nothing the client reads before this could see that: no socket
100/// ends, so no drop clock ever starts, and the session keeps its slot, its
101/// projected row and its host resource for as long as the client runs.
102///
103/// So the reading is taken off the frames themselves. A session whose task is
104/// still bound and unsettled and whose last accepted frame is older than the
105/// protocol's heartbeat interval by [`HEARTBEAT_SILENCE_MARGIN`] has no agent
106/// behind its socket.
107///
108/// Why no task bound is excluded: the plugin stops its heartbeat loop with the
109/// last task it was given, so a task-free session that has gone quiet is an
110/// agent waiting for work by design — the ordinary shape between deliveries.
111/// Sweeping it would retire the very connection the next payload is staged onto.
112/// The unsettled check beside the binding says the same thing about work that
113/// already landed: a settled task has nothing left for its session to answer,
114/// and `release_locked` has already given the binding back.
115///
116/// Why the drop clock's arm never reads this stamp: the two are mutually
117/// exclusive by construction. A re-mount clears `dropped_at`, so a session that
118/// came back is judged by its beat alone, and a session whose connection went
119/// away is judged by the clock alone and never by a stamp its agent can no
120/// longer refresh.
121fn silent_past_window(inner: &DispatchInner, key: &str, slot: &SessionSlot, now: Instant) -> bool {
122    if slot.dropped_at.is_some() || !has_attached_transport(inner, key, slot) {
123        return false;
124    }
125    let Some(task_id) = slot.task_id.as_deref() else {
126        return false;
127    };
128    if stored_task_state(inner, task_id) != TaskState::Pending {
129        return false;
130    }
131    let quiet = HEARTBEAT_INTERVAL * HEARTBEAT_SILENCE_MARGIN;
132    slot.last_beat.is_some_and(|beat| {
133        now.checked_duration_since(beat)
134            .is_some_and(|away| away >= quiet)
135    })
136}
137
138/// Stop holding a session's connection as its transport.
139///
140/// The act a socket ending performs, run here on the client's own verdict
141/// instead: a session judged dead by its silence still holds its socket, and the
142/// connection is not going to end on its own while the agent behind it is
143/// blocked. The binding goes now, so the retirement below runs as the ordinary
144/// one, and a frame from that connection afterwards is refused the way every
145/// frame from a connection no session answers for is — it is served no state,
146/// which is the same door a stale reporter already comes to.
147fn unbind_transports(inner: &mut DispatchInner, key: &str, slot: &SessionSlot) {
148    inner
149        .transports
150        .retain(|served, _| !names_session(key, slot, served));
151}
152
153/// One backend close a retirement owed, carried off the dispatch lock.
154///
155/// A close is a host round trip — a pane kill, a terminal close, an agent reap —
156/// and it can take seconds. The dispatch lock is the one lock every adapter frame,
157/// every report, and every slot of this role queues behind, so a sweep that
158/// retires more than one session at a time must not spend that lock on the hosts
159/// it is calling. A retirement therefore records what it took down here and runs
160/// the closes once it is off the lock.
161pub(super) struct PendingClose {
162    pub(super) backend: Arc<dyn SessionBackend>,
163    pub(super) session: SessionRef,
164    pub(super) reason: crate::backend::CloseReason,
165}
166
167/// Run the closes a retirement collected.
168///
169/// Four of the five callers hold the dispatch lock and can put it down first — the
170/// two sweeps, the goodbye path, and the merged-handoff retire — and each does,
171/// because a close that blocks must not hold every adapter frame of this role
172/// behind it. The fifth is `release_locked`, which runs inside a caller that
173/// already holds the lock for the store work above it and cannot leave; it takes
174/// the close where it stands, which is the behavior that path has always had.
175///
176/// A failed close is reported and not propagated. The slot is gone from this
177/// client's books either way and its row already reads closed, so the caller has
178/// nothing left to undo; an error that travelled back would replace the answer
179/// the caller is waiting for — which sessions left, and therefore must be
180/// published — with a failure to say so.
181pub(super) fn close_retired(pending: Vec<PendingClose>) {
182    for close in pending {
183        if let Err(error) = close.backend.close(&close.session, close.reason, false) {
184            tracing::warn!(
185                task = %close.session.task_id,
186                backend = %close.session.backend,
187                resource = %close.session.backend_ref,
188                error = %error,
189                "session resource retirement failed"
190            );
191        }
192    }
193}
194
195/// Whether this role's scope keeps one session alive after its delivery settles.
196///
197/// `oneshot` is the rule the client has always had: the session served its one
198/// delivery and the slot is done with it. A `task` or `role` session outlives
199/// the delivery that opened it, which is the whole point of the scope, so its
200/// slot stays serving nothing until the scope sends it the next delivery or
201/// `idle_close` releases its process.
202///
203/// A read-only slot is never kept: it is a connection that came back for a task
204/// a newer session already took, and it owns no delivery of this role.
205pub(super) fn keeps_idle(inner: &DispatchInner, key: &str) -> bool {
206    inner
207        .sessions
208        .get(key)
209        .is_some_and(|slot| slot.keeps_idle && !slot.read_only)
210}
211
212/// Retire one task-free session after its transport set becomes empty.
213///
214/// This is the `oneshot` ending, and the ending of every session whose scope
215/// does not keep it: its resource closes because the agent able to run another
216/// task in it has left. [`keeps_idle`] answers for the scopes that keep a
217/// session between deliveries, and those never reach here with work behind them.
218///
219/// The idle slot releases its backend resource because the agent able to run
220/// another task in it has left. An attached transport keeps the resource because
221/// that agent remains reachable. The dispatch lock serializes the final transport
222/// check, reference refresh, lifecycle projection, and slot removal with adapter
223/// binding. The backend close is deliberately NOT one of them: it is recorded in
224/// `pending` for the caller to run once it is off the lock, so a sweep of sessions
225/// cannot hold every frame of this role behind a host round trip. See
226/// [`close_retired`] for the one path that cannot put the lock down.
227pub(super) fn retire_idle_locked(
228    inner: &mut DispatchInner,
229    key: &str,
230    reason: crate::backend::CloseReason,
231    pending: &mut Vec<PendingClose>,
232) -> bool {
233    let Some(slot) = inner.sessions.get(key) else {
234        return false;
235    };
236    if slot.task_id.is_some() || has_attached_transport(inner, key, slot) {
237        return false;
238    }
239
240    let original = slot.session.clone();
241    let task_id = original.task_id.clone();
242    let resource = inner
243        .store
244        .get_session(&task_id)
245        .ok()
246        .flatten()
247        .map(|row| row.resource_state)
248        .unwrap_or_else(|| "detached".to_string());
249    if resource != "detached" && resource != "closed" {
250        let session = match inner.backend.attach(&original) {
251            Ok(refreshed) => {
252                if refreshed != original {
253                    inner.bridge.track_live(refreshed.clone());
254                    if let Some(slot) = inner.sessions.get_mut(key) {
255                        slot.session = refreshed.clone();
256                    }
257                }
258                refreshed
259            }
260            Err(_) => original,
261        };
262        tracing::info!(
263            task = %task_id,
264            backend = %session.backend,
265            resource = %session.backend_ref,
266            ?reason,
267            "retiring idle session resource"
268        );
269        if let Err(error) = feed_resource_closed(&inner.bridge, &inner.store, &task_id) {
270            tracing::warn!(
271                task = %task_id,
272                backend = %session.backend,
273                resource = %session.backend_ref,
274                error = %error,
275                "session resource close projection failed"
276            );
277        }
278        pending.push(PendingClose {
279            backend: Arc::clone(&inner.backend),
280            session,
281            reason,
282        });
283    }
284    if reason == crate::backend::CloseReason::Completed {
285        if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, &task_id) {
286            tracing::warn!(
287                task = %task_id,
288                error = %error,
289                "agent-gone projection failed for a completed session"
290            );
291        }
292    }
293    inner.bridge.untrack_live(&task_id);
294    forget_tools_binding(inner, key);
295    inner.sessions.remove(key);
296    true
297}
298
299/// Give one session's task slot back. Settled tasks enter idle retirement, and
300/// explicit reasons drive the control-close path.
301pub(super) fn release_locked(
302    inner: &mut DispatchInner,
303    task_id: &str,
304    reason: Option<crate::backend::CloseReason>,
305) -> Result<()> {
306    let resource = inner
307        .store
308        .get_session(task_id)?
309        .map(|row| row.resource_state)
310        .unwrap_or_else(|| "detached".to_string());
311    if let Some((key, slot)) = slot_key_serving_task(inner, task_id)
312        .and_then(|key| inner.sessions.get(&key).map(|slot| (key, slot.clone())))
313    {
314        if let Some(reason) = reason {
315            if resource != "detached" && resource != "closed" {
316                feed_resource_closed(&inner.bridge, &inner.store, task_id)?;
317                inner.backend.close(&slot.session, reason, false)?;
318            }
319            // The agent goes with the resource: this path closes a session whose work an
320            // operator ended or whose backend faulted, and the slot below leaves the map in
321            // the same breath. Without this feed the tuple keeps the agent phase its last
322            // beat reported, and `project` answers `working` for a `cancelled` or `failed`
323            // task whenever the agent is not `Gone` — so the mirrored row read `working` for
324            // a session the client had already closed, which is what `onlyne sessions` and
325            // the board showed beside a ledger row that had settled. The reconnect sweep
326            // feeds the same event for the same ending.
327            if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, task_id) {
328                tracing::warn!(
329                    task = %task_id,
330                    error = %error,
331                    "agent-gone projection failed for a closed session"
332                );
333            }
334            inner.bridge.untrack_live(task_id);
335            // The delivery row this session is still holding is answered here,
336            // while the slot that holds its handle is still in the map. The close
337            // takes the slot, and the handle goes with it: a row this client
338            // never answered is a row the server still reads as owed, so the
339            // pull that would have taken it passes and the release of a session
340            // the server judges gone hands it back to the queue — which
341            // dispatches the task a second time and runs it again, as the live
342            // run showed for a task whose work the close had already ended.
343            //
344            // The handle is taken, so the row is answered once: the control
345            // settle watchdog and the reconnect sweep refuse the same handle
346            // the same way, and whichever of the three runs first spends it and
347            // the others find nothing left to answer.
348            let held = inner
349                .sessions
350                .get_mut(&key)
351                .and_then(|slot| slot.msg_id.take());
352            if let Some(msg_id) = held {
353                store_ack(
354                    inner,
355                    AckArgs {
356                        msg_id,
357                        op_id: None,
358                        accepted: false,
359                        reason: Some(close_refusal(reason).to_string()),
360                    },
361                );
362            }
363            forget_tools_binding(inner, &key);
364            inner.sessions.remove(&key);
365        } else {
366            let session_id = inner
367                .sessions
368                .get(&key)
369                .map(|slot| slot.session.task_id.clone())
370                .unwrap_or_else(|| task_id.to_string());
371            let keeps = keeps_idle(inner, &key);
372            let now = Instant::now();
373            if let Some(session) = inner.sessions.get_mut(&key) {
374                session.task_id = None;
375                session.ready = false;
376                session.idle_since = Some(now);
377            }
378            // The binding goes with the delivery. A session that serves several
379            // deliveries in turn serves none between them, and both the hello
380            // claim and the mirrored row read the open binding to say which
381            // delivery a session is on.
382            if let Err(error) = inner.store.release_binding(&session_id, task_id) {
383                tracing::warn!(
384                        session = %session_id,
385                        task = %task_id,
386                        error = %error,
387                        "a settled session's binding was not handed back"
388                );
389            }
390            if keeps {
391                // A `task` or `role` session outlives the delivery that opened
392                // it: that is what the scope is for. It stays idle, with its
393                // process and its row, until the scope sends it the next
394                // delivery or `idle_close` releases the process.
395                tracing::info!(
396                        session = %session_id,
397                        task = %task_id,
398                        "delivery settled; the session stays idle for its scope's next one"
399                );
400            } else {
401                let mut pending = Vec::new();
402                retire_idle_locked(
403                    inner,
404                    &key,
405                    crate::backend::CloseReason::Completed,
406                    &mut pending,
407                );
408                // This function answers to its caller's lock, which is already held
409                // across the store work above, so the close it owes cannot wait for an
410                // unlock this scope cannot perform. Running it here is the same
411                // under-lock call the path has always made; the sweeps that CAN leave
412                // the lock — `retire_dropped_ghosts`, `reclaim_exited_resources`,
413                // `close_all` — are the ones that now defer it.
414                close_retired(pending);
415            }
416        }
417    }
418    inner.stall.forget(task_id);
419    if reason.is_some() && resource == "detached" {
420        inner
421            .store
422            .note_alert(format!("session recycled {task_id}"));
423    }
424    Ok(())
425}
426
427/// The operator's word a control close stands for, as the refusal that answers
428/// the row its session was still holding.
429///
430/// The words are the ones the settle fallback already writes for the same
431/// commands ([`ControlWord::refusal`]), so a row reads the same thing whichever
432/// door refused it, and each names the command that was given rather than the
433/// verdict that command left behind.
434fn close_refusal(reason: crate::backend::CloseReason) -> &'static str {
435    match reason {
436        // The close an operator's `cancel` runs: the `ControlOp::Cancel` arm of
437        // `on_control`, which is also how the server asks for `repair close` and
438        // `repair fail` to reach this client.
439        crate::backend::CloseReason::Cancelled => ControlWord::Cancel.refusal(),
440        // The close an operator's `recycle` runs: the `ControlOp::Recycle` arm
441        // of `on_control`.
442        crate::backend::CloseReason::Operator => ControlWord::Recycle.refusal(),
443        // No control command reaches this branch with another reason: a
444        // `completed`, `fault` or `replaced` close retires an idle slot, and a
445        // `shutdown` close runs `close_all`, neither of which comes through
446        // here. The word is the server's own for a close that named no command
447        // — `repair close` without a `--reason` writes it on the task's row.
448        _ => "operator close",
449    }
450}
451
452/// Close every live session's resource with `reason` and forget the slots.
453///
454/// This is the shutdown path: a stopped client must not leave resources behind
455/// that only it can address, and each backend's own record of the resource —
456/// the Orca tab map included — ends with the session. `budget` bounds the whole
457/// sweep, because an operator's SIGTERM must not turn into a hang while a slow
458/// backend CLI exits; whatever the budget cuts off is reported and dropped
459/// anyway.
460///
461/// The order is deliberate: every slot is taken off this client's books under the
462/// dispatch lock — untracked and removed, so nothing still reads a live session —
463/// and the closes it collected run once the lock is let go. A shutdown that
464/// closed each pane while holding that lock spent the whole budget on the host,
465/// with the lock no adapter frame could reach.
466pub fn close_all(state: &DispatchState, reason: crate::backend::CloseReason, budget: Duration) {
467    let started = Instant::now();
468    let pending: Vec<PendingClose> = {
469        let mut inner = state.inner.lock();
470        let sessions: Vec<(String, SessionRef)> = inner
471            .sessions
472            .iter()
473            .map(|(key, slot)| (key.clone(), slot.session.clone()))
474            .collect();
475        let mut pending = Vec::with_capacity(sessions.len());
476        for (key, session) in sessions {
477            inner.bridge.untrack_live(&session.task_id);
478            forget_tools_binding(&mut inner, &key);
479            inner.sessions.remove(&key);
480            pending.push(PendingClose {
481                backend: Arc::clone(&inner.backend),
482                session,
483                reason,
484            });
485        }
486        pending
487    };
488    for close in pending {
489        if started.elapsed() > budget {
490            tracing::warn!(
491                task = %close.session.task_id,
492                "shutdown close budget reached; the resource is left behind"
493            );
494            continue;
495        }
496        if let Err(error) = close.backend.close(&close.session, close.reason, false) {
497            tracing::warn!(
498                task = %close.session.task_id,
499                error = %error,
500                "session close failed during shutdown"
501            );
502        }
503    }
504}
505
506/// Which way a session's window closed.
507///
508/// Two arms reach the same verdict through different facts, and an operator reading one
509/// aggregate log line cannot tell them apart: one means the connection ended and the agent
510/// stayed away, the other means the connection is still up while nothing the client accepts
511/// arrives over it. The words exist so the log can name which reading retired a session.
512#[derive(Clone, Copy, Debug, PartialEq, Eq)]
513pub enum RetirementArm {
514    /// The plugin connection ended and `[client] reconnect_grace_secs` expired.
515    Dropped,
516    /// The connection stayed up while the session went quiet past the heartbeat window.
517    Silent,
518}
519
520impl RetirementArm {
521    pub fn word(self) -> &'static str {
522        match self {
523            Self::Dropped => "reconnect_grace",
524            Self::Silent => "heartbeat_silence",
525        }
526    }
527}
528
529/// One session the sweep retired, with what it read on the way.
530pub struct Retired {
531    /// The retired session's own id, which a client-held session shares with its task.
532    pub session_id: String,
533    /// The arm that decided it.
534    pub arm: RetirementArm,
535    /// Seconds since this session's last accepted frame.
536    pub quiet_secs: u64,
537    /// Seconds since its connection ended, when one did.
538    pub away_secs: Option<u64>,
539}
540
541/// How long a session has gone without a frame this client accepted.
542fn quiet_secs(slot: &SessionSlot, now: Instant) -> u64 {
543    slot.last_beat
544        .map(|beat| now.saturating_duration_since(beat).as_secs())
545        .unwrap_or(u64::MAX)
546}
547
548/// How long a session's connection has been gone, for a session still waiting on one.
549fn away_secs(slot: &SessionSlot, now: Instant) -> Option<u64> {
550    slot.dropped_at
551        .map(|left| now.saturating_duration_since(left).as_secs())
552}
553
554impl DispatchState {
555    /// Retire tracked resources whose stored lifecycle has reached `Exited`.
556    ///
557    /// The periodic readiness tick calls this after completed work becomes an
558    /// idle slot. Task-free sessions with an attached transport stay bound to
559    /// their host resource, and task-free sessions whose agent has left release it.
560    ///
561    /// The session ids come back because the retirement wrote each one of those
562    /// rows and the server mirrors only what this client reports: the resource
563    /// close and, for a completed session, the agent's exit both moved the row
564    /// this tick found, and a publish cannot run under this lock. The caller is
565    /// handed what to publish, the same answer [`DispatchState::retire_dropped_ghosts`]
566    /// gives the sweep above.
567    pub fn reclaim_exited_resources(&self) -> Vec<String> {
568        let mut inner = self.inner.lock();
569        let candidates: Vec<(String, crate::backend::CloseReason)> = inner
570            .sessions
571            .iter()
572            .filter(|(key, slot)| {
573                // A suspended session is not an exited one. Its process is gone
574                // *because* it was released, its conversation is what the
575                // family's next delivery is for, and the row it published says
576                // `idle` for exactly that reason — `binding_task_state` answers
577                // `Pending` for a slot that serves nothing and whose scope keeps
578                // it, so the two derivations of "is this over" disagree here.
579                //
580                // This sweep asks the other one, and it asks about the *work*:
581                // a suspended session's delivery is finished, so it reads
582                // `Exited` and the slot is retired within a tick. The family's
583                // next delivery then finds nothing to resume and opens a second
584                // conversation for one chain — which is the failure this filter
585                // existed to prevent, reached through the sweep that was meant
586                // to clean up after a session that was already gone.
587                !slot.suspended
588                    && slot.task_id.is_none()
589                    && session_exited(&inner, &slot.session.task_id)
590                    && !has_attached_transport(&inner, key, slot)
591            })
592            .filter_map(|(key, slot)| {
593                stored_close_reason(&inner, &slot.session.task_id)
594                    .map(|reason| (key.clone(), reason))
595            })
596            .collect();
597        let mut retired: Vec<String> = Vec::new();
598        let mut pending: Vec<PendingClose> = Vec::new();
599        for (key, reason) in candidates {
600            // The id that travels is the session's own, the one whose row the
601            // retirement below is about to write.
602            let Some(task_id) = inner
603                .sessions
604                .get(&key)
605                .map(|slot| slot.session.task_id.clone())
606            else {
607                continue;
608            };
609            if retire_idle_locked(&mut inner, &key, reason, &mut pending) {
610                retired.push(task_id);
611            }
612        }
613        // The tick that found a batch of finished sessions owes a host close for
614        // each, and this sweep runs every 250 ms: on the lock, one slow backend
615        // would spend the whole window and every adapter frame queued behind it.
616        drop(inner);
617        close_retired(pending);
618        retired
619    }
620
621    /// Retire the sessions whose plugin connection dropped and never came back, or
622    /// whose connection stayed up while they went quiet, and answer which ones left
623    /// and why.
624    ///
625    /// A connection that ends without a `detach` frame leaves its session tracked
626    /// so an agent that restarts inside `[client] reconnect_grace_secs` finds the
627    /// resource it was using. That promise has to expire: a process that is
628    /// simply gone would otherwise hold a slot, a projected `idle` row, and a live
629    /// host resource forever, and on a role with `max_sessions = 1` it stops every
630    /// later delivery. The window answers for the agent itself, so a session still
631    /// bound to a task goes with it: the plugin connection that would have
632    /// reported the ending is the one that dropped. The agent-gone feed is what
633    /// says the process left — the session's own tuple reaches `Exited` through
634    /// `AgentPhase::Gone` rather than through a task result — and the reason the
635    /// backend is handed is the one `grace_close_reason` reads off what the slot
636    /// still owes.
637    ///
638    /// What the slot owed is settled too: the task a bound session was serving
639    /// ends `failed` here, because the agent that would have reported its ending
640    /// is the one that left. A task with no verdict stays open for the server to
641    /// re-offer and for `open_tasks` to keep reading, and no later caller exists
642    /// to write one.
643    ///
644    /// A slot this client holds read-only is not this sweep's to end, agent gone
645    /// or not: the session id it would feed is the task id, so the ghost's death
646    /// would take the live session's mirror and its delivery row down with it.
647    /// That retirement belongs to `retire_revived`, which runs when the
648    /// completion that answers the held connection merges.
649    ///
650    /// The window has a second way to open, and it is the one a socket cannot
651    /// report: a plugin whose event loop is blocked keeps its connection and
652    /// stops beating, so no socket ends and no clock this sweep could read
653    /// before moved. What such a session leaves behind is a stamp going stale
654    /// while its task stays bound and unsettled, and that is the reading this
655    /// sweep takes now. It is the same window and the same verdict — one clock,
656    /// one retirement, no second threshold beside `[client]
657    /// reconnect_grace_secs` and no fault row of the kind `stall_report_secs`
658    /// records and leaves behind.
659    ///
660    /// The sessions' own ids come back rather than a count, because a retirement
661    /// still owes the server the session's own ending: it is the only writer left
662    /// for that task, and a mirror nobody tells keeps that session's last reading —
663    /// `working`, for one that had beaten — until the server's own observer records
664    /// a fault about it. The publish is
665    /// [`sync_session`](crate::session::dispatch::sync_session)'s, which is the
666    /// report an ordinary ending travels on, and it cannot run under this lock —
667    /// so the caller is handed what to publish instead of a second writer being
668    /// invented here.
669    pub fn retire_dropped_ghosts(&self, now: Instant, grace_secs: u64) -> Vec<Retired> {
670        if grace_secs == 0 {
671            return Vec::new();
672        }
673        let window = Duration::from_secs(grace_secs);
674        let mut inner = self.inner.lock();
675        // The arm is decided from the pair of readings here, while both still
676        // describe the slot: which fact closed the window is what the operator
677        // has to be able to tell apart once the retirement itself is a line in
678        // the log.
679        let due: Vec<(String, RetirementArm)> = inner
680            .sessions
681            .iter()
682            .filter_map(|(key, slot)| {
683                let arm = if dropped_past_window(&inner, key, slot, now, window) {
684                    RetirementArm::Dropped
685                } else if silent_past_window(&inner, key, slot, now) {
686                    RetirementArm::Silent
687                } else {
688                    return None;
689                };
690                Some((key.clone(), arm))
691            })
692            .collect();
693        let mut retired: Vec<Retired> = Vec::new();
694        let mut pending: Vec<PendingClose> = Vec::new();
695        for (key, arm) in due {
696            let Some(slot) = inner.sessions.get(&key).cloned() else {
697                continue;
698            };
699            // A slot this client holds read-only is not this sweep's to end, for
700            // the reason the doc above gives. The demotion alone does not decide
701            // it: the held connection that owns the slot can go without the task
702            // ever completing, and a slot nothing owns any more is what this
703            // window is for.
704            if held_read_only(&inner, &key, &slot) {
705                continue;
706            }
707            // A session judged dead on its silence is the one case where the
708            // connection is still there: the socket has not ended and will not
709            // while the agent behind it is blocked, so the client's own verdict
710            // has to take the binding the way the death of the socket would have.
711            // The retirement below refuses a slot an attached transport still
712            // serves, and that refusal is what this unbinding answers: the
713            // verdict has already been reached here, so the binding goes rather
714            // than the death of the socket that would normally take it.
715            if slot.dropped_at.is_none() {
716                unbind_transports(&mut inner, &key, &slot);
717            }
718            let task_id = slot.session.task_id.clone();
719            // Both the feed and the reason name the session's own task, so the id
720            // that travels is the one the row and the resource are keyed by.
721            let reason = grace_close_reason(&inner, &task_id);
722            // The work this session still owed ends here, and this sweep is the
723            // only writer left to say so: the plugin connection that would have
724            // reported the ending is the one that dropped. A task nobody answers
725            // stays `settled_at IS NULL` forever, so `open_tasks` keeps reading
726            // it and the server keeps re-offering a delivery no client can take.
727            // The binding is what the slot owed — a slot past its window with no
728            // task bound owes nothing — and `failed` is the verdict the close
729            // reason above already carries for it. A row an earlier verdict
730            // settled keeps that one: `settle_task` updates only where
731            // `settled_at IS NULL` and answers `false`.
732            //
733            // The write runs before the agent-gone feed and before the binding
734            // hand-back, both of which end this slot's turn through the sweep:
735            // the id is captured here, and the verdict is on disk before the
736            // only handle on it goes away.
737            if let Some(owed) = slot.task_id.clone() {
738                // A verdict written here answers the task for good, so a
739                // `control` command that is still waiting for its plugin's report
740                // has nothing left to authorise: the note goes with the verdict
741                // that outranks it.
742                inner.control_settles.retain(|noted| noted.task_id != owed);
743                if let Err(error) = inner.store.settle_task(&owed, TaskState::Failed) {
744                    tracing::warn!(
745                        task = %owed,
746                        error = %error,
747                        "the task of a retired ghost was not settled"
748                    );
749                }
750                // The delivery handle this session was holding is spent as a
751                // refusal that names the death, and it is the ledger half of the
752                // verdict above. A row left `in_flight` is handed to a pull no
753                // longer — `pull` passes by a row whose ticket is armed, and a
754                // role-level pull's ticket carries no session id for the release
755                // path to match — so nothing would answer for this task until the
756                // link dropped, and an operator reading `onlyne ledger` would see
757                // a session that has been buried as one still holding its
758                // delivery. The reason is this client's own word for a death
759                // (`SESSION_DEAD`), and a refusal is terminal: the work comes back
760                // through `repair retry`, not by itself.
761                let handle = inner
762                    .sessions
763                    .get_mut(&key)
764                    .and_then(|slot| slot.msg_id.take());
765                if let Some(msg_id) = handle {
766                    store_ack(
767                        &inner,
768                        AckArgs {
769                            msg_id,
770                            op_id: None,
771                            accepted: false,
772                            reason: Some(SESSION_DEAD.to_string()),
773                        },
774                    );
775                }
776            }
777            if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, &task_id) {
778                tracing::warn!(
779                    task = %task_id,
780                    error = %error,
781                    "agent-gone projection failed for a retired ghost"
782                );
783            }
784            // The agent left, so the session owes no task any more: the binding
785            // goes back before the idle retirement takes the slot.
786            if let Some(slot) = inner.sessions.get_mut(&key) {
787                slot.task_id = None;
788            }
789            if retire_idle_locked(&mut inner, &key, reason, &mut pending) {
790                // The id that travels is the session's own, the one whose row was
791                // just fed agent-gone and resource-closed: that row is what the
792                // server mirrors, and its ending is what the caller publishes. The
793                // ages are read off the slot as it stood before the retirement.
794                retired.push(Retired {
795                    session_id: task_id,
796                    arm,
797                    quiet_secs: quiet_secs(&slot, now),
798                    away_secs: away_secs(&slot, now),
799                });
800            }
801        }
802        // Every session this sweep took is off the slot map and fully written down
803        // by now; what remains is the host's own work, and the lock that answers an
804        // agent's next frame is not to be held through it.
805        drop(inner);
806        close_retired(pending);
807        retired
808    }
809}