onlyne_client/session/dispatch/retire.rs
1use super::*;
2
3use super::outbound::store_ack;
4use super::projection::stored_task_state;
5use super::state::{
6 DispatchInner, DispatchState, forget_tools_binding, has_attached_transport, session_exited,
7 slot_key_serving_task,
8};
9use super::transport::{held_read_only, names_session};
10
11/// Reason a session that stopped answering is closed with.
12///
13/// The wire word the ledger and `onlyne sessions` read for a death this client
14/// judged: the reconnect sweep stamps it on the refusal ack that buries the
15/// session's delivery, and an operator's own
16/// `repair fail --reason session_dead` writes the same word.
17pub const SESSION_DEAD: &str = "session_dead";
18
19/// Retire one session. The stored tuple decides whether a live resource
20/// remains to close, and the caller's reason reaches the backend unchanged, so an
21/// operator cancel stops reporting itself as a completion.
22pub fn on_recycled(
23 state: &DispatchState,
24 task_id: &str,
25 reason: crate::backend::CloseReason,
26) -> Result<()> {
27 let mut inner = state.inner.lock();
28 release_locked(&mut inner, task_id, Some(reason))
29}
30
31/// Why the task of one session ended, as the retirement reason the task table
32/// records.
33///
34/// The task's own row is the only place a verdict lives: the session tuple says
35/// nothing about how its work ended, and an open task — `pending`, or no row at
36/// all, which is the same reading — has no reason to retire anything.
37pub(super) fn stored_close_reason(
38 inner: &DispatchInner,
39 task_id: &str,
40) -> Option<crate::backend::CloseReason> {
41 match stored_task_state(inner, task_id) {
42 TaskState::Pending => None,
43 TaskState::Done => Some(crate::backend::CloseReason::Completed),
44 // A blocked delivery leaves the work owed, which is what a fault reason
45 // names here, exactly as it does for a failed one.
46 TaskState::Failed | TaskState::Blocked => Some(crate::backend::CloseReason::Fault),
47 TaskState::Cancelled => Some(crate::backend::CloseReason::Cancelled),
48 }
49}
50
51/// The reason one session the reconnect grace retires is closed with, read from
52/// what this client holds rather than from a settle that never came.
53///
54/// The id is the session's own (`slot.session.task_id`), never the task binding
55/// beside it: the sweep is what feeds that session's row and closes the resource
56/// its agent was holding, and the two ids part company exactly where a slot's
57/// binding is not the session it was born for. A session still owing work closes
58/// as that task's own record reads: a `done` task is a `Completed`, a
59/// `cancelled` one is a `Cancelled`, and a `failed` task — like one that never
60/// settled at all — is a `Fault`, because the work was still owed when the agent
61/// left.
62fn grace_close_reason(inner: &DispatchInner, task_id: &str) -> crate::backend::CloseReason {
63 match stored_task_state(inner, task_id) {
64 TaskState::Done => crate::backend::CloseReason::Completed,
65 TaskState::Pending | TaskState::Failed | TaskState::Blocked => {
66 crate::backend::CloseReason::Fault
67 }
68 TaskState::Cancelled => crate::backend::CloseReason::Cancelled,
69 }
70}
71
72/// Whether one slot is due for the reconnect grace to take it.
73///
74/// The window as it always was: the connection that would have sent this
75/// session's next heartbeat has ended, and the window runs from the moment it
76/// left. A slot a live connection serves is not this arm's to end, because an
77/// attached transport is the one thing that says the agent is still reachable.
78fn dropped_past_window(
79 inner: &DispatchInner,
80 key: &str,
81 slot: &SessionSlot,
82 now: Instant,
83 window: Duration,
84) -> bool {
85 slot.dropped_at.is_some_and(|dropped| {
86 !has_attached_transport(inner, key, slot)
87 && now
88 .checked_duration_since(dropped)
89 .is_some_and(|away| away >= window)
90 })
91}
92
93/// Whether one session's own agent has gone quiet on a socket that is still up.
94///
95/// The window's other door, and it exists precisely because an attached
96/// transport — the one thing the arm above trusts — can lie. A socket that is
97/// still up proves the connection survived; it says nothing about the agent
98/// behind it, and a plugin whose event loop is blocked holds its socket and
99/// stops beating. Nothing the client reads before this could see that: no socket
100/// ends, so no drop clock ever starts, and the session keeps its slot, its
101/// projected row and its host resource for as long as the client runs.
102///
103/// So the reading is taken off the frames themselves. A session whose task is
104/// still bound and unsettled and whose last accepted frame is older than the
105/// protocol's heartbeat interval by [`HEARTBEAT_SILENCE_MARGIN`] has no agent
106/// behind its socket.
107///
108/// Why no task bound is excluded: the plugin stops its heartbeat loop with the
109/// last task it was given, so a task-free session that has gone quiet is an
110/// agent waiting for work by design — the ordinary shape between deliveries.
111/// Sweeping it would retire the very connection the next payload is staged onto.
112/// The unsettled check beside the binding says the same thing about work that
113/// already landed: a settled task has nothing left for its session to answer,
114/// and `release_locked` has already given the binding back.
115///
116/// Why the drop clock's arm never reads this stamp: the two are mutually
117/// exclusive by construction. A re-mount clears `dropped_at`, so a session that
118/// came back is judged by its beat alone, and a session whose connection went
119/// away is judged by the clock alone and never by a stamp its agent can no
120/// longer refresh.
121fn silent_past_window(inner: &DispatchInner, key: &str, slot: &SessionSlot, now: Instant) -> bool {
122 if slot.dropped_at.is_some() || !has_attached_transport(inner, key, slot) {
123 return false;
124 }
125 let Some(task_id) = slot.task_id.as_deref() else {
126 return false;
127 };
128 if stored_task_state(inner, task_id) != TaskState::Pending {
129 return false;
130 }
131 let quiet = HEARTBEAT_INTERVAL * HEARTBEAT_SILENCE_MARGIN;
132 slot.last_beat.is_some_and(|beat| {
133 now.checked_duration_since(beat)
134 .is_some_and(|away| away >= quiet)
135 })
136}
137
138/// Stop holding a session's connection as its transport.
139///
140/// The act a socket ending performs, run here on the client's own verdict
141/// instead: a session judged dead by its silence still holds its socket, and the
142/// connection is not going to end on its own while the agent behind it is
143/// blocked. The binding goes now, so the retirement below runs as the ordinary
144/// one, and a frame from that connection afterwards is refused the way every
145/// frame from a connection no session answers for is — it is served no state,
146/// which is the same door a stale reporter already comes to.
147fn unbind_transports(inner: &mut DispatchInner, key: &str, slot: &SessionSlot) {
148 inner
149 .transports
150 .retain(|served, _| !names_session(key, slot, served));
151}
152
153/// One backend close a retirement owed, carried off the dispatch lock.
154///
155/// A close is a host round trip — a pane kill, a terminal close, an agent reap —
156/// and it can take seconds. The dispatch lock is the one lock every adapter frame,
157/// every report, and every slot of this role queues behind, so a sweep that
158/// retires more than one session at a time must not spend that lock on the hosts
159/// it is calling. A retirement therefore records what it took down here and runs
160/// the closes once it is off the lock.
161pub(super) struct PendingClose {
162 pub(super) backend: Arc<dyn SessionBackend>,
163 pub(super) session: SessionRef,
164 pub(super) reason: crate::backend::CloseReason,
165}
166
167/// Run the closes a retirement collected.
168///
169/// Four of the five callers hold the dispatch lock and can put it down first — the
170/// two sweeps, the goodbye path, and the merged-handoff retire — and each does,
171/// because a close that blocks must not hold every adapter frame of this role
172/// behind it. The fifth is `release_locked`, which runs inside a caller that
173/// already holds the lock for the store work above it and cannot leave; it takes
174/// the close where it stands, which is the behavior that path has always had.
175///
176/// A failed close is reported and not propagated. The slot is gone from this
177/// client's books either way and its row already reads closed, so the caller has
178/// nothing left to undo; an error that travelled back would replace the answer
179/// the caller is waiting for — which sessions left, and therefore must be
180/// published — with a failure to say so.
181pub(super) fn close_retired(pending: Vec<PendingClose>) {
182 for close in pending {
183 if let Err(error) = close.backend.close(&close.session, close.reason, false) {
184 tracing::warn!(
185 task = %close.session.task_id,
186 backend = %close.session.backend,
187 resource = %close.session.backend_ref,
188 error = %error,
189 "session resource retirement failed"
190 );
191 }
192 }
193}
194
195/// Whether this role's scope keeps one session alive after its delivery settles.
196///
197/// `oneshot` is the rule the client has always had: the session served its one
198/// delivery and the slot is done with it. A `task` or `role` session outlives
199/// the delivery that opened it, which is the whole point of the scope, so its
200/// slot stays serving nothing until the scope sends it the next delivery or
201/// `idle_close` releases its process.
202///
203/// A read-only slot is never kept: it is a connection that came back for a task
204/// a newer session already took, and it owns no delivery of this role.
205pub(super) fn keeps_idle(inner: &DispatchInner, key: &str) -> bool {
206 inner
207 .sessions
208 .get(key)
209 .is_some_and(|slot| slot.keeps_idle && !slot.read_only)
210}
211
212/// Retire one task-free session after its transport set becomes empty.
213///
214/// This is the `oneshot` ending, and the ending of every session whose scope
215/// does not keep it: its resource closes because the agent able to run another
216/// task in it has left. [`keeps_idle`] answers for the scopes that keep a
217/// session between deliveries, and those never reach here with work behind them.
218///
219/// The idle slot releases its backend resource because the agent able to run
220/// another task in it has left. An attached transport keeps the resource because
221/// that agent remains reachable. The dispatch lock serializes the final transport
222/// check, reference refresh, lifecycle projection, and slot removal with adapter
223/// binding. The backend close is deliberately NOT one of them: it is recorded in
224/// `pending` for the caller to run once it is off the lock, so a sweep of sessions
225/// cannot hold every frame of this role behind a host round trip. See
226/// [`close_retired`] for the one path that cannot put the lock down.
227pub(super) fn retire_idle_locked(
228 inner: &mut DispatchInner,
229 key: &str,
230 reason: crate::backend::CloseReason,
231 pending: &mut Vec<PendingClose>,
232) -> bool {
233 let Some(slot) = inner.sessions.get(key) else {
234 return false;
235 };
236 if slot.task_id.is_some() || has_attached_transport(inner, key, slot) {
237 return false;
238 }
239
240 let original = slot.session.clone();
241 let task_id = original.task_id.clone();
242 let resource = inner
243 .store
244 .get_session(&task_id)
245 .ok()
246 .flatten()
247 .map(|row| row.resource_state)
248 .unwrap_or_else(|| "detached".to_string());
249 if resource != "detached" && resource != "closed" {
250 let session = match inner.backend.attach(&original) {
251 Ok(refreshed) => {
252 if refreshed != original {
253 inner.bridge.track_live(refreshed.clone());
254 if let Some(slot) = inner.sessions.get_mut(key) {
255 slot.session = refreshed.clone();
256 }
257 }
258 refreshed
259 }
260 Err(_) => original,
261 };
262 tracing::info!(
263 task = %task_id,
264 backend = %session.backend,
265 resource = %session.backend_ref,
266 ?reason,
267 "retiring idle session resource"
268 );
269 if let Err(error) = feed_resource_closed(&inner.bridge, &inner.store, &task_id) {
270 tracing::warn!(
271 task = %task_id,
272 backend = %session.backend,
273 resource = %session.backend_ref,
274 error = %error,
275 "session resource close projection failed"
276 );
277 }
278 pending.push(PendingClose {
279 backend: Arc::clone(&inner.backend),
280 session,
281 reason,
282 });
283 }
284 if reason == crate::backend::CloseReason::Completed {
285 if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, &task_id) {
286 tracing::warn!(
287 task = %task_id,
288 error = %error,
289 "agent-gone projection failed for a completed session"
290 );
291 }
292 }
293 inner.bridge.untrack_live(&task_id);
294 forget_tools_binding(inner, key);
295 inner.sessions.remove(key);
296 true
297}
298
299/// Give one session's task slot back. Settled tasks enter idle retirement, and
300/// explicit reasons drive the control-close path.
301pub(super) fn release_locked(
302 inner: &mut DispatchInner,
303 task_id: &str,
304 reason: Option<crate::backend::CloseReason>,
305) -> Result<()> {
306 let resource = inner
307 .store
308 .get_session(task_id)?
309 .map(|row| row.resource_state)
310 .unwrap_or_else(|| "detached".to_string());
311 if let Some((key, slot)) = slot_key_serving_task(inner, task_id)
312 .and_then(|key| inner.sessions.get(&key).map(|slot| (key, slot.clone())))
313 {
314 if let Some(reason) = reason {
315 if resource != "detached" && resource != "closed" {
316 feed_resource_closed(&inner.bridge, &inner.store, task_id)?;
317 inner.backend.close(&slot.session, reason, false)?;
318 }
319 // The agent goes with the resource: this path closes a session whose work an
320 // operator ended or whose backend faulted, and the slot below leaves the map in
321 // the same breath. Without this feed the tuple keeps the agent phase its last
322 // beat reported, and `project` answers `working` for a `cancelled` or `failed`
323 // task whenever the agent is not `Gone` — so the mirrored row read `working` for
324 // a session the client had already closed, which is what `onlyne sessions` and
325 // the board showed beside a ledger row that had settled. The reconnect sweep
326 // feeds the same event for the same ending.
327 if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, task_id) {
328 tracing::warn!(
329 task = %task_id,
330 error = %error,
331 "agent-gone projection failed for a closed session"
332 );
333 }
334 inner.bridge.untrack_live(task_id);
335 // The delivery row this session is still holding is answered here,
336 // while the slot that holds its handle is still in the map. The close
337 // takes the slot, and the handle goes with it: a row this client
338 // never answered is a row the server still reads as owed, so the
339 // pull that would have taken it passes and the release of a session
340 // the server judges gone hands it back to the queue — which
341 // dispatches the task a second time and runs it again, as the live
342 // run showed for a task whose work the close had already ended.
343 //
344 // The handle is taken, so the row is answered once: the control
345 // settle watchdog and the reconnect sweep refuse the same handle
346 // the same way, and whichever of the three runs first spends it and
347 // the others find nothing left to answer.
348 let held = inner
349 .sessions
350 .get_mut(&key)
351 .and_then(|slot| slot.msg_id.take());
352 if let Some(msg_id) = held {
353 store_ack(
354 inner,
355 AckArgs {
356 msg_id,
357 op_id: None,
358 accepted: false,
359 reason: Some(close_refusal(reason).to_string()),
360 },
361 );
362 }
363 forget_tools_binding(inner, &key);
364 inner.sessions.remove(&key);
365 } else {
366 let session_id = inner
367 .sessions
368 .get(&key)
369 .map(|slot| slot.session.task_id.clone())
370 .unwrap_or_else(|| task_id.to_string());
371 let keeps = keeps_idle(inner, &key);
372 let now = Instant::now();
373 if let Some(session) = inner.sessions.get_mut(&key) {
374 session.task_id = None;
375 session.ready = false;
376 session.idle_since = Some(now);
377 }
378 // The binding goes with the delivery. A session that serves several
379 // deliveries in turn serves none between them, and both the hello
380 // claim and the mirrored row read the open binding to say which
381 // delivery a session is on.
382 if let Err(error) = inner.store.release_binding(&session_id, task_id) {
383 tracing::warn!(
384 session = %session_id,
385 task = %task_id,
386 error = %error,
387 "a settled session's binding was not handed back"
388 );
389 }
390 if keeps {
391 // A `task` or `role` session outlives the delivery that opened
392 // it: that is what the scope is for. It stays idle, with its
393 // process and its row, until the scope sends it the next
394 // delivery or `idle_close` releases the process.
395 tracing::info!(
396 session = %session_id,
397 task = %task_id,
398 "delivery settled; the session stays idle for its scope's next one"
399 );
400 } else {
401 let mut pending = Vec::new();
402 retire_idle_locked(
403 inner,
404 &key,
405 crate::backend::CloseReason::Completed,
406 &mut pending,
407 );
408 // This function answers to its caller's lock, which is already held
409 // across the store work above, so the close it owes cannot wait for an
410 // unlock this scope cannot perform. Running it here is the same
411 // under-lock call the path has always made; the sweeps that CAN leave
412 // the lock — `retire_dropped_ghosts`, `reclaim_exited_resources`,
413 // `close_all` — are the ones that now defer it.
414 close_retired(pending);
415 }
416 }
417 }
418 inner.stall.forget(task_id);
419 if reason.is_some() && resource == "detached" {
420 inner
421 .store
422 .note_alert(format!("session recycled {task_id}"));
423 }
424 Ok(())
425}
426
427/// The operator's word a control close stands for, as the refusal that answers
428/// the row its session was still holding.
429///
430/// The words are the ones the settle fallback already writes for the same
431/// commands ([`ControlWord::refusal`]), so a row reads the same thing whichever
432/// door refused it, and each names the command that was given rather than the
433/// verdict that command left behind.
434fn close_refusal(reason: crate::backend::CloseReason) -> &'static str {
435 match reason {
436 // The close an operator's `cancel` runs: the `ControlOp::Cancel` arm of
437 // `on_control`, which is also how the server asks for `repair close` and
438 // `repair fail` to reach this client.
439 crate::backend::CloseReason::Cancelled => ControlWord::Cancel.refusal(),
440 // The close an operator's `recycle` runs: the `ControlOp::Recycle` arm
441 // of `on_control`.
442 crate::backend::CloseReason::Operator => ControlWord::Recycle.refusal(),
443 // No control command reaches this branch with another reason: a
444 // `completed`, `fault` or `replaced` close retires an idle slot, and a
445 // `shutdown` close runs `close_all`, neither of which comes through
446 // here. The word is the server's own for a close that named no command
447 // — `repair close` without a `--reason` writes it on the task's row.
448 _ => "operator close",
449 }
450}
451
452/// Close every live session's resource with `reason` and forget the slots.
453///
454/// This is the shutdown path: a stopped client must not leave resources behind
455/// that only it can address, and each backend's own record of the resource —
456/// the Orca tab map included — ends with the session. `budget` bounds the whole
457/// sweep, because an operator's SIGTERM must not turn into a hang while a slow
458/// backend CLI exits; whatever the budget cuts off is reported and dropped
459/// anyway.
460///
461/// The order is deliberate: every slot is taken off this client's books under the
462/// dispatch lock — untracked and removed, so nothing still reads a live session —
463/// and the closes it collected run once the lock is let go. A shutdown that
464/// closed each pane while holding that lock spent the whole budget on the host,
465/// with the lock no adapter frame could reach.
466pub fn close_all(state: &DispatchState, reason: crate::backend::CloseReason, budget: Duration) {
467 let started = Instant::now();
468 let pending: Vec<PendingClose> = {
469 let mut inner = state.inner.lock();
470 let sessions: Vec<(String, SessionRef)> = inner
471 .sessions
472 .iter()
473 .map(|(key, slot)| (key.clone(), slot.session.clone()))
474 .collect();
475 let mut pending = Vec::with_capacity(sessions.len());
476 for (key, session) in sessions {
477 inner.bridge.untrack_live(&session.task_id);
478 forget_tools_binding(&mut inner, &key);
479 inner.sessions.remove(&key);
480 pending.push(PendingClose {
481 backend: Arc::clone(&inner.backend),
482 session,
483 reason,
484 });
485 }
486 pending
487 };
488 for close in pending {
489 if started.elapsed() > budget {
490 tracing::warn!(
491 task = %close.session.task_id,
492 "shutdown close budget reached; the resource is left behind"
493 );
494 continue;
495 }
496 if let Err(error) = close.backend.close(&close.session, close.reason, false) {
497 tracing::warn!(
498 task = %close.session.task_id,
499 error = %error,
500 "session close failed during shutdown"
501 );
502 }
503 }
504}
505
506/// Which way a session's window closed.
507///
508/// Two arms reach the same verdict through different facts, and an operator reading one
509/// aggregate log line cannot tell them apart: one means the connection ended and the agent
510/// stayed away, the other means the connection is still up while nothing the client accepts
511/// arrives over it. The words exist so the log can name which reading retired a session.
512#[derive(Clone, Copy, Debug, PartialEq, Eq)]
513pub enum RetirementArm {
514 /// The plugin connection ended and `[client] reconnect_grace_secs` expired.
515 Dropped,
516 /// The connection stayed up while the session went quiet past the heartbeat window.
517 Silent,
518}
519
520impl RetirementArm {
521 pub fn word(self) -> &'static str {
522 match self {
523 Self::Dropped => "reconnect_grace",
524 Self::Silent => "heartbeat_silence",
525 }
526 }
527}
528
529/// One session the sweep retired, with what it read on the way.
530pub struct Retired {
531 /// The retired session's own id, which a client-held session shares with its task.
532 pub session_id: String,
533 /// The arm that decided it.
534 pub arm: RetirementArm,
535 /// Seconds since this session's last accepted frame.
536 pub quiet_secs: u64,
537 /// Seconds since its connection ended, when one did.
538 pub away_secs: Option<u64>,
539}
540
541/// How long a session has gone without a frame this client accepted.
542fn quiet_secs(slot: &SessionSlot, now: Instant) -> u64 {
543 slot.last_beat
544 .map(|beat| now.saturating_duration_since(beat).as_secs())
545 .unwrap_or(u64::MAX)
546}
547
548/// How long a session's connection has been gone, for a session still waiting on one.
549fn away_secs(slot: &SessionSlot, now: Instant) -> Option<u64> {
550 slot.dropped_at
551 .map(|left| now.saturating_duration_since(left).as_secs())
552}
553
554impl DispatchState {
555 /// Retire tracked resources whose stored lifecycle has reached `Exited`.
556 ///
557 /// The periodic readiness tick calls this after completed work becomes an
558 /// idle slot. Task-free sessions with an attached transport stay bound to
559 /// their host resource, and task-free sessions whose agent has left release it.
560 ///
561 /// The session ids come back because the retirement wrote each one of those
562 /// rows and the server mirrors only what this client reports: the resource
563 /// close and, for a completed session, the agent's exit both moved the row
564 /// this tick found, and a publish cannot run under this lock. The caller is
565 /// handed what to publish, the same answer [`DispatchState::retire_dropped_ghosts`]
566 /// gives the sweep above.
567 pub fn reclaim_exited_resources(&self) -> Vec<String> {
568 let mut inner = self.inner.lock();
569 let candidates: Vec<(String, crate::backend::CloseReason)> = inner
570 .sessions
571 .iter()
572 .filter(|(key, slot)| {
573 // A suspended session is not an exited one. Its process is gone
574 // *because* it was released, its conversation is what the
575 // family's next delivery is for, and the row it published says
576 // `idle` for exactly that reason — `binding_task_state` answers
577 // `Pending` for a slot that serves nothing and whose scope keeps
578 // it, so the two derivations of "is this over" disagree here.
579 //
580 // This sweep asks the other one, and it asks about the *work*:
581 // a suspended session's delivery is finished, so it reads
582 // `Exited` and the slot is retired within a tick. The family's
583 // next delivery then finds nothing to resume and opens a second
584 // conversation for one chain — which is the failure this filter
585 // existed to prevent, reached through the sweep that was meant
586 // to clean up after a session that was already gone.
587 !slot.suspended
588 && slot.task_id.is_none()
589 && session_exited(&inner, &slot.session.task_id)
590 && !has_attached_transport(&inner, key, slot)
591 })
592 .filter_map(|(key, slot)| {
593 stored_close_reason(&inner, &slot.session.task_id)
594 .map(|reason| (key.clone(), reason))
595 })
596 .collect();
597 let mut retired: Vec<String> = Vec::new();
598 let mut pending: Vec<PendingClose> = Vec::new();
599 for (key, reason) in candidates {
600 // The id that travels is the session's own, the one whose row the
601 // retirement below is about to write.
602 let Some(task_id) = inner
603 .sessions
604 .get(&key)
605 .map(|slot| slot.session.task_id.clone())
606 else {
607 continue;
608 };
609 if retire_idle_locked(&mut inner, &key, reason, &mut pending) {
610 retired.push(task_id);
611 }
612 }
613 // The tick that found a batch of finished sessions owes a host close for
614 // each, and this sweep runs every 250 ms: on the lock, one slow backend
615 // would spend the whole window and every adapter frame queued behind it.
616 drop(inner);
617 close_retired(pending);
618 retired
619 }
620
621 /// Retire the sessions whose plugin connection dropped and never came back, or
622 /// whose connection stayed up while they went quiet, and answer which ones left
623 /// and why.
624 ///
625 /// A connection that ends without a `detach` frame leaves its session tracked
626 /// so an agent that restarts inside `[client] reconnect_grace_secs` finds the
627 /// resource it was using. That promise has to expire: a process that is
628 /// simply gone would otherwise hold a slot, a projected `idle` row, and a live
629 /// host resource forever, and on a role with `max_sessions = 1` it stops every
630 /// later delivery. The window answers for the agent itself, so a session still
631 /// bound to a task goes with it: the plugin connection that would have
632 /// reported the ending is the one that dropped. The agent-gone feed is what
633 /// says the process left — the session's own tuple reaches `Exited` through
634 /// `AgentPhase::Gone` rather than through a task result — and the reason the
635 /// backend is handed is the one `grace_close_reason` reads off what the slot
636 /// still owes.
637 ///
638 /// What the slot owed is settled too: the task a bound session was serving
639 /// ends `failed` here, because the agent that would have reported its ending
640 /// is the one that left. A task with no verdict stays open for the server to
641 /// re-offer and for `open_tasks` to keep reading, and no later caller exists
642 /// to write one.
643 ///
644 /// A slot this client holds read-only is not this sweep's to end, agent gone
645 /// or not: the session id it would feed is the task id, so the ghost's death
646 /// would take the live session's mirror and its delivery row down with it.
647 /// That retirement belongs to `retire_revived`, which runs when the
648 /// completion that answers the held connection merges.
649 ///
650 /// The window has a second way to open, and it is the one a socket cannot
651 /// report: a plugin whose event loop is blocked keeps its connection and
652 /// stops beating, so no socket ends and no clock this sweep could read
653 /// before moved. What such a session leaves behind is a stamp going stale
654 /// while its task stays bound and unsettled, and that is the reading this
655 /// sweep takes now. It is the same window and the same verdict — one clock,
656 /// one retirement, no second threshold beside `[client]
657 /// reconnect_grace_secs` and no fault row of the kind `stall_report_secs`
658 /// records and leaves behind.
659 ///
660 /// The sessions' own ids come back rather than a count, because a retirement
661 /// still owes the server the session's own ending: it is the only writer left
662 /// for that task, and a mirror nobody tells keeps that session's last reading —
663 /// `working`, for one that had beaten — until the server's own observer records
664 /// a fault about it. The publish is
665 /// [`sync_session`](crate::session::dispatch::sync_session)'s, which is the
666 /// report an ordinary ending travels on, and it cannot run under this lock —
667 /// so the caller is handed what to publish instead of a second writer being
668 /// invented here.
669 pub fn retire_dropped_ghosts(&self, now: Instant, grace_secs: u64) -> Vec<Retired> {
670 if grace_secs == 0 {
671 return Vec::new();
672 }
673 let window = Duration::from_secs(grace_secs);
674 let mut inner = self.inner.lock();
675 // The arm is decided from the pair of readings here, while both still
676 // describe the slot: which fact closed the window is what the operator
677 // has to be able to tell apart once the retirement itself is a line in
678 // the log.
679 let due: Vec<(String, RetirementArm)> = inner
680 .sessions
681 .iter()
682 .filter_map(|(key, slot)| {
683 let arm = if dropped_past_window(&inner, key, slot, now, window) {
684 RetirementArm::Dropped
685 } else if silent_past_window(&inner, key, slot, now) {
686 RetirementArm::Silent
687 } else {
688 return None;
689 };
690 Some((key.clone(), arm))
691 })
692 .collect();
693 let mut retired: Vec<Retired> = Vec::new();
694 let mut pending: Vec<PendingClose> = Vec::new();
695 for (key, arm) in due {
696 let Some(slot) = inner.sessions.get(&key).cloned() else {
697 continue;
698 };
699 // A slot this client holds read-only is not this sweep's to end, for
700 // the reason the doc above gives. The demotion alone does not decide
701 // it: the held connection that owns the slot can go without the task
702 // ever completing, and a slot nothing owns any more is what this
703 // window is for.
704 if held_read_only(&inner, &key, &slot) {
705 continue;
706 }
707 // A session judged dead on its silence is the one case where the
708 // connection is still there: the socket has not ended and will not
709 // while the agent behind it is blocked, so the client's own verdict
710 // has to take the binding the way the death of the socket would have.
711 // The retirement below refuses a slot an attached transport still
712 // serves, and that refusal is what this unbinding answers: the
713 // verdict has already been reached here, so the binding goes rather
714 // than the death of the socket that would normally take it.
715 if slot.dropped_at.is_none() {
716 unbind_transports(&mut inner, &key, &slot);
717 }
718 let task_id = slot.session.task_id.clone();
719 // Both the feed and the reason name the session's own task, so the id
720 // that travels is the one the row and the resource are keyed by.
721 let reason = grace_close_reason(&inner, &task_id);
722 // The work this session still owed ends here, and this sweep is the
723 // only writer left to say so: the plugin connection that would have
724 // reported the ending is the one that dropped. A task nobody answers
725 // stays `settled_at IS NULL` forever, so `open_tasks` keeps reading
726 // it and the server keeps re-offering a delivery no client can take.
727 // The binding is what the slot owed — a slot past its window with no
728 // task bound owes nothing — and `failed` is the verdict the close
729 // reason above already carries for it. A row an earlier verdict
730 // settled keeps that one: `settle_task` updates only where
731 // `settled_at IS NULL` and answers `false`.
732 //
733 // The write runs before the agent-gone feed and before the binding
734 // hand-back, both of which end this slot's turn through the sweep:
735 // the id is captured here, and the verdict is on disk before the
736 // only handle on it goes away.
737 if let Some(owed) = slot.task_id.clone() {
738 // A verdict written here answers the task for good, so a
739 // `control` command that is still waiting for its plugin's report
740 // has nothing left to authorise: the note goes with the verdict
741 // that outranks it.
742 inner.control_settles.retain(|noted| noted.task_id != owed);
743 if let Err(error) = inner.store.settle_task(&owed, TaskState::Failed) {
744 tracing::warn!(
745 task = %owed,
746 error = %error,
747 "the task of a retired ghost was not settled"
748 );
749 }
750 // The delivery handle this session was holding is spent as a
751 // refusal that names the death, and it is the ledger half of the
752 // verdict above. A row left `in_flight` is handed to a pull no
753 // longer — `pull` passes by a row whose ticket is armed, and a
754 // role-level pull's ticket carries no session id for the release
755 // path to match — so nothing would answer for this task until the
756 // link dropped, and an operator reading `onlyne ledger` would see
757 // a session that has been buried as one still holding its
758 // delivery. The reason is this client's own word for a death
759 // (`SESSION_DEAD`), and a refusal is terminal: the work comes back
760 // through `repair retry`, not by itself.
761 let handle = inner
762 .sessions
763 .get_mut(&key)
764 .and_then(|slot| slot.msg_id.take());
765 if let Some(msg_id) = handle {
766 store_ack(
767 &inner,
768 AckArgs {
769 msg_id,
770 op_id: None,
771 accepted: false,
772 reason: Some(SESSION_DEAD.to_string()),
773 },
774 );
775 }
776 }
777 if let Err(error) = feed_agent_gone(&inner.bridge, &inner.store, &task_id) {
778 tracing::warn!(
779 task = %task_id,
780 error = %error,
781 "agent-gone projection failed for a retired ghost"
782 );
783 }
784 // The agent left, so the session owes no task any more: the binding
785 // goes back before the idle retirement takes the slot.
786 if let Some(slot) = inner.sessions.get_mut(&key) {
787 slot.task_id = None;
788 }
789 if retire_idle_locked(&mut inner, &key, reason, &mut pending) {
790 // The id that travels is the session's own, the one whose row was
791 // just fed agent-gone and resource-closed: that row is what the
792 // server mirrors, and its ending is what the caller publishes. The
793 // ages are read off the slot as it stood before the retirement.
794 retired.push(Retired {
795 session_id: task_id,
796 arm,
797 quiet_secs: quiet_secs(&slot, now),
798 away_secs: away_secs(&slot, now),
799 });
800 }
801 }
802 // Every session this sweep took is off the slot map and fully written down
803 // by now; what remains is the host's own work, and the lock that answers an
804 // agent's next frame is not to be held through it.
805 drop(inner);
806 close_retired(pending);
807 retired
808 }
809}