mahbot 0.7.3

An autonomous agentic engineering system that manages software development through role separation, subagents, and deterministic diagnostics.
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
492
493
494
495
496
497
498
499
500
501
502
503
504
505
506
507
508
509
510
511
512
513
514
515
516
517
518
519
520
521
522
523
524
525
526
527
528
529
530
531
532
533
534
535
536
537
538
539
540
541
542
543
544
545
546
547
548
549
550
551
552
553
554
555
556
557
558
559
560
561
562
563
564
565
566
567
568
569
570
571
572
573
574
575
576
577
578
579
580
581
582
583
584
585
586
587
588
589
590
591
592
593
594
595
596
597
598
599
600
601
602
603
604
605
606
607
608
609
610
611
612
613
614
615
616
617
618
619
620
621
622
623
624
625
626
627
628
629
630
631
632
633
634
635
636
637
638
639
640
641
642
643
644
645
646
647
648
649
650
651
652
653
654
655
656
657
658
659
660
661
662
663
664
665
666
667
668
669
670
671
672
673
674
675
676
677
678
679
680
681
682
683
684
685
686
687
688
689
690
691
692
693
694
695
696
697
698
699
700
701
702
703
704
705
706
707
708
709
710
711
712
713
714
715
716
717
718
719
720
721
722
723
724
725
726
727
728
729
730
731
732
733
734
735
736
737
738
739
740
741
742
743
744
745
746
747
748
749
750
751
752
753
754
755
756
757
758
759
760
761
762
763
764
765
766
767
768
769
770
771
772
773
774
775
776
777
778
779
780
781
782
783
784
785
786
787
788
789
790
791
792
793
794
795
796
797
798
799
800
801
802
803
804
805
806
807
808
809
810
811
812
813
814
815
816
817
818
819
820
821
822
823
824
825
826
827
828
829
830
831
832
833
834
835
836
837
838
839
840
841
842
843
844
845
846
847
848
849
850
851
852
853
854
855
856
857
858
859
860
861
862
863
864
865
866
867
868
869
870
871
872
873
874
875
876
877
878
879
880
881
882
883
884
885
886
887
888
889
890
891
892
893
894
895
896
897
898
899
900
901
902
903
904
905
906
907
908
909
910
911
912
913
914
915
916
917
918
919
920
921
922
923
924
925
926
927
928
929
930
931
932
933
934
935
936
937
938
939
940
941
942
943
944
945
946
947
948
949
950
951
952
953
954
955
956
957
958
959
960
961
962
963
964
965
966
967
968
969
970
971
972
973
974
975
976
977
978
979
980
981
982
983
984
985
986
987
988
989
990
991
992
993
994
995
996
997
998
999
1000
1001
1002
1003
1004
1005
1006
1007
1008
1009
1010
1011
1012
1013
1014
1015
1016
1017
1018
1019
1020
1021
1022
1023
1024
1025
1026
1027
1028
1029
1030
1031
1032
1033
1034
1035
1036
1037
1038
1039
1040
1041
1042
1043
1044
1045
1046
1047
1048
1049
1050
1051
1052
1053
1054
1055
1056
1057
1058
1059
1060
1061
1062
1063
1064
1065
1066
1067
1068
1069
1070
1071
1072
1073
1074
1075
1076
1077
1078
1079
1080
1081
1082
1083
1084
1085
1086
1087
1088
1089
1090
1091
1092
1093
1094
1095
1096
1097
1098
1099
1100
1101
1102
1103
1104
1105
1106
1107
1108
1109
1110
1111
1112
1113
1114
1115
1116
1117
1118
1119
1120
1121
1122
1123
1124
1125
1126
1127
1128
1129
1130
1131
1132
1133
1134
1135
1136
1137
1138
1139
1140
1141
1142
1143
1144
1145
1146
1147
1148
1149
1150
1151
1152
1153
1154
1155
1156
1157
1158
1159
1160
1161
1162
1163
1164
1165
1166
1167
1168
1169
1170
1171
1172
1173
1174
1175
1176
1177
1178
1179
1180
1181
1182
1183
1184
1185
1186
1187
1188
1189
1190
1191
1192
1193
1194
1195
1196
1197
1198
1199
1200
1201
1202
1203
1204
1205
1206
1207
1208
1209
1210
1211
1212
1213
1214
1215
1216
1217
1218
1219
1220
1221
1222
1223
1224
1225
1226
1227
1228
1229
1230
1231
1232
1233
1234
1235
1236
1237
1238
1239
1240
1241
1242
1243
1244
1245
1246
1247
1248
1249
1250
1251
1252
1253
1254
1255
1256
1257
1258
1259
1260
1261
1262
1263
1264
1265
1266
1267
1268
1269
1270
1271
1272
1273
1274
1275
1276
1277
1278
1279
1280
1281
1282
1283
1284
1285
1286
1287
1288
1289
1290
1291
1292
1293
1294
1295
1296
1297
1298
1299
1300
1301
1302
1303
1304
1305
1306
1307
1308
1309
1310
1311
1312
1313
1314
1315
1316
1317
1318
1319
1320
1321
1322
1323
1324
1325
1326
1327
1328
1329
1330
1331
1332
1333
1334
1335
1336
1337
1338
1339
1340
1341
1342
1343
1344
1345
1346
1347
1348
1349
1350
1351
1352
1353
1354
1355
1356
1357
1358
1359
1360
1361
1362
1363
1364
1365
1366
1367
1368
1369
1370
1371
1372
1373
1374
1375
1376
1377
1378
1379
1380
1381
1382
1383
1384
1385
1386
1387
1388
1389
1390
1391
1392
1393
1394
1395
1396
1397
1398
1399
1400
1401
1402
1403
1404
1405
1406
1407
1408
1409
1410
1411
1412
1413
1414
1415
1416
1417
1418
1419
1420
1421
1422
1423
1424
1425
1426
1427
1428
1429
1430
1431
1432
1433
1434
1435
1436
1437
1438
1439
1440
1441
1442
1443
1444
1445
1446
1447
1448
1449
1450
//! Global shutdown infrastructure.
//!
//! Provides a global shutdown token and signal handling for graceful daemon
//! shutdown. Used by provider, agent, management, storage, and channel
//! code to race futures against shutdown signals.
//!
//! Extracted from `self_update` where it was a layer violation — shutdown
//! coordination is not self-update.
//!
//! # The exits that run none of the closing path
//!
//! The product's closing path is the library's `gui::Dashboard::save_and_exit` (draft flush,
//! window state, the token fired before the iced runtime drops) and the binary's
//! `shutdown_after_dashboard` (browser release, background-task join, store checkpoint). The exits
//! below run none of it, and are listed here once so that the rest of the tree points at this list
//! instead of restating it:
//!
//! - a platform-forced termination (Force Quit, a `SIGKILL`): nothing can intercept it; a plain
//!   `kill` is SIGTERM, which the product handles once its signal task is installed;
//! - the self-update tail (its own drain and its own exit, `exit(0)` after its checkpoint), which
//!   never reaches `save_and_exit` or `shutdown_after_dashboard`;
//! - the fatal-signal and abort paths (`install_fatal_signal_handlers`, the panic hook): an
//!   `_exit` from the handler, or an unwind that never reaches the teardown;
//! - an iced `run()` that returns an error: the process ends with that error before the teardown
//!   is reached at all;
//! - the system's own log-out, restart or shut-down on macOS: the quit hook hands the platform's
//!   ask straight back and AppKit ends the process inside its run loop (see `macos_quit`), with
//!   the losses that route always had — the last moment of typing, the exit-time tidying that
//!   leaves the next start slower, and the browser release, which the next start performs anyway;
//! - an ordinary macOS quit the hook does not take over — the interface is already past its exit,
//!   or the interception could not be installed: the platform's own handling ends the process
//!   (both are accepted in `macos_quit`'s "Accepted" section);
//! - a SIGTERM that arrives before the signal task exists: the platform's own handling ends the
//!   process the same way;
//! - the exit bound `macos_quit` arms for a taken-over quit: a hard `_exit` with no closing steps
//!   after the closing path has failed to finish within it, the same class as a force quit.
//!
//! A Windows session end is not on that list: the listener stops the daemon like any other stop
//! request, so the closing path runs, bounded by the deadline the platform gave it — except for a
//! notification that arrives before the dashboard subscribes, where the forced stop has no
//! consumer. The CLI subcommands and a launch the instance lock turns away reach none of the
//! closing path either, having exited before the runtime exists. A new exit path that drops the
//! runtime without firing the token belongs on this list.

use crate::util::UnwrapPoison;
use std::sync::{Mutex, OnceLock};
use std::time::{Duration, Instant};
use tokio_util::sync::CancellationToken;
use tracing::info;

// ── Global shutdown token ─────────────────────────────────────────────────

static GLOBAL_SHUTDOWN: OnceLock<CancellationToken> = OnceLock::new();

fn global_shutdown() -> &'static CancellationToken {
    GLOBAL_SHUTDOWN.get_or_init(CancellationToken::new)
}

/// Global graceful-drain state: a `watch` channel of `bool` set by the first
/// shutdown signal (SIGINT / window-close / quit / self-update). Distinct from the
/// cancellation token — during the drain the token is NOT fired, so in-flight
/// LLM calls (which race the token around the HTTP send) survive to complete
/// their current round. Background loops fold this flag into their
/// sleep/shutdown races; the second signal maps to force-cancel.
static DRAIN: OnceLock<tokio::sync::watch::Sender<bool>> = OnceLock::new();

/// The process-lifetime drain sender.
fn drain_sender() -> &'static tokio::sync::watch::Sender<bool> {
    DRAIN.get_or_init(|| tokio::sync::watch::channel(false).0)
}

/// A receiver tracking the current drain value. Each call subscribes fresh, so
/// a waiter that registers before a drain begins is notified the instant it
/// flips; one that registers after reads the current value immediately.
fn drain_receiver() -> tokio::sync::watch::Receiver<bool> {
    drain_sender().subscribe()
}

/// Mark the daemon as draining (graceful-shutdown window). Idempotent.
pub fn drain_begin() {
    drain_sender().send_replace(true);
    info!("Draining: in-flight work completes before exit");
}

/// Whether the graceful-drain window is active.
#[must_use]
pub fn is_draining() -> bool {
    *drain_sender().borrow()
}

/// Whether the daemon is aborting: the shutdown token fired OR the graceful
/// drain is active. Loops gate NEW work on this (during the drain, in-flight
/// work completes but nothing new starts).
#[must_use]
pub fn aborting() -> bool {
    shutdown_token().is_cancelled() || is_draining()
}

/// Force-cancel the drain: fire the global token immediately (in-flight
/// agents are cancelled and boot-resume via status='launched'), then the normal
/// exit path (checkpoint + join + exit) runs.
pub fn force_cancel() {
    global_shutdown().cancel();
}

/// Clear the drain flag. Production code never clears it (drains are
/// one-way); tests use this to restore isolation after asserting drain
/// behavior.
pub fn drain_clear() {
    drain_sender().send_replace(false);
}

/// Get a clone of the global shutdown token.
#[must_use]
pub fn shutdown_token() -> CancellationToken {
    global_shutdown().clone()
}

/// Clean-drain-completion trigger: fires the global token once the drain-watch
/// sees no in-flight work, ending the graceful-drain window. Gracefulness comes
/// from [`drain_begin`] — this does not start a drain (if the token already
/// fired, the watch returns before reaching this). Contrast [`force_cancel`],
/// the abort path that fires the token mid-drain to cancel in-flight work.
pub fn shutdown() {
    global_shutdown().cancel();
}

/// Error returned by [`race_shutdown`] when the global shutdown token fires.
pub struct Shutdown;

/// Race a future against the global shutdown token.
/// Returns `Ok(T)` if the future completes first, `Err(Shutdown)` if shutdown is signaled.
pub async fn race_shutdown<F, T>(fut: F) -> Result<T, Shutdown>
where
    F: std::future::Future<Output = T>,
{
    let token = shutdown_token();
    tokio::select! {
        result = fut => Ok(result),
        () = token.cancelled() => Err(Shutdown),
    }
}

/// Sleep for the given duration, or return early if shutdown is signaled.
/// Returns `true` if the sleep completed normally, `false` if shutdown was signaled.
#[must_use]
pub async fn sleep_or_shutdown(duration: Duration) -> bool {
    race_shutdown(tokio::time::sleep(duration)).await.is_ok()
}

/// Sleep for the given duration, breaking early on the shutdown token OR the
/// graceful-drain flag. Background loops fold the drain into their sleep
/// cycles so they stop spawning new work when the drain begins.
/// Returns `true` if the sleep completed normally, `false` if shutdown or
/// draining was signaled.
#[must_use]
pub async fn sleep_or_shutdown_or_drain(duration: Duration) -> bool {
    let token = shutdown_token();
    let drain = drain_wait();
    tokio::pin!(drain);
    tokio::select! {
        () = token.cancelled() => false,
        () = &mut drain => false,
        () = tokio::time::sleep(duration) => true,
    }
}

/// Completes when the drain flag flips. Event-driven: waits on the drain watch
/// channel rather than polling. Borrowing before awaiting makes it race-free —
/// a drain that began before this waiter registered resolves immediately, and
/// it is cancel-safe for `select!` use.
pub(crate) async fn drain_wait() {
    let mut rx = drain_receiver();
    while !*rx.borrow_and_update() {
        if rx.changed().await.is_err() {
            return; // sender is a process-lifetime static — unreachable
        }
    }
}

// ── Stop requests ─────────────────────────────────────────────────────────

/// A stop request the platform delivered to the daemon.
///
/// Unix has two, on an async signal stream. Windows has three, from the console
/// control handler below, and they are not the same request: only Ctrl+C is the
/// platform's "wind down" gesture. A launch with no console — the windowed product's
/// normal one — receives none of them: the platform delivers no console control event
/// to a process that has no console (see the `console` module). A session end
/// (log-off, shutdown, restart) is a stop request too, but not one of these and not a
/// drain: it comes from the window listener ([`install_session_end_listener`]).
#[cfg(any(windows, test))]
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum StopRequest {
    /// Ctrl+C — the Windows counterpart of SIGINT/SIGTERM.
    Interrupt,
    /// Ctrl+Break — terminates the process by default, with no timeout.
    Break,
    /// The console window is closing, so only a launch that has a console can receive
    /// it; the task manager's "end task" raises the same event. The platform kills the
    /// process when its grace elapses.
    ConsoleClose,
}

#[cfg(windows)]
impl StopRequest {
    /// The name the request is logged and recorded under.
    #[must_use]
    fn label(self) -> &'static str {
        match self {
            Self::Interrupt => "Ctrl+C",
            Self::Break => "Ctrl+Break",
            Self::ConsoleClose => "console close",
        }
    }
}

/// What a stop request does to the process.
#[cfg(any(windows, test))]
#[derive(Clone, Copy, Debug, PartialEq, Eq)]
enum StopAction {
    /// Begin the graceful drain: in-flight work finishes its current round.
    Drain,
    /// Abandon the drain and run the exit path now.
    ForceCancel,
}

/// The request→action rule — the Windows half of the protocol Unix gets from
/// SIGINT/SIGTERM. Ctrl+C drains, unless a drain is already running: then it is
/// the "second request" that force-cancels. Ctrl+Break is that same hard stop in
/// its own right, and a console close is one the platform's kill deadline may cut
/// off — so neither begins a drain.
#[cfg(any(windows, test))]
#[must_use]
fn stop_action(request: StopRequest, draining: bool) -> StopAction {
    if request == StopRequest::Interrupt && !draining {
        StopAction::Drain
    } else {
        StopAction::ForceCancel
    }
}

/// The instant the platform's stop deadline expires: when a stop that carries a grace
/// arrived (a Windows console close or session end) plus the grace it grants. `None`
/// until such a stop arrives.
static STOP_DEADLINE: Mutex<Option<Instant>> = Mutex::new(None);

/// The grace assumed for a stop whose own value cannot be read: the documented default of
/// `SPI_GETHUNGAPPTIMEOUT`, the user-settable parameter behind a console close. A session
/// end has no readable equivalent, so it takes the same conservative bound.
#[cfg(windows)]
const DEFAULT_STOP_GRACE: Duration = Duration::from_secs(5);

/// Record the deadline a stop has to finish before, for the exit path's budget. The
/// earliest one wins: that is the kill the process must beat.
#[cfg(windows)]
fn record_stop_deadline(deadline: Instant) {
    let mut recorded = STOP_DEADLINE.lock().unwrap_poison();
    if recorded.is_none_or(|current| deadline < current) {
        *recorded = Some(deadline);
    }
}

/// The exit path's budget from the platform's stop deadline and the moment a stage
/// starts: half the grace still left, so the checkpoint keeps the rest. A deadline that
/// has passed leaves nothing.
#[must_use]
fn stage_budget(deadline: Instant, now: Instant) -> Duration {
    deadline.saturating_duration_since(now) / 2
}

/// The bound on the exit path's stage before the store checkpoint, for a stop that
/// runs under the platform's own kill deadline (a Windows console close or session
/// end): `None` for every other stop, which no platform deadline bounds (a dashboard
/// window close, a first Ctrl+C, a macOS quit).
///
/// Sampled once, when that stage starts, from the deadline the platform's stops recorded
/// (`record_stop_deadline` keeps the earliest — the kill the process must beat);
/// [`stage_budget`] then takes half of what is still left of it, so the later in the exit
/// path the stage starts, the less it gets. The checkpoint that follows is never bounded —
/// it is the durable step the exit path exists for — so the rest of the grace is left for
/// it. What the bound cuts loses nothing durable: the browser releases a cut flush did not
/// settle are durable records retried at boot, and its closing sweep is best-effort by
/// design.
#[must_use]
pub fn urgent_release_budget() -> Option<Duration> {
    STOP_DEADLINE
        .lock()
        .unwrap_poison()
        .map(|deadline| stage_budget(deadline, Instant::now()))
}

/// The last stop request the platform delivered, for the exit path to name: the Windows stop
/// deliveries (the console protocol loop and the session-end listener) and a macOS quit (the
/// dashboard's exit request) write one — the macOS route only where that request is what drives
/// the exit, so a repeat gesture leaves the record to the request that caused it. `None` when
/// nothing was recorded: a dashboard window close and a Unix signal record nothing, and the exit
/// line then keeps the wording it always had. A `Mutex`, not a `OnceLock`: a later request takes
/// the record over from the one before it.
///
/// The exit line itself is dropped outright — it runs after the iced runtime whose log writer
/// batches — so this makes a stop's reason nameable by that line, not visible: a stop's reason
/// reaches the logs store only where an earlier line named it.
static EXIT_TRIGGER: Mutex<Option<&'static str>> = Mutex::new(None);

/// Record the stop request the exit path is running for, so the exit path can name it.
pub(crate) fn record_exit_trigger(label: &'static str) {
    *EXIT_TRIGGER.lock().unwrap_poison() = Some(label);
}

/// The last recorded stop request, if any was recorded — `EXIT_TRIGGER` above says what writes one.
#[must_use]
pub fn exit_trigger() -> Option<&'static str> {
    *EXIT_TRIGGER.lock().unwrap_poison()
}

// ── Windows session end (log-off, shutdown, restart) ──────────────────────

/// The name of the session end a `WM_ENDSESSION` notification reports — `None` for a
/// notification this daemon must not act on.
///
/// The platform announces a session end twice: first the preliminary question
/// (`WM_QUERYENDSESSION`), which is answered affirmatively and never acted on, then the
/// final notification. `ending` is that notification's `wParam`: it is false when a
/// shutdown that had started is cancelled, so only a true one may stop the daemon.
/// `closes_app_only` is its `lParam`'s `ENDSESSION_CLOSEAPP` bit — the platform asking this
/// application alone to close so that an update can proceed, which ends no session and
/// whose stop would leave the daemon down on a session that continues — and `log_off` is
/// its `ENDSESSION_LOGOFF` bit; without it the end is a machine shutdown or restart, which
/// the flags do not tell apart. The critical end sets neither bit and is a session end like
/// any other.
#[cfg(any(windows, test))]
#[must_use]
fn session_end_label(ending: bool, closes_app_only: bool, log_off: bool) -> Option<&'static str> {
    if !ending || closes_app_only {
        return None;
    }
    Some(if log_off {
        "Windows log-off"
    } else {
        "Windows shutdown/restart"
    })
}

/// Act on a session end: name it for the exit path, bound the stages before the exit-time
/// checkpoint, and force the stop.
///
/// Forced, not drained: the seconds the platform allows cannot hold the graceful drain, so
/// waiting would leave the stop unfinished when the platform ends the process. Nothing is
/// waited for either, and nothing is asked of the platform (no shutdown-blocking reason is
/// declared, which is the one way to ask for more time): this runs on the listener's own
/// thread and returns at once, so the shutdown is never slowed down and no "this program is
/// preventing shutdown" surface can appear. The stop sequence itself (the dashboard's draft
/// and geometry flush, the browser release, the journal checkpoint) then runs where it always
/// does, inside the time the platform already grants.
#[cfg(windows)]
fn stop_for_session_end(label: &'static str) {
    record_exit_trigger(label);
    record_stop_deadline(Instant::now() + DEFAULT_STOP_GRACE);
    info!("{label} — forcing the stop: the platform's seconds cannot hold a drain");
    force_cancel();
}

// ── Signal handling ───────────────────────────────────────────────────────

/// Wait for stop requests, then drive the two-request drain protocol.
///
/// Unix: first signal (SIGINT or SIGTERM) begins the graceful drain via
/// [`drain_begin`] — the global token is NOT fired, so in-flight LLM calls
/// and tool groups complete. Background loops break on the drain flag.
/// The signal streams stay alive (never dropped) so a SECOND signal during
/// the drain maps to [`force_cancel`] — abort the drain, checkpoint, exit.
/// SIGHUP is explicitly ignored so the daemon survives terminal/SSH disconnects.
///
/// Windows: the same protocol, fed by the console control handler installed by
/// `install_console_stop_handler` (each event arrives on a fresh OS thread, so
/// the handler queues it here and, for a closing console, holds that thread; the
/// `console` module documents the platform rules and what start-up does before this
/// loop exists — and why a launch with no console has no events to feed it). Ctrl+C
/// is the drain request, a second Ctrl+C force-cancels, and
/// Ctrl+Break and a console close are force-cancel class outright. The session end
/// (log-off, shutdown, restart) is not a console event and does not come through here at
/// all — it is force-cancel class on its own terms, in `install_session_end_listener`.
///
/// The "second request" is read off the global drain flag rather than a per-source
/// count, so a first Ctrl+C that follows a drain begun elsewhere — the dashboard
/// window close, a platform quit, a self-update's finalizing drain — force-cancels,
/// where on Unix a first SIGINT after those is a no-op. The dashboard's own exit
/// requests read the same flag (see `gui::Dashboard::request_exit`), so this is
/// deliberate and not a defect.
///
/// Returns only when the drain must be abandoned (a force-cancel-class request), naming that
/// request where the platform names it — a Windows console request does, while a Unix signal is
/// named in its own line locally and returns none, leaving the caller's line the fallback it always
/// carried; the clean-drain exit path is driven by the drain-watch task in the binary (fires the
/// token when no in-flight agents or orchestrator calls remain).
pub async fn wait_for_shutdown_signal() -> anyhow::Result<Option<&'static str>> {
    #[cfg(unix)]
    {
        use tokio::signal::unix::{SignalKind, signal};
        use tracing::debug;

        let mut sigint = signal(SignalKind::interrupt())?;
        let mut sigterm = signal(SignalKind::terminate())?;
        let mut sighup = signal(SignalKind::hangup())?;

        let mut first = true;
        loop {
            let signal = tokio::select! {
                _ = sigint.recv() => "SIGINT",
                _ = sigterm.recv() => "SIGTERM",
                _ = sighup.recv() => {
                    debug!("Received SIGHUP, ignoring (daemon stays running)");
                    continue;
                }
            };
            if first {
                info!("Received {signal} — draining (second signal force-cancels)");
                drain_begin();
                first = false;
            } else {
                info!("Received second {signal} — force-cancelling drain");
                return Ok(None);
            }
        }
    }

    #[cfg(windows)]
    {
        loop {
            let request = console::next_request().await?;
            let label = request.label();
            record_exit_trigger(label);
            match stop_action(request, is_draining()) {
                StopAction::Drain => {
                    info!("Received {label} — draining (a second request force-cancels)");
                    drain_begin();
                }
                // The caller's signal task logs this one, naming the request returned here.
                StopAction::ForceCancel => return Ok(Some(label)),
            }
        }
    }
}

// ── Stop-request sources ──────────────────────────────────────────────────

/// Subscribe the process to every stop request the platform can deliver, before boot and before
/// the interface exists: the two Windows sources below and the quit source, the only one of the
/// three that is not platform-gated (the `install_quit_requests` call below states it).
///
/// They have to be up this early because the platform may deliver a stop the moment the window is
/// up — before boot has finished — and the request has to survive until the dashboard subscribes
/// to it.
pub fn install_stop_request_sources() {
    install_console_stop_handler();
    install_session_end_listener();
    install_quit_requests();
}

// ── Windows console control handler ───────────────────────────────────────

/// Install the console stop-request handler — a no-op on macOS/Linux.
///
/// Called from [`install_stop_request_sources`] before boot: the handler needs no runtime (it
/// queues for the async protocol loop, see the `console` module), so the subscription is never
/// what a stop request goes missing on. A registration failure is reported where
/// there is a console (see the `console` module) and never fails a launch — the
/// fallback it costs is the tokio Ctrl+C handler, which (like this one) can only
/// ever fire where a console exists to deliver the event, so a console-less launch
/// loses nothing it had.
fn install_console_stop_handler() {
    #[cfg(windows)]
    console::install();
}

/// Install the Windows session-end listener (log-off, shutdown, restart) — a no-op on
/// macOS/Linux.
///
/// Called from [`install_stop_request_sources`] before boot, next to
/// [`install_console_stop_handler`]: the platform
/// may end the session at any time, and the listener needs no runtime (it forces the stop
/// synchronously, see the `session_end` module). A failure to bring it up is reported, never
/// fatal: the daemon then keeps exactly the stop requests it had.
fn install_session_end_listener() {
    #[cfg(windows)]
    session_end::install();
}

/// The Windows console control handler — the platform's stop-request source
/// wherever a console exists.
///
/// Ctrl+C, Ctrl+Break and the console window closing arrive as console control
/// events, each on a **new OS thread the system creates in this process**. That
/// thread has no async context and a closing console must not be left to a task the
/// console might kill first, so the handler queues the request for
/// [`wait_for_shutdown_signal`] instead of serving it. That loop is spawned only
/// after a successful boot: while it does not exist, a request carrying a platform
/// deadline (the close) is still queued and one that carries none keeps the
/// platform's default handling (see the dead ends below).
///
/// # The platform's rules, and what they force
///
/// - A close event brings a grace the system starts counting down on delivery, from
///   the user-settable `SPI_GETHUNGAPPTIMEOUT` parameter (5000 ms by default, so it
///   is read here and never assumed). The system kills the process when it elapses,
///   and that grace is the only window a close leaves for an exit.
/// - The grace is usable only by NOT returning: the contract is "return TRUE and the
///   system terminates the process". So the close handler holds its thread while the
///   async side runs the exit path, and the system kills the process if that path
///   does not finish — the accepted hard death. Holding the full grace is
///   deliberate: returning sooner only kills the process sooner than the platform
///   would, and the hold ends at the deadline the platform is already charging for.
/// - Ctrl+C and Ctrl+Break have no timeout, and returning TRUE leaves the process
///   running, so once the loop exists they queue and return immediately.
/// - Two rapid requests may reach the process as one, because the console can
///   collapse them before the handler runs (as Unix may) — so a "second request"
///   that is never observed is not a defect. Every event the handler does receive is
///   queued on its own.
///
/// # Session end is not a console event
///
/// `CTRL_LOGOFF_EVENT` and `CTRL_SHUTDOWN_EVENT` are not delivered to a console
/// application that loads the GUI libraries (user32 and gdi32, as this one does for its
/// dashboard): once user32 is loaded, the platform treats the process as one that owns a
/// window and tells it about a log-off, shutdown or restart through window messages
/// instead. This handler therefore declines both events, and the daemon hears a session
/// end through its own hidden window — the `session_end` module below.
///
/// # Accepted dead ends (documented, deliberately not fixed)
///
/// - External termination (the task manager's "end process", `TerminateProcess`): no
///   process can intercept it. Its "end task" is a different command, raises the
///   close event, and does become graceful here.
/// - Process-tree containment: a stop that ends the daemon takes the command trees
///   it started with it — a job object's handles go with the process
///   (`tools::shell::tree`) — while the launches it detaches on purpose (the
///   browser, the browser-automation CLI, the replacement instance of a
///   self-update) survive it. That boundary belongs to those subsystems, not to
///   termination, and is left alone.
/// - Ctrl+C and Ctrl+Break before the loop exists: nothing can act on them, so they
///   keep today's platform default handling, as macOS/Linux does before its signal
///   task registers its streams. Swallowing them instead would make Ctrl+C a dead key
///   wherever boot never reaches the loop — the start-failure screen, which answers its
///   own window's close but has no consumer for a console event at all.
/// - Once the loop exists the handler suppresses the default handling for Ctrl+C and
///   Ctrl+Break, so those would be inert if that loop died while the process lived. Unix
///   is no better: tokio keeps its handler installed for the whole process even once the
///   signal streams are dropped, so a died-out signal task there swallows the signal the
///   same way. No machinery is warranted: the console close and the dashboard window
///   close remain as escapes.
/// - A kill mid-exit, when a close's grace expires first: what the bound cuts is
///   covered by the durability note on [`urgent_release_budget`], and a kill that lands
///   mid-checkpoint cuts only hygiene — committed work is durable on disk. Whatever
///   state that leaves the store in is the next start's to sort out, through the same
///   bring-up gate as any other hard death here: `db::open_store` shape-checks the file
///   and then runs its data-preserving repairs.
///
/// # Remaining silent hard deaths
///
/// Those the dead ends above name, plus a stop request that arrives with the handler
/// unregistered, and a console close during a start-up that never reached the loop — a
/// console event has no consumer while that loop does not exist, and none is added for
/// it. Both of those are console events, so both need a launch that has a console to
/// arise at all; there, and only there, Ctrl+C still reaches tokio's own handler, whose
/// `ctrl_c()` is itself a `SetConsoleCtrlHandler` registration. The dashboard's own window close
/// is not among them: the dashboard consumes it in every state, whether or not boot
/// finished. A session end is not among them: it reaches the process through its own window
/// whether or not this handler is registered — the instance a self-update leaves behind owns
/// no console at all, so no console close can reach it, but it owns a window just the same.
#[cfg(windows)]
mod console {
    use super::{DEFAULT_STOP_GRACE, StopRequest, record_stop_deadline};
    use crate::util::UnwrapPoison;
    use std::collections::VecDeque;
    use std::sync::Mutex;
    use std::sync::atomic::{AtomicBool, Ordering};
    use std::time::{Duration, Instant};
    use tokio::sync::Notify;
    use windows_sys::Win32::Foundation::{FALSE, TRUE};
    use windows_sys::Win32::System::Console::{
        CTRL_BREAK_EVENT, CTRL_C_EVENT, CTRL_CLOSE_EVENT, GetConsoleCP, SetConsoleCtrlHandler,
    };
    use windows_sys::Win32::UI::WindowsAndMessaging::{
        SPI_GETHUNGAPPTIMEOUT, SystemParametersInfoW,
    };

    /// Requests queued by handler threads, drained in order by the protocol loop.
    static REQUESTS: Mutex<VecDeque<StopRequest>> = Mutex::new(VecDeque::new());

    /// Wakes the protocol loop. `notify_one` keeps its permit, so a request
    /// queued before the loop waits is never lost.
    static QUEUED: Notify = Notify::const_new();

    /// Whether the handler is registered. The protocol loop falls back to tokio's
    /// Ctrl+C when it is not, so a failed registration cannot cost a launch that has a
    /// console the stop request it had before; a launch without one receives no
    /// console event either way, which is why the failure is not diagnosed there (see
    /// [`attached_to_console`]). The handler is registered on every launch regardless
    /// — it costs nothing, and a console-subsystem build (the binary's own test
    /// harness) does have a console to serve.
    static INSTALLED: AtomicBool = AtomicBool::new(false);

    /// Whether the protocol loop has started, i.e. whether the queue has a consumer
    /// (see the module docs).
    static LOOP_RUNNING: AtomicBool = AtomicBool::new(false);

    /// Whether this process is attached to a console: without one the platform
    /// delivers no console control event, so a registration that failed has
    /// nothing to lose and is not worth a standing diagnostic.
    fn attached_to_console() -> bool {
        // SAFETY: `GetConsoleCP` only reads the calling process's console input
        // code page. The platform documents zero as the failure value, not "no
        // console"; a launch with no console is one of the ways it fails, and the
        // consequence here is only whether a refused registration is worth a
        // standing diagnostic line.
        unsafe { GetConsoleCP() != 0 }
    }

    /// Register the console control handler for this process.
    pub(super) fn install() {
        // SAFETY: `SetConsoleCtrlHandler` only records the function pointer in
        // this process's handler list; `handler` has the required
        // `extern "system" fn(u32) -> i32` signature and lives for the process's
        // lifetime.
        let installed = unsafe { SetConsoleCtrlHandler(Some(handler), TRUE) != 0 };
        INSTALLED.store(installed, Ordering::SeqCst);
        if !installed && attached_to_console() {
            // Through the boot diagnostic rather than `error!`: this runs before
            // boot opens the stores, so it reaches stderr now and the logs store
            // once tracing exists — where an `error!` would be dropped.
            crate::boot::boot_diagnostic(
                "console control handler not registered — Ctrl+C still stops the daemon \
                 through tokio's own handler; a console close and Ctrl+Break will kill \
                 the process without a shutdown."
                    .to_string(),
            );
        }
    }

    /// The console control handler: queues the request for the async protocol loop
    /// and, for a close, holds this thread (see the module docs).
    extern "system" fn handler(event: u32) -> i32 {
        let Some(request) = request_for(event) else {
            // CTRL_LOGOFF_EVENT / CTRL_SHUTDOWN_EVENT: not ours (see the section above — a
            // session end arrives as a window message, which the `session_end` module takes),
            // so the default handler's termination stands.
            return FALSE;
        };
        // A close event starts the platform's countdown now, so its grace is read here
        // rather than by the async side, and this thread holds until then.
        let close = request == StopRequest::ConsoleClose;
        let deadline = close.then(close_deadline);
        // A close brings a deadline the platform charges for either way, so it is
        // queued even before the loop exists; a request that carries none is left to
        // the platform (see the dead ends on this module).
        if !close && !LOOP_RUNNING.load(Ordering::SeqCst) {
            return FALSE;
        }
        // `unwrap_poison` rather than a panic: a panic on a control-handler thread
        // aborts the process instead of shutting it down.
        REQUESTS.lock().unwrap_poison().push_back(request);
        QUEUED.notify_one();
        if let Some(deadline) = deadline {
            // Hold this thread for the grace the platform is counting down (the module
            // docs say why returning is not an option).
            std::thread::sleep(deadline.saturating_duration_since(Instant::now()));
        }
        TRUE
    }

    /// The next queued request, awaiting one when the queue is empty.
    ///
    /// With no handler registered, Ctrl+C through tokio's own handler is the only
    /// request source left — today's behaviour, kept so a failed registration cannot
    /// cost a launch that has a console the stop request it had. (tokio subscribes
    /// through the same `SetConsoleCtrlHandler`, so this fallback reaches exactly the
    /// launches the handler above would have.) Its failure to subscribe is a real
    /// error, so the caller's signal task reports it rather than inventing a request
    /// — except on a launch with no console, which can receive no console control
    /// event at all and waits forever instead of subscribing.
    pub(super) async fn next_request() -> anyhow::Result<StopRequest> {
        LOOP_RUNNING.store(true, Ordering::SeqCst);
        if !INSTALLED.load(Ordering::SeqCst) {
            if !attached_to_console() {
                // A launch with no console receives no console event at all, so there is
                // nothing to subscribe to: subscribing (tokio's `ctrl_c`) would only turn
                // that into a standing `Signal handler failed to set up` error. The stop
                // sources such a launch really has are the window and the session-end
                // listener.
                return std::future::pending().await;
            }
            tokio::signal::ctrl_c().await?;
            return Ok(StopRequest::Interrupt);
        }
        loop {
            let queued = REQUESTS.lock().unwrap_poison().pop_front();
            if let Some(request) = queued {
                return Ok(request);
            }
            QUEUED.notified().await;
        }
    }

    /// Map a console event onto the request class it belongs to, `None` for the
    /// events this handler deliberately leaves to the default handler.
    fn request_for(event: u32) -> Option<StopRequest> {
        match event {
            CTRL_C_EVENT => Some(StopRequest::Interrupt),
            CTRL_BREAK_EVENT => Some(StopRequest::Break),
            CTRL_CLOSE_EVENT => Some(StopRequest::ConsoleClose),
            _ => None,
        }
    }

    /// The instant this process's closing console stops granting time: the grace read now
    /// (see the module docs), recorded for the exit path's budget and held by this handler
    /// thread until it.
    fn close_deadline() -> Instant {
        let deadline = Instant::now() + close_grace();
        record_stop_deadline(deadline);
        deadline
    }

    /// The grace this process's closing console grants (see the module docs). A stored
    /// `0` is taken at face value — no grace at all, i.e. today's hard kill — so the
    /// exit path cannot run past a deadline that has already passed.
    fn close_grace() -> Duration {
        let mut millis: u32 = 0;
        // SAFETY: `SPI_GETHUNGAPPTIMEOUT` writes a single `u32` through `pvparam`;
        // `uiparam` and `fWinIni` are unused for this action.
        let read =
            unsafe { SystemParametersInfoW(SPI_GETHUNGAPPTIMEOUT, 0, (&raw mut millis).cast(), 0) };
        if read == 0 {
            DEFAULT_STOP_GRACE
        } else {
            Duration::from_millis(u64::from(millis))
        }
    }
}

// ── Windows session-end listener ──────────────────────────────────────────

/// The session-end listener — the platform's stop request for a log-off, shutdown or
/// restart, delivered to a window of this daemon's own.
///
/// # Why a window of its own
///
/// Windows announces a session end through a process's top-level windows, with
/// `WM_QUERYENDSESSION` and then `WM_ENDSESSION` (see the section on the console module
/// above for why the console events do not carry this news here). Nothing in the windowing
/// stack this daemon draws with handles either message and neither offers a hook to borrow
/// the dashboard window's procedure — winit, which iced draws with, sends both to the
/// default procedure, which answers the question with "yes" and ignores the end, leaving
/// the process to be terminated silently. So the daemon creates its own window: hidden, on
/// its own thread, but a real top-level window — which is what the platform's broadcast
/// reaches. A message-only window (`HWND_MESSAGE`) is not a candidate: the broadcast never
/// reaches one.
///
/// # What it does
///
/// - Answers the preliminary question with "go ahead" (a true `BOOL`) and does nothing else
///   with it — [`session_end_label`] says why acting there would be wrong.
/// - On the final notification for a session that really is ending, forces the stop and
///   returns — [`stop_for_session_end`], which owns the platform rules this must respect.
/// - Hands every other message to `DefWindowProcW`: a window procedure that swallowed
///   messages would fail window creation (the window has to answer messages of its own to
///   be created at all) and would refuse the shutdown by answering the question with "no".
///
/// # The time available
///
/// The platform allows seconds — fewer during a critical shutdown — before it ends a process
/// that has not finished, so the stop steps can still be cut off mid-way, the exit-time
/// checkpoint that runs last most of all. What the bound cuts loses nothing durable (see
/// [`urgent_release_budget`]) and in-flight agent work is cut exactly as it is on the other
/// platforms; a process ended by the platform at its deadline is the same class as today's
/// silent hard death.
///
/// # Accepted, deliberately not fixed
///
/// - A session end before the dashboard is ready: the dashboard subscribes to shutdown only
///   once boot has succeeded and nothing else consumes the forced stop, so the platform ends
///   the process as today — unless the owner closes the window during the grace, which the
///   dashboard consumes in every state and which runs the same exit path. A boot that
///   finishes inside the platform's grace does exit properly (the token stays fired for the
///   subscription that then appears); one that fails never does.
/// - A session end while a self-update is finalizing: the update's exit path wins, spawn
///   included. The replacement it starts into a session that is ending is cut off with the
///   session like any other in-flight work, and the next start recovers. Deliberately no guard
///   against that spawn.
/// - Bringing the listener up can fail (the window class cannot be registered, the window
///   cannot be created, the thread cannot be spawned). It is reported through the boot
///   diagnostics and the daemon starts and runs with the stop requests it always had; it
///   must never be fatal.
/// - Fast user switching: it ends no session and sends no notification, so nothing here
///   runs for it and the daemon is meant to keep running.
/// - The stop trace: best-effort, not a durable record — this route forces the stop at once and
///   the exit burst lasts milliseconds, so the listener's own line is dropped with the runtime
///   exactly as the exit path's line is (see `EXIT_TRIGGER`).
#[cfg(windows)]
mod session_end {
    use super::{session_end_label, stop_for_session_end};
    use std::io::Error;
    use std::mem::{size_of, zeroed};
    use windows_sys::Win32::Foundation::{HWND, LPARAM, LRESULT, WPARAM};
    use windows_sys::Win32::System::LibraryLoader::GetModuleHandleW;
    use windows_sys::Win32::UI::WindowsAndMessaging::{
        CreateWindowExW, DefWindowProcW, DispatchMessageW, ENDSESSION_CLOSEAPP, ENDSESSION_LOGOFF,
        GetMessageW, MSG, RegisterClassExW, WM_ENDSESSION, WM_QUERYENDSESSION, WNDCLASSEXW,
    };
    use windows_sys::w;

    /// The window class this listener registers. Nothing looks it up by name: a class needs one.
    const WINDOW_CLASS: windows_sys::core::PCWSTR = w!("MahBotSessionEnd");

    /// Bring the listener up on a thread of its own, which is never joined: a window's
    /// messages are dispatched to the thread that created it, and none of this may run on
    /// the dashboard's thread. A failure is reported, never fatal (see the module docs).
    pub(super) fn install() {
        if let Err(e) = std::thread::Builder::new()
            .name("session-end-listener".to_string())
            .spawn(listen)
        {
            report("thread not spawned", &e);
        }
    }

    /// The listener thread: create the window, then pump its messages for the life of the
    /// process.
    fn listen() {
        // SAFETY: a null module name asks for this process's own module, which every
        // process has; the window class below needs an instance to own its procedure.
        let instance = unsafe { GetModuleHandleW(std::ptr::null()) };
        // SAFETY: zeroed is the documented "no icon, no cursor, no background, no menu"
        // for the fields set here, and the class is fully described for its `cbSize` by
        // the assignments.
        let mut class: WNDCLASSEXW = unsafe { zeroed() };
        // `cbSize` must report this structure's size: that is how the platform tells which
        // of the two structures it is being handed. `allow`, not `expect` — the cast cannot
        // truncate on a 32-bit target, and an unfired expectation is itself a warning.
        #[allow(clippy::cast_possible_truncation)]
        let class_size = size_of::<WNDCLASSEXW>() as u32;
        class.cbSize = class_size;
        class.lpfnWndProc = Some(window_proc);
        class.hInstance = instance;
        class.lpszClassName = WINDOW_CLASS;
        // SAFETY: the class lives for the duration of the call and describes itself
        // completely, which is all the registration reads it for; the procedure it names
        // lives for the life of the process.
        if unsafe { RegisterClassExW(&raw const class) } == 0 {
            report("window class not registered", &Error::last_os_error());
            return;
        }
        // SAFETY: the class is registered to this module, the style is "nothing", the caption
        // is null (never shown), and the parent is null — which is what makes this a hidden
        // *top-level* window, the kind the session-end broadcast reaches (a message-only parent
        // would exclude it). The window is never shown, moved or resized, and it is never
        // destroyed: it lasts as long as the process.
        let window = unsafe {
            CreateWindowExW(
                0,
                WINDOW_CLASS,
                std::ptr::null(),
                0,
                0,
                0,
                0,
                0,
                0,
                0,
                instance,
                std::ptr::null(),
            )
        };
        if window == 0 {
            report("window not created", &Error::last_os_error());
            return;
        }
        // SAFETY: zeroed is a valid initial `MSG`; every field is written by the call
        // below.
        let mut message: MSG = unsafe { zeroed() };
        loop {
            // SAFETY: the buffer is `MSG`-sized and lives on this thread's stack; the
            // filter arguments ask for every message of this thread. The retrieval is also
            // what dispatches messages *sent* to the window from another thread — which is
            // how the platform delivers the two this listener exists for.
            let retrieved = unsafe { GetMessageW(&raw mut message, 0, 0, 0) };
            if retrieved == 0 {
                // `WM_QUIT` — nothing posts one to this thread.
                break;
            }
            if retrieved < 0 {
                report("message loop failed", &Error::last_os_error());
                break;
            }
            // SAFETY: `message` was filled by the call above; the two notifications arrive
            // sent, not queued, so this serves whatever else the window is posted.
            unsafe { DispatchMessageW(&raw const message) };
        }
    }

    /// The listener window's procedure: the two session-end notifications, and the default
    /// handling for everything else (see the module docs for why that is not optional).
    extern "system" fn window_proc(
        window: HWND,
        message: u32,
        wparam: WPARAM,
        lparam: LPARAM,
    ) -> LRESULT {
        match message {
            // The preliminary question, always answered "go ahead" (a true `BOOL`): this
            // daemon never holds up the platform. Never acted on — see the module docs.
            WM_QUERYENDSESSION => LRESULT::from(true),
            WM_ENDSESSION => {
                // The flags are a 32-bit field in the parameter's low half, which the
                // platform may have sign-extended into the parameter's own width. `allow`,
                // not `expect`: the sign-loss lint fires on every target but the truncation
                // one only where the parameter is wider than the field, and an unfired
                // expectation is itself a warning.
                #[allow(clippy::cast_possible_truncation, clippy::cast_sign_loss)]
                let flags = lparam as u32;
                let ending = wparam != 0;
                let closes_app_only = flags & ENDSESSION_CLOSEAPP != 0;
                let log_off = flags & ENDSESSION_LOGOFF != 0;
                if let Some(label) = session_end_label(ending, closes_app_only, log_off) {
                    stop_for_session_end(label);
                }
                0
            }
            // SAFETY: the procedure delegates with its own arguments untouched, on the
            // thread the platform called it on.
            _ => unsafe { DefWindowProcW(window, message, wparam, lparam) },
        }
    }

    /// Report a failure to bring the listener up. Through the boot diagnostic rather than
    /// `warn!`, because this may run before boot opens the stores: it reaches stderr now
    /// and the logs store once tracing exists.
    fn report(what: &str, detail: &Error) {
        crate::boot::boot_diagnostic(format!(
            "session-end listener: {what}: {detail} — a Windows log-off, shutdown or restart \
             will end the daemon without its stop steps"
        ));
    }
}

// ── Application-machinery quit (macOS) ────────────────────────────────────

/// The wording the macOS quit route records for itself in [`EXIT_TRIGGER`]: every ordinary quit
/// ask — ⌘Q, the application menu's Quit item, the Dock's Quit — arrives as the one AppleEvent
/// `macos_quit` takes over, so they share one label.
pub(crate) const MACOS_QUIT_TRIGGER: &str = "macOS quit";

/// Install the source of the platform's quit requests: the quit channel, and on macOS the
/// interception of the platform's own asks.
///
/// Called from [`install_stop_request_sources`] before the interface starts, next to the Windows
/// installers there. The channel is created on every platform so the dashboard's subscription
/// always has a source, but only macOS sends on it — and only the interception is macOS-only.
fn install_quit_requests() {
    let (tx, rx) = tokio::sync::mpsc::unbounded_channel();
    // The receiving half first, so a quit the platform delivers before the dashboard subscribes
    // waits here instead of being lost.
    *QUIT_REQUESTS.lock().unwrap_poison() = Some(rx);
    // The channel exists on every platform so the dashboard's subscription always has a source;
    // only macOS sends, so everywhere else the sender is dropped here and the source stays empty.
    #[cfg(target_os = "macos")]
    macos_quit::install(tx);
    #[cfg(not(target_os = "macos"))]
    drop(tx);
}

/// The receiving half: the dashboard's subscription takes it once, and a quit that arrives
/// before anything has subscribed waits here instead of being lost — the platform asks whether
/// or not the interface is ready for it.
static QUIT_REQUESTS: Mutex<Option<tokio::sync::mpsc::UnboundedReceiver<()>>> = Mutex::new(None);

/// Take the quit-request receiver, for the dashboard's subscription. `None` before
/// [`install_quit_requests`] ran, or once the receiver has been taken: iced spawns a
/// `Subscription::run` recipe — identified by its function pointer — once per process, so the
/// subscriber takes it exactly once.
pub(crate) fn take_quit_requests() -> Option<tokio::sync::mpsc::UnboundedReceiver<()>> {
    QUIT_REQUESTS.lock().unwrap_poison().take()
}

/// The macOS quit interception: the platform's own ordinary quit asks, brought into the
/// product's closing path.
///
/// # Why it exists
///
/// macOS ends an application through its own machinery: its ordinary quit asks — the ones
/// [`crate::gui::Message::QuitRequested`] receives — reach AppKit's `terminate:`, which asks the
/// application delegate (`applicationShouldTerminate:`) whether it may terminate and then
/// terminates the process itself. winit, which iced draws with, installs that delegate and
/// implements no `applicationShouldTerminate:`, so AppKit's default answer stands: the asks never
/// reach the closing path, the window's state is never recorded, the exit-time checkpoint never
/// runs, the browser release is deferred to the next start, and in-flight work is cut off hard.
/// The system's own asks during a log-out, restart or shut-down travel the same route and are
/// deliberately left alone (see below).
///
/// # The hook
///
/// The method is added to winit's delegate class at run time (`class_addMethod`), which leaves
/// winit's delegate object untouched — it has to, because winit panics unless the application's
/// delegate is the one it registered itself (`ApplicationDelegate::get`), so the delegate cannot
/// be replaced. The class does not exist before the event loop is built (iced builds it inside
/// `run`, after `main`'s install call), so a thread of its own waits for it and gives up after
/// `INSTALL_WAIT` — the fallback named under "Accepted" below.
///
/// # Which ask it is
///
/// `kAEQuitReason` on the Quit AppleEvent names why the quit was sent; an ordinary quit carries no
/// reason at all. The delegate's `sender` cannot tell them apart — it is the application itself
/// for every AppleEvent-driven route, the Dock's Quit included. The two are answered differently:
///
/// - An ordinary quit is answered `NSTerminateCancel`: the platform's termination is refused and
///   the closing path takes over, exactly as a window close does (winit answers
///   `windowShouldClose:` with "no" for the same reason). Nothing stays pending on the platform's
///   side, so a **second** quit during the graceful drain reaches this hook again and does what a
///   second window close does — forces that drain, except while a self-update's own drain owns
///   the exit, which neither gesture forces.
/// - A session end — a log-out, restart or shut-down — is not the product's to handle: the ask is
///   answered `NSTerminateNow`, the answer an application that does not implement this method at
///   all gets, and the platform ends the process at once. That exit and its accepted losses are on
///   the list above.
///
/// # The bound
///
/// An ordinary quit is refused at the platform level, so no platform deadline is left to end the
/// process — the product's own has to, or a wedged closing path would leave it running. It cannot
/// live inside the machinery the closing path tears down, so `arm_exit_bound` puts it on a thread
/// that outlives the interface, armed the moment a quit is taken over: `EXIT_BOUND`, the drain cap
/// plus an allowance for the closing steps after it — the last resort that makes "never left
/// running" true.
///
/// # Accepted, deliberately not fixed
///
/// The platform-forced termination and the session end are on the `shutdown` module's list of the
/// exits that run none of the closing path; nothing here changes either. What this module accepts
/// on its own terms, the canonical list naming where each of them ends the process:
///
/// - A quit that arrives with nothing left to act on it — the dashboard's subscription gone
///   because the interface is already past its exit: it is handed back (`NSTerminateNow`), which
///   keeps a quit from being a dead key and the platform from waiting on an answer nothing can
///   give.
/// - A failure to install (the delegate class never appears, a future winit renames it, the class
///   already implements the method, the thread cannot be spawned): an ordinary quit keeps today's
///   behaviour — the platform ends the process — reported, never fatal.
/// - The exit bound firing: a hard end with no closing steps, the same class as a force quit. A
///   quit arriving while a self-update is finalizing is armed like any other, so the bound can cut
///   that hand-off where the exit path would otherwise wait for it: a tail outrunning the
///   allowance looks no different from a wedged one.
#[cfg(target_os = "macos")]
mod macos_quit {
    use objc2::ffi::class_addMethod;
    use objc2::runtime::{AnyClass, AnyObject, Imp, Sel};
    use objc2::sel;
    use objc2_app_kit::NSApplicationTerminateReply;
    use objc2_core_services::{
        kAEQuitAll, kAEQuitReason, kAEReallyLogOut, kAERestart, kAEShutDown,
    };
    use objc2_foundation::NSAppleEventManager;
    use std::ffi::CStr;
    use std::sync::OnceLock;
    use std::sync::atomic::{AtomicBool, Ordering};
    use std::time::{Duration, Instant};
    use tracing::{info, warn};

    /// The sending half of the quit channel; this module is its only producer, and what a quit
    /// carries says nothing — every one of them arrives by the same route.
    static QUIT_TX: OnceLock<tokio::sync::mpsc::UnboundedSender<()>> = OnceLock::new();

    /// Hand a quit to the dashboard. `false` when nothing will consume it — the subscription is
    /// gone — so the caller has to fall back on the platform ending the process.
    fn queue_quit() -> bool {
        QUIT_TX.get().is_some_and(|tx| tx.send(()).is_ok())
    }

    /// The delegate class winit registers for its event loop, and the only one AppKit asks about
    /// termination; a rename in a future winit costs the interception and nothing else.
    const DELEGATE_CLASS: &CStr = c"WinitApplicationDelegate";

    /// `applicationShouldTerminate:`'s type encoding: the reply (`NSUInteger`), the receiver,
    /// the selector, the sender.
    const SHOULD_TERMINATE_TYPES: &CStr = c"Q@:@";

    /// How long the installer waits for winit's delegate class to be registered.
    const INSTALL_WAIT: Duration = Duration::from_secs(60);
    /// How often it looks while waiting.
    const INSTALL_POLL: Duration = Duration::from_millis(5);

    /// The last-resort bound on a quit's exit: the drain cap plus an allowance for the closing
    /// steps that follow it (see the module's bound note).
    const EXIT_BOUND: Duration = Duration::from_secs(crate::jobs::DRAIN_CAP_SECS + 300);

    /// Bring the interception up on a thread of its own, never joined: a failure is reported and
    /// never fatal — an ordinary quit keeps today's behaviour.
    pub(super) fn install(tx: tokio::sync::mpsc::UnboundedSender<()>) {
        let _ = QUIT_TX.set(tx);
        if let Err(e) = std::thread::Builder::new()
            .name("quit-interception".to_string())
            .spawn(install_when_registered)
        {
            report(&format!("thread not spawned: {e}"));
        }
    }

    /// Wait for winit's delegate class, then add the delegate method to it.
    fn install_when_registered() {
        let give_up = Instant::now() + INSTALL_WAIT;
        loop {
            if let Some(class) = AnyClass::get(DELEGATE_CLASS) {
                // SAFETY: the IMP's signature is the method's own (`self`, `_cmd`, `sender`,
                // returning the reply type, an `NSUInteger`), and the encoding string spells it
                // out; the class outlives the process, so the method stays valid for every
                // delegate instance.
                let added = unsafe {
                    class_addMethod(
                        std::ptr::from_ref(class).cast_mut(),
                        sel!(applicationShouldTerminate:),
                        std::mem::transmute::<
                            unsafe extern "C-unwind" fn(
                                *mut AnyObject,
                                Sel,
                                *mut AnyObject,
                            )
                                -> NSApplicationTerminateReply,
                            Imp,
                        >(should_terminate),
                        SHOULD_TERMINATE_TYPES.as_ptr(),
                    )
                };
                if !added.as_bool() {
                    report(
                        "winit's delegate already implements applicationShouldTerminate: — \
                         the platform's own termination stands",
                    );
                }
                // Success is silent, as the Windows installers are: the class appears inside
                // `run`, before boot opens the stores, so nothing would carry a line here — and a
                // takeover is not reported to the owner either (`EXIT_TRIGGER` says what becomes
                // of the exit path's line). What it produces is the closing path itself.
                return;
            }
            if Instant::now() >= give_up {
                report(
                    "winit's delegate class was not registered in time — the platform's own \
                     termination stands",
                );
                return;
            }
            std::thread::sleep(INSTALL_POLL);
        }
    }

    /// The delegate method AppKit asks about termination: an ordinary quit is handed to the
    /// product's closing path, a session end is left to the platform (see the module docs).
    ///
    /// # Safety
    ///
    /// AppKit calls this as `-[WinitApplicationDelegate applicationShouldTerminate:]` on the
    /// main thread: `_this` is the delegate, `_sender` the application object (unused — the ask
    /// is told apart by the AppleEvent the system is handling, not by the sender).
    unsafe extern "C-unwind" fn should_terminate(
        _this: *mut AnyObject,
        _cmd: Sel,
        _sender: *mut AnyObject,
    ) -> NSApplicationTerminateReply {
        if is_session_end() {
            return NSApplicationTerminateReply::TerminateNow;
        }
        if !queue_quit() {
            warn!("the dashboard is past its exit — the platform ends the process");
            return NSApplicationTerminateReply::TerminateNow;
        }
        info!("macOS quit — the platform asked the product to quit");
        arm_exit_bound();
        NSApplicationTerminateReply::TerminateCancel
    }

    /// Whether the platform is ending the session rather than asking the product to quit: its
    /// Quit AppleEvent names a session end with `kAEQuitReason`, and an ordinary quit carries no
    /// reason at all.
    ///
    /// A reason this build does not know — and a session end whose reason the system leaves out —
    /// is read as an ordinary quit. The asymmetry is deliberate: a session end read as an ordinary
    /// one only starts a closing path the platform's own deadline then cuts, while an ordinary
    /// quit read as a session end would leave a quit the user is watching to the platform — the
    /// behaviour this module exists to replace. It does mean a misread session end cancels the
    /// platform's termination, and that the classification rests on Apple's documented reason
    /// codes alone.
    fn is_session_end() -> bool {
        /// The reason codes read as a session end: the documented values of `kAEQuitReason`
        /// (`AERegistry.h`) — shutdown, restart, log-out, and the system's quit-all, the last of
        /// which is read as a session end here rather than as an ordinary quit.
        const SESSION_ENDS: [u32; 4] = [kAEShutDown, kAERestart, kAEReallyLogOut, kAEQuitAll];
        NSAppleEventManager::sharedAppleEventManager()
            .currentAppleEvent()
            .and_then(|event| event.attributeDescriptorForKeyword(kAEQuitReason))
            .is_some_and(|reason| SESSION_ENDS.contains(&reason.typeCodeValue()))
    }

    /// Arm [`EXIT_BOUND`] on the first quit the product takes over.
    fn arm_exit_bound() {
        static ARMED: AtomicBool = AtomicBool::new(false);
        if ARMED.swap(true, Ordering::SeqCst) {
            return;
        }
        if let Err(e) = std::thread::Builder::new()
            .name("quit-exit-bound".to_string())
            .spawn(|| {
                std::thread::sleep(EXIT_BOUND);
                report(&format!(
                    "the closing path did not finish within {EXIT_BOUND:?} of a quit — ending \
                     the process"
                ));
                // SAFETY: `_exit` ends the process at once, from any thread. Nothing is
                // waited for: this is the last resort, the same class of end as a force quit.
                unsafe { libc::_exit(1) }
            })
        {
            report(&format!("exit bound not armed: {e}"));
        }
    }

    /// Report a failure of the interception, or the exit bound firing, through the boot diagnostic,
    /// as the session-end listener's `report` does: its stderr leg is the one that carries it (an
    /// install failure can run before the logs store exists, and the exit-bound report is followed
    /// at once by the `_exit`).
    fn report(what: &str) {
        crate::boot::boot_diagnostic(format!("quit interception: {what}"));
    }
}

// ── Fatal signal handlers ─────────────────────────────────────────────────

/// Install bare-metal signal handlers for fatal signals (SIGBUS, SIGABRT).
///
/// These are separate from the tokio-based [`wait_for_shutdown_signal`] — that
/// handles graceful shutdown (SIGINT/SIGTERM). The handlers here catch
/// *unexpected* fatal signals that would otherwise kill the process silently
/// with no diagnostic output.
///
/// On first call, installs handlers via `libc::signal`. Safe to call
/// multiple times — only the first call installs handlers.
///
/// The handlers write a one-line diagnostic message to stderr using the
/// async-signal-safe `write(2)` syscall, then `_exit(1)`. No heap allocation,
/// no locks, no stdio — safe to call from within a signal handler.
#[cfg(unix)]
pub fn install_fatal_signal_handlers() {
    use std::sync::Once;
    static INSTALLED: Once = Once::new();
    INSTALLED.call_once(|| {
        // SAFETY: `libc::signal` is async-signal-safe. The handler functions
        // use STATIC string constants (no heap allocation) and call only
        // `libc::write` (raw syscall) and `libc::_exit` — both
        // async-signal-safe per POSIX.
        unsafe {
            libc::signal(
                libc::SIGBUS,
                fatal_signal_handler as *const () as libc::sighandler_t,
            );
            libc::signal(
                libc::SIGABRT,
                fatal_signal_handler as *const () as libc::sighandler_t,
            );
        }
    });
}

#[cfg(unix)]
const SIGBUS_MSG: &str = "mahbot: caught SIGBUS (bus error), terminating\n";
#[cfg(unix)]
const SIGABRT_MSG: &str = "mahbot: caught SIGABRT (abort), terminating\n";

#[cfg(unix)]
extern "C" fn fatal_signal_handler(sig: i32) {
    let msg = match sig {
        libc::SIGBUS => SIGBUS_MSG,
        libc::SIGABRT => SIGABRT_MSG,
        _ => "mahbot: caught unknown fatal signal, terminating\n",
    };
    // SAFETY: write(2) and _exit(2) are async-signal-safe per POSIX.
    unsafe {
        let _ = libc::write(
            libc::STDERR_FILENO,
            msg.as_ptr().cast::<libc::c_void>(),
            msg.len(),
        );
        libc::_exit(1);
    }
}

#[cfg(not(unix))]
pub fn install_fatal_signal_handlers() {
    // No-op on non-Unix platforms.
}

// ── Panic hook ────────────────────────────────────────────────────────────

thread_local! {
    /// Set while a contained pass ([`contain_panics`]) runs on this thread.
    static CONTAINED_PANIC: std::cell::Cell<bool> = const { std::cell::Cell::new(false) };
}

/// Run `f`, reporting a panic it raises nowhere: the boot hook skips its report
/// for a panic contained here, and the panic is caught instead ([`Err`] carries
/// its payload).
///
/// For a caller that contains *many* panics by design — the per-page PDF text
/// pass, where one malformed page panics — the default hook would otherwise
/// turn one failed document into one report per failed page on stderr.
pub(crate) fn contain_panics<T>(f: impl FnOnce() -> T) -> std::thread::Result<T> {
    let previous = CONTAINED_PANIC.with(|flag| flag.replace(true));
    let result = std::panic::catch_unwind(std::panic::AssertUnwindSafe(f));
    CONTAINED_PANIC.with(|flag| flag.set(previous));
    result
}

/// Install a global panic hook that prints a timestamped marker line before
/// the default panic report, so panics captured in update.log (the replacement
/// daemon's stderr) are time-attributable. The default hook (message +
/// backtrace) still runs — except for a panic contained by [`contain_panics`],
/// which is reported through the caller's own result instead.
pub fn install_panic_hook() {
    let default_hook = std::panic::take_hook();
    std::panic::set_hook(Box::new(move |info| {
        if CONTAINED_PANIC.with(std::cell::Cell::get) {
            return;
        }
        crate::boot::timestamped_stderr(&info.to_string());
        default_hook(info);
    }));
}

// ── Tests ─────────────────────────────────────────────────────────────────

#[cfg(test)]
mod tests {
    use super::*;

    #[test]
    fn only_the_first_interrupt_drains() {
        for (request, draining, expected) in [
            (StopRequest::Interrupt, false, StopAction::Drain),
            (StopRequest::Interrupt, true, StopAction::ForceCancel),
            (StopRequest::Break, false, StopAction::ForceCancel),
            (StopRequest::ConsoleClose, false, StopAction::ForceCancel),
        ] {
            assert_eq!(
                stop_action(request, draining),
                expected,
                "{request:?}, draining={draining}"
            );
        }
    }

    #[test]
    fn only_the_final_session_end_notification_stops_the_daemon() {
        // The final notification for a session that really is ending — a log-off, or a
        // shutdown/restart (a critical shutdown carries neither flag).
        assert_eq!(
            session_end_label(true, false, true),
            Some("Windows log-off")
        );
        assert_eq!(
            session_end_label(true, false, false),
            Some("Windows shutdown/restart")
        );
        // The question's "not ending" answer: a shutdown under way was cancelled, so a
        // stop here would leave the daemon half-stopped for a session that continues.
        assert_eq!(session_end_label(false, false, false), None);
        // `ENDSESSION_CLOSEAPP`: the platform asking this application alone to close so
        // that an update can proceed, which ends no session.
        assert_eq!(session_end_label(true, true, false), None);
    }

    #[test]
    fn the_release_budget_samples_half_the_remaining_grace() {
        let now = Instant::now();
        let deadline = now + Duration::from_secs(5);
        assert_eq!(stage_budget(deadline, now), Duration::from_millis(2500));
        // Sampled from what is left when the stage starts, not from the original
        // grace: the later that is, the less the stage gets.
        assert_eq!(
            stage_budget(deadline, now + Duration::from_secs(4)),
            Duration::from_millis(500)
        );
        // A grace already spent leaves nothing to spend, so the stage is cut at once
        // and the exit path goes straight to the checkpoint.
        assert_eq!(stage_budget(now, now), Duration::ZERO);
    }

    #[test]
    fn a_contained_panic_is_not_reported() {
        /// Only the marker this test raises is counted: the hook stays
        /// transparent for every other panic, so a test running in parallel keeps
        /// its ordinary report.
        const MARKER: &str = "mahbot: contained-panic test marker";
        static REPORTED: std::sync::atomic::AtomicUsize = std::sync::atomic::AtomicUsize::new(0);

        type PanicHook = Box<dyn Fn(&std::panic::PanicHookInfo<'_>) + Send + Sync>;

        /// Puts the process hook back even when an assertion fails: a counting
        /// hook left installed would drop every later panic report.
        struct RestoreHook(Option<std::sync::Arc<PanicHook>>);

        impl Drop for RestoreHook {
            fn drop(&mut self) {
                if let Some(hook) = self.0.take() {
                    std::panic::set_hook(Box::new(move |info| hook.as_ref()(info)));
                }
            }
        }

        let previous = std::sync::Arc::new(std::panic::take_hook());
        let _restore = RestoreHook(Some(std::sync::Arc::clone(&previous)));
        std::panic::set_hook(Box::new(move |info| {
            // A formatted panic carries a `String` payload; a literal one a `&str`.
            let payload = info.payload();
            let text = payload
                .downcast_ref::<&str>()
                .copied()
                .or_else(|| payload.downcast_ref::<String>().map(String::as_str));
            if text == Some(MARKER) {
                REPORTED.fetch_add(1, std::sync::atomic::Ordering::Relaxed);
            } else {
                previous.as_ref()(info);
            }
        }));
        install_panic_hook();
        assert!(
            contain_panics(|| panic!("{MARKER}")).is_err(),
            "the panic is still returned to the caller"
        );
        assert_eq!(
            REPORTED.load(std::sync::atomic::Ordering::Relaxed),
            0,
            "a contained panic must not reach the hook: one document conversion \
             cannot report once per failing page"
        );
        let _ = std::panic::catch_unwind(|| panic!("{MARKER}"));
        assert_eq!(
            REPORTED.load(std::sync::atomic::Ordering::Relaxed),
            1,
            "every other panic is reported as before"
        );
    }
}