rs_teststand_websocket/client/backoff.rs
1//! How hard to try when a host is not answering yet.
2
3use std::time::Duration;
4
5/// How hard to try when a host is not answering yet.
6///
7/// A host restarts, and every panel connected to it tries to come back. Without
8/// spreading those attempts out they arrive together and knock the host over
9/// again, so the delay grows and carries a random fraction. The growth stops a
10/// dead host being hammered; the randomness stops a live one being hit by a
11/// crowd.
12#[derive(Debug, Clone, Copy, PartialEq, Eq)]
13pub struct Backoff {
14 /// Wait before the second attempt. Doubles from there.
15 pub first: Duration,
16 /// Ceiling for the wait, before jitter.
17 pub longest: Duration,
18 /// How many attempts in total, including the first.
19 pub attempts: u32,
20}
21
22impl Default for Backoff {
23 /// A second, doubling, capped at thirty, giving up after ten attempts.
24 ///
25 /// Ten attempts spans roughly four minutes, which covers a host restart
26 /// without leaving a caller waiting on one that is never coming back.
27 fn default() -> Self {
28 Self {
29 first: Duration::from_secs(1),
30 longest: Duration::from_secs(30),
31 attempts: 10,
32 }
33 }
34}
35
36impl Backoff {
37 /// How long to wait before the very first attempt.
38 ///
39 /// RFC 6455 section 7.2.3 asks for a random delay here specifically, and
40 /// suggests somewhere between zero and five seconds. This uses that range
41 /// rather than [`first`](Self::first), because the point is to scatter a
42 /// crowd of clients that all woke at once, not to pace one client's
43 /// retries.
44 #[must_use]
45 pub fn first_delay() -> Duration {
46 /// The upper end the specification suggests.
47 const SPREAD: Duration = Duration::from_secs(5);
48
49 let nanos = std::time::SystemTime::now()
50 .duration_since(std::time::UNIX_EPOCH)
51 .map_or(0, |since| since.subsec_nanos());
52 let ceiling = u64::try_from(SPREAD.as_nanos()).unwrap_or(u64::MAX);
53 Duration::from_nanos(u64::from(nanos) % ceiling.max(1))
54 }
55
56 /// How long to wait before attempt `attempt`, counting the first as 0.
57 ///
58 /// Doubling, capped, then up to a quarter added on top. The jitter comes
59 /// from the clock rather than a random number generator, which keeps a
60 /// dependency out for a value that only has to differ between processes.
61 #[must_use]
62 pub fn delay(self, attempt: u32) -> Duration {
63 let doubled = self
64 .first
65 .saturating_mul(1_u32.checked_shl(attempt.min(16)).unwrap_or(u32::MAX));
66 let capped = doubled.min(self.longest);
67
68 let spread = capped / 4;
69 if spread.is_zero() {
70 return capped;
71 }
72 let nanos = std::time::SystemTime::now()
73 .duration_since(std::time::UNIX_EPOCH)
74 .map_or(0, |since| since.subsec_nanos());
75 // `spread` is at most a quarter of `longest`, so it fits a u64 for any
76 // sane ceiling; the fallback keeps that true if one is ever set absurd.
77 let ceiling = u64::try_from(spread.as_nanos()).unwrap_or(u64::MAX);
78 capped + Duration::from_nanos(u64::from(nanos) % ceiling.max(1))
79 }
80}