Skip to main content

rs_teststand_websocket/client/
backoff.rs

1//! How hard to try when a host is not answering yet.
2
3use std::time::Duration;
4
5/// How hard to try when a host is not answering yet.
6///
7/// A host restarts, and every panel connected to it tries to come back. Without
8/// spreading those attempts out they arrive together and knock the host over
9/// again, so the delay grows and carries a random fraction. The growth stops a
10/// dead host being hammered; the randomness stops a live one being hit by a
11/// crowd.
12#[derive(Debug, Clone, Copy, PartialEq, Eq)]
13pub struct Backoff {
14    /// Wait before the second attempt. Doubles from there.
15    pub first: Duration,
16    /// Ceiling for the wait, before jitter.
17    pub longest: Duration,
18    /// How many attempts in total, including the first.
19    pub attempts: u32,
20}
21
22impl Default for Backoff {
23    /// A second, doubling, capped at thirty, giving up after ten attempts.
24    ///
25    /// Ten attempts spans roughly four minutes, which covers a host restart
26    /// without leaving a caller waiting on one that is never coming back.
27    fn default() -> Self {
28        Self {
29            first: Duration::from_secs(1),
30            longest: Duration::from_secs(30),
31            attempts: 10,
32        }
33    }
34}
35
36impl Backoff {
37    /// How long to wait before the very first attempt.
38    ///
39    /// RFC 6455 section 7.2.3 asks for a random delay here specifically, and
40    /// suggests somewhere between zero and five seconds. This uses that range
41    /// rather than [`first`](Self::first), because the point is to scatter a
42    /// crowd of clients that all woke at once, not to pace one client's
43    /// retries.
44    #[must_use]
45    pub fn first_delay() -> Duration {
46        /// The upper end the specification suggests.
47        const SPREAD: Duration = Duration::from_secs(5);
48
49        let nanos = std::time::SystemTime::now()
50            .duration_since(std::time::UNIX_EPOCH)
51            .map_or(0, |since| since.subsec_nanos());
52        let ceiling = u64::try_from(SPREAD.as_nanos()).unwrap_or(u64::MAX);
53        Duration::from_nanos(u64::from(nanos) % ceiling.max(1))
54    }
55
56    /// How long to wait before attempt `attempt`, counting the first as 0.
57    ///
58    /// Doubling, capped, then up to a quarter added on top. The jitter comes
59    /// from the clock rather than a random number generator, which keeps a
60    /// dependency out for a value that only has to differ between processes.
61    #[must_use]
62    pub fn delay(self, attempt: u32) -> Duration {
63        let doubled = self
64            .first
65            .saturating_mul(1_u32.checked_shl(attempt.min(16)).unwrap_or(u32::MAX));
66        let capped = doubled.min(self.longest);
67
68        let spread = capped / 4;
69        if spread.is_zero() {
70            return capped;
71        }
72        let nanos = std::time::SystemTime::now()
73            .duration_since(std::time::UNIX_EPOCH)
74            .map_or(0, |since| since.subsec_nanos());
75        // `spread` is at most a quarter of `longest`, so it fits a u64 for any
76        // sane ceiling; the fallback keeps that true if one is ever set absurd.
77        let ceiling = u64::try_from(spread.as_nanos()).unwrap_or(u64::MAX);
78        capped + Duration::from_nanos(u64::from(nanos) % ceiling.max(1))
79    }
80}