pitboard-core 0.5.2

The engine behind pitboard: parking and restoring Claude Code and Codex logins. Serves pitboard's own front ends.
Documentation
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
397
398
399
400
401
402
403
404
405
406
407
408
409
410
411
412
413
414
415
416
417
418
419
420
421
422
423
424
425
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
457
458
459
460
461
462
463
464
465
466
467
468
469
470
471
472
473
474
475
476
477
478
479
480
481
482
483
484
485
486
487
488
489
490
491
//! The crash matrix: every durable step of every change, killed, recovered, and checked.
//!
//! What this project promises is that an interrupted change leaves a machine a later run
//! can make sense of, and nobody loses a login. Until now that was tested by planting a
//! journal file and a vault item describing a crash that never happened, which tests
//! `reconcile` and not the sequence that produced what `reconcile` is handed. The two
//! orphan windows this roadmap found, in enrolling and in renewing, were exactly that
//! shape and survived every one of those tests.
//!
//! So each case here kills a real change at a real point with [`crate::fault`], runs
//! recovery, and asserts invariants rather than a particular outcome. There is more than
//! one right answer to being killed halfway; there is only one set of things that must be
//! true afterwards.
//!
//! Every case then runs recovery a second time, because recovery that is not idempotent is
//! a machine that cannot be fixed by running the command again, which is the only
//! instruction a person is ever given.

use super::harness::{
    NOW, POINTS, codex_machine, document, hold, machine, owner, recover, signed_in,
};
use super::*;
use crate::api::scripted::{ScriptedApi, Trouble};
use crate::fault;
use crate::store::memory::Fault;
use serde_json::json;
use std::sync::Arc;

/// Kill a switch at every durable step, recover, and check. Then recover again, because a
/// recovery that only works once leaves a machine nobody can fix.
///
/// For every tool, through the same invariants: what must be true after a crash is a fact
/// about parking a login, and a tool whose park may never be a copy is exactly the one
/// where getting it wrong costs most.
#[test]
fn a_switch_killed_at_any_step_recovers_to_something_whole() {
    type Make = fn(&str) -> super::harness::Machine;
    let machines: [(&str, Make); 2] = [("claude", machine), ("codex", codex_machine)];
    for (tool, make) in machines {
        for point in POINTS {
            let m = make(&point.replace('.', "-"));

            let settled = settle(&m.ctx, None).expect("nothing to recover yet").0;
            let died = fault::killing(point, || switch(settled, &m.key("there")));
            assert_eq!(
                died.unwrap_err(),
                point,
                "{tool}: the switch must reach {point} on this machine, or the case proves \
                 nothing"
            );

            let at = format!("{tool}, {point}");
            recover(&m).unwrap_or_else(|e| panic!("{at}: recovery refused: {e}"));
            hold(&m, &at);

            recover(&m).unwrap_or_else(|e| panic!("{at}: the second recovery refused: {e}"));
            hold(&m, &format!("{at}, recovered twice"));
        }
    }
}

/// The same kills, with Anthropic unreachable afterwards. Recovery decides what an
/// interrupted switch did from which side the live credential's refresh token came from,
/// and both sides were fingerprinted when the record was written, so the common case is a
/// comparison and not a round trip. That is what makes a switch recoverable on a plane.
#[test]
fn a_switch_killed_with_nobody_to_ask_is_recovered_from_the_record() {
    for point in POINTS {
        let m = machine(&format!("offline-{}", point.replace('.', "-")));

        let settled = settle(&m.ctx, None).expect("nothing to recover yet").0;
        let died = fault::killing(point, || {
            switch(
                settled,
                &crate::state::Key::new(crate::provider::ProviderId::Claude, "there"),
            )
        });
        assert_eq!(died.unwrap_err(), point);

        // Anthropic goes away. A scripted api answers Unauthorized for tokens it does not
        // know, so telling it to be unreachable for both is how it goes offline.
        let offline = ScriptedApi::new();
        let ctx = m.ctx.clone().with_scripted_api(Arc::clone(&offline));
        offline.token_trouble("access-here-refresh", Trouble::Offline);
        offline.token_trouble("access-there-refresh", Trouble::Offline);

        // Which side the live credential came from is written in the record as two
        // fingerprints, so this settles without asking anyone. Dropped at once: a settled
        // machine holds pitboard's exclusivity lock until it is.
        let decided = settle(&ctx, None).is_ok();
        assert!(
            decided,
            "{point}: recovery should read the live login's fingerprint rather than \
             needing Anthropic"
        );
        assert_eq!(
            offline.calls(),
            0,
            "{point}: and it should not have asked at all"
        );
        hold(&m, &format!("{point}, recovered with no network"));

        // Still true when Anthropic comes back, and still true run twice.
        recover(&m).unwrap_or_else(|e| panic!("{point}: recovery refused once back: {e}"));
        hold(&m, &format!("{point}, then online"));
    }
}

/// The fingerprints narrow the network dependency; they do not remove it. Claude Code
/// rotating the token inside the seconds of an interrupted switch leaves a live credential
/// matching neither side, which is exactly when there is nothing to read off and Anthropic
/// has to be asked. With nobody to ask, the only right answer is to change nothing.
#[test]
fn a_switch_whose_token_rotated_while_it_was_interrupted_still_needs_anthropic() {
    let m = machine("rotated-offline");
    let settled = settle(&m.ctx, None).expect("nothing to recover yet").0;
    let died = fault::killing("switch.park_recorded", || {
        switch(
            settled,
            &crate::state::Key::new(crate::provider::ProviderId::Claude, "there"),
        )
    });
    assert_eq!(died.unwrap_err(), "switch.park_recorded");

    // Claude Code refreshes the login it still believes is signed in, so the slot now holds
    // a token neither side of the record fingerprints to.
    m.mem.live().plant(
        &m.service,
        &json!({"claudeAiOauth": {
            "refreshToken": "rotated-since",
            "accessToken": "access-here-refresh",
            "expiresAt": (NOW + 3600) * 1000,
            "refreshTokenExpiresAt": (NOW + 30 * 86_400) * 1000,
        }})
        .to_string(),
    );

    let before = m.mem.vault().services();
    let live_before = m.mem.live().peek(&m.service);
    let offline = ScriptedApi::new();
    let ctx = m.ctx.clone().with_scripted_api(Arc::clone(&offline));
    offline.token_trouble("access-here-refresh", Trouble::Offline);

    assert!(
        settle(&ctx, None).is_err(),
        "with nothing to read off and nobody to ask, this must not be guessed at"
    );
    assert_eq!(m.mem.vault().services(), before, "nothing may be deleted");
    assert_eq!(m.mem.live().peek(&m.service), live_before);
    assert!(
        journal::pending(&ctx),
        "and the record is kept for a later run"
    );
}

/// Enrolling by signing in writes a login into the vault before anything names it. Killed
/// in that window, the item is an orphan: never renewed, never deleted, and on macOS not
/// listable by any tool the user has.
#[test]
fn enrolling_killed_between_the_write_and_the_record_leaves_nothing_unnamed() {
    for point in ["enroll.park_stored", "enroll.park_recorded"] {
        let m = machine(&point.replace('.', "-"));
        m.api.owned_by("access-third-refresh", owner("third"));

        let settled = settle(&m.ctx, None).expect("nothing to recover").0;
        let login = enroll::planted(&m.ctx, ProviderId::Claude, document("third-refresh"))
            .expect("a sign-in");
        let died = fault::killing(point, || {
            enroll(
                settled,
                &crate::state::Key::new(crate::provider::ProviderId::Claude, "third"),
                Some(login),
            )
        });
        assert_eq!(died.unwrap_err(), point);

        recover(&m).unwrap_or_else(|e| panic!("{point}: recovery refused: {e}"));
        hold(&m, point);
    }
}

/// Signing in again to the account in use writes its new login in place of the old one,
/// then records it, and parks nothing. Killed after either, the account is signed in with
/// the login that was written and every other account still has its own.
#[test]
fn signing_in_again_killed_at_any_step_recovers_to_something_whole() {
    type Make = fn(&str) -> super::harness::Machine;
    let machines: [(&str, Make); 2] = [("claude", machine), ("codex", codex_machine)];
    for (tool, make) in machines {
        for point in ["enroll.installed", "enroll.recorded"] {
            let m = make(&format!("again-{}", point.replace('.', "-")));
            let login = signed_in(&m, "here", "here-refresh-2");

            let settled = settle(&m.ctx, None).expect("nothing to recover yet").0;
            let died = fault::killing(point, || enroll(settled, &m.key("here"), Some(login)));
            assert_eq!(
                died.unwrap_err(),
                point,
                "{tool}: the sign-in must reach {point} on this machine, or the case proves \
                 nothing"
            );

            let at = format!("{tool}, {point}");
            recover(&m).unwrap_or_else(|e| panic!("{at}: recovery refused: {e}"));
            hold(&m, &at);
            let live = m.live().expect("a login in use");
            assert_eq!(
                crate::provider::of(m.which).fingerprint(&live),
                store::fingerprint("here-refresh-2"),
                "{at}: the new login is the one in use"
            );

            recover(&m).unwrap_or_else(|e| panic!("{at}: the second recovery refused: {e}"));
            hold(&m, &format!("{at}, recovered twice"));
        }
    }
}

/// The one Claude Code write this lock cannot exclude. Measured in 2.1.278: a `/logout`
/// that has given up waiting deletes the credential with nothing held. If it lands just
/// after the install, the incoming login is gone, and pitboard used to print "Switched to
/// work" and exit 0 over an account that was signed out.
#[test]
fn a_switch_whose_login_was_removed_again_does_not_report_a_switch() {
    let m = machine("did-not-hold");
    let parked_before = m.mem.vault().services();

    let settled = settle(&m.ctx, None).expect("nothing to recover yet").0;
    m.mem.live().fault(&m.service, Fault::DeletedAfterWrite);
    let failed = switch(
        settled,
        &crate::state::Key::new(crate::provider::ProviderId::Claude, "there"),
    )
    .expect_err("the login did not stay");
    assert!(
        matches!(failed, Error::SwitchDidNotHold { .. }),
        "got {failed:?}"
    );

    let state = state::load(&m.ctx).expect("state");
    // Both logins are still here: the one that was parked on the way out, and the one that
    // was being installed. Neither may be thrown away over a write that did not hold.
    assert!(
        state
            .get(&crate::state::Key::new(
                crate::provider::ProviderId::Claude,
                "here"
            ))
            .expect("account")
            .parked
            .is_some(),
        "the outgoing login was parked and stays parked"
    );
    assert!(
        state
            .get(&crate::state::Key::new(
                crate::provider::ProviderId::Claude,
                "there"
            ))
            .expect("account")
            .parked
            .is_some(),
        "the incoming login is not discarded over a switch that did not stand"
    );
    assert!(m.mem.vault().services().len() > parked_before.len());
    assert!(
        !journal::pending(&m.ctx),
        "and there is nothing half-done to finish"
    );
    hold(&m, "did not hold");
}

/// A switch that could not find out what it did keeps everything, including its record of
/// intent, so a later run with a store that answers decides. Nothing here is a crash: this
/// is the ordinary shape of a machine whose keychain is locked.
#[test]
fn a_switch_that_cannot_read_the_store_back_keeps_every_copy_and_its_record() {
    let m = machine("unverified");
    let before = m.mem.live().peek(&m.service).expect("a live login");
    let parked_before = m.mem.vault().services();

    // The keychain locks partway through, which is what a screen lock does. The reads the
    // switch makes before it writes still answer; the write and everything after it do not.
    let settled = settle(&m.ctx, None).expect("nothing to recover yet").0;
    m.mem.live().fault(&m.service, Fault::LocksOnWrite);
    let failed = switch(
        settled,
        &crate::state::Key::new(crate::provider::ProviderId::Claude, "there"),
    )
    .expect_err("a keychain that locked partway");
    assert!(
        matches!(failed, Error::SwitchUnverified { .. }),
        "got {failed:?}"
    );

    assert!(
        journal::pending(&m.ctx),
        "the record of intent stays, because nobody can say what happened"
    );
    assert_eq!(
        m.mem.live().peek(&m.service).as_deref(),
        Some(before.as_str()),
        "and nothing was written"
    );
    let state = state::load(&m.ctx).expect("state");
    assert!(
        state.discarded.is_empty(),
        "nothing may be listed for deletion on a guess"
    );
    assert!(
        m.mem.vault().services().len() >= parked_before.len(),
        "and no parked login was thrown away"
    );

    // Once the store answers again, the same machine recovers and holds together.
    m.mem.live().heal_all();
    recover(&m).expect("recovery once the keychain is unlocked");
    hold(&m, "unverified, then unlocked");
}

/// The other window the roadmap named. A renewal reserves a name, writes the fresh login
/// into it, and records it; killed between the write and the record, the copy is an orphan,
/// and the renewal runs inside every plain `pitboard`.
#[test]
fn renewing_killed_between_the_write_and_the_record_leaves_nothing_unnamed() {
    let m = machine("renew-park-stored");
    // The parked login is due: its access token has lapsed.
    let mut state = state::load(&m.ctx).expect("state");
    let park = state
        .get(&crate::state::Key::new(
            crate::provider::ProviderId::Claude,
            "there",
        ))
        .expect("account")
        .parked
        .clone()
        .expect("parked");
    m.mem.vault().plant(
        &park.service,
        &json!({
            "accessToken": "access-there-refresh",
            "refreshToken": "there-refresh",
            "expiresAt": (NOW - 60) * 1000,
            "refreshTokenExpiresAt": (NOW + 30 * 86_400) * 1000
        })
        .to_string(),
    );
    state.park(
        &crate::state::Key::new(crate::provider::ProviderId::Claude, "there"),
        park::describe(
            crate::provider::ProviderId::Claude,
            &park.service,
            NOW,
            &json!({
                "refreshToken": "there-refresh",
                "expiresAt": (NOW - 60) * 1000,
                "refreshTokenExpiresAt": (NOW + 30 * 86_400) * 1000
            }),
        ),
    );
    state::save(&m.ctx, &state).expect("saved");
    m.api.renews(
        "there-refresh",
        crate::api::Renewed {
            access_token: "access-fresh".into(),
            refresh_token: Some("fresh".into()),
            expires_in: 3600,
            refresh_token_expires_in: Some(30 * 86_400),
            scopes: None,
            at: None,
        },
    );

    let died = fault::killing("renew.park_stored", || renew::renew_parked(&m.ctx));
    assert_eq!(died.unwrap_err(), "renew.park_stored");

    recover(&m).expect("recovery");
    hold(&m, "renew.park_stored");

    // Which chain survived is the whole point. Anthropic spent `there-refresh` when it
    // answered; the copy written just before the kill is the only one that still works.
    let kept = state::load(&m.ctx)
        .expect("state")
        .get(&crate::state::Key::new(
            crate::provider::ProviderId::Claude,
            "there",
        ))
        .and_then(|a| a.parked.clone())
        .expect("`there` still holds a login");
    assert_eq!(
        kept.refresh_fingerprint,
        crate::provider::claude::document::fingerprint_of(&json!({"refreshToken": "fresh"})),
        "the fresh login is kept and the spent one dropped, not the other way round"
    );
    assert_eq!(m.mem.vault().services(), vec![kept.service]);
}

/// A renewal whose answer is written and whose record cannot be saved keeps what it wrote.
/// The service has already spent the chain the record still names, so deleting the fresh
/// copy, which this once did, left the account nothing that works. The next change gives
/// it back.
#[test]
fn a_renewal_that_cannot_record_its_answer_keeps_it_for_the_next_run() {
    use std::os::unix::fs::PermissionsExt;
    let m = machine("renew-save-fails");
    let key = crate::state::Key::new(crate::provider::ProviderId::Claude, "there");
    let mut state = state::load(&m.ctx).expect("state");
    let held = state.get(&key).unwrap().parked.clone().unwrap();
    let lapsed = json!({
        "accessToken": "access-there-refresh",
        "refreshToken": "there-refresh",
        "expiresAt": (NOW - 60) * 1000,
        "refreshTokenExpiresAt": (NOW + 30 * 86_400) * 1000
    });
    m.mem.vault().plant(&held.service, &lapsed.to_string());
    state.park(
        &key,
        park::describe(
            crate::provider::ProviderId::Claude,
            &held.service,
            NOW,
            &lapsed,
        ),
    );
    state::save(&m.ctx, &state).expect("saved");
    m.api.renews(
        "there-refresh",
        crate::api::Renewed {
            access_token: "access-fresh".into(),
            refresh_token: Some("fresh".into()),
            expires_in: 3600,
            refresh_token_expires_in: Some(30 * 86_400),
            scopes: None,
            at: None,
        },
    );

    // pitboard's home goes read-only once the fresh login is in the vault, so the record
    // of it cannot be written.
    let home = crate::home::dir(&m.ctx);
    let locked = home.clone();
    let outcomes = fault::meanwhile(
        "renew.park_stored",
        move || {
            std::fs::set_permissions(&locked, std::fs::Permissions::from_mode(0o500)).unwrap();
        },
        || renew::renew_parked(&m.ctx),
    );
    std::fs::set_permissions(&home, std::fs::Permissions::from_mode(0o700)).unwrap();
    assert!(
        outcomes
            .iter()
            .any(|(_, r)| matches!(r, renew::Renewal::Failed(_))),
        "the save failed, and says so"
    );

    recover(&m).expect("the next change");
    let kept = state::load(&m.ctx)
        .expect("state")
        .get(&key)
        .and_then(|a| a.parked.clone())
        .expect("`there` still holds a login");
    assert_eq!(
        kept.refresh_fingerprint,
        crate::provider::claude::document::fingerprint_of(&json!({"refreshToken": "fresh"})),
    );
    hold(&m, "after a renewal that could not be recorded");
}

/// Forgetting deletes the account before deleting its park. Killed between the two, the
/// park is listed for deletion and a later run finishes it.
#[test]
fn forgetting_killed_after_the_record_still_deletes_the_park() {
    let m = machine("forget-recorded");
    let settled = settle(&m.ctx, None).expect("nothing to recover").0;

    let died = fault::killing("forget.recorded", || {
        forget::forget(
            settled,
            &crate::state::Key::new(crate::provider::ProviderId::Claude, "there"),
        )
    });
    assert_eq!(died.unwrap_err(), "forget.recorded");

    recover(&m).expect("recovery");
    hold(&m, "forget.recorded");
    assert!(
        m.mem.vault().services().is_empty(),
        "a forgotten account's login is deleted, not left in the vault"
    );
}