# shellcheck shell=bash
# Canonical control-plane roll: fetch → verify → anchor → install → assert →
# restart. yubaba + kamaji are the supervised pair; yah-scryer (0.8.32),
# passway + passway-demux (0.8.33) and the turso-backup durability helpers
# (0.8.37) ride the same tarball, each conditional on the tarball carrying it
# so a rollback to an older release still succeeds.
#
# THIS FILE IS THE ONE COPY. Three callers consume these exact bytes:
#   1. `build_install_script` (control_plane_install.rs) include_str!s it for the
#      SSH transport (`rollout::apply::apply_over_ssh`).
#   2. …and for the mesh transport (yubaba `POST /self-update`, run as root in a
#      systemd-run transient unit).
#   3. `scripts/roll-node.sh` reads it off disk and pipes it to `ssh … bash -s`.
# Every caller prepends a prologue setting the four variables below and nothing
# else. Adding a fourth copy of this logic is how the fleet drifts — don't.
#
# Required from the caller's prologue:
#   URL   release tarball URL, from a SIGNED release manifest
#   SHA   that tarball's sha256, from the SAME manifest
#   VER   the version being installed (labels the log line)
#   SUDO  "sudo" when the executing user is not root, empty when it is
#
# Never touches durable state: no /var/lib/yah-cloud/identity.json (wiping the
# ed25519 host identity forces a re-TOFU and breaks hostkey-drift detection —
# the R589 gotcha), no raft log dir. A roll moves /usr/local/bin bytes + unit
# files, nothing else. `script_never_touches_durable_state` is the guard.
set -euo pipefail
: "${URL:?URL must be set from a signed release manifest}"
: "${SHA:?SHA must be set from the same signed release manifest}"
: "${VER:?VER must be set}"
SUDO="${SUDO-}"
STAMP="$(date -u +%Y%m%d)"
WORK="$(mktemp -d /tmp/yah-roll.XXXXXX)"
trap 'rm -rf "$WORK"' EXIT
cd "$WORK"

echo "== fetch + verify (sha256 from signed manifest) =="
curl -fsSL -o pair.tar.gz "$URL"
printf '%s  pair.tar.gz\n' "$SHA" | sha256sum -c -
mkdir x && tar -xzf pair.tar.gz -C x
D="$(find x -maxdepth 1 -type d -name 'yubaba-*' | head -1)"
[ -n "$D" ] || { echo 'tarball layout unexpected: no yubaba-* dir' >&2; exit 1; }

echo "== rollback anchors (.rollback-$STAMP) =="
# The convention these boxes already carry: /usr/local/bin/kamaji.rollback-YYYYMMDD,
# the shape both the passway roll and the 2026-07-21 kamaji roll left behind.
# Written BEFORE anything is replaced, so the anchor is the build that was live
# going in.
#
# Never overwritten within the same UTC day, and that is the load-bearing part:
# on a second run the anchor must still hold the build that was live before the
# FIRST roll of the day. Re-anchoring would quietly overwrite the escape hatch
# with the very binary you are trying to escape from.
#
# A missing target is not an error — a first install on a fresh box has nothing
# to anchor. The unit files are anchored alongside the binaries under the same
# stamp: rolling the binaries back while leaving new units in place is not a
# rollback, and the tarball ships all five together.
anchor() { # path
  [ -e "$1" ] || return 0
  if [ ! -e "$1.rollback-$STAMP" ]; then
    $SUDO cp -p "$1" "$1.rollback-$STAMP"
    echo "  anchored $1.rollback-$STAMP"
  else
    echo "  kept existing $1.rollback-$STAMP (pre-roll build for today)"
  fi
}
anchor /usr/local/bin/yubaba
anchor /usr/local/bin/kamaji
anchor /usr/local/bin/yah-scryer
anchor /usr/local/bin/passway
anchor /usr/local/bin/passway-demux
anchor /usr/local/bin/passway-http-router
anchor /usr/local/bin/passway-graceful-upgrade
anchor /usr/local/bin/turso-backup-hydrate
anchor /usr/local/bin/turso-backup-tail
anchor /etc/systemd/system/yubaba.slice
anchor /etc/systemd/system/kamaji.service
anchor /etc/systemd/system/yubaba.service
anchor /etc/systemd/system/yah-scryer.service
# The durability drop-in is anchored like any other file this script replaces.
# A `.rollback-YYYYMMDD` sibling inside a `.d` directory is inert — systemd
# reads only `*.conf` there — so the anchor cannot itself change the unit.
anchor /etc/systemd/system/kamaji.service.d/50-durability-helpers.conf
# R858-T24's credential + flag drop-ins, anchored the same way.
anchor /etc/systemd/system/kamaji.service.d/51-headscale-durability-cred.conf
anchor /etc/systemd/system/yubaba.service.d/50-headscale-durability.conf

echo "== install (atomic, yubaba + kamaji as one pair) =="
# Stage next to the target on the SAME filesystem, then rename. A rename within
# a filesystem is atomic, so a half-written binary/unit is never observable.
install_atomic() { # src mode dest
  $SUDO install -m"$2" "$1" "$3.roll-new.$$"
  $SUDO mv -f "$3.roll-new.$$" "$3"
}
install_atomic "$D/yubaba"         0755 /usr/local/bin/yubaba
install_atomic "$D/kamaji"         0755 /usr/local/bin/kamaji
install_atomic "$D/yubaba.slice"   0644 /etc/systemd/system/yubaba.slice
install_atomic "$D/kamaji.service" 0644 /etc/systemd/system/kamaji.service
install_atomic "$D/yubaba.service" 0644 /etc/systemd/system/yubaba.service
# yah-scryer rides the same tarball from 0.8.32 (R556-F6 gate (b), A049: each
# node runs its own scryer). CONDITIONAL on the tarball actually carrying it so
# a rollback to a pre-scryer release still succeeds — it leaves whatever scryer
# is already installed in place rather than failing on a missing member. The
# scryer's own durable state (/var/lib/yah/scryer, events.db) is never touched
# here; the unit's StateDirectory owns it.
HAS_SCRYER=0
if [ -e "$D/yah-scryer" ]; then
  HAS_SCRYER=1
  install_atomic "$D/yah-scryer"         0755 /usr/local/bin/yah-scryer
  install_atomic "$D/yah-scryer.service" 0644 /etc/systemd/system/yah-scryer.service
fi
# passway + passway-demux ride the same tarball from 0.8.33 (R870-B2). Before
# this they had NO distribution path at all — the live doors were hand-cross-
# built and scp'd (R853-T2) — so a fleet roll could not carry an ingress fix.
# Conditional for the same reason as scryer: a rollback to a pre-0.8.33 release
# must still succeed.
#
# NO UNIT IS INSTALLED AND NOTHING IS RESTARTED, deliberately, and both halves
# matter. The unit name is not knowable from here: the two live origins run
# passway.service (south) and passway-test.service (east) against per-node env
# files, so there is nothing this script could name. And a restart is not free —
# passway cannot hot-swap (tls.rs "The reload gap": TlsSettings is static), so
# `systemctl restart` DROPS IN-FLIGHT CONNECTIONS on a public front door. The
# install is a rename, so a running door keeps serving from its open inode and
# picks the new bytes up on the operator's next restart. Staging bytes without
# cutting live traffic is the correct default for the :443 tier; R870-T3 is
# wiring the graceful PASSWAY_UPGRADE handoff that makes a restart safe.
HAS_PASSWAY=0
if [ -e "$D/passway" ]; then
  HAS_PASSWAY=1
  install_atomic "$D/passway"       0755 /usr/local/bin/passway
  install_atomic "$D/passway-demux" 0755 /usr/local/bin/passway-demux
fi
# passway-http-router — the :80 tier (R870-F1) — joined at 0.8.34, one release
# AFTER the passway pair, so it gets its OWN conditional rather than riding
# HAS_PASSWAY. Rolling a 0.8.33 tarball with this script must still succeed, and
# it would not if a missing member were assumed present because a sibling was.
HAS_HTTP_ROUTER=0
if [ -e "$D/passway-http-router" ]; then
  HAS_HTTP_ROUTER=1
  install_atomic "$D/passway-http-router" 0755 /usr/local/bin/passway-http-router
fi
# passway-graceful-upgrade (R870-T3) — the ExecReload= that turns a cert
# rotation into a process swap instead of a connection-dropping restart. A
# SCRIPT, so its own conditional again rather than riding a sibling's.
#
# The script is fleet-wide and carries no node state, which is why it rides the
# roll. The DROP-IN that arms it
# (app/yah/cli/resources/passway-graceful-upgrade.conf) deliberately does NOT:
# it lands in /etc/systemd/system/<unit>.service.d/, the per-node unit name
# differs between doors, and this script must never touch a door's unit
# configuration — the same rule that keeps passway.service out of the tarball.
HAS_UPGRADE_HELPER=0
if [ -e "$D/passway-graceful-upgrade" ]; then
  HAS_UPGRADE_HELPER=1
  install_atomic "$D/passway-graceful-upgrade" 0755 /usr/local/bin/passway-graceful-upgrade
fi
# turso-backup-hydrate + turso-backup-tail (R858-F17), from 0.8.37 — the two
# helpers kamaji execs to restore and then continuously tail a workload's SQLite
# state. kamaji HARD-REFUSES to deploy any workload declaring
# `yah.durability.tier` when either is missing (kamaji-bin/src/hydrate.rs,
# src/tail.rs), so a node without them is a node where durability can never be
# switched on — declaring a tier there takes the service DOWN instead of backing
# it up.
#
# R858-T21: until this block they were placed ONLY by provisioning
# (stand-up-yubaba.sh's "durability helpers" block, mirror.yml's turso-backup
# block), so a node that was ROLLED rather than freshly stood up could never
# acquire them, and R858-F17's own "cut a release, roll it, then verify the
# helpers are on the node" instruction could not pass on any rolled node. This
# is a transcription of stand-up-yubaba.sh's block, down to the drop-in's bytes.
#
# Conditional for the same reason as every block above: a rollback to a
# pre-0.8.37 release must still succeed, just without durability.
#
# THE DROP-IN USES `Environment=`, NEVER `ExecStart=`. A drop-in that redeclares
# ExecStart= silently drops every flag the unit added after it — measured on
# us-south-001, half of the 2026-09-03 outage (the 20-bundle.conf note in
# yubaba's litestream.rs). It no-ops until a workload actually declares a tier,
# so laying it down on every node this script touches is safe, and it is
# staged-then-renamed like everything else here rather than written in place.
HAS_DURABILITY_HELPERS=0
if [ -e "$D/turso-backup-hydrate" ] && [ -e "$D/turso-backup-tail" ]; then
  HAS_DURABILITY_HELPERS=1
  install_atomic "$D/turso-backup-hydrate" 0755 /usr/local/bin/turso-backup-hydrate
  install_atomic "$D/turso-backup-tail"    0755 /usr/local/bin/turso-backup-tail
  printf '[Service]\nEnvironment=KAMAJI_HYDRATE_HELPER=/usr/local/bin/turso-backup-hydrate\nEnvironment=KAMAJI_TAIL_HELPER=/usr/local/bin/turso-backup-tail\n' \
    > "$WORK/50-durability-helpers.conf"
  $SUDO mkdir -p /etc/systemd/system/kamaji.service.d
  install_atomic "$WORK/50-durability-helpers.conf" 0644 \
    /etc/systemd/system/kamaji.service.d/50-durability-helpers.conf
else
  echo "  turso-backup helpers absent from this tarball — durability-declaring"
  echo "  workloads will refuse to deploy on this node (R858-F17)"
fi
# R858-T24 — the credential AND the flag, in ONE conditional, never separately.
# This relay is a 37-hour and a 14-hour outage caused by the same defect twice:
# a declaration (`yah.durability.tier` / YUBABA_HEADSCALE_DURABILITY) shipped
# ahead of its prerequisite. kamaji hard-refuses a tier-declaring workload when
# either helper above is missing (hydrate.rs/tail.rs), and even with both
# helpers present, the hydrate helper hard-refuses without S3_ACCESS_KEY /
# S3_SECRET_KEY — so setting the flag without the credential reproduces the
# exact failure this whole relay exists to close.
#
# This script CANNOT MINT the credential — it is not in the release tarball —
# so it can only detect an operator-placed credential file and wire it,
# mirroring how /etc/yah-cloud/cert-store.env is delivered for the cert store
# (us-east-001.toml:148-153: no in-tree writer places that file either, an
# operator does, and yubaba.service.d/50-cert-store.conf only references it).
HEADSCALE_DURABILITY_CRED=/etc/yah-cloud/headscale-durability.env
HEADSCALE_DURABILITY_ON=0
if [ "$HAS_DURABILITY_HELPERS" = 1 ] && [ -f "$HEADSCALE_DURABILITY_CRED" ]; then
  # kamaji leg: EnvironmentFile= for the secret pair (S3_ACCESS_KEY /
  # S3_SECRET_KEY), plus the non-secret R2 endpoint/region as plain
  # Environment= lines in the same drop-in — never ExecStart=.
  printf '[Service]\nEnvironmentFile=%s\nEnvironment=S3_ENDPOINT=https://3948dc292e724e71b0deefde0ea95999.r2.cloudflarestorage.com\nEnvironment=S3_REGION=auto\n' \
    "$HEADSCALE_DURABILITY_CRED" > "$WORK/51-headscale-durability-cred.conf"
  install_atomic "$WORK/51-headscale-durability-cred.conf" 0644 \
    /etc/systemd/system/kamaji.service.d/51-headscale-durability-cred.conf

  # yubaba leg: the flag itself, gated on the SAME `if` as the credential above
  # — this is the whole point, not a stylistic choice. --headscale-durability /
  # YUBABA_HEADSCALE_DURABILITY defaults OFF (headscale_appliance.rs); this is
  # the only place in the tree that turns it on, and it can't fire without the
  # credential leg above having just run in this same pass.
  $SUDO mkdir -p /etc/systemd/system/yubaba.service.d
  printf '[Service]\nEnvironment=YUBABA_HEADSCALE_DURABILITY=1\n' \
    > "$WORK/50-headscale-durability.conf"
  install_atomic "$WORK/50-headscale-durability.conf" 0644 \
    /etc/systemd/system/yubaba.service.d/50-headscale-durability.conf
  HEADSCALE_DURABILITY_ON=1
elif [ "$HAS_DURABILITY_HELPERS" = 1 ]; then
  echo "  $HEADSCALE_DURABILITY_CRED absent — headscale durability tier stays OFF."
  echo "  Place the yah-headscale-scoped R2 credential there (S3_ACCESS_KEY /"
  echo "  S3_SECRET_KEY) to turn it on; YUBABA_HEADSCALE_DURABILITY is NOT set"
  echo "  without it (R858-T24)."
fi

echo "== assert by CONTENT, not by version string =="
# `--version` prints the workspace version baked in at build time, which says
# nothing about whether the bytes on disk are the ones you just shipped:
# us-east-001 reported kamaji 0.8.22 while carrying none of the 0.8.22 tree's
# code (R746-T3). Hash the installed file against the file extracted from the
# tarball whose sha256 the signed manifest already vouched for, and the chain
# manifest → tarball → extracted → installed closes with no version string in
# it anywhere.
assert_installed_bytes() { # extracted installed
  local want got
  want="$(sha256sum "$1" | awk '{print $1}')"
  got="$(sha256sum "$2" | awk '{print $1}')"
  if [ "$want" != "$got" ]; then
    echo "content assertion FAILED for $2: tarball has $want, installed file has $got" >&2
    exit 1
  fi
  echo "  $2 sha256=$got"
}
assert_installed_bytes "$D/yubaba" /usr/local/bin/yubaba
assert_installed_bytes "$D/kamaji" /usr/local/bin/kamaji
if [ "$HAS_SCRYER" = 1 ]; then
  assert_installed_bytes "$D/yah-scryer" /usr/local/bin/yah-scryer
fi
if [ "$HAS_PASSWAY" = 1 ]; then
  assert_installed_bytes "$D/passway"       /usr/local/bin/passway
  assert_installed_bytes "$D/passway-demux" /usr/local/bin/passway-demux
fi
if [ "$HAS_HTTP_ROUTER" = 1 ]; then
  assert_installed_bytes "$D/passway-http-router" /usr/local/bin/passway-http-router
fi
if [ "$HAS_UPGRADE_HELPER" = 1 ]; then
  assert_installed_bytes "$D/passway-graceful-upgrade" /usr/local/bin/passway-graceful-upgrade
fi
if [ "$HAS_DURABILITY_HELPERS" = 1 ]; then
  assert_installed_bytes "$D/turso-backup-hydrate" /usr/local/bin/turso-backup-hydrate
  assert_installed_bytes "$D/turso-backup-tail"    /usr/local/bin/turso-backup-tail
fi

echo "== restart supervision tree (kamaji then yubaba, W154 order) =="
$SUDO systemctl daemon-reload
$SUDO systemctl restart kamaji.service
$SUDO systemctl restart yubaba.service
# Scryer sits outside the W154 order — it neither drives nor is driven by the
# pair (A049: located by yubaba, not driven by it), so it restarts last.
# `enable` covers the first install; on later rolls it is a no-op.
if [ "$HAS_SCRYER" = 1 ]; then
  $SUDO systemctl enable yah-scryer.service
  $SUDO systemctl restart yah-scryer.service
fi
if [ "$HAS_PASSWAY" = 1 ]; then
  echo "  passway + passway-demux bytes are STAGED, not live — this roll deliberately"
  echo "  does not restart the front door (see the install block above)."
  if [ "$HAS_UPGRADE_HELPER" = 1 ]; then
    echo "  To activate WITHOUT a blip, on a door carrying"
    echo "  /etc/systemd/system/<unit>.service.d/passway-graceful-upgrade.conf:"
    echo "    systemctl reload <unit>     # R870-T3: process swap, listeners handed over"
    echo "  Without that drop-in the only verb is \`systemctl restart\`, which drops"
    echo "  every in-flight connection on :443."
  else
    echo "  Restart the node's own passway unit when a :443 blip is acceptable."
  fi
fi
if [ "$HAS_DURABILITY_HELPERS" = 1 ]; then
  # Live on THIS roll, not the next one: the drop-in landed before the
  # daemon-reload above, and kamaji was restarted after it, so the running
  # kamaji already carries KAMAJI_HYDRATE_HELPER / KAMAJI_TAIL_HELPER.
  echo "  durability helpers installed; kamaji restarted above with"
  echo "  KAMAJI_{HYDRATE,TAIL}_HELPER set (R858-T21)"
fi
if [ "$HEADSCALE_DURABILITY_ON" = 1 ]; then
  echo "  headscale durability tier ON: kamaji restarted above with S3_ACCESS_KEY/"
  echo "  S3_SECRET_KEY from $HEADSCALE_DURABILITY_CRED, yubaba restarted above"
  echo "  with YUBABA_HEADSCALE_DURABILITY=1 (R858-T24)"
fi
if [ "$HAS_HTTP_ROUTER" = 1 ]; then
  echo "  passway-http-router bytes are STAGED, not live, for the same reason — and"
  echo "  a :80 restart is the cheap one: it terminates nothing, holds no cert, and"
  echo "  the traffic it drops is redirects a client immediately retries."
fi
echo "installed target=$VER yubaba=$(/usr/local/bin/yubaba --version 2>/dev/null) kamaji=$(/usr/local/bin/kamaji --version 2>/dev/null)"
