diff --git a/scripts/redeploy-fleetd.sh b/scripts/redeploy-fleetd.sh index 2a32ac5..08be7af 100755 --- a/scripts/redeploy-fleetd.sh +++ b/scripts/redeploy-fleetd.sh @@ -5,7 +5,7 @@ # A merge is not a deployment: the running daemon holds the jar it was started with, so code merged # to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon". # -# This script exists to turn six remembered traps into one auditable command: +# This script exists to turn eight remembered traps into one auditable command: # # 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped. # 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty: @@ -27,6 +27,21 @@ # restart of the OLD jar. So this script detects whether the agent is loaded and, only then, # swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control # at any moment is whichever one you asked to act, never both. +# 7. fleetd #492 — a systemd --user unit is a THIRD possible supervisor (seen on a second host): +# Restart=on-failure treats this JVM's SIGTERM exit code (143, per CB-594 above) as a failure +# too, so a bare `kill` there would race systemd's own restart of the OLD jar exactly like +# launchd would. This script now tells launchd, systemd, and "genuinely unsupervised" apart as +# three different answers, drives whichever one it finds through its own control plane +# (`launchctl` / `systemctl --user`), and REFUSES outright — never falls back to `kill` — when +# it finds a supervision signal it cannot map to exactly one of the two it knows how to drive. +# A wrong guess here is how two daemons end up running against one herdr session. Follow-up: +# "not currently loaded" is not the same fact as "unsupervised" — a unit that is installed but +# activating/failed/pending-restart, or a `systemctl` call that could not answer at all (e.g. +# no user-bus access), both now read as a fifth answer, "unclear", and REFUSE the same way +# "ambiguous" does, rather than silently falling through to "none". +# 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if +# more than one is alive. That is the one thing none of the checks above (healthz 200, jar id, +# the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon. # # Usage: # scripts/redeploy-fleetd.sh # build, confirm, restart, verify @@ -55,13 +70,42 @@ HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start LAUNCHD_LABEL='dev.ltms.fleetd' LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist" +# fleetd #492: the systemd --user unit this script must not fight with either (see trap 7 above). +# Measured on the second host: `systemctl --user cat fleetd` names the unit "fleetd" (not +# "dev.ltms.fleetd" — systemd user units here are not namespaced the way the launchd label is). +SYSTEMD_UNIT='fleetd' + +# fleetd #492 follow-up: detect_supervisor packs TWO values (kind, detail) onto the one stdout +# line that survives its $(...) call — see the constraints comment above that function. This is +# the separator between them: the ASCII "unit separator" byte, chosen because it never occurs in +# any of the prose detail strings and needs no escaping in a `case`/glob pattern. +SUPERVISOR_DETAIL_SEP=$'\x1f' + +# fleetd #492 follow-up: set by systemd_loaded/systemd_installed when the underlying `systemctl` +# call could not answer cleanly — it exited non-zero AND wrote something to stderr, which is a real +# tool failure (e.g. it cannot reach the user bus over a non-lingering ssh session), never the same +# fact as a clean negative answer ("not active", no stderr). Initialized here, not just inside the +# probes, so detect_supervisor can read them under `set -u` even before either probe has ever run, +# and so a test that stubs a probe with a plain `return 0`/`return 1` body (leaving these untouched) +# reads a deterministic 0 rather than whatever a previous probe call left behind. +SYSTEMD_LOADED_ERRORED=0 +SYSTEMD_INSTALLED_ERRORED=0 +# fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL is the specific supervisor/reason that +# require_drivable_supervisor's die() names on an "unclear" answer. Deliberately NOT pre-declared +# here (unlike the two flags above): it is set only by the real call site, right after it unpacks +# detect_supervisor's stdout (see the constraints comment above detect_supervisor). If that call +# site is ever skipped or broken, a bare `set -u` reference to this variable in +# require_drivable_supervisor must fail loudly with "unbound variable" — a pre-declared empty +# default would instead silently print an empty reason, hiding exactly the value this ticket +# exists to surface. + DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0 for arg in "$@"; do case "$arg" in --yes|-y) ASSUME_YES=1 ;; --no-build) DO_BUILD=0 ;; --check) CHECK_ONLY=1 ;; - -h|--help) sed -n '3,37p' "${BASH_SOURCE[0]}"; exit 0 ;; + -h|--help) sed -n '3,48p' "${BASH_SOURCE[0]}"; exit 0 ;; *) echo "unknown option: $arg (try --help)" >&2; exit 2 ;; esac done @@ -79,6 +123,193 @@ running_pid() { pgrep -f "$PATTERN" || true; } launchd_installed() { [ -f "$LAUNCHD_PLIST" ]; } launchd_loaded() { launchctl list "$LAUNCHD_LABEL" >/dev/null 2>&1; } +# fleetd #492: same two questions for systemd --user. Kept as separate, overridable functions +# (never an inline `systemctl` call at each use site) so a test on a box with no systemd at all +# (this repo is developed on macOS) can substitute each one independently — the same seam +# launchd_installed/launchd_loaded above already use. +# +# fleetd #492 follow-up: both functions used to throw `systemctl`'s stderr straight into +# /dev/null, which meant "systemctl answered no" and "systemctl could not answer at all" (e.g. it +# cannot reach the user bus over a non-lingering ssh session) looked identical — both a plain +# nonzero exit. They now capture stderr separately and set their own *_ERRORED flag ONLY when the +# call exited non-zero AND wrote something to stderr — a real tool failure, never a clean "not +# installed"/"not active" answer (which exits non-zero with empty stderr). detect_supervisor reads +# the flag right after calling the probe, so a probe that could not answer routes to "unclear", +# never silently becomes "none". +# +# "installed": a unit FILE by this name exists, regardless of its current state — the systemd +# analogue of the plist file existing on disk. `list-unit-files` reads unit definitions without +# depending on runtime state, so this stays read-only and safe under --check. +systemd_installed() { + SYSTEMD_INSTALLED_ERRORED=0 + command -v systemctl >/dev/null 2>&1 || return 1 + local err_file out rc=0 + if ! err_file="$(mktemp -t systemd-installed-err)"; then + SYSTEMD_INSTALLED_ERRORED=1 + return 1 + fi + out="$(systemctl --user list-unit-files "$SYSTEMD_UNIT.service" --no-legend 2>"$err_file")" || rc=$? + if [ "$rc" -ne 0 ]; then + if [ -s "$err_file" ]; then + SYSTEMD_INSTALLED_ERRORED=1 + fi + rm -f "$err_file" + return "$rc" + fi + rm -f "$err_file" + printf '%s' "$out" | grep -q . +} +# "loaded": systemd currently supervises this unit as an active job — the systemd analogue of +# `launchctl list