From dcd505286f035395e25cccc6923524a6ad649391 Mon Sep 17 00:00:00 2001 From: Dai Ha Date: Sat, 12 Sep 2026 09:00:59 +0700 Subject: [PATCH 1/3] fleetd #492: teach redeploy-fleetd.sh systemd --user as a third supervisor launchd, systemd --user, and unsupervised are three different answers, not two. Refuse (die) rather than fall through to kill+nohup when a supervisor is detected that this script cannot drive (e.g. both signals fire at once), and add a post-restart check that fails the run if more than one fleetd process is alive. detect_supervisor()/require_drivable_supervisor()/ count_daemon_pids()/assert_single_daemon() are pure, overridable functions so scripts/test-redeploy-fleetd.sh can exercise them without a real launchd or systemd. --- scripts/redeploy-fleetd.sh | 271 +++++++++++++++++++++++++------- scripts/test-redeploy-fleetd.sh | 80 ++++++++++ 2 files changed, 297 insertions(+), 54 deletions(-) diff --git a/scripts/redeploy-fleetd.sh b/scripts/redeploy-fleetd.sh index 2a32ac5..e034029 100755 --- a/scripts/redeploy-fleetd.sh +++ b/scripts/redeploy-fleetd.sh @@ -5,7 +5,7 @@ # A merge is not a deployment: the running daemon holds the jar it was started with, so code merged # to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon". # -# This script exists to turn six remembered traps into one auditable command: +# This script exists to turn eight remembered traps into one auditable command: # # 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped. # 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty: @@ -27,6 +27,17 @@ # restart of the OLD jar. So this script detects whether the agent is loaded and, only then, # swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control # at any moment is whichever one you asked to act, never both. +# 7. fleetd #492 — a systemd --user unit is a THIRD possible supervisor (seen on a second host): +# Restart=on-failure treats this JVM's SIGTERM exit code (143, per CB-594 above) as a failure +# too, so a bare `kill` there would race systemd's own restart of the OLD jar exactly like +# launchd would. This script now tells launchd, systemd, and "genuinely unsupervised" apart as +# three different answers, drives whichever one it finds through its own control plane +# (`launchctl` / `systemctl --user`), and REFUSES outright — never falls back to `kill` — when +# it finds a supervision signal it cannot map to exactly one of the two it knows how to drive. +# A wrong guess here is how two daemons end up running against one herdr session. +# 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if +# more than one is alive. That is the one thing none of the checks above (healthz 200, jar id, +# the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon. # # Usage: # scripts/redeploy-fleetd.sh # build, confirm, restart, verify @@ -55,13 +66,18 @@ HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start LAUNCHD_LABEL='dev.ltms.fleetd' LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist" +# fleetd #492: the systemd --user unit this script must not fight with either (see trap 7 above). +# Measured on the second host: `systemctl --user cat fleetd` names the unit "fleetd" (not +# "dev.ltms.fleetd" — systemd user units here are not namespaced the way the launchd label is). +SYSTEMD_UNIT='fleetd' + DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0 for arg in "$@"; do case "$arg" in --yes|-y) ASSUME_YES=1 ;; --no-build) DO_BUILD=0 ;; --check) CHECK_ONLY=1 ;; - -h|--help) sed -n '3,37p' "${BASH_SOURCE[0]}"; exit 0 ;; + -h|--help) sed -n '3,48p' "${BASH_SOURCE[0]}"; exit 0 ;; *) echo "unknown option: $arg (try --help)" >&2; exit 2 ;; esac done @@ -79,6 +95,91 @@ running_pid() { pgrep -f "$PATTERN" || true; } launchd_installed() { [ -f "$LAUNCHD_PLIST" ]; } launchd_loaded() { launchctl list "$LAUNCHD_LABEL" >/dev/null 2>&1; } +# fleetd #492: same two questions for systemd --user. Kept as separate, overridable functions +# (never an inline `systemctl` call at each use site) so a test on a box with no systemd at all +# (this repo is developed on macOS) can substitute each one independently — the same seam +# launchd_installed/launchd_loaded above already use. +# +# "installed": a unit FILE by this name exists, regardless of its current state — the systemd +# analogue of the plist file existing on disk. `list-unit-files` reads unit definitions without +# depending on runtime state, so this stays read-only and safe under --check. +systemd_installed() { + command -v systemctl >/dev/null 2>&1 \ + && systemctl --user list-unit-files "$SYSTEMD_UNIT.service" --no-legend 2>/dev/null | grep -q . +} +# "loaded": systemd currently supervises this unit as an active job — the systemd analogue of +# `launchctl list