#!/usr/bin/env bash # # Rebuild and restart the fleetd daemon. # # A merge is not a deployment: the running daemon holds the jar it was started with, so code merged # to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon". # # This script exists to turn eight remembered traps into one auditable command: # # 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped. # 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty: # WORKER_GITEA_TOKEN (workers cannot open a PR) and AI_GATEWAY_TOKEN (401 at llm.ltms.dev). # Both are read from the DAEMON's own environment at spawn time, so a value added to # secrets.sh after startup is absent. Nothing logs this here, so the script checks and says # so — and since CB-594, fleetd's own startup log says so too, by env var name. # 3. An old daemon that never actually died looks identical from the outside, so the script waits # for the process to exit and for the port to free before it starts a new one. # 4. "It started" is not "it works": the script polls /healthz until it answers, and reports the # herdr protocol number, because healthz can be green while every spawn fails on a protocol # mismatch. # 5. Restarting under live members drops their tickets, so the script refuses unless you confirm # the fleet is drained. # 6. CB-594 — the launchd agent (deploy/dev.ltms.fleetd.plist), if installed and loaded, is a # SECOND supervisor: its KeepAlive.SuccessfulExit=false restarts the daemon on any nonzero # exit, and a bare SIGTERM makes this JVM exit 143 even with its shutdown hook running to # completion (measured — see the CB-594 report). A plain `kill` here would race launchd's own # restart of the OLD jar. So this script detects whether the agent is loaded and, only then, # swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control # at any moment is whichever one you asked to act, never both. # 7. fleetd #492 — a systemd --user unit is a THIRD possible supervisor (seen on a second host): # Restart=on-failure treats this JVM's SIGTERM exit code (143, per CB-594 above) as a failure # too, so a bare `kill` there would race systemd's own restart of the OLD jar exactly like # launchd would. This script now tells launchd, systemd, and "genuinely unsupervised" apart as # three different answers, drives whichever one it finds through its own control plane # (`launchctl` / `systemctl --user`), and REFUSES outright — never falls back to `kill` — when # it finds a supervision signal it cannot map to exactly one of the two it knows how to drive. # A wrong guess here is how two daemons end up running against one herdr session. Follow-up: # "not currently loaded" is not the same fact as "unsupervised" — a unit that is installed but # activating/failed/pending-restart, or a `systemctl` call that could not answer at all (e.g. # no user-bus access), both now read as a fifth answer, "unclear", and REFUSE the same way # "ambiguous" does, rather than silently falling through to "none". # 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if # more than one is alive. That is the one thing none of the checks above (healthz 200, jar id, # the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon. # # Usage: # scripts/redeploy-fleetd.sh # build, confirm, restart, verify # scripts/redeploy-fleetd.sh --yes # skip the drain confirmation (fleet already checked) # scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk # scripts/redeploy-fleetd.sh --check # report state and exit; changes nothing # # Exits non-zero on any failure. A failed build never stops the running daemon. set -euo pipefail REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" MODULE="$REPO/fleetd" JAR="$MODULE/target/fleetd.jar" OUT="$MODULE/fleetd.out" # Matches BOTH the absolute form and the relative `java -jar target/fleetd.jar` a hand-start # produces from inside fleetd/. Anchoring on the absolute path alone was a real bug: the daemon # restarted correctly and the script still reported "no process appeared", because it launched with # a relative path and then looked for an absolute one. PATTERN='target/fleetd.jar' HEALTH='http://127.0.0.1:8765/healthz' STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start # CB-594: the launchd agent this script must not fight with (see trap 6 above). LAUNCHD_LABEL='dev.ltms.fleetd' LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist" # fleetd #492: the systemd --user unit this script must not fight with either (see trap 7 above). # Measured on the second host: `systemctl --user cat fleetd` names the unit "fleetd" (not # "dev.ltms.fleetd" — systemd user units here are not namespaced the way the launchd label is). SYSTEMD_UNIT='fleetd' # fleetd #492 follow-up: detect_supervisor packs TWO values (kind, detail) onto the one stdout # line that survives its $(...) call — see the constraints comment above that function. This is # the separator between them: the ASCII "unit separator" byte, chosen because it never occurs in # any of the prose detail strings and needs no escaping in a `case`/glob pattern. SUPERVISOR_DETAIL_SEP=$'\x1f' # fleetd #492 follow-up: set by systemd_loaded/systemd_installed when the underlying `systemctl` # call could not answer cleanly — it exited non-zero AND wrote something to stderr, which is a real # tool failure (e.g. it cannot reach the user bus over a non-lingering ssh session), never the same # fact as a clean negative answer ("not active", no stderr). Initialized here, not just inside the # probes, so detect_supervisor can read them under `set -u` even before either probe has ever run, # and so a test that stubs a probe with a plain `return 0`/`return 1` body (leaving these untouched) # reads a deterministic 0 rather than whatever a previous probe call left behind. SYSTEMD_LOADED_ERRORED=0 SYSTEMD_INSTALLED_ERRORED=0 # fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL is the specific supervisor/reason that # require_drivable_supervisor's die() names on an "unclear" answer. Deliberately NOT pre-declared # here (unlike the two flags above): it is set only by the real call site, right after it unpacks # detect_supervisor's stdout (see the constraints comment above detect_supervisor). If that call # site is ever skipped or broken, a bare `set -u` reference to this variable in # require_drivable_supervisor must fail loudly with "unbound variable" — a pre-declared empty # default would instead silently print an empty reason, hiding exactly the value this ticket # exists to surface. DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0 for arg in "$@"; do case "$arg" in --yes|-y) ASSUME_YES=1 ;; --no-build) DO_BUILD=0 ;; --check) CHECK_ONLY=1 ;; -h|--help) sed -n '3,48p' "${BASH_SOURCE[0]}"; exit 0 ;; *) echo "unknown option: $arg (try --help)" >&2; exit 2 ;; esac done say() { printf '\n\033[1m== %s\033[0m\n' "$*"; } ok() { printf ' ok %s\n' "$*"; } warn() { printf ' WARN %s\n' "$*"; } die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; } jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; } running_pid() { pgrep -f "$PATTERN" || true; } # `launchctl list