#!/usr/bin/env bash # # Rebuild and restart the fleetd daemon. # # A merge is not a deployment: the running daemon holds the jar it was started with, so code merged # to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon". # # This script exists to turn eight remembered traps into one auditable command: # # 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped. # 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty: # WORKER_GITEA_TOKEN (workers cannot open a PR) and AI_GATEWAY_TOKEN (401 at llm.ltms.dev). # Both are read from the DAEMON's own environment at spawn time, so a value added to # secrets.sh after startup is absent. Nothing logs this here, so the script checks and says # so — and since CB-594, fleetd's own startup log says so too, by env var name. # 3. An old daemon that never actually died looks identical from the outside, so the script waits # for the process to exit and for the port to free before it starts a new one. # 4. "It started" is not "it works": the script polls /healthz until it answers, and reports the # herdr protocol number, because healthz can be green while every spawn fails on a protocol # mismatch. # 5. Restarting under live members drops their tickets, so the script refuses unless you confirm # the fleet is drained. # 6. CB-594 — the launchd agent (deploy/dev.ltms.fleetd.plist), if installed and loaded, is a # SECOND supervisor: its KeepAlive.SuccessfulExit=false restarts the daemon on any nonzero # exit, and a bare SIGTERM makes this JVM exit 143 even with its shutdown hook running to # completion (measured — see the CB-594 report). A plain `kill` here would race launchd's own # restart of the OLD jar. So this script detects whether the agent is loaded and, only then, # swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control # at any moment is whichever one you asked to act, never both. # 7. fleetd #492 — a systemd --user unit is a THIRD possible supervisor (seen on a second host): # Restart=on-failure treats this JVM's SIGTERM exit code (143, per CB-594 above) as a failure # too, so a bare `kill` there would race systemd's own restart of the OLD jar exactly like # launchd would. This script now tells launchd, systemd, and "genuinely unsupervised" apart as # three different answers, drives whichever one it finds through its own control plane # (`launchctl` / `systemctl --user`), and REFUSES outright — never falls back to `kill` — when # it finds a supervision signal it cannot map to exactly one of the two it knows how to drive. # A wrong guess here is how two daemons end up running against one herdr session. Follow-up: # "not currently loaded" is not the same fact as "unsupervised" — a unit that is installed but # activating/failed/pending-restart, or a `systemctl` call that could not answer at all (e.g. # no user-bus access), both now read as a fifth answer, "unclear", and REFUSE the same way # "ambiguous" does, rather than silently falling through to "none". # 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if # more than one is alive. That is the one thing none of the checks above (healthz 200, jar id, # the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon. # # Usage: # scripts/redeploy-fleetd.sh # build, confirm, restart, verify # scripts/redeploy-fleetd.sh --yes # skip the drain confirmation (fleet already checked) # scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk # scripts/redeploy-fleetd.sh --check # report state and exit; changes nothing # # Exits non-zero on any failure. A failed build never stops the running daemon. set -euo pipefail REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" MODULE="$REPO/fleetd" JAR="$MODULE/target/fleetd.jar" # fleetd #493: never build into the path a running process holds. The build writes here first # (Maven's shade plugin has finalName=fleetd, so `clean install` still lands its output at # target/fleetd.jar — that part is unchanged and out of this script's control), but this script # now moves it out to JAR_STAGED immediately, and only swaps it back to JAR (a plain `mv`, so a # rename, never a byte-by-byte overwrite) after the OLD daemon has been confirmed exited. See # stage_built_jar/swap_staged_jar below. JAR_STAGED="$MODULE/target/fleetd-new.jar" OUT="$MODULE/fleetd.out" # Matches BOTH the absolute form and the relative `java -jar target/fleetd.jar` a hand-start # produces from inside fleetd/. Anchoring on the absolute path alone was a real bug: the daemon # restarted correctly and the script still reported "no process appeared", because it launched with # a relative path and then looked for an absolute one. PATTERN='target/fleetd.jar' HEALTH='http://127.0.0.1:8765/healthz' STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start # CB-594: the launchd agent this script must not fight with (see trap 6 above). LAUNCHD_LABEL='dev.ltms.fleetd' LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist" # fleetd #492: the systemd --user unit this script must not fight with either (see trap 7 above). # Measured on the second host: `systemctl --user cat fleetd` names the unit "fleetd" (not # "dev.ltms.fleetd" — systemd user units here are not namespaced the way the launchd label is). SYSTEMD_UNIT='fleetd' # fleetd #492 follow-up: detect_supervisor packs TWO values (kind, detail) onto the one stdout # line that survives its $(...) call — see the constraints comment above that function. This is # the separator between them: the ASCII "unit separator" byte, chosen because it never occurs in # any of the prose detail strings and needs no escaping in a `case`/glob pattern. SUPERVISOR_DETAIL_SEP=$'\x1f' # fleetd #492 follow-up: set by systemd_loaded/systemd_installed when the underlying `systemctl` # call could not answer cleanly — it exited non-zero AND wrote something to stderr, which is a real # tool failure (e.g. it cannot reach the user bus over a non-lingering ssh session), never the same # fact as a clean negative answer ("not active", no stderr). Initialized here, not just inside the # probes, so detect_supervisor can read them under `set -u` even before either probe has ever run, # and so a test that stubs a probe with a plain `return 0`/`return 1` body (leaving these untouched) # reads a deterministic 0 rather than whatever a previous probe call left behind. SYSTEMD_LOADED_ERRORED=0 SYSTEMD_INSTALLED_ERRORED=0 # fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL is the specific supervisor/reason that # require_drivable_supervisor's die() names on an "unclear" answer. Deliberately NOT pre-declared # here (unlike the two flags above): it is set only by the real call site, right after it unpacks # detect_supervisor's stdout (see the constraints comment above detect_supervisor). If that call # site is ever skipped or broken, a bare `set -u` reference to this variable in # require_drivable_supervisor must fail loudly with "unbound variable" — a pre-declared empty # default would instead silently print an empty reason, hiding exactly the value this ticket # exists to surface. DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0 for arg in "$@"; do case "$arg" in --yes|-y) ASSUME_YES=1 ;; --no-build) DO_BUILD=0 ;; --check) CHECK_ONLY=1 ;; -h|--help) sed -n '3,48p' "${BASH_SOURCE[0]}"; exit 0 ;; *) echo "unknown option: $arg (try --help)" >&2; exit 2 ;; esac done say() { printf '\n\033[1m== %s\033[0m\n' "$*"; } ok() { printf ' ok %s\n' "$*"; } warn() { printf ' WARN %s\n' "$*"; } die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; } # Reports the hash of $JAR by default, or of whatever path is passed — used to report the STAGED # jar right after a build (before it has been swapped in) without ever changing what a bare # `jar_id` (no args) means: the live path, $JAR. --check and the final "pid ..., jar ..." line # both call it with no args on purpose, so neither can ever be fooled by a leftover staged file. jar_id() { local f="${1:-$JAR}"; [ -f "$f" ] && shasum -a 256 "$f" | cut -c1-12 || echo "absent"; } running_pid() { pgrep -f "$PATTERN" || true; } # fleetd #493 — three small, independently testable pieces of "never build into the path a # running process holds": # # stage_built_jar moves the jar Maven just produced OUT of the live path and onto the staging # path, immediately after a successful build. Dies (leaving the OLD daemon # untouched — this runs before the stop step) if Maven reported success but # left no jar behind, or if the move itself fails. # require_no_build_jar the --no-build path never builds or stages anything: it must find a # jar already sitting at the live path from an earlier successful run, and # die with the same truthful message this script has always used if not. # wait_for_daemon_exit polls running_pid() for up to $1 seconds and reports whether the OLD # daemon actually exited — extracted to its own function so the main flow # can be relied on to call swap_staged_jar only AFTER this returns success, # and so a test can prove that ordering by reading the script's own source. # swap_staged_jar the actual swap: a plain `mv` of the staged jar onto the live path. Called # only once the OLD daemon is confirmed gone (see wait_for_daemon_exit above), # so this is never a write into a path a running process holds — by the time # it runs, nothing holds that path anymore. If it fails, the caller must not # start a new daemon: die() below already refuses that by exiting the script. stage_built_jar() { [ -f "$JAR" ] || die "build succeeded but produced no jar at $JAR — cannot stage it for restart. The running daemon was NOT touched." mv -f "$JAR" "$JAR_STAGED" \ || die "could not move the freshly built jar from $JAR to the staging path $JAR_STAGED. The running daemon was NOT touched." } require_no_build_jar() { [ -f "$JAR" ] || die "no jar at $JAR — run without --no-build" } wait_for_daemon_exit() { local timeout="$1" _i for _i in $(seq "$timeout"); do [ -z "$(running_pid)" ] && return 0 sleep 1 done [ -z "$(running_pid)" ] } swap_staged_jar() { local staged="$1" live="$2" [ -f "$staged" ] || die "no staged jar at $staged to swap in — the daemon was NOT started." mv -f "$staged" "$live" \ || die "could not move the staged jar from $staged into place at $live — the daemon was NOT started. The built jar is still sitting at $staged; a manual 'mv \"$staged\" \"$live\"' may recover this once you find out why the move failed." } # `launchctl list