#!/usr/bin/env bash # # Rebuild and restart the fleetd daemon. # # A merge is not a deployment: the running daemon holds the jar it was started with, so code merged # to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon". # # This script exists to turn six remembered traps into one auditable command: # # 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped. # 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty: # WORKER_GITEA_TOKEN (workers cannot open a PR) and AI_GATEWAY_TOKEN (401 at llm.ltms.dev). # Both are read from the DAEMON's own environment at spawn time, so a value added to # secrets.sh after startup is absent. Nothing logs this here, so the script checks and says # so — and since CB-594, fleetd's own startup log says so too, by env var name. # 3. An old daemon that never actually died looks identical from the outside, so the script waits # for the process to exit and for the port to free before it starts a new one. # 4. "It started" is not "it works": the script polls /healthz until it answers, and reports the # herdr protocol number, because healthz can be green while every spawn fails on a protocol # mismatch. # 5. Restarting under live members drops their tickets, so the script refuses unless you confirm # the fleet is drained. # 6. CB-594 — the launchd agent (deploy/dev.ltms.fleetd.plist), if installed and loaded, is a # SECOND supervisor: its KeepAlive.SuccessfulExit=false restarts the daemon on any nonzero # exit, and a bare SIGTERM makes this JVM exit 143 even with its shutdown hook running to # completion (measured — see the CB-594 report). A plain `kill` here would race launchd's own # restart of the OLD jar. So this script detects whether the agent is loaded and, only then, # swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control # at any moment is whichever one you asked to act, never both. # # Usage: # scripts/redeploy-fleetd.sh # build, confirm, restart, verify # scripts/redeploy-fleetd.sh --yes # skip the drain confirmation (fleet already checked) # scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk # scripts/redeploy-fleetd.sh --check # report state and exit; changes nothing # # Exits non-zero on any failure. A failed build never stops the running daemon. set -euo pipefail REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" MODULE="$REPO/fleetd" JAR="$MODULE/target/fleetd.jar" OUT="$MODULE/fleetd.out" # Matches BOTH the absolute form and the relative `java -jar target/fleetd.jar` a hand-start # produces from inside fleetd/. Anchoring on the absolute path alone was a real bug: the daemon # restarted correctly and the script still reported "no process appeared", because it launched with # a relative path and then looked for an absolute one. PATTERN='target/fleetd.jar' HEALTH='http://127.0.0.1:8765/healthz' STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start # CB-594: the launchd agent this script must not fight with (see trap 6 above). LAUNCHD_LABEL='dev.ltms.fleetd' LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist" DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0 for arg in "$@"; do case "$arg" in --yes|-y) ASSUME_YES=1 ;; --no-build) DO_BUILD=0 ;; --check) CHECK_ONLY=1 ;; -h|--help) sed -n '3,37p' "${BASH_SOURCE[0]}"; exit 0 ;; *) echo "unknown option: $arg (try --help)" >&2; exit 2 ;; esac done say() { printf '\n\033[1m== %s\033[0m\n' "$*"; } ok() { printf ' ok %s\n' "$*"; } warn() { printf ' WARN %s\n' "$*"; } die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; } jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; } running_pid() { pgrep -f "$PATTERN" || true; } # `launchctl list