#!/usr/bin/env bash # # Rebuild and restart the bridged daemon. # # A merge is not a deployment: the running daemon holds the jar it was started with, so code merged # to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon". # # This script exists to turn five remembered traps into one auditable command: # # 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped. # 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty: # WORKER_GITEA_TOKEN (workers cannot open a PR) and AI_GATEWAY_TOKEN (401 at llm.ltms.dev). # Both are read from the DAEMON's own environment at spawn time, so a value added to # secrets.sh after startup is absent. Nothing logs this, so the script checks and says so. # 3. An old daemon that never actually died looks identical from the outside, so the script waits # for the process to exit and for the port to free before it starts a new one. # 4. "It started" is not "it works": the script polls /healthz until it answers, and reports the # herdr protocol number, because healthz can be green while every spawn fails on a protocol # mismatch. # 5. Restarting under live members drops their tickets, so the script refuses unless you confirm # the fleet is drained. # # Usage: # scripts/redeploy-bridged.sh # build, confirm, restart, verify # scripts/redeploy-bridged.sh --yes # skip the drain confirmation (fleet already checked) # scripts/redeploy-bridged.sh --no-build # restart the jar already on disk # scripts/redeploy-bridged.sh --check # report state and exit; changes nothing # # Exits non-zero on any failure. A failed build never stops the running daemon. set -euo pipefail REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" BRIDGED="$REPO/bridged" JAR="$BRIDGED/target/bridged.jar" OUT="$BRIDGED/bridged.out" # Matches BOTH the absolute form and the relative `java -jar target/bridged.jar` a hand-start # produces from inside bridged/. Anchoring on the absolute path alone was a real bug: the daemon # restarted correctly and the script still reported "no process appeared", because it launched with # a relative path and then looked for an absolute one. PATTERN='target/bridged.jar' HEALTH='http://127.0.0.1:8765/healthz' STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0 for arg in "$@"; do case "$arg" in --yes|-y) ASSUME_YES=1 ;; --no-build) DO_BUILD=0 ;; --check) CHECK_ONLY=1 ;; -h|--help) sed -n '3,30p' "${BASH_SOURCE[0]}"; exit 0 ;; *) echo "unknown option: $arg (try --help)" >&2; exit 2 ;; esac done say() { printf '\n\033[1m== %s\033[0m\n' "$*"; } ok() { printf ' ok %s\n' "$*"; } warn() { printf ' WARN %s\n' "$*"; } die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; } jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; } running_pid() { pgrep -f "$PATTERN" || true; } # ---------------------------------------------------------------- report state say "current state" OLD_PID="$(running_pid)" if [ -n "$OLD_PID" ]; then ok "daemon running, pid $OLD_PID" else warn "no daemon running — this will be a cold start" fi ok "jar on disk: $(jar_id) ($([ -f "$JAR" ] && date -r "$JAR" '+%Y-%m-%d %H:%M:%S' || echo 'none'))" ok "HEAD: $(git -C "$REPO" log --oneline -1)" # The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started # below. Never prints the value — only whether it resolved. if zsh -lc '[ -n "${WORKER_GITEA_TOKEN:-}" ]' 2>/dev/null; then ok "WORKER_GITEA_TOKEN resolves in a login shell" else warn "WORKER_GITEA_TOKEN is EMPTY in a login shell." warn "The daemon will start fine and workers will silently fail to open PRs." warn "Fix \${SHARED_ENV}/tools/secrets.sh before relying on worker checkpoints." fi # Same trap, second variable (CB-591). A profile's `tokenEnv:` is resolved from the DAEMON's own # process environment by HerdrPeerLauncher.resolveEnv, so a token added to secrets.sh after the # daemon started is simply absent. The launcher then injects an empty token and llm.ltms.dev answers # 401 — long after the restart, and with nothing tying the two together. if zsh -lc '[ -n "${AI_GATEWAY_TOKEN:-}" ]' 2>/dev/null; then ok "AI_GATEWAY_TOKEN resolves in a login shell" else warn "AI_GATEWAY_TOKEN is EMPTY in a login shell." warn "Any profile whose tokenEnv is AI_GATEWAY_TOKEN will get an empty token and 401 at the gateway." warn "This only matters once a profile points at llm.ltms.dev — harmless before that." fi if [ "$CHECK_ONLY" = 1 ]; then say "--check: nothing changed" exit 0 fi # ---------------------------------------------------------------------- build # Deliberately before the stop: a failed build must never leave the fleet down. if [ "$DO_BUILD" = 1 ]; then say "build" BUILD_LOG="$(mktemp -t bridged-build)" echo " log: $BUILD_LOG" if ! mvn -f "$BRIDGED/pom.xml" clean install > "$BUILD_LOG" 2>&1; then grep -E 'ERROR|BUILD FAILURE|Tests run:.*Failures: [1-9]|Tests run:.*Errors: [1-9]' "$BUILD_LOG" \ | head -20 || true die "build failed — the running daemon was NOT touched. Full log: $BUILD_LOG" fi grep -E '^\[INFO\] Tests run:.*Failures' "$BUILD_LOG" | tail -1 | sed 's/^\[INFO\] / /' || true ok "BUILD SUCCESS" ok "jar now: $(jar_id)" else say "build skipped (--no-build)" fi [ -f "$JAR" ] || die "no jar at $JAR — run without --no-build" # ----------------------------------------------------------------- drain gate if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then say "drain check" echo " A restart drops every in-flight ticket and rendezvous. A member's report" echo " is NOT recoverable once its ticket is gone." echo echo " Confirm with bridge_list that no members are live, and bridge_poll anything" echo " you still want, BEFORE continuing." echo read -r -p " Fleet drained? type yes to restart: " reply [ "$reply" = "yes" ] || die "aborted — nothing changed" fi # ------------------------------------------------------------------ stop if [ -n "$OLD_PID" ]; then say "stop" RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" # verify a FRESH line appears later kill "$OLD_PID" for _ in $(seq "$STOP_WAIT"); do [ -z "$(running_pid)" ] && break sleep 1 done if [ -n "$(running_pid)" ]; then die "pid $OLD_PID still alive after ${STOP_WAIT}s. Not escalating to kill -9 automatically: the shutdown hook releases sessions and worktrees in order, and killing it hard can leave worktrees and panes behind. Investigate, then kill -9 by hand if you accept that." fi ok "pid $OLD_PID exited" else RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" fi # ------------------------------------------------------------------ start # Login shell (zsh -l) is what puts the secrets on the daemon's environment. cwd must be bridged/ # because the daemon resolves bridged.yaml, logs/ and target/ relative to it. say "start" # Absolute jar path so `ps` names which checkout is running; cwd still bridged/ because the daemon # resolves bridged.yaml, logs/ and target/ relative to it. ( cd "$BRIDGED" && zsh -lc "nohup java -jar '$JAR' >> bridged.out 2>&1 &" ) for _ in $(seq 10); do NEW_PID="$(running_pid)" [ -n "$NEW_PID" ] && break sleep 1 done [ -n "${NEW_PID:-}" ] || die "no process appeared. Last lines of $OUT: $(tail -20 "$OUT" 2>/dev/null)" [ "$NEW_PID" != "${OLD_PID:-}" ] || die "pid unchanged ($NEW_PID) — the old daemon never died" ok "started, pid $NEW_PID" # ------------------------------------------------------------------ verify say "verify" HEALTH_BODY="" for _ in $(seq "$HEALTH_WAIT"); do if HEALTH_BODY="$(curl -fsS --max-time 2 "$HEALTH" 2>/dev/null)"; then break; fi HEALTH_BODY="" sleep 1 done if [ -z "$HEALTH_BODY" ]; then # 503 still means the daemon is up — it means herdr is unreachable. Say which. CODE="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "$HEALTH" 2>/dev/null || echo 000)" if [ "$CODE" = "503" ]; then warn "/healthz answers 503 degraded — the daemon is up but herdr is unreachable." warn "Spawns will fail. Check herdr before delegating anything." curl -s --max-time 2 "$HEALTH" 2>/dev/null | head -3 || true else die "/healthz never answered within ${HEALTH_WAIT}s (last code: $CODE). Last lines of $OUT: $(tail -30 "$OUT" 2>/dev/null)" fi else ok "/healthz 200 — $HEALTH_BODY" warn "healthz green only proves herdr ANSWERS. If its protocol number changed, spawns can still" warn "fail — prove a real spawn before trusting the fleet." fi # A fresh listening line, strictly after the restart mark. An old daemon that never died would # otherwise let an old line pass for a new one. if tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -q 'bridged listening'; then ok "$(tail -n "+$((RESTART_MARK + 1))" "$OUT" | grep 'bridged listening' | tail -1)" else warn "no fresh 'bridged listening' line after the restart — check $OUT yourself" fi # Config keys the daemon accepted or deferred at boot. This is usually WHY you restarted. say "config at boot" tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null \ | grep -iE 'deferred|classification:|fleet health:|coverage' | tail -8 | sed 's/^/ /' \ || echo " (nothing reported)" # Errors since the restart, anchored to the marker so old noise cannot leak in. ERRS="$(tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -cE ' (ERROR|SEVERE) ' || true)" say "result" ok "pid $NEW_PID, jar $(jar_id)" if [ "${ERRS:-0}" -gt 0 ]; then warn "$ERRS ERROR lines since restart:" tail -n "+$((RESTART_MARK + 1))" "$OUT" | grep -E ' (ERROR|SEVERE) ' | tail -5 | sed 's/^/ /' else ok "no ERROR lines since restart" fi echo echo " Next: call bridge_whoami and confirm it still answers 'primary'. A lead whose tab label" echo " no longer matches fleet.leaders.*.tab is demoted to worker and refuses orchestration." echo