428 lines
22 KiB
Bash
Executable File
428 lines
22 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
#
|
|
# Rebuild and restart the fleetd daemon.
|
|
#
|
|
# A merge is not a deployment: the running daemon holds the jar it was started with, so code merged
|
|
# to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon".
|
|
#
|
|
# This script exists to turn six remembered traps into one auditable command:
|
|
#
|
|
# 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped.
|
|
# 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty:
|
|
# WORKER_GITEA_TOKEN (workers cannot open a PR) and AI_GATEWAY_TOKEN (401 at llm.ltms.dev).
|
|
# Both are read from the DAEMON's own environment at spawn time, so a value added to
|
|
# secrets.sh after startup is absent. Nothing logs this here, so the script checks and says
|
|
# so — and since CB-594, fleetd's own startup log says so too, by env var name.
|
|
# 3. An old daemon that never actually died looks identical from the outside, so the script waits
|
|
# for the process to exit and for the port to free before it starts a new one.
|
|
# 4. "It started" is not "it works": the script polls /healthz until it answers, and reports the
|
|
# herdr protocol number, because healthz can be green while every spawn fails on a protocol
|
|
# mismatch.
|
|
# 5. Restarting under live members drops their tickets, so the script refuses unless you confirm
|
|
# the fleet is drained.
|
|
# 6. CB-594 — the launchd agent (deploy/dev.ltms.fleetd.plist), if installed and loaded, is a
|
|
# SECOND supervisor: its KeepAlive.SuccessfulExit=false restarts the daemon on any nonzero
|
|
# exit, and a bare SIGTERM makes this JVM exit 143 even with its shutdown hook running to
|
|
# completion (measured — see the CB-594 report). A plain `kill` here would race launchd's own
|
|
# restart of the OLD jar. So this script detects whether the agent is loaded and, only then,
|
|
# swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control
|
|
# at any moment is whichever one you asked to act, never both.
|
|
#
|
|
# Usage:
|
|
# scripts/redeploy-fleetd.sh # build, confirm, restart, verify
|
|
# scripts/redeploy-fleetd.sh --yes # skip the drain confirmation (fleet already checked)
|
|
# scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk
|
|
# scripts/redeploy-fleetd.sh --check # report state and exit; changes nothing
|
|
#
|
|
# Exits non-zero on any failure. A failed build never stops the running daemon.
|
|
|
|
set -euo pipefail
|
|
|
|
REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
MODULE="$REPO/fleetd"
|
|
JAR="$MODULE/target/fleetd.jar"
|
|
OUT="$MODULE/fleetd.out"
|
|
# Matches BOTH the absolute form and the relative `java -jar target/fleetd.jar` a hand-start
|
|
# produces from inside fleetd/. Anchoring on the absolute path alone was a real bug: the daemon
|
|
# restarted correctly and the script still reported "no process appeared", because it launched with
|
|
# a relative path and then looked for an absolute one.
|
|
PATTERN='target/fleetd.jar'
|
|
HEALTH='http://127.0.0.1:8765/healthz'
|
|
STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure
|
|
HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start
|
|
|
|
# CB-594: the launchd agent this script must not fight with (see trap 6 above).
|
|
LAUNCHD_LABEL='dev.ltms.fleetd'
|
|
LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist"
|
|
|
|
DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0
|
|
for arg in "$@"; do
|
|
case "$arg" in
|
|
--yes|-y) ASSUME_YES=1 ;;
|
|
--no-build) DO_BUILD=0 ;;
|
|
--check) CHECK_ONLY=1 ;;
|
|
-h|--help) sed -n '3,37p' "${BASH_SOURCE[0]}"; exit 0 ;;
|
|
*) echo "unknown option: $arg (try --help)" >&2; exit 2 ;;
|
|
esac
|
|
done
|
|
|
|
say() { printf '\n\033[1m== %s\033[0m\n' "$*"; }
|
|
ok() { printf ' ok %s\n' "$*"; }
|
|
warn() { printf ' WARN %s\n' "$*"; }
|
|
die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; }
|
|
|
|
jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; }
|
|
running_pid() { pgrep -f "$PATTERN" || true; }
|
|
# `launchctl list <label>` exits 0 iff the label is loaded (registered with launchd) — true whether
|
|
# or not it is currently running, which is exactly "supervision is active" for our purposes. Read-
|
|
# only: neither helper below changes anything, so both are also safe under --check.
|
|
launchd_installed() { [ -f "$LAUNCHD_PLIST" ]; }
|
|
launchd_loaded() { launchctl list "$LAUNCHD_LABEL" >/dev/null 2>&1; }
|
|
|
|
# CB-600: the script computes its own log path from where it sits on disk (REPO, above); the
|
|
# plist hard-codes an absolute StandardOutPath. Nothing forced the two to agree — if this script
|
|
# were ever run from a checkout other than the one the loaded plist names, launchd would start and
|
|
# log the daemon correctly, while every check below (the fresh "fleetd listening" line, the
|
|
# ERROR-count scan) would read a different, empty or stale file and the script would report a
|
|
# clean restart while the daemon crash-loops. Pure and side-effect-free besides `die`/`ok` — reads
|
|
# the two paths, resolves them, compares — so it never touches launchd or the daemon and can be
|
|
# exercised by sourcing this script (see the SOURCED guard below) without installing the agent.
|
|
check_log_path_matches_plist() {
|
|
local script_out="$1" plist_path="$2"
|
|
local plist_out resolved_out resolved_plist_out
|
|
# Checked by exit status, not by emptiness: on a missing file/key PlistBuddy exits nonzero but
|
|
# still writes a message ("File Doesn't Exist, Will Create: ...") that command substitution
|
|
# would happily capture as if it were the real value — testing only `-z` missed that case.
|
|
if ! plist_out="$(/usr/libexec/PlistBuddy -c 'Print :StandardOutPath' "$plist_path" 2>/dev/null)" \
|
|
|| [ -z "$plist_out" ]; then
|
|
die "launchd agent is loaded but PlistBuddy could not read StandardOutPath from
|
|
$plist_path
|
|
— cannot verify the daemon logs where this script is about to look. Fix the plist before
|
|
redeploying supervised."
|
|
fi
|
|
resolved_out="$(cd "$(dirname "$script_out")" 2>/dev/null && pwd -P)/$(basename "$script_out")" || true
|
|
resolved_plist_out="$(cd "$(dirname "$plist_out")" 2>/dev/null && pwd -P)/$(basename "$plist_out")" || true
|
|
if [ -z "$resolved_out" ] || [ -z "$resolved_plist_out" ] || [ "$resolved_out" != "$resolved_plist_out" ]; then
|
|
die "log path mismatch — this script reads
|
|
$script_out (resolved: ${resolved_out:-<directory does not exist>})
|
|
but the loaded plist's StandardOutPath is
|
|
$plist_out (resolved: ${resolved_plist_out:-<directory does not exist>})
|
|
Under supervision the daemon writes to the PLIST's path, not necessarily this script's — every
|
|
post-restart check below (the fresh 'fleetd listening' line, the ERROR-count scan) would read
|
|
the wrong file and could report a clean restart while the daemon crash-loops. Fix the mismatch
|
|
(move this checkout to match the plist, or edit the plist's StandardOutPath/StandardErrorPath)
|
|
before redeploying supervised."
|
|
fi
|
|
ok "log path check: script and plist agree ($resolved_out)"
|
|
}
|
|
|
|
# Classify ERROR lines in one fresh log region. AMQP failure messages now include the connection
|
|
# name, so a recovery can clear only errors for its own connection. A candidate with neither name
|
|
# remains unexplained: it must never be quieted by a recovery on the other connection.
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line pending_inbox=0 pending_lead_mailbox=0
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*|*'AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery!'*)
|
|
pending_inbox=$((pending_inbox + 1))
|
|
;;
|
|
*'AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred'*|*'AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!'*)
|
|
pending_lead_mailbox=$((pending_lead_mailbox + 1))
|
|
;;
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*'AMQP connection recovered; cleared held replies for fresh redelivery'*)
|
|
if [ "$pending_inbox" -gt 0 ]; then
|
|
pending_inbox=$((pending_inbox - 1))
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
fi
|
|
;;
|
|
*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*)
|
|
if [ "$pending_lead_mailbox" -gt 0 ]; then
|
|
pending_lead_mailbox=$((pending_lead_mailbox - 1))
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
fi
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
|
|
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending_inbox + pending_lead_mailbox))
|
|
}
|
|
|
|
# CB-600: sourceable for testing. When this file is SOURCED (not executed) it stops here — nothing
|
|
# below runs — so a test harness can `source` it to call check_log_path_matches_plist (or the
|
|
# other pure helpers above) against a throwaway plist fixture without ever reaching the mutating
|
|
# flow (build/stop/start) or touching the real daemon or launchd. On a normal `./redeploy-fleetd.sh`
|
|
# invocation `(return 0 2>/dev/null)` fails (return is illegal at top level of an executed script),
|
|
# so this whole block is a no-op and every line below still runs exactly as before.
|
|
if (return 0 2>/dev/null); then
|
|
return 0
|
|
fi
|
|
|
|
# ---------------------------------------------------------------- report state
|
|
|
|
say "current state"
|
|
OLD_PID="$(running_pid)"
|
|
if [ -n "$OLD_PID" ]; then
|
|
ok "daemon running, pid $OLD_PID"
|
|
else
|
|
warn "no daemon running — this will be a cold start"
|
|
fi
|
|
ok "jar on disk: $(jar_id) ($([ -f "$JAR" ] && date -r "$JAR" '+%Y-%m-%d %H:%M:%S' || echo 'none'))"
|
|
ok "HEAD: $(git -C "$REPO" log --oneline -1)"
|
|
|
|
# CB-594: supervision state. Installed and loaded are different facts — a copied-but-never-loaded
|
|
# plist supervises nothing, and a loaded label with no file backing it (rare, but possible after an
|
|
# edited/moved plist) is still what launchd will act on.
|
|
if launchd_installed; then
|
|
ok "launchd agent installed: $LAUNCHD_PLIST"
|
|
else
|
|
warn "launchd agent NOT installed (no supervision — a crash will not restart the daemon)."
|
|
fi
|
|
SUPERVISED=0
|
|
if launchd_loaded; then
|
|
SUPERVISED=1
|
|
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
|
# CB-600: fail loudly here, before ANY other check runs, if this script and the loaded plist
|
|
# would read different log files — every check after this point is worthless otherwise.
|
|
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
|
else
|
|
warn "launchd agent not loaded — this script is the only thing that will restart the daemon."
|
|
fi
|
|
|
|
# The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started
|
|
# below. Never prints the value — only whether it resolved.
|
|
if zsh -lc '[ -n "${WORKER_GITEA_TOKEN:-}" ]' 2>/dev/null; then
|
|
ok "WORKER_GITEA_TOKEN resolves in a login shell"
|
|
else
|
|
warn "WORKER_GITEA_TOKEN is EMPTY in a login shell."
|
|
warn "The daemon will start fine and workers will silently fail to open PRs."
|
|
warn "Fix \${SHARED_ENV}/tools/secrets.sh before relying on worker checkpoints."
|
|
fi
|
|
|
|
# Same trap, second variable (CB-591). A profile's `tokenEnv:` is resolved from the DAEMON's own
|
|
# process environment by HerdrPeerLauncher.resolveEnv, so a token added to secrets.sh after the
|
|
# daemon started is simply absent. The launcher then injects an empty token and llm.ltms.dev answers
|
|
# 401 — long after the restart, and with nothing tying the two together.
|
|
if zsh -lc '[ -n "${AI_GATEWAY_TOKEN:-}" ]' 2>/dev/null; then
|
|
ok "AI_GATEWAY_TOKEN resolves in a login shell"
|
|
else
|
|
warn "AI_GATEWAY_TOKEN is EMPTY in a login shell."
|
|
warn "Any profile whose tokenEnv is AI_GATEWAY_TOKEN will get an empty token and 401 at the gateway."
|
|
warn "This only matters once a profile points at llm.ltms.dev — harmless before that."
|
|
fi
|
|
|
|
# Third variable, same trap (CB-635). broker.uriEnv names the env var holding the AMQP URI, so the
|
|
# password stays out of fleetd.yaml — but that moves the failure into the environment. If the
|
|
# variable is empty the daemon still starts: since #152 it warns and falls back to the in-memory
|
|
# reply inbox, so nothing crashes and replies simply stop surviving a restart. Only this check says
|
|
# so before the fact. Read the name out of the config so a renamed key cannot make the check lie.
|
|
BROKER_URI_ENV=$(sed -n 's/^[[:space:]]*uriEnv:[[:space:]]*\([A-Za-z_][A-Za-z0-9_]*\).*/\1/p' "$MODULE/fleetd.yaml" | head -1)
|
|
if [ -z "$BROKER_URI_ENV" ]; then
|
|
ok "no broker.uriEnv configured — reply inbox is in-memory by design"
|
|
elif zsh -lc "[ -n \"\${$BROKER_URI_ENV:-}\" ]" 2>/dev/null; then
|
|
ok "$BROKER_URI_ENV (broker.uriEnv) resolves in a login shell"
|
|
else
|
|
warn "$BROKER_URI_ENV (broker.uriEnv) is EMPTY in a login shell."
|
|
warn "The daemon will start and fall back to the IN-MEMORY reply inbox."
|
|
warn "Replies stop surviving a restart — a held report is lost, not delayed."
|
|
fi
|
|
|
|
if [ "$CHECK_ONLY" = 1 ]; then
|
|
say "--check: nothing changed"
|
|
exit 0
|
|
fi
|
|
|
|
# ---------------------------------------------------------------------- build
|
|
# Deliberately before the stop: a failed build must never leave the fleet down.
|
|
|
|
if [ "$DO_BUILD" = 1 ]; then
|
|
say "build"
|
|
BUILD_LOG="$(mktemp -t fleetd-build)"
|
|
echo " log: $BUILD_LOG"
|
|
if ! mvn -f "$MODULE/pom.xml" clean install > "$BUILD_LOG" 2>&1; then
|
|
grep -E 'ERROR|BUILD FAILURE|Tests run:.*Failures: [1-9]|Tests run:.*Errors: [1-9]' "$BUILD_LOG" \
|
|
| head -20 || true
|
|
die "build failed — the running daemon was NOT touched. Full log: $BUILD_LOG"
|
|
fi
|
|
grep -E '^\[INFO\] Tests run:.*Failures' "$BUILD_LOG" | tail -1 | sed 's/^\[INFO\] / /' || true
|
|
ok "BUILD SUCCESS"
|
|
ok "jar now: $(jar_id)"
|
|
else
|
|
say "build skipped (--no-build)"
|
|
fi
|
|
|
|
[ -f "$JAR" ] || die "no jar at $JAR — run without --no-build"
|
|
|
|
# ----------------------------------------------------------------- drain gate
|
|
|
|
if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
|
say "drain check"
|
|
echo " A restart drops every in-flight ticket and rendezvous. A member's report"
|
|
echo " is NOT recoverable once its ticket is gone."
|
|
echo
|
|
echo " Confirm with fleet_list that no members are live, and fleet_poll anything"
|
|
echo " you still want, BEFORE continuing."
|
|
echo
|
|
read -r -p " Fleet drained? type yes to restart: " reply
|
|
[ "$reply" = "yes" ] || die "aborted — nothing changed"
|
|
fi
|
|
|
|
# ------------------------------------------------------------------ stop
|
|
#
|
|
# CB-594: when SUPERVISED, launchd owns the stop — never a raw `kill` here. A bare SIGTERM makes
|
|
# this JVM exit 143 even with its shutdown hook running to completion (verified separately: a
|
|
# throwaway Java process with an equivalent shutdown hook, sent SIGTERM from a login shell that
|
|
# could `wait` on it directly, reported exit code 143 every time — never 0). launchd's
|
|
# KeepAlive.SuccessfulExit=false treats any nonzero exit as a crash and restarts the OLD jar,
|
|
# which would race this script's own restart of the NEW one. `launchctl unload` avoids that race
|
|
# by deregistering the job first, so no KeepAlive is left armed when the process actually stops.
|
|
|
|
if [ -n "$OLD_PID" ]; then
|
|
say "stop"
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" # verify a FRESH line appears later
|
|
if [ "$SUPERVISED" = 1 ]; then
|
|
echo " supervision is ON: using 'launchctl unload' (not kill) so launchd's own KeepAlive"
|
|
echo " cannot restart the OLD jar out from under this script — see the CB-594 comment above."
|
|
launchctl unload -w "$LAUNCHD_PLIST" \
|
|
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
|
|
else
|
|
kill "$OLD_PID"
|
|
fi
|
|
for _ in $(seq "$STOP_WAIT"); do
|
|
[ -z "$(running_pid)" ] && break
|
|
sleep 1
|
|
done
|
|
if [ -n "$(running_pid)" ]; then
|
|
die "pid $OLD_PID still alive after ${STOP_WAIT}s. Not escalating to kill -9 automatically:
|
|
the shutdown hook releases sessions and worktrees in order, and killing it hard can
|
|
leave worktrees and panes behind. Investigate, then kill -9 by hand if you accept that."
|
|
fi
|
|
ok "pid $OLD_PID exited"
|
|
elif [ "$SUPERVISED" = 1 ]; then
|
|
# Loaded but not currently running (e.g. throttled after a crash loop). Unload it anyway so the
|
|
# start step below does a clean load, never a load stacked on an already-loaded label.
|
|
say "stop"
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
|
launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true
|
|
ok "launchd agent unloaded (was already not running)"
|
|
else
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
|
fi
|
|
|
|
# ------------------------------------------------------------------ start
|
|
# Unsupervised: login shell (zsh -l) is what puts the secrets on the daemon's environment, and cwd
|
|
# must be fleetd/ because the daemon resolves fleetd.yaml, logs/ and target/ relative to it.
|
|
# Supervised: launchd does both — deploy/dev.ltms.fleetd.plist points ProgramArguments at
|
|
# scripts/fleetd-launchd-wrapper.sh (CB-594), which is what execs the login shell in launchd's
|
|
# place, and WorkingDirectory in the plist already pins fleetd/.
|
|
|
|
say "start"
|
|
if [ "$SUPERVISED" = 1 ]; then
|
|
echo " supervision is ON: using 'launchctl load' so launchd starts and keeps supervising this"
|
|
echo " process, instead of a manual nohup that launchd would know nothing about."
|
|
# CB-600: 'launchctl unload -w' above already persisted Disabled=true for this label. A load -w
|
|
# that succeeds clears it; a load -w that FAILS leaves the agent both stopped and disabled — worse
|
|
# than before this script ran, because a later reboot or login will not bring it back either. One
|
|
# retry covers a transient race (e.g. launchd not yet fully done deregistering); if it still fails,
|
|
# die with the exact recovery command rather than a bare "failed".
|
|
if ! launchctl load -w "$LAUNCHD_PLIST" 2>/dev/null; then
|
|
warn "launchctl load failed on the first attempt — retrying once after a short pause"
|
|
sleep 2
|
|
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed twice.
|
|
The agent is now STOPPED and DISABLED — it will NOT come back on its own, not even after a
|
|
reboot or login, because 'launchctl unload -w' above persisted Disabled=true and load -w
|
|
never got the chance to clear it. Recover with:
|
|
launchctl load -w \"$LAUNCHD_PLIST\"
|
|
If that still fails, check 'launchctl list $LAUNCHD_LABEL', validate the plist with
|
|
'plutil -lint \"$LAUNCHD_PLIST\"', and check $OUT before assuming a retry will succeed."
|
|
fi
|
|
else
|
|
# Absolute jar path so `ps` names which checkout is running.
|
|
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
|
fi
|
|
|
|
for _ in $(seq 10); do
|
|
NEW_PID="$(running_pid)"
|
|
[ -n "$NEW_PID" ] && break
|
|
sleep 1
|
|
done
|
|
[ -n "${NEW_PID:-}" ] || die "no process appeared. Last lines of $OUT:
|
|
$(tail -20 "$OUT" 2>/dev/null)"
|
|
[ "$NEW_PID" != "${OLD_PID:-}" ] || die "pid unchanged ($NEW_PID) — the old daemon never died"
|
|
ok "started, pid $NEW_PID"
|
|
|
|
# ------------------------------------------------------------------ verify
|
|
|
|
say "verify"
|
|
|
|
HEALTH_BODY=""
|
|
for _ in $(seq "$HEALTH_WAIT"); do
|
|
if HEALTH_BODY="$(curl -fsS --max-time 2 "$HEALTH" 2>/dev/null)"; then break; fi
|
|
HEALTH_BODY=""
|
|
sleep 1
|
|
done
|
|
|
|
if [ -z "$HEALTH_BODY" ]; then
|
|
# 503 still means the daemon is up — it means herdr is unreachable. Say which.
|
|
CODE="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "$HEALTH" 2>/dev/null || echo 000)"
|
|
if [ "$CODE" = "503" ]; then
|
|
warn "/healthz answers 503 degraded — the daemon is up but herdr is unreachable."
|
|
warn "Spawns will fail. Check herdr before delegating anything."
|
|
curl -s --max-time 2 "$HEALTH" 2>/dev/null | head -3 || true
|
|
else
|
|
die "/healthz never answered within ${HEALTH_WAIT}s (last code: $CODE). Last lines of $OUT:
|
|
$(tail -30 "$OUT" 2>/dev/null)"
|
|
fi
|
|
else
|
|
ok "/healthz 200 — $HEALTH_BODY"
|
|
warn "healthz green only proves herdr ANSWERS. If its protocol number changed, spawns can still"
|
|
warn "fail — prove a real spawn before trusting the fleet."
|
|
fi
|
|
|
|
# A fresh listening line, strictly after the restart mark. An old daemon that never died would
|
|
# otherwise let an old line pass for a new one.
|
|
if tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -q 'fleetd listening'; then
|
|
ok "$(tail -n "+$((RESTART_MARK + 1))" "$OUT" | grep 'fleetd listening' | tail -1)"
|
|
else
|
|
warn "no fresh 'fleetd listening' line after the restart — check $OUT yourself"
|
|
fi
|
|
|
|
# Config keys the daemon accepted or deferred at boot. This is usually WHY you restarted.
|
|
say "config at boot"
|
|
tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null \
|
|
| grep -iE 'deferred|classification:|fleet health:|coverage' | tail -8 | sed 's/^/ /' \
|
|
|| echo " (nothing reported)"
|
|
|
|
# Errors since the restart, anchored to the marker so old noise cannot leak in. Keep the fresh
|
|
# region in a file because the classifier must preserve the order of errors and recoveries.
|
|
FRESH_LOG="$(mktemp -t fleetd-fresh-log)"
|
|
trap 'rm -f "$FRESH_LOG"' EXIT
|
|
tail -n "+$((RESTART_MARK + 1))" "$OUT" > "$FRESH_LOG" 2>/dev/null || true
|
|
classify_amqp_connection_errors "$FRESH_LOG"
|
|
say "result"
|
|
ok "pid $NEW_PID, jar $(jar_id)"
|
|
if [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then
|
|
ok "no ERROR lines since restart"
|
|
elif [ "$REDEPLOY_UNEXPLAINED_ERRORS" -eq 0 ]; then
|
|
ok "$REDEPLOY_RECOVERED_AMQP_ERRORS AMQP connection reset ERROR lines recovered since restart"
|
|
else
|
|
warn "$REDEPLOY_ERROR_COUNT ERROR lines since restart:"
|
|
grep -E ' (ERROR|SEVERE) ' "$FRESH_LOG" | tail -5 | sed 's/^/ /'
|
|
fi
|
|
echo
|
|
echo " Next: call fleet_whoami and confirm it still answers 'primary'. A lead whose tab label"
|
|
echo " no longer matches fleet.leaders.*.tab is demoted to worker and refuses orchestration."
|
|
echo
|