5fe02b7c98
Found by running it. The script launched the daemon with a relative jar path (cwd is bridged/) but detected it with an absolute one, so pgrep never matched. The daemon restarted correctly and booted clean, and the script still failed with 'no process appeared' — the worst shape of bug for a deploy tool, because it invites a second restart on a daemon that is already healthy. Detection now matches both path forms, and the launch uses the absolute path so ps names which checkout is running.
220 lines
9.1 KiB
Bash
Executable File
220 lines
9.1 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
#
|
|
# Rebuild and restart the bridged daemon.
|
|
#
|
|
# A merge is not a deployment: the running daemon holds the jar it was started with, so code merged
|
|
# to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon".
|
|
#
|
|
# This script exists to turn five remembered traps into one auditable command:
|
|
#
|
|
# 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped.
|
|
# 2. The daemon must start from a LOGIN shell, or WORKER_GITEA_TOKEN is empty and workers cannot
|
|
# open a PR. Nothing in the daemon logs this, so the script checks it and says so out loud.
|
|
# 3. An old daemon that never actually died looks identical from the outside, so the script waits
|
|
# for the process to exit and for the port to free before it starts a new one.
|
|
# 4. "It started" is not "it works": the script polls /healthz until it answers, and reports the
|
|
# herdr protocol number, because healthz can be green while every spawn fails on a protocol
|
|
# mismatch.
|
|
# 5. Restarting under live members drops their tickets, so the script refuses unless you confirm
|
|
# the fleet is drained.
|
|
#
|
|
# Usage:
|
|
# scripts/redeploy-bridged.sh # build, confirm, restart, verify
|
|
# scripts/redeploy-bridged.sh --yes # skip the drain confirmation (fleet already checked)
|
|
# scripts/redeploy-bridged.sh --no-build # restart the jar already on disk
|
|
# scripts/redeploy-bridged.sh --check # report state and exit; changes nothing
|
|
#
|
|
# Exits non-zero on any failure. A failed build never stops the running daemon.
|
|
|
|
set -euo pipefail
|
|
|
|
REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
BRIDGED="$REPO/bridged"
|
|
JAR="$BRIDGED/target/bridged.jar"
|
|
OUT="$BRIDGED/bridged.out"
|
|
# Matches BOTH the absolute form and the relative `java -jar target/bridged.jar` a hand-start
|
|
# produces from inside bridged/. Anchoring on the absolute path alone was a real bug: the daemon
|
|
# restarted correctly and the script still reported "no process appeared", because it launched with
|
|
# a relative path and then looked for an absolute one.
|
|
PATTERN='target/bridged.jar'
|
|
HEALTH='http://127.0.0.1:8765/healthz'
|
|
STOP_WAIT=30 # seconds to wait for a clean exit before reporting failure
|
|
HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start
|
|
|
|
DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0
|
|
for arg in "$@"; do
|
|
case "$arg" in
|
|
--yes|-y) ASSUME_YES=1 ;;
|
|
--no-build) DO_BUILD=0 ;;
|
|
--check) CHECK_ONLY=1 ;;
|
|
-h|--help) sed -n '3,30p' "${BASH_SOURCE[0]}"; exit 0 ;;
|
|
*) echo "unknown option: $arg (try --help)" >&2; exit 2 ;;
|
|
esac
|
|
done
|
|
|
|
say() { printf '\n\033[1m== %s\033[0m\n' "$*"; }
|
|
ok() { printf ' ok %s\n' "$*"; }
|
|
warn() { printf ' WARN %s\n' "$*"; }
|
|
die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; }
|
|
|
|
jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; }
|
|
running_pid() { pgrep -f "$PATTERN" || true; }
|
|
|
|
# ---------------------------------------------------------------- report state
|
|
|
|
say "current state"
|
|
OLD_PID="$(running_pid)"
|
|
if [ -n "$OLD_PID" ]; then
|
|
ok "daemon running, pid $OLD_PID"
|
|
else
|
|
warn "no daemon running — this will be a cold start"
|
|
fi
|
|
ok "jar on disk: $(jar_id) ($([ -f "$JAR" ] && date -r "$JAR" '+%Y-%m-%d %H:%M:%S' || echo 'none'))"
|
|
ok "HEAD: $(git -C "$REPO" log --oneline -1)"
|
|
|
|
# The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started
|
|
# below. Never prints the value — only whether it resolved.
|
|
if zsh -lc '[ -n "${WORKER_GITEA_TOKEN:-}" ]' 2>/dev/null; then
|
|
ok "WORKER_GITEA_TOKEN resolves in a login shell"
|
|
else
|
|
warn "WORKER_GITEA_TOKEN is EMPTY in a login shell."
|
|
warn "The daemon will start fine and workers will silently fail to open PRs."
|
|
warn "Fix \${SHARED_ENV}/tools/secrets.sh before relying on worker checkpoints."
|
|
fi
|
|
|
|
if [ "$CHECK_ONLY" = 1 ]; then
|
|
say "--check: nothing changed"
|
|
exit 0
|
|
fi
|
|
|
|
# ---------------------------------------------------------------------- build
|
|
# Deliberately before the stop: a failed build must never leave the fleet down.
|
|
|
|
if [ "$DO_BUILD" = 1 ]; then
|
|
say "build"
|
|
BUILD_LOG="$(mktemp -t bridged-build)"
|
|
echo " log: $BUILD_LOG"
|
|
if ! mvn -f "$BRIDGED/pom.xml" clean install > "$BUILD_LOG" 2>&1; then
|
|
grep -E 'ERROR|BUILD FAILURE|Tests run:.*Failures: [1-9]|Tests run:.*Errors: [1-9]' "$BUILD_LOG" \
|
|
| head -20 || true
|
|
die "build failed — the running daemon was NOT touched. Full log: $BUILD_LOG"
|
|
fi
|
|
grep -E '^\[INFO\] Tests run:.*Failures' "$BUILD_LOG" | tail -1 | sed 's/^\[INFO\] / /' || true
|
|
ok "BUILD SUCCESS"
|
|
ok "jar now: $(jar_id)"
|
|
else
|
|
say "build skipped (--no-build)"
|
|
fi
|
|
|
|
[ -f "$JAR" ] || die "no jar at $JAR — run without --no-build"
|
|
|
|
# ----------------------------------------------------------------- drain gate
|
|
|
|
if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
|
say "drain check"
|
|
echo " A restart drops every in-flight ticket and rendezvous. A member's report"
|
|
echo " is NOT recoverable once its ticket is gone."
|
|
echo
|
|
echo " Confirm with bridge_list that no members are live, and bridge_poll anything"
|
|
echo " you still want, BEFORE continuing."
|
|
echo
|
|
read -r -p " Fleet drained? type yes to restart: " reply
|
|
[ "$reply" = "yes" ] || die "aborted — nothing changed"
|
|
fi
|
|
|
|
# ------------------------------------------------------------------ stop
|
|
|
|
if [ -n "$OLD_PID" ]; then
|
|
say "stop"
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" # verify a FRESH line appears later
|
|
kill "$OLD_PID"
|
|
for _ in $(seq "$STOP_WAIT"); do
|
|
[ -z "$(running_pid)" ] && break
|
|
sleep 1
|
|
done
|
|
if [ -n "$(running_pid)" ]; then
|
|
die "pid $OLD_PID still alive after ${STOP_WAIT}s. Not escalating to kill -9 automatically:
|
|
the shutdown hook releases sessions and worktrees in order, and killing it hard can
|
|
leave worktrees and panes behind. Investigate, then kill -9 by hand if you accept that."
|
|
fi
|
|
ok "pid $OLD_PID exited"
|
|
else
|
|
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
|
fi
|
|
|
|
# ------------------------------------------------------------------ start
|
|
# Login shell (zsh -l) is what puts the secrets on the daemon's environment. cwd must be bridged/
|
|
# because the daemon resolves bridged.yaml, logs/ and target/ relative to it.
|
|
|
|
say "start"
|
|
# Absolute jar path so `ps` names which checkout is running; cwd still bridged/ because the daemon
|
|
# resolves bridged.yaml, logs/ and target/ relative to it.
|
|
( cd "$BRIDGED" && zsh -lc "nohup java -jar '$JAR' >> bridged.out 2>&1 &" )
|
|
|
|
for _ in $(seq 10); do
|
|
NEW_PID="$(running_pid)"
|
|
[ -n "$NEW_PID" ] && break
|
|
sleep 1
|
|
done
|
|
[ -n "${NEW_PID:-}" ] || die "no process appeared. Last lines of $OUT:
|
|
$(tail -20 "$OUT" 2>/dev/null)"
|
|
[ "$NEW_PID" != "${OLD_PID:-}" ] || die "pid unchanged ($NEW_PID) — the old daemon never died"
|
|
ok "started, pid $NEW_PID"
|
|
|
|
# ------------------------------------------------------------------ verify
|
|
|
|
say "verify"
|
|
|
|
HEALTH_BODY=""
|
|
for _ in $(seq "$HEALTH_WAIT"); do
|
|
if HEALTH_BODY="$(curl -fsS --max-time 2 "$HEALTH" 2>/dev/null)"; then break; fi
|
|
HEALTH_BODY=""
|
|
sleep 1
|
|
done
|
|
|
|
if [ -z "$HEALTH_BODY" ]; then
|
|
# 503 still means the daemon is up — it means herdr is unreachable. Say which.
|
|
CODE="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "$HEALTH" 2>/dev/null || echo 000)"
|
|
if [ "$CODE" = "503" ]; then
|
|
warn "/healthz answers 503 degraded — the daemon is up but herdr is unreachable."
|
|
warn "Spawns will fail. Check herdr before delegating anything."
|
|
curl -s --max-time 2 "$HEALTH" 2>/dev/null | head -3 || true
|
|
else
|
|
die "/healthz never answered within ${HEALTH_WAIT}s (last code: $CODE). Last lines of $OUT:
|
|
$(tail -30 "$OUT" 2>/dev/null)"
|
|
fi
|
|
else
|
|
ok "/healthz 200 — $HEALTH_BODY"
|
|
warn "healthz green only proves herdr ANSWERS. If its protocol number changed, spawns can still"
|
|
warn "fail — prove a real spawn before trusting the fleet."
|
|
fi
|
|
|
|
# A fresh listening line, strictly after the restart mark. An old daemon that never died would
|
|
# otherwise let an old line pass for a new one.
|
|
if tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -q 'bridged listening'; then
|
|
ok "$(tail -n "+$((RESTART_MARK + 1))" "$OUT" | grep 'bridged listening' | tail -1)"
|
|
else
|
|
warn "no fresh 'bridged listening' line after the restart — check $OUT yourself"
|
|
fi
|
|
|
|
# Config keys the daemon accepted or deferred at boot. This is usually WHY you restarted.
|
|
say "config at boot"
|
|
tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null \
|
|
| grep -iE 'deferred|classification:|fleet health:|coverage' | tail -8 | sed 's/^/ /' \
|
|
|| echo " (nothing reported)"
|
|
|
|
# Errors since the restart, anchored to the marker so old noise cannot leak in.
|
|
ERRS="$(tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -cE ' (ERROR|SEVERE) ' || true)"
|
|
say "result"
|
|
ok "pid $NEW_PID, jar $(jar_id)"
|
|
if [ "${ERRS:-0}" -gt 0 ]; then
|
|
warn "$ERRS ERROR lines since restart:"
|
|
tail -n "+$((RESTART_MARK + 1))" "$OUT" | grep -E ' (ERROR|SEVERE) ' | tail -5 | sed 's/^/ /'
|
|
else
|
|
ok "no ERROR lines since restart"
|
|
fi
|
|
echo
|
|
echo " Next: call bridge_whoami and confirm it still answers 'primary'. A lead whose tab label"
|
|
echo " no longer matches fleet.leaders.*.tab is demoted to worker and refuses orchestration."
|
|
echo
|