fleetd #555: lift 8 main-flow decisions into tested predicate/dispatch functions #565
+253
-138
@@ -806,6 +806,229 @@ refuse_drain_gate() {
|
||||
die "$(drain_gate_refusal "$do_build" "$staged_path")"
|
||||
}
|
||||
|
||||
# fleetd #555 — the main flow itself was almost entirely untestable: sourcing this file (the SOURCED
|
||||
# guard below) stops before a single line of the report/build/drain/stop/swap/start/verify sequence
|
||||
# ever runs, and every decision in that sequence used to live as a bare `if`/`case` written directly
|
||||
# into the main flow rather than inside a function. Three tests reached that flow at all
|
||||
# (test_swap_ordered_after_wait_and_before_start, test_refuse_drain_gate_call_site_present,
|
||||
# test_report_shutdown_drain_call_site_present), and all three do it by grepping this file's own
|
||||
# source for a call site — which proves the call site exists, never that the guard around it still
|
||||
# reaches it. Every function below follows the swap_if_built/refuse_drain_gate shape those two
|
||||
# tickets (#521/#528) already established: the decision (a small, separately-tested predicate) and
|
||||
# the action it gates live together in ONE function, and the main flow calls that function
|
||||
# unconditionally — so there is no bare guard left in the main flow for a future edit to invert
|
||||
# silently. See test_no_untested_main_flow_conditionals in test-redeploy-fleetd.sh for the structural
|
||||
# guard that keeps a ninth bare conditional from arriving the same way these eight did.
|
||||
|
||||
# Ticket item 1 — `--check` must stay genuinely read-only. Inverting the guard used to mean --check
|
||||
# performs a real redeploy; now the guard is should_stop_for_check, and stop_if_check_only is the
|
||||
# only thing the main flow calls.
|
||||
should_stop_for_check() {
|
||||
local check_only="$1"
|
||||
[ "$check_only" = 1 ]
|
||||
}
|
||||
|
||||
stop_if_check_only() {
|
||||
local check_only="$1"
|
||||
should_stop_for_check "$check_only" || return 0
|
||||
say "--check: nothing changed"
|
||||
exit 0
|
||||
}
|
||||
|
||||
# Ticket items 2 and 3 — the drain-gate entry (`-n "$OLD_PID" && "$ASSUME_YES" = 0`) and the reply
|
||||
# comparison (`"$reply" != "yes"`). drain_gate_required and drain_confirmed are the two predicates;
|
||||
# run_drain_gate is the only thing the main flow calls, and it is the one place that reads $reply at
|
||||
# all, so a test can drive it end-to-end over stdin.
|
||||
drain_gate_required() {
|
||||
local old_pid="$1" assume_yes="$2"
|
||||
[ -n "$old_pid" ] && [ "$assume_yes" = 0 ]
|
||||
}
|
||||
|
||||
drain_confirmed() {
|
||||
[ "$1" = "yes" ]
|
||||
}
|
||||
|
||||
run_drain_gate() {
|
||||
local old_pid="$1" assume_yes="$2" do_build="$3" staged_path="$4" reply
|
||||
drain_gate_required "$old_pid" "$assume_yes" || return 0
|
||||
say "drain check"
|
||||
echo " A restart drops every in-flight ticket and rendezvous. A member's report"
|
||||
echo " is NOT recoverable once its ticket is gone."
|
||||
echo
|
||||
echo " Confirm with fleet_list that no members are live, and fleet_poll anything"
|
||||
echo " you still want, BEFORE continuing."
|
||||
echo
|
||||
read -r -p " Fleet drained? type yes to restart: " reply
|
||||
drain_confirmed "$reply" || refuse_drain_gate "$do_build" "$staged_path"
|
||||
}
|
||||
|
||||
# Ticket item 4 — the report-state dispatch on $SUPERVISOR_KIND. Inverting this used to report the
|
||||
# wrong supervisor and, for the launchd arm specifically, skip check_log_path_matches_plist — whose
|
||||
# own comment already says every check after it is worthless if it never runs.
|
||||
report_supervisor_state() {
|
||||
local kind="$1"
|
||||
SUPERVISED=0
|
||||
case "$kind" in
|
||||
launchd)
|
||||
SUPERVISED=1
|
||||
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
||||
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
||||
;;
|
||||
systemd)
|
||||
SUPERVISED=1
|
||||
ok "systemd --user unit active ($SYSTEMD_UNIT) — systemd supervises this daemon"
|
||||
;;
|
||||
none)
|
||||
warn "no supervisor loaded — this script is the only thing that will restart the daemon."
|
||||
;;
|
||||
*)
|
||||
warn "unrecognised supervisor kind: '$kind' — detect_supervisor returned a value this block does not know; continuing to report the rest of the state."
|
||||
;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Ticket item 5 — the stop dispatch on $SUPERVISOR_KIND. Inverting this kills a supervised daemon
|
||||
# with a raw `kill` instead of `launchctl unload`/`systemctl --user stop`, so the supervisor revives
|
||||
# the OLD jar — the exact CB-594/#492 failure both branches exist to prevent. Each mechanism is its
|
||||
# own thin, overridable function (same seam launchd_installed/launchd_loaded already use) so a test
|
||||
# can prove dispatch_stop picks the right one without ever calling real launchctl/systemctl/kill.
|
||||
stop_via_launchd() {
|
||||
echo " supervision is ON (launchd): using 'launchctl unload' (not kill) so launchd's own"
|
||||
echo " KeepAlive cannot restart the OLD jar out from under this script — see the CB-594"
|
||||
echo " comment above."
|
||||
launchctl unload -w "$LAUNCHD_PLIST" \
|
||||
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
|
||||
}
|
||||
|
||||
stop_via_systemd() {
|
||||
echo " supervision is ON (systemd --user): using 'systemctl --user stop' (not kill) so"
|
||||
echo " systemd's own Restart=on-failure cannot restart the OLD jar out from under this"
|
||||
echo " script — see the fleetd #492 comment above."
|
||||
systemctl --user stop "$SYSTEMD_UNIT" \
|
||||
|| die "'systemctl --user stop $SYSTEMD_UNIT' failed — the daemon may still be under supervision; investigate before retrying"
|
||||
}
|
||||
|
||||
stop_via_kill() {
|
||||
local pid="$1"
|
||||
kill "$pid"
|
||||
}
|
||||
|
||||
dispatch_stop() {
|
||||
local kind="$1" pid="$2"
|
||||
case "$kind" in
|
||||
launchd) stop_via_launchd ;;
|
||||
systemd) stop_via_systemd ;;
|
||||
none) stop_via_kill "$pid" ;;
|
||||
*)
|
||||
die "detect_supervisor returned an unrecognized value '$kind' at the stop step —
|
||||
refusing to guess how to stop a daemon under an unknown supervisor. The daemon was NOT
|
||||
stopped." ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Ticket item 6 — the start dispatch on $SUPERVISOR_KIND. Inverting this starts the daemon by the
|
||||
# wrong mechanism; for launchd the branch also carries the retry that keeps a failed load from
|
||||
# leaving the agent stopped-and-disabled. Same shape as the stop dispatch above.
|
||||
start_via_launchd() {
|
||||
echo " supervision is ON (launchd): using 'launchctl load' so launchd starts and keeps"
|
||||
echo " supervising this process, instead of a manual nohup that launchd would know nothing"
|
||||
echo " about."
|
||||
# CB-600: 'launchctl unload -w' above already persisted Disabled=true for this label. A load -w
|
||||
# that succeeds clears it; a load -w that FAILS leaves the agent both stopped and disabled — worse
|
||||
# than before this script ran, because a later reboot or login will not bring it back either. One
|
||||
# retry covers a transient race (e.g. launchd not yet fully done deregistering); if it still fails,
|
||||
# die with the exact recovery command rather than a bare "failed".
|
||||
if ! launchctl load -w "$LAUNCHD_PLIST" 2>/dev/null; then
|
||||
warn "launchctl load failed on the first attempt — retrying once after a short pause"
|
||||
sleep 2
|
||||
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed twice.
|
||||
The agent is now STOPPED and DISABLED — it will NOT come back on its own, not even after a
|
||||
reboot or login, because 'launchctl unload -w' above persisted Disabled=true and load -w
|
||||
never got the chance to clear it. Recover with:
|
||||
launchctl load -w \"$LAUNCHD_PLIST\"
|
||||
If that still fails, check 'launchctl list $LAUNCHD_LABEL', validate the plist with
|
||||
'plutil -lint \"$LAUNCHD_PLIST\"', and check $OUT before assuming a retry will succeed."
|
||||
fi
|
||||
}
|
||||
|
||||
start_via_systemd() {
|
||||
echo " supervision is ON (systemd --user): using 'systemctl --user start' so systemd starts"
|
||||
echo " and keeps supervising this process, instead of a manual nohup it would know nothing"
|
||||
echo " about."
|
||||
systemctl --user start "$SYSTEMD_UNIT" || die "'systemctl --user start $SYSTEMD_UNIT' failed.
|
||||
Check 'systemctl --user status $SYSTEMD_UNIT' and $OUT before assuming a retry will succeed."
|
||||
}
|
||||
|
||||
start_via_none() {
|
||||
# Absolute jar path so `ps` names which checkout is running.
|
||||
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
||||
}
|
||||
|
||||
dispatch_start() {
|
||||
local kind="$1"
|
||||
case "$kind" in
|
||||
launchd) start_via_launchd ;;
|
||||
systemd) start_via_systemd ;;
|
||||
none) start_via_none ;;
|
||||
*)
|
||||
die "detect_supervisor returned an unrecognized value '$kind' at the start step —
|
||||
refusing to guess how to start a daemon under an unknown supervisor. The daemon was NOT
|
||||
started." ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Ticket item 7 — the HEALTH_BODY poll loop and the `[ -z "$HEALTH_BODY" ]` branch. Inverting the
|
||||
# latter (health_is_up) makes a dead daemon report /healthz 200 or a live one report failure.
|
||||
# poll_health_body is pulled out too so the loop's own `if ...; then break; fi` is inside a function
|
||||
# rather than sitting bare in the main flow.
|
||||
poll_health_body() {
|
||||
local url="$1" wait_s="$2" body="" _i
|
||||
for _i in $(seq "$wait_s"); do
|
||||
if body="$(curl -fsS --max-time 2 "$url" 2>/dev/null)"; then
|
||||
printf '%s' "$body"
|
||||
return 0
|
||||
fi
|
||||
sleep 1
|
||||
done
|
||||
return 1
|
||||
}
|
||||
|
||||
health_is_up() {
|
||||
[ -n "$1" ]
|
||||
}
|
||||
|
||||
report_health() {
|
||||
local body="$1" code="$2" out_file="$3" wait_s="$4"
|
||||
if health_is_up "$body"; then
|
||||
ok "/healthz 200 — $body"
|
||||
warn "healthz green only proves herdr ANSWERS. If its protocol number changed, spawns can still"
|
||||
warn "fail — prove a real spawn before trusting the fleet."
|
||||
return 0
|
||||
fi
|
||||
# 503 still means the daemon is up — it means herdr is unreachable. Say which.
|
||||
if [ "$code" = "503" ]; then
|
||||
warn "/healthz answers 503 degraded — the daemon is up but herdr is unreachable."
|
||||
warn "Spawns will fail. Check herdr before delegating anything."
|
||||
curl -s --max-time 2 "$HEALTH" 2>/dev/null | head -3 || true
|
||||
else
|
||||
die "/healthz never answered within ${wait_s}s (last code: $code). Last lines of $out_file:
|
||||
$(tail -30 "$out_file" 2>/dev/null)"
|
||||
fi
|
||||
}
|
||||
|
||||
# Ticket item 8 — `HAD_OLD_PID=0; [ -n "$OLD_PID" ] && HAD_OLD_PID=1`. Measured safe under `set -e`
|
||||
# at both bash 3.2.57 and 5.x (see the header comment trap 9 discussion in the ticket) — not a `set
|
||||
# -e` hazard, but still an untested computation feeding report_shutdown_drain's own four-way
|
||||
# decision. compute_had_old_pid makes the mapping itself directly testable.
|
||||
compute_had_old_pid() {
|
||||
local old_pid="$1"
|
||||
if [ -n "$old_pid" ]; then
|
||||
echo 1
|
||||
else
|
||||
echo 0
|
||||
fi
|
||||
}
|
||||
|
||||
# CB-600: sourceable for testing. When this file is SOURCED (not executed) it stops here — nothing
|
||||
# below runs — so a test harness can `source` it to call check_log_path_matches_plist (or the
|
||||
# other pure helpers above) against a throwaway plist fixture without ever reaching the mutating
|
||||
@@ -854,30 +1077,11 @@ SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
||||
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
||||
require_drivable_supervisor "$SUPERVISOR_KIND"
|
||||
ok "supervisor detected: $SUPERVISOR_KIND"
|
||||
SUPERVISED=0
|
||||
case "$SUPERVISOR_KIND" in
|
||||
launchd)
|
||||
SUPERVISED=1
|
||||
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
||||
# CB-600: fail loudly here, before ANY other check runs, if this script and the loaded plist
|
||||
# would read different log files — every check after this point is worthless otherwise.
|
||||
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
||||
;;
|
||||
systemd)
|
||||
SUPERVISED=1
|
||||
ok "systemd --user unit active ($SYSTEMD_UNIT) — systemd supervises this daemon"
|
||||
;;
|
||||
none)
|
||||
warn "no supervisor loaded — this script is the only thing that will restart the daemon."
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block only DISPLAYS state, it changes nothing yet — so a value
|
||||
# it doesn't recognise gets reported, not an abort that goes silent on exactly the state most
|
||||
# worth seeing. (Unreachable today: require_drivable_supervisor above already died on
|
||||
# "ambiguous"/"unclear" before this case runs. Guards the value nobody has invented yet.)
|
||||
warn "unrecognised supervisor kind: '$SUPERVISOR_KIND' — detect_supervisor returned a value this block does not know; continuing to report the rest of the state."
|
||||
;;
|
||||
esac
|
||||
# fleetd #555: the dispatch below used to be a bare `case` written directly here — see
|
||||
# report_supervisor_state above for why it is now the only thing this line calls. (Unreachable
|
||||
# today for the `*)` arm: require_drivable_supervisor above already died on "ambiguous"/"unclear"
|
||||
# before this call runs. Guards the value nobody has invented yet.)
|
||||
report_supervisor_state "$SUPERVISOR_KIND"
|
||||
|
||||
# The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started
|
||||
# below. Never prints the value — only whether it resolved.
|
||||
@@ -917,10 +1121,9 @@ else
|
||||
warn "Replies stop surviving a restart — a held report is lost, not delayed."
|
||||
fi
|
||||
|
||||
if [ "$CHECK_ONLY" = 1 ]; then
|
||||
say "--check: nothing changed"
|
||||
exit 0
|
||||
fi
|
||||
# fleetd #555: should_stop_for_check/stop_if_check_only above — --check must stay genuinely
|
||||
# read-only, and this is the only place that decides whether to stop here.
|
||||
stop_if_check_only "$CHECK_ONLY"
|
||||
|
||||
# ---------------------------------------------------------------------- build
|
||||
# Deliberately before the stop: a failed build must never leave the fleet down.
|
||||
@@ -954,23 +1157,11 @@ fi
|
||||
|
||||
# ----------------------------------------------------------------- drain gate
|
||||
|
||||
if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
||||
say "drain check"
|
||||
echo " A restart drops every in-flight ticket and rendezvous. A member's report"
|
||||
echo " is NOT recoverable once its ticket is gone."
|
||||
echo
|
||||
echo " Confirm with fleet_list that no members are live, and fleet_poll anything"
|
||||
echo " you still want, BEFORE continuing."
|
||||
echo
|
||||
read -r -p " Fleet drained? type yes to restart: " reply
|
||||
if [ "$reply" != "yes" ]; then
|
||||
# fleetd #493 / #517 / #528: "nothing changed" would be a lie once a build has run and staged a
|
||||
# jar — see drain_gate_refusal above for the full decision and why each of its four cases reads
|
||||
# the way it does. refuse_drain_gate composes that message AND calls die itself, so this guard
|
||||
# has nothing left of its own to get wrong beyond whether it calls refuse_drain_gate at all.
|
||||
refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"
|
||||
fi
|
||||
fi
|
||||
# fleetd #555: run_drain_gate above is drain_gate_required + the prompt + drain_confirmed, called
|
||||
# unconditionally — it returns immediately when the gate is not required, and composes/dies through
|
||||
# refuse_drain_gate itself when the reply does not confirm. See #493/#517/#528 for why "nothing
|
||||
# changed" would be a lie once a build has staged a jar.
|
||||
run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"
|
||||
|
||||
# ------------------------------------------------------------------ stop
|
||||
#
|
||||
@@ -988,33 +1179,9 @@ fi
|
||||
if [ -n "$OLD_PID" ]; then
|
||||
say "stop"
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" # verify a FRESH line appears later
|
||||
case "$SUPERVISOR_KIND" in
|
||||
launchd)
|
||||
echo " supervision is ON (launchd): using 'launchctl unload' (not kill) so launchd's own"
|
||||
echo " KeepAlive cannot restart the OLD jar out from under this script — see the CB-594"
|
||||
echo " comment above."
|
||||
launchctl unload -w "$LAUNCHD_PLIST" \
|
||||
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
|
||||
;;
|
||||
systemd)
|
||||
echo " supervision is ON (systemd --user): using 'systemctl --user stop' (not kill) so"
|
||||
echo " systemd's own Restart=on-failure cannot restart the OLD jar out from under this"
|
||||
echo " script — see the fleetd #492 comment above."
|
||||
systemctl --user stop "$SYSTEMD_UNIT" \
|
||||
|| die "'systemctl --user stop $SYSTEMD_UNIT' failed — the daemon may still be under supervision; investigate before retrying"
|
||||
;;
|
||||
none)
|
||||
kill "$OLD_PID"
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block ACTS (stops the daemon one specific way per kind) — an
|
||||
# unrecognised value must never fall through to a default action, silently picking the wrong
|
||||
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
||||
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
||||
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the stop step —
|
||||
refusing to guess how to stop a daemon under an unknown supervisor. The daemon was NOT
|
||||
stopped." ;;
|
||||
esac
|
||||
# fleetd #555: dispatch_stop above — the launchd/systemd/none/* case lives there now, so this
|
||||
# line is the only thing left in the main flow to get wrong.
|
||||
dispatch_stop "$SUPERVISOR_KIND" "$OLD_PID"
|
||||
if ! wait_for_daemon_exit "$STOP_WAIT"; then
|
||||
die "pid $OLD_PID still alive after ${STOP_WAIT}s. Not escalating to kill -9 automatically:
|
||||
the shutdown hook releases sessions and worktrees in order, and killing it hard can
|
||||
@@ -1063,48 +1230,9 @@ swap_if_built "$DO_BUILD"
|
||||
# above) and WorkingDirectory is already pinned to fleetd/.
|
||||
|
||||
say "start"
|
||||
case "$SUPERVISOR_KIND" in
|
||||
launchd)
|
||||
echo " supervision is ON (launchd): using 'launchctl load' so launchd starts and keeps"
|
||||
echo " supervising this process, instead of a manual nohup that launchd would know nothing"
|
||||
echo " about."
|
||||
# CB-600: 'launchctl unload -w' above already persisted Disabled=true for this label. A load -w
|
||||
# that succeeds clears it; a load -w that FAILS leaves the agent both stopped and disabled — worse
|
||||
# than before this script ran, because a later reboot or login will not bring it back either. One
|
||||
# retry covers a transient race (e.g. launchd not yet fully done deregistering); if it still fails,
|
||||
# die with the exact recovery command rather than a bare "failed".
|
||||
if ! launchctl load -w "$LAUNCHD_PLIST" 2>/dev/null; then
|
||||
warn "launchctl load failed on the first attempt — retrying once after a short pause"
|
||||
sleep 2
|
||||
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed twice.
|
||||
The agent is now STOPPED and DISABLED — it will NOT come back on its own, not even after a
|
||||
reboot or login, because 'launchctl unload -w' above persisted Disabled=true and load -w
|
||||
never got the chance to clear it. Recover with:
|
||||
launchctl load -w \"$LAUNCHD_PLIST\"
|
||||
If that still fails, check 'launchctl list $LAUNCHD_LABEL', validate the plist with
|
||||
'plutil -lint \"$LAUNCHD_PLIST\"', and check $OUT before assuming a retry will succeed."
|
||||
fi
|
||||
;;
|
||||
systemd)
|
||||
echo " supervision is ON (systemd --user): using 'systemctl --user start' so systemd starts"
|
||||
echo " and keeps supervising this process, instead of a manual nohup it would know nothing"
|
||||
echo " about."
|
||||
systemctl --user start "$SYSTEMD_UNIT" || die "'systemctl --user start $SYSTEMD_UNIT' failed.
|
||||
Check 'systemctl --user status $SYSTEMD_UNIT' and $OUT before assuming a retry will succeed."
|
||||
;;
|
||||
none)
|
||||
# Absolute jar path so `ps` names which checkout is running.
|
||||
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block ACTS (starts the daemon one specific way per kind) — an
|
||||
# unrecognised value must never fall through to a default action, silently picking the wrong
|
||||
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
||||
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
||||
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the start step —
|
||||
refusing to guess how to start a daemon under an unknown supervisor. The daemon was NOT
|
||||
started." ;;
|
||||
esac
|
||||
# fleetd #555: dispatch_start above — the launchd/systemd/none/* case (and the launchd retry) live
|
||||
# there now, so this line is the only thing left in the main flow to get wrong.
|
||||
dispatch_start "$SUPERVISOR_KIND"
|
||||
|
||||
for _ in $(seq 10); do
|
||||
NEW_PID="$(running_pid)"
|
||||
@@ -1120,29 +1248,14 @@ ok "started, pid $NEW_PID"
|
||||
|
||||
say "verify"
|
||||
|
||||
HEALTH_BODY=""
|
||||
for _ in $(seq "$HEALTH_WAIT"); do
|
||||
if HEALTH_BODY="$(curl -fsS --max-time 2 "$HEALTH" 2>/dev/null)"; then break; fi
|
||||
HEALTH_BODY=""
|
||||
sleep 1
|
||||
done
|
||||
|
||||
if [ -z "$HEALTH_BODY" ]; then
|
||||
# 503 still means the daemon is up — it means herdr is unreachable. Say which.
|
||||
CODE="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "$HEALTH" 2>/dev/null || echo 000)"
|
||||
if [ "$CODE" = "503" ]; then
|
||||
warn "/healthz answers 503 degraded — the daemon is up but herdr is unreachable."
|
||||
warn "Spawns will fail. Check herdr before delegating anything."
|
||||
curl -s --max-time 2 "$HEALTH" 2>/dev/null | head -3 || true
|
||||
else
|
||||
die "/healthz never answered within ${HEALTH_WAIT}s (last code: $CODE). Last lines of $OUT:
|
||||
$(tail -30 "$OUT" 2>/dev/null)"
|
||||
fi
|
||||
else
|
||||
ok "/healthz 200 — $HEALTH_BODY"
|
||||
warn "healthz green only proves herdr ANSWERS. If its protocol number changed, spawns can still"
|
||||
warn "fail — prove a real spawn before trusting the fleet."
|
||||
fi
|
||||
# fleetd #555: poll_health_body/health_is_up/report_health above. HEALTH_CODE is only ever
|
||||
# consulted by report_health when the body came back empty; `|| true` on both assignments is the
|
||||
# same "an absent/failing command substitution must not kill the script under set -e" idiom the
|
||||
# swap/drain helpers already rely on (see poll_health_body's own comment).
|
||||
HEALTH_BODY="$(poll_health_body "$HEALTH" "$HEALTH_WAIT")" || true
|
||||
HEALTH_CODE="000"
|
||||
[ -n "$HEALTH_BODY" ] || HEALTH_CODE="$(curl -s -o /dev/null -w '%{http_code}' --max-time 2 "$HEALTH" 2>/dev/null || echo 000)"
|
||||
report_health "$HEALTH_BODY" "$HEALTH_CODE" "$OUT" "$HEALTH_WAIT"
|
||||
|
||||
# A fresh listening line, strictly after the restart mark. An old daemon that never died would
|
||||
# otherwise let an old line pass for a new one.
|
||||
@@ -1185,7 +1298,9 @@ classify_amqp_connection_errors "$FRESH_LOG"
|
||||
# "cannot tell"). HAD_OLD_PID crosses in whether a previous daemon was actually stopped this run;
|
||||
# see the function's own comment for why that matters.
|
||||
say "previous daemon's shutdown drain"
|
||||
HAD_OLD_PID=0; [ -n "$OLD_PID" ] && HAD_OLD_PID=1
|
||||
# fleetd #555: compute_had_old_pid above — the mapping itself is now directly testable rather than
|
||||
# an inline `&&` idiom sitting bare in the main flow.
|
||||
HAD_OLD_PID="$(compute_had_old_pid "$OLD_PID")"
|
||||
report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"
|
||||
|
||||
# fleetd #492: checked here, after healthz and the fresh-log check have both had time to run, so a
|
||||
|
||||
@@ -198,7 +198,7 @@ STUB
|
||||
# own random suffix, which is exactly why six such sites survived undetected here — GNU mktemp
|
||||
# (every Linux distribution) refuses a template with fewer than three `X`s and exits non-zero. There
|
||||
# is no BSD-vs-GNU seam to stub on this Mac, so this is a source-text check rather than a
|
||||
# behavioural one, the same shape as test_refuse_drain_gate_call_site_present above. Anchored on
|
||||
# behavioural one, the same shape as test_run_drain_gate_call_site_present above. Anchored on
|
||||
# `mktemp -t ` (with the trailing space) so it inspects only the `-t`-style templates this ticket is
|
||||
# about, never the `mktemp -d` calls this file and test-probe-member-credentials.sh already use
|
||||
# (both already carry their own `XXXXXX` and are a different mktemp mode entirely).
|
||||
@@ -892,14 +892,489 @@ test_refuse_drain_gate_no_build_staged_absent() {
|
||||
# flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural
|
||||
# tests above would even notice.
|
||||
#
|
||||
# fleetd #555 — the main flow's own call site moved: it used to read
|
||||
# `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` directly; it now reads
|
||||
# `run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"`, and run_drain_gate (tested
|
||||
# directly below by test_run_drain_gate_*) is what calls refuse_drain_gate with its own local names.
|
||||
# This grep now pins THAT call site — the thing that would go missing if a future edit deleted the
|
||||
# main flow's call to run_drain_gate altogether, the same residual gap #521/#528 already accepted for
|
||||
# swap_if_built/refuse_drain_gate (sourcing stops before the main flow runs, so no test in this file
|
||||
# can do better than reading the source for this one specific gap).
|
||||
#
|
||||
# The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail
|
||||
# the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below
|
||||
# ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body).
|
||||
test_refuse_drain_gate_call_site_present() {
|
||||
test_run_drain_gate_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
call_line="$(grep -Fn 'run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's refuse_drain_gate call site in redeploy-fleetd.sh"
|
||||
|| fail "could not find the main flow's run_drain_gate call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 item 1 — `--check` must stay genuinely read-only: should_stop_for_check is the
|
||||
# predicate, stop_if_check_only is the only thing the main flow calls.
|
||||
test_should_stop_for_check_true_when_check_only() {
|
||||
should_stop_for_check 1 || fail "should_stop_for_check must be true when CHECK_ONLY=1"
|
||||
}
|
||||
|
||||
test_should_stop_for_check_false_when_not_check_only() {
|
||||
should_stop_for_check 0 && fail "should_stop_for_check must be false when CHECK_ONLY=0"
|
||||
return 0
|
||||
}
|
||||
|
||||
# Captured via $(...): stop_if_check_only's own exit only ends this subshell (same reason the
|
||||
# unload/stop "tolerates clean negative" tests above capture this way), so this test's own message
|
||||
# is what reaches the report if the exit ever stops happening.
|
||||
test_stop_if_check_only_exits_zero_and_says_nothing_changed() {
|
||||
local output rc=0
|
||||
output="$(stop_if_check_only 1)" || rc=$?
|
||||
[ "$rc" -eq 0 ] || fail "stop_if_check_only 1 must exit 0, got $rc"
|
||||
printf '%s' "$output" | grep -qF -- '--check: nothing changed' \
|
||||
|| fail "stop_if_check_only 1 did not print the --check message"
|
||||
}
|
||||
|
||||
test_stop_if_check_only_returns_without_exiting_when_not_check_only() {
|
||||
local output
|
||||
output="$(stop_if_check_only 0; echo REACHED)"
|
||||
printf '%s' "$output" | grep -qF 'REACHED' \
|
||||
|| fail "stop_if_check_only 0 must return, not exit — the caller's next line never ran"
|
||||
}
|
||||
|
||||
test_stop_if_check_only_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'stop_if_check_only "$CHECK_ONLY"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's stop_if_check_only call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 items 2 and 3 — the drain-gate entry (`-n "$OLD_PID" && "$ASSUME_YES" = 0`) and the
|
||||
# reply comparison (`"$reply" != "yes"`).
|
||||
test_drain_gate_required_true_when_pid_and_not_assume_yes() {
|
||||
drain_gate_required "123" 0 || fail "drain_gate_required must be true with an old pid and ASSUME_YES=0"
|
||||
}
|
||||
|
||||
test_drain_gate_required_false_when_no_pid() {
|
||||
drain_gate_required "" 0 && fail "drain_gate_required must be false with no old pid"
|
||||
return 0
|
||||
}
|
||||
|
||||
test_drain_gate_required_false_when_assume_yes() {
|
||||
drain_gate_required "123" 1 && fail "drain_gate_required must be false when ASSUME_YES=1"
|
||||
return 0
|
||||
}
|
||||
|
||||
test_drain_confirmed_true_on_yes() {
|
||||
drain_confirmed "yes" || fail "drain_confirmed must be true on exactly 'yes'"
|
||||
}
|
||||
|
||||
test_drain_confirmed_false_on_anything_else() {
|
||||
drain_confirmed "no" && fail "drain_confirmed must be false on 'no'"
|
||||
drain_confirmed "" && fail "drain_confirmed must be false on empty input"
|
||||
return 0
|
||||
}
|
||||
|
||||
# run_drain_gate end-to-end, driving real stdin through `read` the same way the real prompt does.
|
||||
# Redirected from a FILE, never a pipe: `printf ... | run_drain_gate ...` would run the function as
|
||||
# the last stage of a pipeline, which bash runs in a subshell by default (no `lastpipe`), so any
|
||||
# global this function set (DIED_CALLED/DIED_MESSAGE below) would be lost the instant the pipe
|
||||
# closed. `< file` redirection on a plain function call carries no such subshell.
|
||||
test_run_drain_gate_skips_prompt_when_not_required() {
|
||||
local output
|
||||
output="$(run_drain_gate "" 0 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)"
|
||||
if printf '%s' "$output" | grep -qF 'Fleet drained?'; then
|
||||
fail "run_drain_gate prompted even though drain_gate_required should have been false (no old pid)"
|
||||
fi
|
||||
output="$(run_drain_gate "123" 1 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)"
|
||||
if printf '%s' "$output" | grep -qF 'Fleet drained?'; then
|
||||
fail "run_drain_gate prompted even though ASSUME_YES=1 should have skipped the gate"
|
||||
fi
|
||||
}
|
||||
|
||||
test_run_drain_gate_confirmed_reply_does_not_refuse() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_die_recorder
|
||||
printf 'yes\n' > "$TMP/drain-reply-yes.txt"
|
||||
run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-yes.txt" \
|
||||
> "$TMP/drain-gate-yes-output" 2>&1
|
||||
[ "$DIED_CALLED" = 0 ] \
|
||||
|| fail "run_drain_gate must not refuse when the reply confirms ('yes')"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_run_drain_gate_declined_reply_refuses() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_die_recorder
|
||||
printf 'no\n' > "$TMP/drain-reply-no.txt"
|
||||
run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-no.txt" \
|
||||
> "$TMP/drain-gate-no-output" 2>&1
|
||||
[ "$DIED_CALLED" = 1 ] \
|
||||
|| fail "run_drain_gate must refuse (call die via refuse_drain_gate) when the reply does not confirm"
|
||||
printf '%s' "$DIED_MESSAGE" | grep -qF 'aborted' \
|
||||
|| fail "run_drain_gate's refusal message does not read as an abort: $DIED_MESSAGE"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 item 4 — the report-state dispatch on $SUPERVISOR_KIND. Inverting this used to report
|
||||
# the wrong supervisor and, for the launchd arm specifically, skip check_log_path_matches_plist.
|
||||
CHECK_LOG_PATH_CALLED=0
|
||||
stub_check_log_path_recorder() {
|
||||
CHECK_LOG_PATH_CALLED=0
|
||||
check_log_path_matches_plist() { CHECK_LOG_PATH_CALLED=1; }
|
||||
}
|
||||
|
||||
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_check_log_path_recorder
|
||||
SUPERVISED=0
|
||||
report_supervisor_state "launchd" > "$TMP/report-supervisor-launchd-output" 2>&1
|
||||
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state launchd must set SUPERVISED=1"
|
||||
[ "$CHECK_LOG_PATH_CALLED" = 1 ] \
|
||||
|| fail "report_supervisor_state launchd must call check_log_path_matches_plist"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_report_supervisor_state_systemd_sets_supervised_without_log_path_check() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_check_log_path_recorder
|
||||
SUPERVISED=0
|
||||
report_supervisor_state "systemd" > "$TMP/report-supervisor-systemd-output" 2>&1
|
||||
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state systemd must set SUPERVISED=1"
|
||||
[ "$CHECK_LOG_PATH_CALLED" = 0 ] \
|
||||
|| fail "report_supervisor_state systemd must NOT call check_log_path_matches_plist"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_report_supervisor_state_none_leaves_supervised_zero() {
|
||||
SUPERVISED=1
|
||||
report_supervisor_state "none" > "$TMP/report-supervisor-none-output" 2>&1
|
||||
[ "$SUPERVISED" = 0 ] || fail "report_supervisor_state none must leave SUPERVISED=0"
|
||||
grep -qF 'no supervisor loaded' "$TMP/report-supervisor-none-output" \
|
||||
|| fail "report_supervisor_state none did not print the unsupervised message"
|
||||
}
|
||||
|
||||
test_report_supervisor_state_unrecognized_kind_warns_and_continues() {
|
||||
SUPERVISED=1
|
||||
report_supervisor_state "bogus-kind" > "$TMP/report-supervisor-bogus-output" 2>&1
|
||||
[ "$SUPERVISED" = 0 ] || fail "report_supervisor_state must reset SUPERVISED for an unrecognized kind"
|
||||
grep -qF "unrecognised supervisor kind: 'bogus-kind'" "$TMP/report-supervisor-bogus-output" \
|
||||
|| fail "report_supervisor_state did not warn about the unrecognized kind"
|
||||
}
|
||||
|
||||
test_report_supervisor_state_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'report_supervisor_state "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's report_supervisor_state call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 item 5 — the stop dispatch on $SUPERVISOR_KIND. Inverting this kills a supervised
|
||||
# daemon with a raw kill instead of launchctl/systemctl, reviving the OLD jar (CB-594/#492). Each
|
||||
# mechanism is stubbed to record which one ran, so a swapped or dropped arm shows up directly.
|
||||
STOP_DISPATCH_CALLED=""
|
||||
stub_stop_dispatch_recorders() {
|
||||
STOP_DISPATCH_CALLED=""
|
||||
stop_via_launchd() { STOP_DISPATCH_CALLED="launchd"; }
|
||||
stop_via_systemd() { STOP_DISPATCH_CALLED="systemd"; }
|
||||
stop_via_kill() { STOP_DISPATCH_CALLED="kill:$1"; }
|
||||
}
|
||||
|
||||
test_dispatch_stop_launchd_calls_stop_via_launchd() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_stop_dispatch_recorders
|
||||
dispatch_stop "launchd" "123"
|
||||
assert_equals "launchd" "$STOP_DISPATCH_CALLED" "dispatch_stop launchd"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_stop_systemd_calls_stop_via_systemd() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_stop_dispatch_recorders
|
||||
dispatch_stop "systemd" "123"
|
||||
assert_equals "systemd" "$STOP_DISPATCH_CALLED" "dispatch_stop systemd"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_stop_none_calls_stop_via_kill_with_pid() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_stop_dispatch_recorders
|
||||
dispatch_stop "none" "456"
|
||||
assert_equals "kill:456" "$STOP_DISPATCH_CALLED" "dispatch_stop none"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_stop_unrecognized_kind_dies() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_stop_dispatch_recorders
|
||||
stub_die_recorder
|
||||
dispatch_stop "bogus-kind" "789"
|
||||
[ "$DIED_CALLED" = 1 ] || fail "dispatch_stop must die on an unrecognized kind"
|
||||
[ -z "$STOP_DISPATCH_CALLED" ] \
|
||||
|| fail "dispatch_stop must not call any stop mechanism on an unrecognized kind"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_stop_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'dispatch_stop "$SUPERVISOR_KIND" "$OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's dispatch_stop call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 item 6 — the start dispatch on $SUPERVISOR_KIND. Same shape as the stop dispatch.
|
||||
START_DISPATCH_CALLED=""
|
||||
stub_start_dispatch_recorders() {
|
||||
START_DISPATCH_CALLED=""
|
||||
start_via_launchd() { START_DISPATCH_CALLED="launchd"; }
|
||||
start_via_systemd() { START_DISPATCH_CALLED="systemd"; }
|
||||
start_via_none() { START_DISPATCH_CALLED="none"; }
|
||||
}
|
||||
|
||||
test_dispatch_start_launchd_calls_start_via_launchd() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_start_dispatch_recorders
|
||||
dispatch_start "launchd"
|
||||
assert_equals "launchd" "$START_DISPATCH_CALLED" "dispatch_start launchd"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_start_systemd_calls_start_via_systemd() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_start_dispatch_recorders
|
||||
dispatch_start "systemd"
|
||||
assert_equals "systemd" "$START_DISPATCH_CALLED" "dispatch_start systemd"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_start_none_calls_start_via_none() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_start_dispatch_recorders
|
||||
dispatch_start "none"
|
||||
assert_equals "none" "$START_DISPATCH_CALLED" "dispatch_start none"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_start_unrecognized_kind_dies() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_start_dispatch_recorders
|
||||
stub_die_recorder
|
||||
dispatch_start "bogus-kind"
|
||||
[ "$DIED_CALLED" = 1 ] || fail "dispatch_start must die on an unrecognized kind"
|
||||
[ -z "$START_DISPATCH_CALLED" ] \
|
||||
|| fail "dispatch_start must not call any start mechanism on an unrecognized kind"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_dispatch_start_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'dispatch_start "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's dispatch_start call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 item 7 — the HEALTH_BODY poll loop and the `[ -z "$HEALTH_BODY" ]` branch. Inverting
|
||||
# health_is_up makes a dead daemon report /healthz 200 or a live one report failure.
|
||||
test_health_is_up_true_with_body() {
|
||||
health_is_up "some body" || fail "health_is_up must be true with a non-empty body"
|
||||
}
|
||||
|
||||
test_health_is_up_false_with_empty_body() {
|
||||
health_is_up "" && fail "health_is_up must be false with an empty body"
|
||||
return 0
|
||||
}
|
||||
|
||||
# fleetd #555: stub_die_recorder replaces die() globally for the rest of the process, the same as
|
||||
# every other user of it in this file — re-source before AND after so a later test (e.g.
|
||||
# test_unload_launchd_if_loaded_dies_on_real_failure, which needs the REAL die() to actually exit)
|
||||
# never inherits a stub left over from here.
|
||||
test_report_health_up_reports_ok_and_never_dies() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_die_recorder
|
||||
local output
|
||||
output="$(report_health '{"status":"ok"}' "200" "$TMP/fake.out" 60)"
|
||||
[ "$DIED_CALLED" = 0 ] || fail "report_health must not die when the body is non-empty"
|
||||
printf '%s' "$output" | grep -qF '/healthz 200' \
|
||||
|| fail "report_health did not report the healthy body"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_report_health_degraded_503_warns_and_never_dies() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_die_recorder
|
||||
local output
|
||||
output="$(report_health "" "503" "$TMP/fake.out" 60 2>&1)"
|
||||
[ "$DIED_CALLED" = 0 ] || fail "report_health must not die on a 503 (degraded, not dead)"
|
||||
printf '%s' "$output" | grep -qF '503 degraded' \
|
||||
|| fail "report_health did not report the 503-degraded case"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_report_health_dead_dies() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
stub_die_recorder
|
||||
report_health "" "000" "$TMP/fake.out" 60 > "$TMP/report-health-dead-output" 2>&1
|
||||
[ "$DIED_CALLED" = 1 ] || fail "report_health must die when the body is empty and the code is not 503"
|
||||
printf '%s' "$DIED_MESSAGE" | grep -qF 'never answered' \
|
||||
|| fail "report_health die message does not say healthz never answered"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_poll_health_body_returns_nonzero_when_unreachable() {
|
||||
local rc=0
|
||||
poll_health_body "http://127.0.0.1:1/healthz" 1 > /dev/null || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "poll_health_body must return non-zero when the health endpoint is unreachable"
|
||||
}
|
||||
|
||||
test_report_health_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'report_health "$HEALTH_BODY" "$HEALTH_CODE" "$OUT" "$HEALTH_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's report_health call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 item 8 — `HAD_OLD_PID=0; [ -n "$OLD_PID" ] && HAD_OLD_PID=1`. Measured safe under
|
||||
# `set -e` at both bash 3.2.57 and 5.x (the ticket's own header discussion); still an untested
|
||||
# computation feeding report_shutdown_drain's four-way decision until now.
|
||||
test_compute_had_old_pid_zero_when_empty() {
|
||||
assert_equals "0" "$(compute_had_old_pid "")" "compute_had_old_pid empty pid"
|
||||
}
|
||||
|
||||
test_compute_had_old_pid_one_when_present() {
|
||||
assert_equals "1" "$(compute_had_old_pid "12345")" "compute_had_old_pid present pid"
|
||||
}
|
||||
|
||||
test_compute_had_old_pid_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'HAD_OLD_PID="$(compute_had_old_pid "$OLD_PID")"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's compute_had_old_pid call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #555 — the structural guard the ticket actually asks for: "pick the shape, not the eight
|
||||
# sites." A future ninth bare main-flow conditional must fail THIS test on sight, not slip through
|
||||
# with every function-level test (67, now many more) still green.
|
||||
#
|
||||
# Scope: from the SOURCED guard onward — the ticket's own "main flow" (report state / build / drain
|
||||
# / stop / swap / start / verify). The `for arg in "$@"; do case "$arg" in ... esac; done` option
|
||||
# parser above the guard is out of scope, the same way it sits outside the ticket's own eight-item
|
||||
# table: it runs even when this file is sourced for tests, is not part of the mutating procedure,
|
||||
# and none of the eight decisions the ticket names live there.
|
||||
#
|
||||
# For every line whose first non-blank token is `if`, `elif`, or `case`, and which sits OUTSIDE any
|
||||
# function body, this demands an EXACT match — full trimmed line, never a regex anchor (a regex can
|
||||
# match inside unrelated text this same mutation could introduce, e.g. a comment) — against
|
||||
# MAIN_FLOW_ALLOWED_CONDITIONALS below. That allowlist is a closed set of the report-only/display
|
||||
# conditionals already in the file (bare state prints: is the daemon running, is X installed, does
|
||||
# this secret resolve — none of them gate a build/stop/start/swap decision) plus the two
|
||||
# loaded-but-not-running elif branches that are the SAME shape as fleetd #504 items 2/3/4, one
|
||||
# function over, and explicitly out of THIS ticket's scope per its own "Related" section.
|
||||
#
|
||||
# Every one of the eight decisions the ticket names has been lifted into its own predicate+action
|
||||
# function above this point in the file (should_stop_for_check, drain_gate_required/run_drain_gate,
|
||||
# report_supervisor_state, dispatch_stop, dispatch_start, health_is_up/report_health,
|
||||
# compute_had_old_pid) — none of their guards appear in this list any more, which is what makes
|
||||
# their specific inversions untestable-by-construction now: there is no bare guard left in the main
|
||||
# flow for a future edit to invert silently. A NEW bare conditional (the ninth) is, by construction,
|
||||
# not in this list, so this test goes red the instant one is added, naming the exact line — passing
|
||||
# requires either lifting the decision into a function (with its own predicate test, same shape as
|
||||
# above) or explicitly extending the allowlist, which is a diff a reviewer sees and can question.
|
||||
#
|
||||
# Function-body detection relies on this file's one consistent convention: a function opens as
|
||||
# `name() {` alone on its own line and closes as a bare `}` alone on its own line — verified: every
|
||||
# multi-line function in this file follows it, and the handful of one-liners like `say() { ...; }`
|
||||
# sit above the SOURCED guard, outside the region this scans. Indentation is NOT used to decide
|
||||
# "inside a function": a bare `if` inside a top-level `for`/`while` loop body (indented, but not
|
||||
# inside any function) would be exactly as reportable as one at column 0 — this file currently has
|
||||
# none, but a future one that hid inside a loop must not get a free pass for being indented.
|
||||
MAIN_FLOW_ALLOWED_CONDITIONALS=(
|
||||
'if [ -n "$OLD_PID" ]; then'
|
||||
'if launchd_installed; then'
|
||||
'if systemd_installed; then'
|
||||
'if zsh -lc '\''[ -n "${WORKER_GITEA_TOKEN:-}" ]'\'' 2>/dev/null; then'
|
||||
'if zsh -lc '\''[ -n "${AI_GATEWAY_TOKEN:-}" ]'\'' 2>/dev/null; then'
|
||||
'if [ -z "$BROKER_URI_ENV" ]; then'
|
||||
'elif zsh -lc "[ -n \"\${$BROKER_URI_ENV:-}\" ]" 2>/dev/null; then'
|
||||
'if [ "$DO_BUILD" = 1 ]; then'
|
||||
'if ! mvn -f "$MODULE/pom.xml" clean install > "$BUILD_LOG" 2>&1; then'
|
||||
'if ! wait_for_daemon_exit "$STOP_WAIT"; then'
|
||||
'elif [ "$SUPERVISOR_KIND" = "launchd" ]; then'
|
||||
'elif [ "$SUPERVISOR_KIND" = "systemd" ]; then'
|
||||
'if tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -q '\''fleetd listening'\''; then'
|
||||
'if ! FRESH_LOG="$(capture_fresh_log_region "$OUT" "$RESTART_MARK")"; then'
|
||||
'if [ "$REDEPLOY_AMQP_CHECK_SKIPPED" -eq 1 ]; then'
|
||||
'elif [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then'
|
||||
'if [ "$REDEPLOY_DRAIN_STATE" = "complete" ] || [ "$REDEPLOY_DRAIN_STATE" = "n/a" ]; then'
|
||||
'elif [ "$REDEPLOY_UNEXPLAINED_ERRORS" -eq 0 ]; then'
|
||||
)
|
||||
|
||||
# Emits "<lineno>\tCOND\t<trimmed line>" for every if/elif/case found outside a function, and
|
||||
# "<lineno>\tFUNC\t<line>" for every function DEFINITION found after guard_line — from guard_line
|
||||
# onward. A separate function (rather than inlined into the test) so the deliberate-mutation proof
|
||||
# in the PR description can call it directly against a scratch copy of the script.
|
||||
#
|
||||
# fleetd #555 rework (comment 17012): a function defined after the SOURCED guard can never be
|
||||
# reached by `source`-ing this script (sourcing returns before the main flow, and before any code
|
||||
# below the guard runs), so anything inside such a function is untestable by construction — the
|
||||
# depth tracker below would otherwise read it as "inside a function, therefore fine" and wave every
|
||||
# conditional in it through unexamined. Emitting FUNC records lets the caller flag the function
|
||||
# itself as the violation, independently of whether its body happens to contain a conditional.
|
||||
mainflow_bare_conditionals() {
|
||||
local src="$1" guard_line="$2" in_func=0 lineno=0 line trimmed
|
||||
while IFS= read -r line || [ -n "$line" ]; do
|
||||
lineno=$((lineno + 1))
|
||||
[ "$lineno" -le "$guard_line" ] && continue
|
||||
if [[ "$line" =~ ^[A-Za-z_][A-Za-z0-9_]*\(\)[[:space:]]*\{[[:space:]]*$ ]]; then
|
||||
in_func=1
|
||||
printf '%d\tFUNC\t%s\n' "$lineno" "$line"
|
||||
continue
|
||||
fi
|
||||
if [ "$in_func" = 1 ] && [[ "$line" =~ ^\}[[:space:]]*$ ]]; then
|
||||
in_func=0
|
||||
continue
|
||||
fi
|
||||
if [ "$in_func" = 0 ]; then
|
||||
trimmed="$(printf '%s' "$line" | sed -E 's/^[[:space:]]+//')"
|
||||
if [[ "$trimmed" =~ ^(if|elif|case)[[:space:]] ]]; then
|
||||
printf '%d\tCOND\t%s\n' "$lineno" "$trimmed"
|
||||
fi
|
||||
fi
|
||||
done < "$src"
|
||||
}
|
||||
|
||||
test_no_untested_main_flow_conditionals() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" guard_line
|
||||
guard_line="$(grep -Fn 'if (return 0 2>/dev/null); then' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$guard_line" ] || { fail "could not find the SOURCED guard in redeploy-fleetd.sh"; return 1; }
|
||||
|
||||
local violations=0 report="" found_line found_kind found_text allowed candidate
|
||||
while IFS=$'\t' read -r found_line found_kind found_text; do
|
||||
[ -n "$found_line" ] || continue
|
||||
if [ "$found_kind" = "FUNC" ]; then
|
||||
# fleetd #555 rework: a function defined after the SOURCED guard line can never be sourced
|
||||
# by this suite, so it can never be tested — that is a violation on its own, regardless of
|
||||
# what its body contains or whether the allowlist would otherwise excuse a bare conditional
|
||||
# inside it.
|
||||
violations=$((violations + 1))
|
||||
report="$report
|
||||
line $found_line: function defined after the SOURCED guard (line $guard_line) — it cannot be sourced, so it cannot be tested: $found_text"
|
||||
continue
|
||||
fi
|
||||
allowed=0
|
||||
for candidate in "${MAIN_FLOW_ALLOWED_CONDITIONALS[@]}"; do
|
||||
if [ "$found_text" = "$candidate" ]; then
|
||||
allowed=1
|
||||
break
|
||||
fi
|
||||
done
|
||||
if [ "$allowed" = 0 ]; then
|
||||
violations=$((violations + 1))
|
||||
report="$report
|
||||
line $found_line: $found_text"
|
||||
fi
|
||||
done < <(mainflow_bare_conditionals "$src" "$guard_line")
|
||||
|
||||
if [ "$violations" -gt 0 ]; then
|
||||
fail "found $violations untested main-flow if/elif/case line(s) or function definition(s) after the SOURCED guard, not lifted into a tested predicate function and not in MAIN_FLOW_ALLOWED_CONDITIONALS:$report"
|
||||
fi
|
||||
}
|
||||
|
||||
# fleetd #504 — the "loaded but not currently running" branches for launchd/systemd used to run
|
||||
@@ -983,7 +1458,7 @@ STUB
|
||||
|| fail "stop_systemd_if_loaded must tolerate a clean already-stopped answer (non-zero exit, empty stderr): $output"
|
||||
}
|
||||
|
||||
# Closes the same gap test_refuse_drain_gate_call_site_present closes for the drain gate: the four
|
||||
# Closes the same gap test_run_drain_gate_call_site_present closes for the drain gate: the four
|
||||
# tests above call unload_launchd_if_loaded/stop_systemd_if_loaded directly, and sourcing stops
|
||||
# before the main flow ever runs (the SOURCED guard), so none of them can prove the main flow still
|
||||
# CALLS these two functions instead of the original bare `2>/dev/null || true`. A source-text check,
|
||||
@@ -1363,7 +1838,7 @@ test_report_shutdown_drain_na_wins_over_missing_log() {
|
||||
|
||||
# fleetd #512 — closes the gap none of the seven tests above can: they call report_shutdown_drain
|
||||
# directly, and sourcing stops before the main flow ever runs (the SOURCED guard), so none of them
|
||||
# can prove the main flow still calls it at all. Same shape as test_refuse_drain_gate_call_site_present
|
||||
# can prove the main flow still calls it at all. Same shape as test_run_drain_gate_call_site_present
|
||||
# and test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site, plus
|
||||
# an ordering check against its neighbours in the verify/result flow.
|
||||
test_report_shutdown_drain_call_site_present() {
|
||||
@@ -1446,7 +1921,46 @@ test_refuse_drain_gate_build_ran_staged_present
|
||||
test_refuse_drain_gate_build_ran_staged_absent
|
||||
test_refuse_drain_gate_no_build_staged_present
|
||||
test_refuse_drain_gate_no_build_staged_absent
|
||||
test_refuse_drain_gate_call_site_present
|
||||
test_run_drain_gate_call_site_present
|
||||
test_should_stop_for_check_true_when_check_only
|
||||
test_should_stop_for_check_false_when_not_check_only
|
||||
test_stop_if_check_only_exits_zero_and_says_nothing_changed
|
||||
test_stop_if_check_only_returns_without_exiting_when_not_check_only
|
||||
test_stop_if_check_only_call_site_present
|
||||
test_drain_gate_required_true_when_pid_and_not_assume_yes
|
||||
test_drain_gate_required_false_when_no_pid
|
||||
test_drain_gate_required_false_when_assume_yes
|
||||
test_drain_confirmed_true_on_yes
|
||||
test_drain_confirmed_false_on_anything_else
|
||||
test_run_drain_gate_skips_prompt_when_not_required
|
||||
test_run_drain_gate_confirmed_reply_does_not_refuse
|
||||
test_run_drain_gate_declined_reply_refuses
|
||||
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path
|
||||
test_report_supervisor_state_systemd_sets_supervised_without_log_path_check
|
||||
test_report_supervisor_state_none_leaves_supervised_zero
|
||||
test_report_supervisor_state_unrecognized_kind_warns_and_continues
|
||||
test_report_supervisor_state_call_site_present
|
||||
test_dispatch_stop_launchd_calls_stop_via_launchd
|
||||
test_dispatch_stop_systemd_calls_stop_via_systemd
|
||||
test_dispatch_stop_none_calls_stop_via_kill_with_pid
|
||||
test_dispatch_stop_unrecognized_kind_dies
|
||||
test_dispatch_stop_call_site_present
|
||||
test_dispatch_start_launchd_calls_start_via_launchd
|
||||
test_dispatch_start_systemd_calls_start_via_systemd
|
||||
test_dispatch_start_none_calls_start_via_none
|
||||
test_dispatch_start_unrecognized_kind_dies
|
||||
test_dispatch_start_call_site_present
|
||||
test_health_is_up_true_with_body
|
||||
test_health_is_up_false_with_empty_body
|
||||
test_report_health_up_reports_ok_and_never_dies
|
||||
test_report_health_degraded_503_warns_and_never_dies
|
||||
test_report_health_dead_dies
|
||||
test_poll_health_body_returns_nonzero_when_unreachable
|
||||
test_report_health_call_site_present
|
||||
test_compute_had_old_pid_zero_when_empty
|
||||
test_compute_had_old_pid_one_when_present
|
||||
test_compute_had_old_pid_call_site_present
|
||||
test_no_untested_main_flow_conditionals
|
||||
test_unload_launchd_if_loaded_dies_on_real_failure
|
||||
test_unload_launchd_if_loaded_tolerates_clean_negative
|
||||
test_stop_systemd_if_loaded_dies_on_real_failure
|
||||
|
||||
Reference in New Issue
Block a user