Files
fleetd/scripts/test-redeploy-fleetd.sh
Dai Ha 8f80d267a0
CI / shell-tests (pull_request) Successful in 9s
CI / contract (pull_request) Successful in 1m20s
CI / build (pull_request) Successful in 3m3s
fleetd #555 rework: catch function definitions after the SOURCED guard
Comment 17012 on #555 found a hole in test_no_untested_main_flow_conditionals:
the guard's function-body detection treats anything inside a function as
"fine, out of scope for this scan" — but a function DEFINED after the
SOURCED guard line can never be reached by sourcing this script (sourcing
stops before the main flow runs), so its body is untestable by construction
while still reading to the guard as safely inside a function.

mainflow_bare_conditionals now also emits a FUNC record for every function
opened after the guard line (reusing the same open-brace detection already
used for depth tracking), and test_no_untested_main_flow_conditionals treats
any such record as a violation on its own, independent of what the function's
body contains or whether the allowlist would otherwise excuse a bare
conditional inside it.

Proof (redeploy-fleetd.sh restored to 4ffacc5185807d39720a3484d85b922413806eb5347318265bd8897dfd61e8d9
after each):

- CONTROL — a bare conditional appended to the main flow is still caught:
  EXIT=1, "found 1 untested main-flow if/elif/case line(s) ... line 1341:
  if [ "$MY_CONTROL_BARE" = 1 ]; then :; fi"
- CANDIDATE — the same conditional wrapped in a function defined after the
  boundary, previously invisible (EXIT=0), is now caught: EXIT=1, "line 1341:
  function defined after the SOURCED guard (line 1038) — it cannot be
  sourced, so it cannot be tested: newfunc_below_the_boundary() {"

Full suite re-run green on both bash 5.3.9 and /bin/bash 3.2.57 (macOS
system bash): exit 0, 0 FAIL lines, reached the final PASS line, on both.

No change to redeploy-fleetd.sh; the 8 lifted decisions, their mutation
proofs, and the allowlist all stand as before.
2026-09-12 17:11:11 +07:00

1995 lines
107 KiB
Bash
Executable File

#!/usr/bin/env bash
# Self-contained checks for the pure log classifier in redeploy-fleetd.sh.
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")"
trap 'rm -rf "$TMP"' EXIT
# Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon.
source "$ROOT/scripts/redeploy-fleetd.sh"
fail() {
printf 'FAIL: %s\n' "$*" >&2
return 1
}
assert_equals() {
local expected="$1" actual="$2" description="$3"
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
}
classify_fixture() {
local name="$1"
classify_amqp_connection_errors "$TMP/$name"
}
# fleetd #492 follow-up: detect_supervisor's stdout is now "kind<SEP>detail" (see the constraints
# comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about
# the kind must split it out with the SAME in-shell parameter expansion the real call site (:438)
# uses, never a bare string comparison against the raw output.
supervisor_kind_of() {
printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}"
}
supervisor_detail_of() {
printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}"
}
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
# being naturally absent: this machine may itself be running a real fleetd under launchd right now
# (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving
# launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state
# instead of the fixture.
test_detect_supervisor_launchd_only() {
launchd_installed() { return 0; }
launchd_loaded() { return 0; }
systemd_installed() { return 1; }
systemd_loaded() { return 1; }
assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection"
}
test_detect_supervisor_systemd_only() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 0; }
systemd_loaded() { return 0; }
assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection"
}
test_detect_supervisor_none() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 1; }
systemd_loaded() { return 1; }
assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection"
}
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
# auto-restart too, and every one of those is a host that IS under systemd.
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 0; } # the unit file IS there
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
local raw
raw="$(detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none"
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \
|| fail "detail does not name the systemd unit it found installed-but-not-loaded"
}
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
# removed, or about to be reloaded) must not read as "no supervisor" either.
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
launchd_installed() { return 0; } # the plist IS there
launchd_loaded() { return 1; } # launchctl list says not loaded
systemd_installed() { return 1; }
systemd_loaded() { return 1; }
local raw
raw="$(detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none"
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \
|| fail "detail does not name the launchd label it found installed-but-not-loaded"
}
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
# the same fact as "no supervisor is loaded".
test_detect_supervisor_systemd_probe_error_is_unclear() {
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
# of those tests), and a bash function definition is global for the rest of the process — without
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
# a real `systemctl` call at all.
source "$ROOT/scripts/redeploy-fleetd.sh"
local bin_dir result rc=0
bin_dir="$TMP/stub-bin-systemctl-errors"
mkdir -p "$bin_dir"
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
echo "Failed to connect to bus: No such file or directory" >&2
exit 1
STUB
chmod +x "$bin_dir/systemctl"
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none"
local detail
detail="$(supervisor_detail_of "$result")"
printf '%s' "$detail" | grep -qF "$SYSTEMD_UNIT" \
|| fail "detail does not name the systemd unit whose probe errored"
# fleetd #545: this is the PROBE-RAN-AND-ANSWERED-BADLY case (systemctl actually executed and
# wrote to stderr) — it must carry that story and never the SET-UP-FAILED story (mktemp never
# even ran here), or the two "unclear" causes have collapsed back into one message that asserts a
# cause it did not measure, which is the exact defect this ticket exists to fix.
printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \
|| fail "detail does not say systemctl ran and answered with stderr: $detail"
printf '%s' "$detail" | grep -qF "could not even be set up" \
&& fail "detail wrongly claims the probe could not be set up, but systemctl actually ran and answered on stderr: $detail"
return 0
}
# fleetd #545 — the companion case to the probe-error test above: here `mktemp` itself fails
# (whatever the reason — the historical bug was a GNU-mktemp-rejects-a-template-with-no-Xs case,
# but this stub simulates ANY reason the probe's own stderr-capture temp file cannot be created,
# e.g. a full or unwritable temp dir) and `systemctl` is never invoked at all. Before this ticket,
# this collapsed into the SAME "systemctl exited non-zero and reported an error on stderr" detail
# as the sibling test above, which asserts a cause (systemctl ran and answered badly) that was
# never measured, because systemctl never ran. This proves the SET-UP-FAILED detail is distinct and
# does not claim systemctl said anything.
test_detect_supervisor_systemd_probe_setup_failure_is_unclear() {
# Re-source first for the same reason test_detect_supervisor_systemd_probe_error_is_unclear does:
# restore the REAL probe bodies before driving them through a stub PATH.
source "$ROOT/scripts/redeploy-fleetd.sh"
local bin_dir result rc=0
bin_dir="$TMP/stub-bin-mktemp-fails"
mkdir -p "$bin_dir"
# A systemctl stub that would fail loudly if it were ever actually invoked — proves the mktemp
# failure short-circuits the probe before systemctl runs, not merely that this test forgot to
# supply a working systemctl.
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
echo "systemctl must never run when mktemp already failed" >&2
exit 1
STUB
chmod +x "$bin_dir/systemctl"
cat > "$bin_dir/mktemp" <<'STUB'
#!/usr/bin/env bash
echo "mktemp: cannot create temp file" >&2
exit 1
STUB
chmod +x "$bin_dir/mktemp"
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
[ "$rc" -ne 0 ] \
|| fail "systemd_loaded must not report loaded=true when its own mktemp setup failed"
assert_equals "2" "$SYSTEMD_LOADED_ERRORED" \
"systemd_loaded must flag a SETUP failure (2), distinct from a probe-answered-with-stderr failure (1)"
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe setup failure must read as unclear, not none"
local detail
detail="$(supervisor_detail_of "$result")"
printf '%s' "$detail" | grep -qF "could not even be set up" \
|| fail "detail does not say the probe could not be SET UP: $detail"
printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \
&& fail "detail wrongly asserts systemctl exited non-zero and reported an error on stderr, but systemctl was never run: $detail"
return 0
}
# fleetd #545 — source-text check: every `mktemp -t` template in redeploy-fleetd.sh must contain an
# `X` placeholder. BSD mktemp (macOS) tolerates a bare template with no `X`s and just appends its
# own random suffix, which is exactly why six such sites survived undetected here — GNU mktemp
# (every Linux distribution) refuses a template with fewer than three `X`s and exits non-zero. There
# is no BSD-vs-GNU seam to stub on this Mac, so this is a source-text check rather than a
# behavioural one, the same shape as test_run_drain_gate_call_site_present above. Anchored on
# `mktemp -t ` (with the trailing space) so it inspects only the `-t`-style templates this ticket is
# about, never the `mktemp -d` calls this file and test-probe-member-credentials.sh already use
# (both already carry their own `XXXXXX` and are a different mktemp mode entirely).
test_mktemp_dash_t_templates_have_x_placeholders() {
local src="$ROOT/scripts/redeploy-fleetd.sh" bad
bad="$(grep -n 'mktemp -t ' "$src" | grep -v 'XXX' || true)"
[ -z "$bad" ] \
|| fail "mktemp -t template(s) with no X placeholder (fails under GNU coreutils): $bad"
}
# fleetd #550 — the shape, not the named lines: #545 showed the exact same failure mode (a
# macOS-only idiom used with no portable fallback) spread from two sites to six across 91 commits
# before anyone tested the SHAPE rather than specific lines. This is the shasum sibling: any script
# under scripts/ that actually INVOKES the macOS-only hasher to compute a hash (as opposed to
# merely probing whether it exists with `command -v`, or mentioning it in prose) must also check
# for the portable one first in that same file — the prefer-portable-fall-back-to-macOS-only idiom
# probe-member-credentials.sh:273-279 and this ticket's own hash256 helper both follow.
#
# The needle is built from two concatenated pieces, deliberately never written as one literal
# string in this file: written whole, it would match THIS CHECK'S OWN source line once the loop
# below reaches this very file, and the check would then "pass" by matching itself rather than any
# real invocation elsewhere — a zero-findings result indistinguishable from a clean file.
test_no_unguarded_macos_only_hasher_calls() {
local needle f bad="" usage
needle='shasum'; needle="$needle -a"
for f in "$ROOT"/scripts/*.sh; do
[ -f "$f" ] || continue
usage="$(grep -Fn "$needle" "$f" || true)"
if [ -n "$usage" ]; then
grep -q 'command -v sha256sum' "$f" \
|| bad="$bad$(basename "$f") "
fi
done
[ -z "$bad" ] \
|| fail "script(s) invoke the macOS-only hasher with no portable-hasher-first fallback guard in the same file: $bad"
}
# fleetd #552 — the shape, not the one line just fixed: a `mktemp` failure AFTER the daemon has
# already been restarted must never be a bare, unguarded assignment again, anywhere in the script,
# so the next one added is caught too — not just line 1113 as it stood at 26f380a. Anchored on the
# real "start" section (the restart call itself), the same anchor test_swap_ordered_after_wait_and_
# before_start above already uses for "before the start section". Needle built from two concatenated
# pieces, the same trick test_no_unguarded_macos_only_hasher_calls uses above, so this check's own
# description of the pattern it looks for can never become a match for itself.
test_no_unguarded_mktemp_assignments_after_restart() {
local src="$ROOT/scripts/redeploy-fleetd.sh" start_line needle bad
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$start_line" ] || fail "could not find the restart call's start section in redeploy-fleetd.sh"
needle='^[[:space:]]*[A-Za-z_][A-Za-z0-9_]*="'
needle="$needle"'\$\(mktemp'
bad="$(grep -nE "$needle" "$src" | awk -F: -v start="$start_line" '$1+0>start')"
[ -z "$bad" ] \
|| fail "unguarded 'VAR=\"\$(mktemp ...)\"' assignment(s) after the restart call (fleetd #552): $bad"
}
# fleetd #552 — capture_fresh_log_region itself. Sourcing stops before the main flow can be driven
# directly (the SOURCED guard below), so this exercises the extracted function the real call site
# now uses, the same stub-mktemp-on-PATH technique
# test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses above.
test_capture_fresh_log_region_control() {
printf 'line one\nline two\nline three\n' > "$TMP/fresh-log-src.out"
local result
result="$(capture_fresh_log_region "$TMP/fresh-log-src.out" 0)"
[ -n "$result" ] || fail "capture_fresh_log_region returned nothing on a working mktemp"
[ -f "$result" ] || fail "capture_fresh_log_region did not create the file it named"
assert_equals "$(printf 'line one\nline two\nline three')" "$(cat "$result")" \
"captured region did not carry the whole source file (restart_mark=0)"
rm -f "$result"
}
test_capture_fresh_log_region_mktemp_failure_returns_nonzero_and_prints_nothing() {
local bin_dir rc=0 out
bin_dir="$TMP/stub-bin-fresh-log-mktemp-fails"; mkdir -p "$bin_dir"
cat > "$bin_dir/mktemp" <<'STUB'
#!/usr/bin/env bash
echo "mktemp: cannot create temp file" >&2
exit 1
STUB
chmod +x "$bin_dir/mktemp"
printf 'line one\n' > "$TMP/fresh-log-src2.out"
out="$(PATH="$bin_dir:$PATH" capture_fresh_log_region "$TMP/fresh-log-src2.out" 0)" && rc=0 || rc=$?
[ "$rc" -ne 0 ] \
|| fail "capture_fresh_log_region must return non-zero when its own mktemp fails"
[ -z "$out" ] \
|| fail "capture_fresh_log_region printed a path even though mktemp failed: $out"
}
# fleetd #552 acceptance: "the exit status after a stubbed mktemp failure must still report the
# restart's own outcome". Sourcing stops before the main flow runs, so this drives the REAL
# call-site shape (the `if ! FRESH_LOG="$(capture_fresh_log_region ...)"; then warn; fi` guard that
# now stands where the old unguarded assignment did) in a subshell, under the SAME `set -euo
# pipefail` the real script runs under, with `mktemp` stubbed to fail on PATH. A regression back to
# the old unguarded `FRESH_LOG="$(mktemp ...)"` shape would abort this subshell before it reaches
# SUBSHELL_REACHED_END, and this test would go red with a non-zero rc.
test_fresh_log_capture_failure_does_not_abort_under_sete() {
local bin_dir rc=0 out
bin_dir="$TMP/stub-bin-call-site-mktemp-fails"; mkdir -p "$bin_dir"
cat > "$bin_dir/mktemp" <<'STUB'
#!/usr/bin/env bash
echo "mktemp: cannot create temp file" >&2
exit 1
STUB
chmod +x "$bin_dir/mktemp"
printf 'line one\n' > "$TMP/fresh-log-callsite.out"
out="$(
PATH="$bin_dir:$PATH"
set -euo pipefail
FRESH_LOG=""
trap 'rm -f "$FRESH_LOG"' EXIT
if ! FRESH_LOG="$(capture_fresh_log_region "$TMP/fresh-log-callsite.out" 0)"; then
echo "WARN skipped"
fi
echo "SUBSHELL_REACHED_END"
)" && rc=0 || rc=$?
[ "$rc" -eq 0 ] \
|| fail "the guarded call-site shape aborted under set -e when mktemp failed (rc=$rc) — a successful restart must not be reported as a failure"
printf '%s' "$out" | grep -qF 'SUBSHELL_REACHED_END' \
|| fail "the subshell aborted before reaching the end — a post-restart mktemp failure must not abort the script"
printf '%s' "$out" | grep -qF 'WARN skipped' \
|| fail "the guarded call site did not report the post-restart check as skipped when mktemp failed"
}
# fleetd #552 item 4 (the fourth reader): the result section must check REDEPLOY_AMQP_CHECK_SKIPPED
# and say so BEFORE it ever consults REDEPLOY_ERROR_COUNT — otherwise a skipped capture (which
# leaves REDEPLOY_ERROR_COUNT at its untouched 0) falls through the same "no ERROR lines" gate a
# genuinely clean restart uses, and (because REDEPLOY_DRAIN_STATE is "skipped", not "complete"/"n/a")
# prints nothing at all: a silence indistinguishable from the clean bill of health this ticket exists
# to prevent. Sourcing stops before the main flow runs, so this is a source-text ordering check, the
# same shape as test_report_shutdown_drain_ordered_after_classify_and_before_result above.
# fleetd #552 — closes the gap test_no_unguarded_mktemp_assignments_after_restart cannot: after
# extracting the mktemp into capture_fresh_log_region, the SAME class of hazard (an unguarded
# command-substitution assignment that can abort the script under set -e, because capture_fresh_
# log_region still returns non-zero on its own mktemp failure) would reappear if the main flow's
# call to it ever lost its own "if !" guard — and that line no longer contains the literal text
# "mktemp", so the sweep above cannot see it. Pin the real call site directly instead, the same
# shape test_swap_ordered_after_wait_and_before_start above uses for its own call site.
test_capture_fresh_log_region_call_site_is_guarded() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'if ! FRESH_LOG="$(capture_fresh_log_region "$OUT" "$RESTART_MARK")"; then' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's guarded capture_fresh_log_region call site in redeploy-fleetd.sh"
}
test_result_section_checks_amqp_skip_before_no_error_lines() {
local src="$ROOT/scripts/redeploy-fleetd.sh" result_line skip_line ok_line
result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)"
skip_line="$(grep -Fn '"$REDEPLOY_AMQP_CHECK_SKIPPED" -eq 1' "$src" | tail -1 | cut -d: -f1 || true)"
ok_line="$(grep -Fn 'ok "no ERROR lines since restart"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$result_line" ] || fail "could not find the result section in redeploy-fleetd.sh"
[ -n "$skip_line" ] || fail "could not find a REDEPLOY_AMQP_CHECK_SKIPPED check in redeploy-fleetd.sh"
[ -n "$ok_line" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh"
[ "$skip_line" -gt "$result_line" ] \
|| fail "the REDEPLOY_AMQP_CHECK_SKIPPED check (line $skip_line) is not inside the result section (starts line $result_line)"
[ "$skip_line" -lt "$ok_line" ] \
|| fail "the REDEPLOY_AMQP_CHECK_SKIPPED check (line $skip_line) is not checked before 'no ERROR lines since restart' (line $ok_line)"
}
# fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a
# hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not
# the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real
# caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout
# survives that boundary, so kind AND detail must both cross on it — this test proves they do.
test_require_drivable_supervisor_refuses_unclear() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 0; }
systemd_loaded() { return 1; }
local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0
# Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand.
SUPERVISOR_RAW="$(detect_supervisor)"
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal"
[ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \
|| fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell"
printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \
|| fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded"
output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
# Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate
# recovery instructions name the unit unconditionally either way ("systemctl --user status
# $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only
# the specific detail string proves the crossed value, not the boilerplate, reached the message.
printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \
|| fail "refusal message does not contain the specific detail that crossed the subshell boundary"
}
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
# `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that
# forks a subshell, so its exit() only ends the subshell and this test script keeps running under
# `set -e`.
test_require_drivable_supervisor_refuses_ambiguous() {
local output rc=0
output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor"
printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \
|| fail "refusal message does not name the launchd label it found"
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|| fail "refusal message does not name the systemd unit it found"
}
test_require_drivable_supervisor_accepts_known_kinds() {
require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor"
require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor"
require_drivable_supervisor "none" || fail "refused the unsupervised case"
}
# fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor
# produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see
# it because either daemon alone satisfies them.
test_count_daemon_pids() {
assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list"
assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid"
assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids"
}
test_assert_single_daemon_accepts_one_pid() {
assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid"
}
test_assert_single_daemon_rejects_two_pids() {
local output rc=0
output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids"
printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found"
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
}
# fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports
# $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call
# must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR.
# Two files with different content, so a default pointed at the wrong one reports the wrong hash
# rather than accidentally matching.
test_jar_id_defaults_to_live_and_reports_explicit_path() {
local dir saved_jar="$JAR" saved_staged="$JAR_STAGED"
local live_hash staged_hash default_result explicit_result
dir="$TMP/jar-id"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
printf 'live jar bytes' > "$JAR"
printf 'staged jar bytes, not the same content' > "$JAR_STAGED"
# fleetd #550: this reference hash must be computed the same portable way jar_id() itself now
# computes one — a bare, unguarded call to the macOS-only hasher here was exactly the item-2
# defect, dying with "command not found" on any Linux runner that has no such hasher at all.
live_hash="$(hash256 "$JAR")"
staged_hash="$(hash256 "$JAR_STAGED")"
default_result="$(jar_id)"
explicit_result="$(jar_id "$JAR_STAGED")"
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
[ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same"
assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR"
assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR"
}
# fleetd #550 — closes a gap the test above leaves open. That test's own reference hash is now ALSO
# computed by calling hash256 (needed for item 2: the old bare macOS-only-hasher call there was the
# Linux crash), so its subject (jar_id, via hash256) and its reference (also hash256) share one
# instrument — they agree no matter which algorithm hash256 actually runs, so a mutation that swaps
# BOTH of hash256's arms for the wrong algorithm is invisible to it. This test's expected value
# comes from neither hasher: it is the published SHA-256 test vector for the 3-byte input "abc"
# (no trailing newline), written here as a literal constant, so it can still tell "hashed
# correctly" from "hashed, just with the wrong algorithm" — which is what this whole ticket is
# about.
test_hash256_computes_a_real_sha256() {
local dir f result
dir="$TMP/hash256-known-vector"; mkdir -p "$dir"
f="$dir/abc.txt"
printf 'abc' > "$f"
result="$(hash256 "$f")"
# SHA-256("abc") = ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad, the standard
# FIPS 180 test vector — first 12 hex chars, matching hash256's own `cut -c1-12`.
assert_equals "ba7816bf8f01" "$result" "hash256 of the literal 3-byte input 'abc' must be the known SHA-256 prefix, not some other algorithm's"
}
# fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511)
# proves both halves of the present-file contract but never exercises the missing-file path. This
# word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean`
# deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points
# operators at `--check` for exactly this. Covers both the no-argument default and an explicit path,
# since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both.
test_jar_id_reports_absent_for_missing_file() {
local saved_jar="$JAR" dir default_result explicit_result
dir="$TMP/jar-id-absent"; mkdir -p "$dir"
JAR="$dir/does-not-exist.jar"
[ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR"
default_result="$(jar_id)"
explicit_result="$(jar_id "$dir/also-does-not-exist.jar")"
JAR="$saved_jar"
assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist"
assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent"
}
# fleetd #550 — the whole point of this ticket: a jar that IS there but could not be hashed must
# never read the same as a jar that is not there at all. Drives the REAL hash256/jar_id bodies
# (never stubbed) through a stub PATH that contains neither of the two hashers this script knows —
# same technique test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses for `mktemp`,
# except here the stub directory is used to REPLACE PATH rather than prepend to it, because the
# point is to make BOTH hashers unreachable, not to intercept one specific command while leaving
# everything else on the real PATH reachable. `[ -f ... ]` and the shell's own `command`/`echo`
# builtins need no PATH at all, so this is safe even with PATH reduced to an empty directory.
test_jar_id_reports_unhashable_when_no_hasher_on_path() {
local bin_dir dir saved_jar="$JAR" default_result explicit_result
bin_dir="$TMP/stub-bin-no-hasher"; mkdir -p "$bin_dir"
dir="$TMP/jar-id-no-hasher"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"
printf 'a real jar that exists but nothing here can hash' > "$JAR"
[ -f "$JAR" ] || fail "test fixture error: \$JAR does not exist at $JAR"
default_result="$(PATH="$bin_dir" jar_id)"
explicit_result="$(PATH="$bin_dir" jar_id "$JAR")"
JAR="$saved_jar"
[ "$default_result" != "absent" ] \
|| fail "jar_id reported absent for a file that exists, only because no hasher was on PATH"
[ "$explicit_result" != "absent" ] \
|| fail "jar_id (explicit path) reported absent for a file that exists, only because no hasher was on PATH"
# A 12-char hex hash is the OTHER wrong answer here: with no hasher at all, nothing could have
# produced one, so a value that merely happens to look like one would mean the stub failed to
# hide the real hashers rather than that jar_id degraded correctly.
printf '%s' "$default_result" | grep -Eq '^[0-9a-f]{12}$' \
&& fail "test fixture error: PATH stub did not actually hide the real hasher(s) — got what looks like a real hash"
assert_equals "unhashable" "$default_result" "jar_id with no hasher on PATH must report a third, distinct state — never absent, never a hash"
assert_equals "unhashable" "$explicit_result" "jar_id (explicit path) with no hasher on PATH must report the same third state"
}
# fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar
# are exercised directly against real files on disk (not stubs), because the whole point is file
# behavior (does the content move, does the source disappear, does a failure leave both sides
# intact) that a stubbed function cannot prove.
test_stage_built_jar_moves_off_live_path() {
local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED"
dir="$TMP/stage-ok"; mkdir -p "$dir"
jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar"
printf 'built jar bytes' > "$jar"
JAR="$jar"; JAR_STAGED="$staged"
stage_built_jar || fail "stage_built_jar rejected a real build output"
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
[ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar"
[ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged"
grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content"
}
test_stage_built_jar_dies_when_build_produced_nothing() {
local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED"
dir="$TMP/stage-missing"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
output="$(stage_built_jar 2>&1)" || rc=$?
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
[ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output"
printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \
|| fail "refusal message does not name the missing jar path"
}
test_swap_staged_jar_moves_staged_onto_live() {
local dir staged live
dir="$TMP/swap-ok"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar"
printf 'swapped jar bytes' > "$staged"
swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar"
[ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged"
[ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live"
grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content"
}
# The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on
# failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes
# inside a command substitution to contain that exit to a subshell.
test_swap_staged_jar_dies_without_staged_file() {
local dir output rc=0
dir="$TMP/swap-missing"; mkdir -p "$dir"
output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar"
[ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged"
printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \
|| fail "refusal message does not name the missing staged path"
}
test_swap_staged_jar_dies_when_mv_fails() {
local dir staged live output rc=0
dir="$TMP/swap-fail"; mkdir -p "$dir/src"
staged="$dir/src/fleetd-new.jar"
printf 'fake jar bytes' > "$staged"
live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails
output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv"
[ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails"
[ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed"
printf '%s' "$output" | grep -qF "$staged" \
|| fail "refusal message does not name the staged path that could not be moved"
}
# --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this
# path) and must still die with the exact wording documented in the script's own header comment.
test_require_no_build_jar_dies_when_absent() {
local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar"
JAR="$missing"
output="$(require_no_build_jar 2>&1)" || rc=$?
JAR="$saved_jar"
[ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar"
printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \
|| fail "refusal message does not match the documented --no-build wording"
}
test_require_no_build_jar_accepts_present_jar() {
local saved_jar="$JAR" dir
dir="$TMP/no-build-present"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"
printf 'existing jar' > "$JAR"
require_no_build_jar || fail "require_no_build_jar rejected an existing jar"
JAR="$saved_jar"
}
# wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while
# running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so
# the timeout-loop test does not actually wait out its budget.
test_wait_for_daemon_exit_returns_true_once_pid_clears() {
# running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so
# a plain shell variable it increments would reset on each call instead of accumulating. Count in
# a file instead, which is the one thing that actually survives across those subshells.
local counter_file="$TMP/wait-exit-calls" final_calls
printf '0' > "$counter_file"
running_pid() {
local n
n="$(cat "$counter_file")"
n=$((n + 1))
printf '%s' "$n" > "$counter_file"
if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi
}
sleep() { :; }
wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared"
final_calls="$(cat "$counter_file")"
[ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid"
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
}
test_wait_for_daemon_exit_times_out_if_pid_never_clears() {
local rc=0
running_pid() { printf '4242'; }
sleep() { :; }
wait_for_daemon_exit 3 || rc=$?
[ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared"
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
}
# fleetd #521 — the swap step's guard, at two levels.
#
# The first two tests call the predicate should_swap() directly. They pin its logic, and that is all
# they pin. On their own they did NOT close #521, and this was measured rather than argued: with the
# main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left
# this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs
# the swap consult the predicate at all. Extracting the decision had moved the untested decision up
# a level, not removed it.
#
# So the last two tests call swap_if_built() — the function the main flow actually calls, holding the
# guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the
# guard is removed, inverted, or stops being consulted.
#
# What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow
# altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing
# stops before the main flow runs, so no test in this file can invoke it.
test_should_swap_true_when_build_ran() {
should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true"
}
test_should_swap_false_when_build_skipped() {
if should_swap 0; then
fail "should_swap 0 (--no-build; nothing was staged this run) must return false"
fi
}
# Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is
# global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id
# overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear),
# and again at the END, so their own stubs do not leak into every test that runs after them.
test_swap_if_built_performs_the_swap_when_build_ran() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local marker="$TMP/swap-if-built-ran"
rm -f "$marker"
swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; }
jar_id() { printf 'stubbed\n'; }
swap_if_built 1 > /dev/null
[ -f "$marker" ] \
|| fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_swap_if_built_skips_the_swap_when_build_skipped() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local marker="$TMP/swap-if-built-skipped"
rm -f "$marker"
swap_staged_jar() { printf 'swapped\n' > "$marker"; }
jar_id() { printf 'stubbed\n'; }
swap_if_built 0 > /dev/null
if [ -f "$marker" ]; then
fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did"
fi
source "$ROOT/scripts/redeploy-fleetd.sh"
}
# fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the
# main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee
# itself — as opposed to the pure functions it's built from — can only be checked by reading the
# script's own call sites, the same way test_recovery_patterns_match_source below checks Java
# source shape instead of behavior it cannot invoke directly.
#
# Two details about the three greps below, both of which have already gone wrong here.
#
# The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not
# `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside
# swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using
# it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)"
# — a true statement about a function definition, and nothing at all about the order of the steps.
#
# Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the
# pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set
# -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to
# report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no
# name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak.
test_swap_ordered_after_wait_and_before_start() {
local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line
wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)"
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh"
[ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh"
[ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh"
[ "$swap_line" -gt "$wait_line" ] \
|| fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)"
[ "$swap_line" -lt "$start_line" ] \
|| fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)"
}
# fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator
# declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)"
# to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has
# already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal
# ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start
# above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow),
# so the only way to pin its exact wording is to read the source.
test_drain_gate_abort_message_says_no_no_build() {
local src="$ROOT/scripts/redeploy-fleetd.sh" msg
msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")"
[ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh"
if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then
fail "abort message still claims a rerun WITH --no-build can finish the restart"
fi
printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \
|| fail "abort message does not tell the operator to rerun without --no-build"
printf '%s' "$msg" | grep -qF 'no longer at the live path' \
|| fail "abort message does not say why --no-build cannot finish the restart"
}
# fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was
# a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's
# own file for the wording, which stays in the file even if the `if` guarding it is mutated to
# `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead,
# so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT
# alongside these, not replaced, because it catches a different regression (a re-wording that still
# reaches the right branch would not change which case fires here, but would still be worth pinning).
test_drain_gate_refusal_build_ran_staged_present() {
local dir staged result
dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'staged jar bytes' > "$staged"
result="$(drain_gate_refusal 1 "$staged")"
printf '%s' "$result" | grep -qF "$staged" \
|| fail "build-ran+staged-present refusal does not name the staged jar path"
printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \
|| fail "build-ran+staged-present refusal does not tell the operator how to finish the restart"
if printf '%s' "$result" | grep -qF 'nothing changed'; then
fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved"
fi
}
test_drain_gate_refusal_build_ran_staged_absent() {
local dir result
dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir"
result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")"
assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording"
}
# --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar
# found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the
# comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning.
test_drain_gate_refusal_no_build_staged_present() {
local dir staged result
dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'leftover staged jar bytes' > "$staged"
result="$(drain_gate_refusal 0 "$staged")"
assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed"
}
test_drain_gate_refusal_no_build_staged_absent() {
local dir result
dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir"
result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")"
assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording"
}
# fleetd #528 — the four tests above pin drain_gate_refusal(), and that is ALL they pin: they call
# the predicate directly and never touch the main flow's call site. That was measured to be not
# enough, the same way test_should_swap_true_when_build_ran/test_should_swap_false_when_build_skipped
# were not enough for #521: with the main flow reading `die "$(drain_gate_refusal "$DO_BUILD"
# "$JAR_STAGED")"`, replacing that whole line with a flat `die "aborted — nothing changed"` left this
# suite at exit 0 with zero FAIL lines and byte-identical output to a clean run. Nothing above could
# tell the difference, because none of it calls anything at or above the call site itself.
#
# So these four call refuse_drain_gate() — the function the main flow actually calls, holding the
# composed message and the die() together — with die() stubbed to RECORD whether it was called and
# with what message, instead of exiting the process. That fails if refuse_drain_gate stops consulting
# drain_gate_refusal, mangles what it passes it, or simply never calls die.
#
# What none of these four can catch: deleting the `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` line
# from the main flow altogether — see the comment above refuse_drain_gate in redeploy-fleetd.sh for
# why no test in this file can do better than that (sourcing stops before the main flow runs).
DIED_CALLED=0
DIED_MESSAGE=""
stub_die_recorder() {
DIED_CALLED=0
DIED_MESSAGE=""
die() { DIED_CALLED=1; DIED_MESSAGE="$*"; }
}
test_refuse_drain_gate_build_ran_staged_present() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir staged
dir="$TMP/refuse-drain-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'staged jar bytes' > "$staged"
stub_die_recorder
refuse_drain_gate 1 "$staged"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate build-ran+staged-present must call die, and did not"
printf '%s' "$DIED_MESSAGE" | grep -qF "$staged" \
|| fail "refuse_drain_gate build-ran+staged-present die message does not name the staged jar"
printf '%s' "$DIED_MESSAGE" | grep -qF 'Rerun WITHOUT --no-build' \
|| fail "refuse_drain_gate build-ran+staged-present die message is missing the rerun instruction"
if printf '%s' "$DIED_MESSAGE" | grep -qF 'nothing changed'; then
fail "refuse_drain_gate build-ran+staged-present must not claim nothing changed — the jar already moved"
fi
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_refuse_drain_gate_build_ran_staged_absent() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir
dir="$TMP/refuse-drain-build-no-staged"; mkdir -p "$dir"
stub_die_recorder
refuse_drain_gate 1 "$dir/fleetd-new.jar"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate build-ran+staged-absent must call die, and did not"
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate build-ran+staged-absent die message"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_refuse_drain_gate_no_build_staged_present() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir staged
dir="$TMP/refuse-drain-no-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'leftover staged jar bytes' > "$staged"
stub_die_recorder
refuse_drain_gate 0 "$staged"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate no-build+staged-present must call die, and did not"
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-present die message"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_refuse_drain_gate_no_build_staged_absent() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir
dir="$TMP/refuse-drain-no-build-no-staged"; mkdir -p "$dir"
stub_die_recorder
refuse_drain_gate 0 "$dir/fleetd-new.jar"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate no-build+staged-absent must call die, and did not"
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-absent die message"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
# fleetd #528 — closes the one gap the four behavioural tests above cannot: they call
# refuse_drain_gate directly, and sourcing stops before the main flow ever runs (the SOURCED guard),
# so none of them can prove the main flow still CALLS refuse_drain_gate at all. Same shape as
# test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site. This is
# what actually kills the item-1 mutation from the ticket — replacing the main flow's call with a
# flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural
# tests above would even notice.
#
# fleetd #555 — the main flow's own call site moved: it used to read
# `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` directly; it now reads
# `run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"`, and run_drain_gate (tested
# directly below by test_run_drain_gate_*) is what calls refuse_drain_gate with its own local names.
# This grep now pins THAT call site — the thing that would go missing if a future edit deleted the
# main flow's call to run_drain_gate altogether, the same residual gap #521/#528 already accepted for
# swap_if_built/refuse_drain_gate (sourcing stops before the main flow runs, so no test in this file
# can do better than reading the source for this one specific gap).
#
# The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail
# the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below
# ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body).
test_run_drain_gate_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's run_drain_gate call site in redeploy-fleetd.sh"
}
# fleetd #555 item 1 — `--check` must stay genuinely read-only: should_stop_for_check is the
# predicate, stop_if_check_only is the only thing the main flow calls.
test_should_stop_for_check_true_when_check_only() {
should_stop_for_check 1 || fail "should_stop_for_check must be true when CHECK_ONLY=1"
}
test_should_stop_for_check_false_when_not_check_only() {
should_stop_for_check 0 && fail "should_stop_for_check must be false when CHECK_ONLY=0"
return 0
}
# Captured via $(...): stop_if_check_only's own exit only ends this subshell (same reason the
# unload/stop "tolerates clean negative" tests above capture this way), so this test's own message
# is what reaches the report if the exit ever stops happening.
test_stop_if_check_only_exits_zero_and_says_nothing_changed() {
local output rc=0
output="$(stop_if_check_only 1)" || rc=$?
[ "$rc" -eq 0 ] || fail "stop_if_check_only 1 must exit 0, got $rc"
printf '%s' "$output" | grep -qF -- '--check: nothing changed' \
|| fail "stop_if_check_only 1 did not print the --check message"
}
test_stop_if_check_only_returns_without_exiting_when_not_check_only() {
local output
output="$(stop_if_check_only 0; echo REACHED)"
printf '%s' "$output" | grep -qF 'REACHED' \
|| fail "stop_if_check_only 0 must return, not exit — the caller's next line never ran"
}
test_stop_if_check_only_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'stop_if_check_only "$CHECK_ONLY"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's stop_if_check_only call site in redeploy-fleetd.sh"
}
# fleetd #555 items 2 and 3 — the drain-gate entry (`-n "$OLD_PID" && "$ASSUME_YES" = 0`) and the
# reply comparison (`"$reply" != "yes"`).
test_drain_gate_required_true_when_pid_and_not_assume_yes() {
drain_gate_required "123" 0 || fail "drain_gate_required must be true with an old pid and ASSUME_YES=0"
}
test_drain_gate_required_false_when_no_pid() {
drain_gate_required "" 0 && fail "drain_gate_required must be false with no old pid"
return 0
}
test_drain_gate_required_false_when_assume_yes() {
drain_gate_required "123" 1 && fail "drain_gate_required must be false when ASSUME_YES=1"
return 0
}
test_drain_confirmed_true_on_yes() {
drain_confirmed "yes" || fail "drain_confirmed must be true on exactly 'yes'"
}
test_drain_confirmed_false_on_anything_else() {
drain_confirmed "no" && fail "drain_confirmed must be false on 'no'"
drain_confirmed "" && fail "drain_confirmed must be false on empty input"
return 0
}
# run_drain_gate end-to-end, driving real stdin through `read` the same way the real prompt does.
# Redirected from a FILE, never a pipe: `printf ... | run_drain_gate ...` would run the function as
# the last stage of a pipeline, which bash runs in a subshell by default (no `lastpipe`), so any
# global this function set (DIED_CALLED/DIED_MESSAGE below) would be lost the instant the pipe
# closed. `< file` redirection on a plain function call carries no such subshell.
test_run_drain_gate_skips_prompt_when_not_required() {
local output
output="$(run_drain_gate "" 0 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)"
if printf '%s' "$output" | grep -qF 'Fleet drained?'; then
fail "run_drain_gate prompted even though drain_gate_required should have been false (no old pid)"
fi
output="$(run_drain_gate "123" 1 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)"
if printf '%s' "$output" | grep -qF 'Fleet drained?'; then
fail "run_drain_gate prompted even though ASSUME_YES=1 should have skipped the gate"
fi
}
test_run_drain_gate_confirmed_reply_does_not_refuse() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
printf 'yes\n' > "$TMP/drain-reply-yes.txt"
run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-yes.txt" \
> "$TMP/drain-gate-yes-output" 2>&1
[ "$DIED_CALLED" = 0 ] \
|| fail "run_drain_gate must not refuse when the reply confirms ('yes')"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_run_drain_gate_declined_reply_refuses() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
printf 'no\n' > "$TMP/drain-reply-no.txt"
run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-no.txt" \
> "$TMP/drain-gate-no-output" 2>&1
[ "$DIED_CALLED" = 1 ] \
|| fail "run_drain_gate must refuse (call die via refuse_drain_gate) when the reply does not confirm"
printf '%s' "$DIED_MESSAGE" | grep -qF 'aborted' \
|| fail "run_drain_gate's refusal message does not read as an abort: $DIED_MESSAGE"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
# fleetd #555 item 4 — the report-state dispatch on $SUPERVISOR_KIND. Inverting this used to report
# the wrong supervisor and, for the launchd arm specifically, skip check_log_path_matches_plist.
CHECK_LOG_PATH_CALLED=0
stub_check_log_path_recorder() {
CHECK_LOG_PATH_CALLED=0
check_log_path_matches_plist() { CHECK_LOG_PATH_CALLED=1; }
}
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_check_log_path_recorder
SUPERVISED=0
report_supervisor_state "launchd" > "$TMP/report-supervisor-launchd-output" 2>&1
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state launchd must set SUPERVISED=1"
[ "$CHECK_LOG_PATH_CALLED" = 1 ] \
|| fail "report_supervisor_state launchd must call check_log_path_matches_plist"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_supervisor_state_systemd_sets_supervised_without_log_path_check() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_check_log_path_recorder
SUPERVISED=0
report_supervisor_state "systemd" > "$TMP/report-supervisor-systemd-output" 2>&1
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state systemd must set SUPERVISED=1"
[ "$CHECK_LOG_PATH_CALLED" = 0 ] \
|| fail "report_supervisor_state systemd must NOT call check_log_path_matches_plist"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_supervisor_state_none_leaves_supervised_zero() {
SUPERVISED=1
report_supervisor_state "none" > "$TMP/report-supervisor-none-output" 2>&1
[ "$SUPERVISED" = 0 ] || fail "report_supervisor_state none must leave SUPERVISED=0"
grep -qF 'no supervisor loaded' "$TMP/report-supervisor-none-output" \
|| fail "report_supervisor_state none did not print the unsupervised message"
}
test_report_supervisor_state_unrecognized_kind_warns_and_continues() {
SUPERVISED=1
report_supervisor_state "bogus-kind" > "$TMP/report-supervisor-bogus-output" 2>&1
[ "$SUPERVISED" = 0 ] || fail "report_supervisor_state must reset SUPERVISED for an unrecognized kind"
grep -qF "unrecognised supervisor kind: 'bogus-kind'" "$TMP/report-supervisor-bogus-output" \
|| fail "report_supervisor_state did not warn about the unrecognized kind"
}
test_report_supervisor_state_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'report_supervisor_state "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's report_supervisor_state call site in redeploy-fleetd.sh"
}
# fleetd #555 item 5 — the stop dispatch on $SUPERVISOR_KIND. Inverting this kills a supervised
# daemon with a raw kill instead of launchctl/systemctl, reviving the OLD jar (CB-594/#492). Each
# mechanism is stubbed to record which one ran, so a swapped or dropped arm shows up directly.
STOP_DISPATCH_CALLED=""
stub_stop_dispatch_recorders() {
STOP_DISPATCH_CALLED=""
stop_via_launchd() { STOP_DISPATCH_CALLED="launchd"; }
stop_via_systemd() { STOP_DISPATCH_CALLED="systemd"; }
stop_via_kill() { STOP_DISPATCH_CALLED="kill:$1"; }
}
test_dispatch_stop_launchd_calls_stop_via_launchd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
dispatch_stop "launchd" "123"
assert_equals "launchd" "$STOP_DISPATCH_CALLED" "dispatch_stop launchd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_systemd_calls_stop_via_systemd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
dispatch_stop "systemd" "123"
assert_equals "systemd" "$STOP_DISPATCH_CALLED" "dispatch_stop systemd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_none_calls_stop_via_kill_with_pid() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
dispatch_stop "none" "456"
assert_equals "kill:456" "$STOP_DISPATCH_CALLED" "dispatch_stop none"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_unrecognized_kind_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
stub_die_recorder
dispatch_stop "bogus-kind" "789"
[ "$DIED_CALLED" = 1 ] || fail "dispatch_stop must die on an unrecognized kind"
[ -z "$STOP_DISPATCH_CALLED" ] \
|| fail "dispatch_stop must not call any stop mechanism on an unrecognized kind"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'dispatch_stop "$SUPERVISOR_KIND" "$OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's dispatch_stop call site in redeploy-fleetd.sh"
}
# fleetd #555 item 6 — the start dispatch on $SUPERVISOR_KIND. Same shape as the stop dispatch.
START_DISPATCH_CALLED=""
stub_start_dispatch_recorders() {
START_DISPATCH_CALLED=""
start_via_launchd() { START_DISPATCH_CALLED="launchd"; }
start_via_systemd() { START_DISPATCH_CALLED="systemd"; }
start_via_none() { START_DISPATCH_CALLED="none"; }
}
test_dispatch_start_launchd_calls_start_via_launchd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
dispatch_start "launchd"
assert_equals "launchd" "$START_DISPATCH_CALLED" "dispatch_start launchd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_systemd_calls_start_via_systemd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
dispatch_start "systemd"
assert_equals "systemd" "$START_DISPATCH_CALLED" "dispatch_start systemd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_none_calls_start_via_none() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
dispatch_start "none"
assert_equals "none" "$START_DISPATCH_CALLED" "dispatch_start none"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_unrecognized_kind_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
stub_die_recorder
dispatch_start "bogus-kind"
[ "$DIED_CALLED" = 1 ] || fail "dispatch_start must die on an unrecognized kind"
[ -z "$START_DISPATCH_CALLED" ] \
|| fail "dispatch_start must not call any start mechanism on an unrecognized kind"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'dispatch_start "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's dispatch_start call site in redeploy-fleetd.sh"
}
# fleetd #555 item 7 — the HEALTH_BODY poll loop and the `[ -z "$HEALTH_BODY" ]` branch. Inverting
# health_is_up makes a dead daemon report /healthz 200 or a live one report failure.
test_health_is_up_true_with_body() {
health_is_up "some body" || fail "health_is_up must be true with a non-empty body"
}
test_health_is_up_false_with_empty_body() {
health_is_up "" && fail "health_is_up must be false with an empty body"
return 0
}
# fleetd #555: stub_die_recorder replaces die() globally for the rest of the process, the same as
# every other user of it in this file — re-source before AND after so a later test (e.g.
# test_unload_launchd_if_loaded_dies_on_real_failure, which needs the REAL die() to actually exit)
# never inherits a stub left over from here.
test_report_health_up_reports_ok_and_never_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
local output
output="$(report_health '{"status":"ok"}' "200" "$TMP/fake.out" 60)"
[ "$DIED_CALLED" = 0 ] || fail "report_health must not die when the body is non-empty"
printf '%s' "$output" | grep -qF '/healthz 200' \
|| fail "report_health did not report the healthy body"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_health_degraded_503_warns_and_never_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
local output
output="$(report_health "" "503" "$TMP/fake.out" 60 2>&1)"
[ "$DIED_CALLED" = 0 ] || fail "report_health must not die on a 503 (degraded, not dead)"
printf '%s' "$output" | grep -qF '503 degraded' \
|| fail "report_health did not report the 503-degraded case"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_health_dead_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
report_health "" "000" "$TMP/fake.out" 60 > "$TMP/report-health-dead-output" 2>&1
[ "$DIED_CALLED" = 1 ] || fail "report_health must die when the body is empty and the code is not 503"
printf '%s' "$DIED_MESSAGE" | grep -qF 'never answered' \
|| fail "report_health die message does not say healthz never answered"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_poll_health_body_returns_nonzero_when_unreachable() {
local rc=0
poll_health_body "http://127.0.0.1:1/healthz" 1 > /dev/null || rc=$?
[ "$rc" -ne 0 ] || fail "poll_health_body must return non-zero when the health endpoint is unreachable"
}
test_report_health_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'report_health "$HEALTH_BODY" "$HEALTH_CODE" "$OUT" "$HEALTH_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's report_health call site in redeploy-fleetd.sh"
}
# fleetd #555 item 8 — `HAD_OLD_PID=0; [ -n "$OLD_PID" ] && HAD_OLD_PID=1`. Measured safe under
# `set -e` at both bash 3.2.57 and 5.x (the ticket's own header discussion); still an untested
# computation feeding report_shutdown_drain's four-way decision until now.
test_compute_had_old_pid_zero_when_empty() {
assert_equals "0" "$(compute_had_old_pid "")" "compute_had_old_pid empty pid"
}
test_compute_had_old_pid_one_when_present() {
assert_equals "1" "$(compute_had_old_pid "12345")" "compute_had_old_pid present pid"
}
test_compute_had_old_pid_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'HAD_OLD_PID="$(compute_had_old_pid "$OLD_PID")"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's compute_had_old_pid call site in redeploy-fleetd.sh"
}
# fleetd #555 — the structural guard the ticket actually asks for: "pick the shape, not the eight
# sites." A future ninth bare main-flow conditional must fail THIS test on sight, not slip through
# with every function-level test (67, now many more) still green.
#
# Scope: from the SOURCED guard onward — the ticket's own "main flow" (report state / build / drain
# / stop / swap / start / verify). The `for arg in "$@"; do case "$arg" in ... esac; done` option
# parser above the guard is out of scope, the same way it sits outside the ticket's own eight-item
# table: it runs even when this file is sourced for tests, is not part of the mutating procedure,
# and none of the eight decisions the ticket names live there.
#
# For every line whose first non-blank token is `if`, `elif`, or `case`, and which sits OUTSIDE any
# function body, this demands an EXACT match — full trimmed line, never a regex anchor (a regex can
# match inside unrelated text this same mutation could introduce, e.g. a comment) — against
# MAIN_FLOW_ALLOWED_CONDITIONALS below. That allowlist is a closed set of the report-only/display
# conditionals already in the file (bare state prints: is the daemon running, is X installed, does
# this secret resolve — none of them gate a build/stop/start/swap decision) plus the two
# loaded-but-not-running elif branches that are the SAME shape as fleetd #504 items 2/3/4, one
# function over, and explicitly out of THIS ticket's scope per its own "Related" section.
#
# Every one of the eight decisions the ticket names has been lifted into its own predicate+action
# function above this point in the file (should_stop_for_check, drain_gate_required/run_drain_gate,
# report_supervisor_state, dispatch_stop, dispatch_start, health_is_up/report_health,
# compute_had_old_pid) — none of their guards appear in this list any more, which is what makes
# their specific inversions untestable-by-construction now: there is no bare guard left in the main
# flow for a future edit to invert silently. A NEW bare conditional (the ninth) is, by construction,
# not in this list, so this test goes red the instant one is added, naming the exact line — passing
# requires either lifting the decision into a function (with its own predicate test, same shape as
# above) or explicitly extending the allowlist, which is a diff a reviewer sees and can question.
#
# Function-body detection relies on this file's one consistent convention: a function opens as
# `name() {` alone on its own line and closes as a bare `}` alone on its own line — verified: every
# multi-line function in this file follows it, and the handful of one-liners like `say() { ...; }`
# sit above the SOURCED guard, outside the region this scans. Indentation is NOT used to decide
# "inside a function": a bare `if` inside a top-level `for`/`while` loop body (indented, but not
# inside any function) would be exactly as reportable as one at column 0 — this file currently has
# none, but a future one that hid inside a loop must not get a free pass for being indented.
MAIN_FLOW_ALLOWED_CONDITIONALS=(
'if [ -n "$OLD_PID" ]; then'
'if launchd_installed; then'
'if systemd_installed; then'
'if zsh -lc '\''[ -n "${WORKER_GITEA_TOKEN:-}" ]'\'' 2>/dev/null; then'
'if zsh -lc '\''[ -n "${AI_GATEWAY_TOKEN:-}" ]'\'' 2>/dev/null; then'
'if [ -z "$BROKER_URI_ENV" ]; then'
'elif zsh -lc "[ -n \"\${$BROKER_URI_ENV:-}\" ]" 2>/dev/null; then'
'if [ "$DO_BUILD" = 1 ]; then'
'if ! mvn -f "$MODULE/pom.xml" clean install > "$BUILD_LOG" 2>&1; then'
'if ! wait_for_daemon_exit "$STOP_WAIT"; then'
'elif [ "$SUPERVISOR_KIND" = "launchd" ]; then'
'elif [ "$SUPERVISOR_KIND" = "systemd" ]; then'
'if tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -q '\''fleetd listening'\''; then'
'if ! FRESH_LOG="$(capture_fresh_log_region "$OUT" "$RESTART_MARK")"; then'
'if [ "$REDEPLOY_AMQP_CHECK_SKIPPED" -eq 1 ]; then'
'elif [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then'
'if [ "$REDEPLOY_DRAIN_STATE" = "complete" ] || [ "$REDEPLOY_DRAIN_STATE" = "n/a" ]; then'
'elif [ "$REDEPLOY_UNEXPLAINED_ERRORS" -eq 0 ]; then'
)
# Emits "<lineno>\tCOND\t<trimmed line>" for every if/elif/case found outside a function, and
# "<lineno>\tFUNC\t<line>" for every function DEFINITION found after guard_line — from guard_line
# onward. A separate function (rather than inlined into the test) so the deliberate-mutation proof
# in the PR description can call it directly against a scratch copy of the script.
#
# fleetd #555 rework (comment 17012): a function defined after the SOURCED guard can never be
# reached by `source`-ing this script (sourcing returns before the main flow, and before any code
# below the guard runs), so anything inside such a function is untestable by construction — the
# depth tracker below would otherwise read it as "inside a function, therefore fine" and wave every
# conditional in it through unexamined. Emitting FUNC records lets the caller flag the function
# itself as the violation, independently of whether its body happens to contain a conditional.
mainflow_bare_conditionals() {
local src="$1" guard_line="$2" in_func=0 lineno=0 line trimmed
while IFS= read -r line || [ -n "$line" ]; do
lineno=$((lineno + 1))
[ "$lineno" -le "$guard_line" ] && continue
if [[ "$line" =~ ^[A-Za-z_][A-Za-z0-9_]*\(\)[[:space:]]*\{[[:space:]]*$ ]]; then
in_func=1
printf '%d\tFUNC\t%s\n' "$lineno" "$line"
continue
fi
if [ "$in_func" = 1 ] && [[ "$line" =~ ^\}[[:space:]]*$ ]]; then
in_func=0
continue
fi
if [ "$in_func" = 0 ]; then
trimmed="$(printf '%s' "$line" | sed -E 's/^[[:space:]]+//')"
if [[ "$trimmed" =~ ^(if|elif|case)[[:space:]] ]]; then
printf '%d\tCOND\t%s\n' "$lineno" "$trimmed"
fi
fi
done < "$src"
}
test_no_untested_main_flow_conditionals() {
local src="$ROOT/scripts/redeploy-fleetd.sh" guard_line
guard_line="$(grep -Fn 'if (return 0 2>/dev/null); then' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$guard_line" ] || { fail "could not find the SOURCED guard in redeploy-fleetd.sh"; return 1; }
local violations=0 report="" found_line found_kind found_text allowed candidate
while IFS=$'\t' read -r found_line found_kind found_text; do
[ -n "$found_line" ] || continue
if [ "$found_kind" = "FUNC" ]; then
# fleetd #555 rework: a function defined after the SOURCED guard line can never be sourced
# by this suite, so it can never be tested — that is a violation on its own, regardless of
# what its body contains or whether the allowlist would otherwise excuse a bare conditional
# inside it.
violations=$((violations + 1))
report="$report
line $found_line: function defined after the SOURCED guard (line $guard_line) — it cannot be sourced, so it cannot be tested: $found_text"
continue
fi
allowed=0
for candidate in "${MAIN_FLOW_ALLOWED_CONDITIONALS[@]}"; do
if [ "$found_text" = "$candidate" ]; then
allowed=1
break
fi
done
if [ "$allowed" = 0 ]; then
violations=$((violations + 1))
report="$report
line $found_line: $found_text"
fi
done < <(mainflow_bare_conditionals "$src" "$guard_line")
if [ "$violations" -gt 0 ]; then
fail "found $violations untested main-flow if/elif/case line(s) or function definition(s) after the SOURCED guard, not lifted into a tested predicate function and not in MAIN_FLOW_ALLOWED_CONDITIONALS:$report"
fi
}
# fleetd #504 — the "loaded but not currently running" branches for launchd/systemd used to run
# `launchctl unload`/`systemctl --user stop` with `2>/dev/null || true` and print `ok`
# unconditionally, so a real supervisor failure (e.g. it cannot reach launchd/the systemd user bus)
# read exactly like a harmless already-stopped answer. unload_launchd_if_loaded/
# stop_systemd_if_loaded (redeploy-fleetd.sh, right after systemd_loaded) apply systemd_loaded's own
# "capture stderr separately — only a non-zero exit WITH stderr is a real failure" pattern to the
# WRITE side. Both call the real `launchctl`/`systemctl` binaries directly (they are not overridable
# wrapper functions the way launchd_loaded/systemd_loaded are), so these tests put a stub binary
# first on PATH — the same technique test_detect_supervisor_systemd_probe_error_is_unclear above
# already uses for `systemctl`.
test_unload_launchd_if_loaded_dies_on_real_failure() {
local bin_dir output rc=0 saved_plist="$LAUNCHD_PLIST"
bin_dir="$TMP/stub-bin-launchctl-error"; mkdir -p "$bin_dir"
cat > "$bin_dir/launchctl" <<'STUB'
#!/usr/bin/env bash
echo "Could not find specified service" >&2
exit 1
STUB
chmod +x "$bin_dir/launchctl"
LAUNCHD_PLIST="$TMP/fake-fail.plist"
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
LAUNCHD_PLIST="$saved_plist"
[ "$rc" -ne 0 ] \
|| fail "unload_launchd_if_loaded must die when launchctl exits non-zero AND writes to stderr"
printf '%s' "$output" | grep -qF 'launchctl unload' \
|| fail "die message does not name the failing launchctl unload command"
}
# Captured via $(...) rather than called bare: unload_launchd_if_loaded's own die() does a hard
# `exit`, and calling it directly at this level would let a regression that makes it die on this
# clean-negative case kill the WHOLE suite before the `|| fail` below ever ran — printing die's own
# message instead of this test's. Inside a command substitution, that `exit` only ends the subshell
# (a-guard-is-defeated-by-its-calling-context: the same reason the *_dies_on_real_failure tests
# above capture this way), so this test's own message is what actually reaches the report.
test_unload_launchd_if_loaded_tolerates_clean_negative() {
local bin_dir saved_plist="$LAUNCHD_PLIST" output rc=0
bin_dir="$TMP/stub-bin-launchctl-noop"; mkdir -p "$bin_dir"
cat > "$bin_dir/launchctl" <<'STUB'
#!/usr/bin/env bash
exit 1
STUB
chmod +x "$bin_dir/launchctl"
LAUNCHD_PLIST="$TMP/fake-noop.plist"
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
LAUNCHD_PLIST="$saved_plist"
[ "$rc" -eq 0 ] \
|| fail "unload_launchd_if_loaded must tolerate a clean already-unloaded answer (non-zero exit, empty stderr): $output"
}
test_stop_systemd_if_loaded_dies_on_real_failure() {
local bin_dir output rc=0
bin_dir="$TMP/stub-bin-systemctl-stop-error"; mkdir -p "$bin_dir"
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
echo "Failed to connect to bus: No such file or directory" >&2
exit 1
STUB
chmod +x "$bin_dir/systemctl"
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
[ "$rc" -ne 0 ] \
|| fail "stop_systemd_if_loaded must die when systemctl exits non-zero AND writes to stderr"
printf '%s' "$output" | grep -qF 'systemctl --user stop' \
|| fail "die message does not name the failing systemctl --user stop command"
}
# Same subshell-capture reasoning as test_unload_launchd_if_loaded_tolerates_clean_negative above:
# stop_systemd_if_loaded's own die() does a hard `exit`, so this must run inside $(...) or a
# regression here would kill the whole suite with die's message instead of this test's.
test_stop_systemd_if_loaded_tolerates_clean_negative() {
local bin_dir output rc=0
bin_dir="$TMP/stub-bin-systemctl-stop-noop"; mkdir -p "$bin_dir"
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
exit 1
STUB
chmod +x "$bin_dir/systemctl"
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
[ "$rc" -eq 0 ] \
|| fail "stop_systemd_if_loaded must tolerate a clean already-stopped answer (non-zero exit, empty stderr): $output"
}
# Closes the same gap test_run_drain_gate_call_site_present closes for the drain gate: the four
# tests above call unload_launchd_if_loaded/stop_systemd_if_loaded directly, and sourcing stops
# before the main flow ever runs (the SOURCED guard), so none of them can prove the main flow still
# CALLS these two functions instead of the original bare `2>/dev/null || true`. A source-text check,
# like test_swap_ordered_after_wait_and_before_start. The call-site needle is anchored (`^ name$`)
# so it cannot be satisfied by the comment lines above each call site that merely mention the
# function by name.
test_stop_branches_call_tolerant_helpers_not_bare_or_true() {
local src="$ROOT/scripts/redeploy-fleetd.sh" unload_call_line stop_call_line
unload_call_line="$(grep -n '^ unload_launchd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
stop_call_line="$(grep -n '^ stop_systemd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$unload_call_line" ] \
|| fail "could not find the main flow's call to unload_launchd_if_loaded in redeploy-fleetd.sh"
[ -n "$stop_call_line" ] \
|| fail "could not find the main flow's call to stop_systemd_if_loaded in redeploy-fleetd.sh"
if grep -qF 'launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true' "$src"; then
fail "the bare 'launchctl unload ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
fi
if grep -qF 'systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true' "$src"; then
fail "the bare 'systemctl --user stop ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
fi
}
test_no_errors() {
cat > "$TMP/no-errors.log" <<'LOG'
2026-09-05 12:00:00 INFO fleetd listening
LOG
classify_fixture no-errors.log
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total"
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained"
# fleetd #552 control: a REAL, readable log region with nothing in it must never look like a
# region that could not be captured at all.
assert_equals 0 "$REDEPLOY_AMQP_CHECK_SKIPPED" "no-errors must not report the check as skipped"
}
# fleetd #552 — the third state: "" (capture_fresh_log_region's own sentinel for "the post-restart
# region could never be captured") must never look like "read a file with nothing in it", which is
# exactly what REDEPLOY_ERROR_COUNT staying at 0 already means on its own (see test_no_errors above).
test_classify_amqp_connection_errors_reports_skipped_when_log_missing() {
classify_amqp_connection_errors "" > "$TMP/classify-skipped-output" 2>&1
assert_equals 1 "$REDEPLOY_AMQP_CHECK_SKIPPED" "an empty log_file must set REDEPLOY_AMQP_CHECK_SKIPPED=1"
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "a skipped check must still leave REDEPLOY_ERROR_COUNT at its initialized 0 (the flag, not the count, carries the distinction)"
grep -qF 'could not be captured' "$TMP/classify-skipped-output" \
|| fail "classify_amqp_connection_errors did not say the post-restart log region could not be captured"
}
test_recovery_patterns_match_source() {
grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|| fail "AMQP failure pattern no longer matches source"
grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|| fail "reply-inbox recovery pattern no longer matches source"
grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \
|| fail "lead-mailbox recovery pattern no longer matches source"
}
test_attributed_recovered_connection_error() {
cat > "$TMP/attributed-recovered.log" <<'LOG'
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
LOG
classify_fixture attributed-recovered.log
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total"
assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors"
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained"
}
test_source_derived_error_shapes_recover_by_connection() {
# These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check
# after redeploy because the new code has not yet written a production line.
cat > "$TMP/source-derived.log" <<'LOG'
17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery!
17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!
17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
LOG
classify_fixture source-derived.log
assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total"
assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered"
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained"
}
test_cross_connection_unattributable_errors_stay_loud() {
# This candidate has neither stable connection name, so LeadMailbox recovery must not consume it.
cat > "$TMP/cross-unattributable.log" <<'LOG'
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
LOG
classify_fixture cross-unattributable.log
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total"
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered"
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained"
}
test_attributed_cross_connection_errors_stay_loud() {
# LeadMailbox recovery cannot heal AmqpReplyInbox errors.
cat > "$TMP/cross-attributed.log" <<'LOG'
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
LOG
classify_fixture cross-attributed.log
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total"
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered"
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained"
}
test_attributed_unrecovered_connection_error() {
cat > "$TMP/unrecovered.log" <<'LOG'
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
LOG
classify_fixture unrecovered.log
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total"
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors"
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained"
}
test_other_error_is_unexplained() {
cat > "$TMP/other-error.log" <<'LOG'
2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed
2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
LOG
classify_fixture other-error.log
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total"
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained"
}
test_recovery_requirement_mutation_is_caught() {
classify_amqp_connection_errors() {
local log_file="$1" line
REDEPLOY_ERROR_COUNT=0
REDEPLOY_RECOVERED_AMQP_ERRORS=0
REDEPLOY_UNEXPLAINED_ERRORS=0
while IFS= read -r line || [ -n "$line" ]; do
case "$line" in
*' ERROR '*|*' SEVERE '*)
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
case "$line" in
*'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*)
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
esac
done < "$log_file"
}
if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then
fail "mutation accepted an unrecovered connection error"
fi
grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \
|| fail "mutation failed without the expected assertion"
printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n'
}
test_shared_counter_mutation_is_caught() {
classify_amqp_connection_errors() {
local log_file="$1" line pending=0
REDEPLOY_ERROR_COUNT=0
REDEPLOY_RECOVERED_AMQP_ERRORS=0
REDEPLOY_UNEXPLAINED_ERRORS=0
while IFS= read -r line || [ -n "$line" ]; do
case "$line" in
*' ERROR '*|*' SEVERE '*)
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
case "$line" in
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
case "$line" in
*'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
*'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*)
if [ "$pending" -gt 0 ]; then
pending=$((pending - 1))
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
fi
;;
esac
done < "$log_file"
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending))
}
if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then
fail "shared counter mutation accepted cross-connection recovery"
fi
grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \
|| fail "shared counter mutation failed without the expected assertion"
printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n'
}
test_unattributable_quiet_mutation_is_caught() {
classify_amqp_connection_errors() {
local log_file="$1" line
REDEPLOY_ERROR_COUNT=0
REDEPLOY_RECOVERED_AMQP_ERRORS=0
REDEPLOY_UNEXPLAINED_ERRORS=0
while IFS= read -r line || [ -n "$line" ]; do
case "$line" in
*' ERROR '*|*' SEVERE '*)
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
case "$line" in
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
esac
done < "$log_file"
}
if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then
fail "unattributable mutation accepted an unknown connection"
fi
grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \
|| fail "unattributable mutation failed without the expected assertion"
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
}
# fleetd #512 part 2 — the negative check (scan_uncaught_exceptions). The heart of this half of the
# ticket: a fixture with the uncaught-exception shape and NO line carrying an ERROR token at all,
# proving the scan finds it without one. A fixture that also carried an ERROR line would pass for
# the wrong reason.
test_scan_uncaught_exceptions_finds_shape_without_error_token() {
cat > "$TMP/scan-died.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
LOG
local error_count
error_count="$(grep -c ' ERROR ' "$TMP/scan-died.log" || true)"
[ "$error_count" = "0" ] \
|| fail "test fixture error: scan-died.log unexpectedly carries an ERROR token"
scan_uncaught_exceptions "$TMP/scan-died.log"
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "scan must find the exception without an ERROR token"
printf '%s' "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" | grep -qF 'NoClassDefFoundError' \
|| fail "scan did not capture the matching line as the sample"
}
test_scan_uncaught_exceptions_clean_control() {
cat > "$TMP/scan-clean.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=0 abandoned=0 (still BUSY at the shutdown deadline)
LOG
scan_uncaught_exceptions "$TMP/scan-clean.log"
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "clean control must find no uncaught exception"
assert_equals "" "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" "clean control sample must be empty"
}
# fleetd #512 part 2 — the positive check (find_drain_complete_line). Both halves of #522's line:
# present, and absent.
test_find_drain_complete_line_present() {
cat > "$TMP/drain-line-present.log" <<'LOG'
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=2 abandoned=1 (still BUSY at the shutdown deadline)
LOG
find_drain_complete_line "$TMP/drain-line-present.log"
printf '%s' "$REDEPLOY_DRAIN_COMPLETE_LINE" | grep -qF 'released=2 abandoned=1' \
|| fail "find_drain_complete_line did not capture the present line"
}
test_find_drain_complete_line_absent() {
cat > "$TMP/drain-line-absent.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
LOG
find_drain_complete_line "$TMP/drain-line-absent.log"
assert_equals "" "$REDEPLOY_DRAIN_COMPLETE_LINE" "find_drain_complete_line must report empty when absent"
}
# fleetd #512 part 2 — report_shutdown_drain, the composite decision+action function the main flow
# calls unconditionally (same shape as swap_if_built/refuse_drain_gate, #521/#528). These four cover
# the four outcomes named in the ticket's "trap": complete, died, unknown ("cannot tell" — neither a
# pass nor a failure), and n/a (no previous daemon was actually stopped this run).
#
# Deliberately NOT run inside `$(...)`: report_shutdown_drain sets REDEPLOY_DRAIN_STATE as a global
# side effect that these tests need to read back afterward, and a command substitution forks a
# subshell that global assignment would not survive (the exact trap documented above
# detect_supervisor in redeploy-fleetd.sh, for the same reason). Plain output redirection to a file
# does not fork a subshell, so it is used to capture what was printed instead.
test_report_shutdown_drain_died_without_error_token() {
cat > "$TMP/drain-died.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
LOG
local error_count
error_count="$(grep -c ' ERROR ' "$TMP/drain-died.log" || true)"
[ "$error_count" = "0" ] \
|| fail "test fixture error: drain-died.log unexpectedly carries an ERROR token"
report_shutdown_drain "$TMP/drain-died.log" 1 > "$TMP/drain-died-output" 2>&1
assert_equals "died" "$REDEPLOY_DRAIN_STATE" "died fixture must set REDEPLOY_DRAIN_STATE=died"
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "died fixture uncaught-exception count"
grep -qF 'NoClassDefFoundError' "$TMP/drain-died-output" \
|| fail "report_shutdown_drain did not report the uncaught-exception shape it found"
grep -qF 'DIED' "$TMP/drain-died-output" \
|| fail "report_shutdown_drain did not report the drain as DIED"
}
test_report_shutdown_drain_complete_control() {
cat > "$TMP/drain-complete.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=3 abandoned=0 (still BUSY at the shutdown deadline)
LOG
report_shutdown_drain "$TMP/drain-complete.log" 1 > "$TMP/drain-complete-output" 2>&1
assert_equals "complete" "$REDEPLOY_DRAIN_STATE" "complete-control fixture must set REDEPLOY_DRAIN_STATE=complete"
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "complete-control fixture must find no uncaught exception"
grep -qF 'released=3 abandoned=0' "$TMP/drain-complete-output" \
|| fail "report_shutdown_drain did not report the drain-complete counts"
}
test_report_shutdown_drain_unknown_cannot_tell() {
cat > "$TMP/drain-unknown.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
LOG
report_shutdown_drain "$TMP/drain-unknown.log" 1 > "$TMP/drain-unknown-output" 2>&1
assert_equals "unknown" "$REDEPLOY_DRAIN_STATE" "cannot-tell fixture must set REDEPLOY_DRAIN_STATE=unknown"
grep -qF 'cannot tell' "$TMP/drain-unknown-output" \
|| fail "report_shutdown_drain did not say it could not tell"
if grep -qF ' ok' "$TMP/drain-unknown-output"; then
fail "cannot-tell outcome must not be printed via ok() — it is neither a pass nor a failure"
fi
}
# A cold start (or a restart where nothing was actually stopped) has no previous-daemon shutdown
# window to have an opinion about at all. This fixture's log content looks exactly like a died drain
# — proving the had_previous_daemon=0 gate is actually consulted, not merely documented: without it,
# this would misreport "died" or "unknown" on every clean cold start.
test_report_shutdown_drain_no_previous_daemon_is_na() {
cat > "$TMP/drain-na.log" <<'LOG'
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
LOG
report_shutdown_drain "$TMP/drain-na.log" 0 > "$TMP/drain-na-output" 2>&1
assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon fixture must set REDEPLOY_DRAIN_STATE=n/a even though the log content looks like a died drain"
grep -qF 'nothing to check' "$TMP/drain-na-output" \
|| fail "report_shutdown_drain did not report that there was nothing to check"
}
# fleetd #552 — a fifth outcome: log_file="" (capture_fresh_log_region's own sentinel) with a
# previous daemon that WAS stopped this run. Distinct from "unknown" (a log was read and neither
# signal was found in it) — here there was no log to read at all.
test_report_shutdown_drain_skipped_when_log_missing() {
report_shutdown_drain "" 1 > "$TMP/drain-skipped-output" 2>&1
assert_equals "skipped" "$REDEPLOY_DRAIN_STATE" "an empty log_file with a previous daemon must set REDEPLOY_DRAIN_STATE=skipped"
grep -qF 'could not be captured' "$TMP/drain-skipped-output" \
|| fail "report_shutdown_drain did not say the post-restart log region could not be captured"
if grep -qF ' ok' "$TMP/drain-skipped-output"; then
fail "the skipped outcome must not be printed via ok() — it is neither a pass nor a failure"
fi
}
# fleetd #552 — proves the had_previous_daemon=0 check really does run BEFORE the missing-log check:
# with BOTH conditions true at once, n/a must win, because a cold start has nothing to check
# regardless of whether the log region could be captured.
test_report_shutdown_drain_na_wins_over_missing_log() {
report_shutdown_drain "" 0 > "$TMP/drain-na-and-missing-output" 2>&1
assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon must win over a missing log region"
}
# fleetd #512 — closes the gap none of the seven tests above can: they call report_shutdown_drain
# directly, and sourcing stops before the main flow ever runs (the SOURCED guard), so none of them
# can prove the main flow still calls it at all. Same shape as test_run_drain_gate_call_site_present
# and test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site, plus
# an ordering check against its neighbours in the verify/result flow.
test_report_shutdown_drain_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's report_shutdown_drain call site in redeploy-fleetd.sh"
}
test_report_shutdown_drain_ordered_after_classify_and_before_result() {
local src="$ROOT/scripts/redeploy-fleetd.sh" classify_line drain_line result_line
classify_line="$(grep -Fn 'classify_amqp_connection_errors "$FRESH_LOG"' "$src" | tail -1 | cut -d: -f1 || true)"
drain_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$classify_line" ] || fail "could not find the classify_amqp_connection_errors call site"
[ -n "$drain_line" ] || fail "could not find the report_shutdown_drain call site"
[ -n "$result_line" ] || fail "could not find the result section"
[ "$drain_line" -gt "$classify_line" ] \
|| fail "report_shutdown_drain (line $drain_line) is not after classify_amqp_connection_errors (line $classify_line)"
[ "$drain_line" -lt "$result_line" ] \
|| fail "report_shutdown_drain (line $drain_line) is not before the result section (line $result_line)"
}
# fleetd #512 item 4 — the summary line must not read as reassurance when the shutdown-drain check
# found something wrong (or could not tell). Sourcing stops before the main flow runs, so this is a
# source-text check like test_drain_gate_abort_message_says_no_no_build above.
test_no_error_lines_message_gated_by_drain_state() {
local src="$ROOT/scripts/redeploy-fleetd.sh" block
block="$(grep -B2 -F 'ok "no ERROR lines since restart"' "$src")"
[ -n "$block" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh"
printf '%s' "$block" | grep -qF 'REDEPLOY_DRAIN_STATE' \
|| fail "'no ERROR lines since restart' is not guarded by the shutdown-drain outcome (fleetd #512 item 4)"
}
test_detect_supervisor_launchd_only
test_detect_supervisor_systemd_only
test_detect_supervisor_none
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
test_detect_supervisor_systemd_probe_error_is_unclear
test_detect_supervisor_systemd_probe_setup_failure_is_unclear
test_mktemp_dash_t_templates_have_x_placeholders
test_no_unguarded_macos_only_hasher_calls
test_no_unguarded_mktemp_assignments_after_restart
test_capture_fresh_log_region_control
test_capture_fresh_log_region_mktemp_failure_returns_nonzero_and_prints_nothing
test_fresh_log_capture_failure_does_not_abort_under_sete
test_capture_fresh_log_region_call_site_is_guarded
test_result_section_checks_amqp_skip_before_no_error_lines
test_require_drivable_supervisor_refuses_ambiguous
test_require_drivable_supervisor_refuses_unclear
test_require_drivable_supervisor_accepts_known_kinds
test_count_daemon_pids
test_assert_single_daemon_accepts_one_pid
test_assert_single_daemon_rejects_two_pids
test_jar_id_defaults_to_live_and_reports_explicit_path
test_hash256_computes_a_real_sha256
test_jar_id_reports_absent_for_missing_file
test_jar_id_reports_unhashable_when_no_hasher_on_path
test_stage_built_jar_moves_off_live_path
test_stage_built_jar_dies_when_build_produced_nothing
test_swap_staged_jar_moves_staged_onto_live
test_swap_staged_jar_dies_without_staged_file
test_swap_staged_jar_dies_when_mv_fails
test_should_swap_true_when_build_ran
test_should_swap_false_when_build_skipped
test_swap_if_built_performs_the_swap_when_build_ran
test_swap_if_built_skips_the_swap_when_build_skipped
test_require_no_build_jar_dies_when_absent
test_require_no_build_jar_accepts_present_jar
test_wait_for_daemon_exit_returns_true_once_pid_clears
test_wait_for_daemon_exit_times_out_if_pid_never_clears
test_swap_ordered_after_wait_and_before_start
test_drain_gate_abort_message_says_no_no_build
test_drain_gate_refusal_build_ran_staged_present
test_drain_gate_refusal_build_ran_staged_absent
test_drain_gate_refusal_no_build_staged_present
test_drain_gate_refusal_no_build_staged_absent
test_refuse_drain_gate_build_ran_staged_present
test_refuse_drain_gate_build_ran_staged_absent
test_refuse_drain_gate_no_build_staged_present
test_refuse_drain_gate_no_build_staged_absent
test_run_drain_gate_call_site_present
test_should_stop_for_check_true_when_check_only
test_should_stop_for_check_false_when_not_check_only
test_stop_if_check_only_exits_zero_and_says_nothing_changed
test_stop_if_check_only_returns_without_exiting_when_not_check_only
test_stop_if_check_only_call_site_present
test_drain_gate_required_true_when_pid_and_not_assume_yes
test_drain_gate_required_false_when_no_pid
test_drain_gate_required_false_when_assume_yes
test_drain_confirmed_true_on_yes
test_drain_confirmed_false_on_anything_else
test_run_drain_gate_skips_prompt_when_not_required
test_run_drain_gate_confirmed_reply_does_not_refuse
test_run_drain_gate_declined_reply_refuses
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path
test_report_supervisor_state_systemd_sets_supervised_without_log_path_check
test_report_supervisor_state_none_leaves_supervised_zero
test_report_supervisor_state_unrecognized_kind_warns_and_continues
test_report_supervisor_state_call_site_present
test_dispatch_stop_launchd_calls_stop_via_launchd
test_dispatch_stop_systemd_calls_stop_via_systemd
test_dispatch_stop_none_calls_stop_via_kill_with_pid
test_dispatch_stop_unrecognized_kind_dies
test_dispatch_stop_call_site_present
test_dispatch_start_launchd_calls_start_via_launchd
test_dispatch_start_systemd_calls_start_via_systemd
test_dispatch_start_none_calls_start_via_none
test_dispatch_start_unrecognized_kind_dies
test_dispatch_start_call_site_present
test_health_is_up_true_with_body
test_health_is_up_false_with_empty_body
test_report_health_up_reports_ok_and_never_dies
test_report_health_degraded_503_warns_and_never_dies
test_report_health_dead_dies
test_poll_health_body_returns_nonzero_when_unreachable
test_report_health_call_site_present
test_compute_had_old_pid_zero_when_empty
test_compute_had_old_pid_one_when_present
test_compute_had_old_pid_call_site_present
test_no_untested_main_flow_conditionals
test_unload_launchd_if_loaded_dies_on_real_failure
test_unload_launchd_if_loaded_tolerates_clean_negative
test_stop_systemd_if_loaded_dies_on_real_failure
test_stop_systemd_if_loaded_tolerates_clean_negative
test_stop_branches_call_tolerant_helpers_not_bare_or_true
test_no_errors
test_classify_amqp_connection_errors_reports_skipped_when_log_missing
test_recovery_patterns_match_source
test_attributed_recovered_connection_error
test_source_derived_error_shapes_recover_by_connection
test_cross_connection_unattributable_errors_stay_loud
test_attributed_cross_connection_errors_stay_loud
test_attributed_unrecovered_connection_error
test_other_error_is_unexplained
test_recovery_requirement_mutation_is_caught
test_shared_counter_mutation_is_caught
test_unattributable_quiet_mutation_is_caught
test_scan_uncaught_exceptions_finds_shape_without_error_token
test_scan_uncaught_exceptions_clean_control
test_find_drain_complete_line_present
test_find_drain_complete_line_absent
test_report_shutdown_drain_died_without_error_token
test_report_shutdown_drain_complete_control
test_report_shutdown_drain_unknown_cannot_tell
test_report_shutdown_drain_no_previous_daemon_is_na
test_report_shutdown_drain_skipped_when_log_missing
test_report_shutdown_drain_na_wins_over_missing_log
test_report_shutdown_drain_call_site_present
test_report_shutdown_drain_ordered_after_classify_and_before_result
test_no_error_lines_message_gated_by_drain_state
printf 'PASS: redeploy log classifier\n'