Files
fleetd/scripts/test-redeploy-fleetd.sh
T
Dai Ha 856dfc6318
CI / shell-tests (pull_request) Failing after 7s
CI / contract (pull_request) Successful in 51s
CI / build (pull_request) Successful in 2m15s
fleetd #603 review: close the untested fall-through path
PR review (comment 17358) found a real gap via mutation testing: replacing
the "pid never found" warn with die "no process appeared" left the whole
suite green, because neither existing test drove the case the fall-through
exists for -- running_pid() never finds anything (as its own doc comment
says it eventually will) while /healthz answers anyway.

Adds test_await_daemon_started_pid_never_found_but_healthy_warns_and_survives:
running_pid always empty, poll_health_body succeeds. Asserts DIED_CALLED=0,
the warn line is emitted, and NEW_PID stays empty (the honest "could not
establish this" answer, never a guessed pid).

Verified both halves myself: reverting the warn to die "no process appeared"
turns this one test red (FAIL: await_daemon_started must not die...);
restoring it returns the suite to green.
2026-09-20 16:28:53 +07:00

2264 lines
123 KiB
Bash
Executable File

#!/usr/bin/env bash
# Self-contained checks for the pure log classifier in redeploy-fleetd.sh.
set -euo pipefail
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")"
trap 'rm -rf "$TMP"' EXIT
# Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon.
source "$ROOT/scripts/redeploy-fleetd.sh"
fail() {
printf 'FAIL: %s\n' "$*" >&2
return 1
}
assert_equals() {
local expected="$1" actual="$2" description="$3"
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
}
classify_fixture() {
local name="$1"
classify_amqp_connection_errors "$TMP/$name"
}
# fleetd #492 follow-up: detect_supervisor's stdout is now "kind<SEP>detail" (see the constraints
# comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about
# the kind must split it out with the SAME in-shell parameter expansion the real call site (:438)
# uses, never a bare string comparison against the raw output.
supervisor_kind_of() {
printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}"
}
supervisor_detail_of() {
printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}"
}
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
# being naturally absent: this machine may itself be running a real fleetd under launchd right now
# (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving
# launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state
# instead of the fixture.
test_detect_supervisor_launchd_only() {
launchd_installed() { return 0; }
launchd_loaded() { return 0; }
systemd_installed() { return 1; }
systemd_loaded() { return 1; }
assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection"
}
test_detect_supervisor_systemd_only() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 0; }
systemd_loaded() { return 0; }
assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection"
}
test_detect_supervisor_none() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 1; }
systemd_loaded() { return 1; }
assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection"
}
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
# auto-restart too, and every one of those is a host that IS under systemd.
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 0; } # the unit file IS there
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
local raw
raw="$(detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none"
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \
|| fail "detail does not name the systemd unit it found installed-but-not-loaded"
}
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
# removed, or about to be reloaded) must not read as "no supervisor" either.
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
launchd_installed() { return 0; } # the plist IS there
launchd_loaded() { return 1; } # launchctl list says not loaded
systemd_installed() { return 1; }
systemd_loaded() { return 1; }
local raw
raw="$(detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none"
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \
|| fail "detail does not name the launchd label it found installed-but-not-loaded"
}
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
# the same fact as "no supervisor is loaded".
test_detect_supervisor_systemd_probe_error_is_unclear() {
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
# of those tests), and a bash function definition is global for the rest of the process — without
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
# a real `systemctl` call at all.
source "$ROOT/scripts/redeploy-fleetd.sh"
local bin_dir result rc=0
bin_dir="$TMP/stub-bin-systemctl-errors"
mkdir -p "$bin_dir"
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
echo "Failed to connect to bus: No such file or directory" >&2
exit 1
STUB
chmod +x "$bin_dir/systemctl"
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none"
local detail
detail="$(supervisor_detail_of "$result")"
printf '%s' "$detail" | grep -qF "$SYSTEMD_UNIT" \
|| fail "detail does not name the systemd unit whose probe errored"
# fleetd #545: this is the PROBE-RAN-AND-ANSWERED-BADLY case (systemctl actually executed and
# wrote to stderr) — it must carry that story and never the SET-UP-FAILED story (mktemp never
# even ran here), or the two "unclear" causes have collapsed back into one message that asserts a
# cause it did not measure, which is the exact defect this ticket exists to fix.
printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \
|| fail "detail does not say systemctl ran and answered with stderr: $detail"
printf '%s' "$detail" | grep -qF "could not even be set up" \
&& fail "detail wrongly claims the probe could not be set up, but systemctl actually ran and answered on stderr: $detail"
return 0
}
# fleetd #545 — the companion case to the probe-error test above: here `mktemp` itself fails
# (whatever the reason — the historical bug was a GNU-mktemp-rejects-a-template-with-no-Xs case,
# but this stub simulates ANY reason the probe's own stderr-capture temp file cannot be created,
# e.g. a full or unwritable temp dir) and `systemctl` is never invoked at all. Before this ticket,
# this collapsed into the SAME "systemctl exited non-zero and reported an error on stderr" detail
# as the sibling test above, which asserts a cause (systemctl ran and answered badly) that was
# never measured, because systemctl never ran. This proves the SET-UP-FAILED detail is distinct and
# does not claim systemctl said anything.
test_detect_supervisor_systemd_probe_setup_failure_is_unclear() {
# Re-source first for the same reason test_detect_supervisor_systemd_probe_error_is_unclear does:
# restore the REAL probe bodies before driving them through a stub PATH.
source "$ROOT/scripts/redeploy-fleetd.sh"
local bin_dir result rc=0
bin_dir="$TMP/stub-bin-mktemp-fails"
mkdir -p "$bin_dir"
# A systemctl stub that would fail loudly if it were ever actually invoked — proves the mktemp
# failure short-circuits the probe before systemctl runs, not merely that this test forgot to
# supply a working systemctl.
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
echo "systemctl must never run when mktemp already failed" >&2
exit 1
STUB
chmod +x "$bin_dir/systemctl"
cat > "$bin_dir/mktemp" <<'STUB'
#!/usr/bin/env bash
echo "mktemp: cannot create temp file" >&2
exit 1
STUB
chmod +x "$bin_dir/mktemp"
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
[ "$rc" -ne 0 ] \
|| fail "systemd_loaded must not report loaded=true when its own mktemp setup failed"
assert_equals "2" "$SYSTEMD_LOADED_ERRORED" \
"systemd_loaded must flag a SETUP failure (2), distinct from a probe-answered-with-stderr failure (1)"
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe setup failure must read as unclear, not none"
local detail
detail="$(supervisor_detail_of "$result")"
printf '%s' "$detail" | grep -qF "could not even be set up" \
|| fail "detail does not say the probe could not be SET UP: $detail"
printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \
&& fail "detail wrongly asserts systemctl exited non-zero and reported an error on stderr, but systemctl was never run: $detail"
return 0
}
# fleetd #545 — source-text check: every `mktemp -t` template in redeploy-fleetd.sh must contain an
# `X` placeholder. BSD mktemp (macOS) tolerates a bare template with no `X`s and just appends its
# own random suffix, which is exactly why six such sites survived undetected here — GNU mktemp
# (every Linux distribution) refuses a template with fewer than three `X`s and exits non-zero. There
# is no BSD-vs-GNU seam to stub on this Mac, so this is a source-text check rather than a
# behavioural one, the same shape as test_run_drain_gate_call_site_present above. Anchored on
# `mktemp -t ` (with the trailing space) so it inspects only the `-t`-style templates this ticket is
# about, never the `mktemp -d` calls this file and test-probe-member-credentials.sh already use
# (both already carry their own `XXXXXX` and are a different mktemp mode entirely).
test_mktemp_dash_t_templates_have_x_placeholders() {
local src="$ROOT/scripts/redeploy-fleetd.sh" bad
bad="$(grep -n 'mktemp -t ' "$src" | grep -v 'XXX' || true)"
[ -z "$bad" ] \
|| fail "mktemp -t template(s) with no X placeholder (fails under GNU coreutils): $bad"
}
# fleetd #550 — the shape, not the named lines: #545 showed the exact same failure mode (a
# macOS-only idiom used with no portable fallback) spread from two sites to six across 91 commits
# before anyone tested the SHAPE rather than specific lines. This is the shasum sibling: any script
# under scripts/ that actually INVOKES the macOS-only hasher to compute a hash (as opposed to
# merely probing whether it exists with `command -v`, or mentioning it in prose) must also check
# for the portable one first in that same file — the prefer-portable-fall-back-to-macOS-only idiom
# probe-member-credentials.sh:273-279 and this ticket's own hash256 helper both follow.
#
# The needle is built from two concatenated pieces, deliberately never written as one literal
# string in this file: written whole, it would match THIS CHECK'S OWN source line once the loop
# below reaches this very file, and the check would then "pass" by matching itself rather than any
# real invocation elsewhere — a zero-findings result indistinguishable from a clean file.
test_no_unguarded_macos_only_hasher_calls() {
local needle f bad="" usage
needle='shasum'; needle="$needle -a"
for f in "$ROOT"/scripts/*.sh; do
[ -f "$f" ] || continue
usage="$(grep -Fn "$needle" "$f" || true)"
if [ -n "$usage" ]; then
grep -q 'command -v sha256sum' "$f" \
|| bad="$bad$(basename "$f") "
fi
done
[ -z "$bad" ] \
|| fail "script(s) invoke the macOS-only hasher with no portable-hasher-first fallback guard in the same file: $bad"
}
# fleetd #552 — the shape, not the one line just fixed: a `mktemp` failure AFTER the daemon has
# already been restarted must never be a bare, unguarded assignment again, anywhere in the script,
# so the next one added is caught too — not just line 1113 as it stood at 26f380a. Anchored on the
# real "start" section (the restart call itself), the same anchor test_swap_ordered_after_wait_and_
# before_start above already uses for "before the start section". Needle built from two concatenated
# pieces, the same trick test_no_unguarded_macos_only_hasher_calls uses above, so this check's own
# description of the pattern it looks for can never become a match for itself.
test_no_unguarded_mktemp_assignments_after_restart() {
local src="$ROOT/scripts/redeploy-fleetd.sh" start_line needle bad
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$start_line" ] || fail "could not find the restart call's start section in redeploy-fleetd.sh"
needle='^[[:space:]]*[A-Za-z_][A-Za-z0-9_]*="'
needle="$needle"'\$\(mktemp'
bad="$(grep -nE "$needle" "$src" | awk -F: -v start="$start_line" '$1+0>start')"
[ -z "$bad" ] \
|| fail "unguarded 'VAR=\"\$(mktemp ...)\"' assignment(s) after the restart call (fleetd #552): $bad"
}
# fleetd #552 — capture_fresh_log_region itself. Sourcing stops before the main flow can be driven
# directly (the SOURCED guard below), so this exercises the extracted function the real call site
# now uses, the same stub-mktemp-on-PATH technique
# test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses above.
test_capture_fresh_log_region_control() {
printf 'line one\nline two\nline three\n' > "$TMP/fresh-log-src.out"
local result
result="$(capture_fresh_log_region "$TMP/fresh-log-src.out" 0)"
[ -n "$result" ] || fail "capture_fresh_log_region returned nothing on a working mktemp"
[ -f "$result" ] || fail "capture_fresh_log_region did not create the file it named"
assert_equals "$(printf 'line one\nline two\nline three')" "$(cat "$result")" \
"captured region did not carry the whole source file (restart_mark=0)"
rm -f "$result"
}
test_capture_fresh_log_region_mktemp_failure_returns_nonzero_and_prints_nothing() {
local bin_dir rc=0 out
bin_dir="$TMP/stub-bin-fresh-log-mktemp-fails"; mkdir -p "$bin_dir"
cat > "$bin_dir/mktemp" <<'STUB'
#!/usr/bin/env bash
echo "mktemp: cannot create temp file" >&2
exit 1
STUB
chmod +x "$bin_dir/mktemp"
printf 'line one\n' > "$TMP/fresh-log-src2.out"
out="$(PATH="$bin_dir:$PATH" capture_fresh_log_region "$TMP/fresh-log-src2.out" 0)" && rc=0 || rc=$?
[ "$rc" -ne 0 ] \
|| fail "capture_fresh_log_region must return non-zero when its own mktemp fails"
[ -z "$out" ] \
|| fail "capture_fresh_log_region printed a path even though mktemp failed: $out"
}
# fleetd #552 acceptance: "the exit status after a stubbed mktemp failure must still report the
# restart's own outcome". Sourcing stops before the main flow runs, so this drives the REAL
# call-site shape (the `if ! FRESH_LOG="$(capture_fresh_log_region ...)"; then warn; fi` guard that
# now stands where the old unguarded assignment did) in a subshell, under the SAME `set -euo
# pipefail` the real script runs under, with `mktemp` stubbed to fail on PATH. A regression back to
# the old unguarded `FRESH_LOG="$(mktemp ...)"` shape would abort this subshell before it reaches
# SUBSHELL_REACHED_END, and this test would go red with a non-zero rc.
test_fresh_log_capture_failure_does_not_abort_under_sete() {
local bin_dir rc=0 out
bin_dir="$TMP/stub-bin-call-site-mktemp-fails"; mkdir -p "$bin_dir"
cat > "$bin_dir/mktemp" <<'STUB'
#!/usr/bin/env bash
echo "mktemp: cannot create temp file" >&2
exit 1
STUB
chmod +x "$bin_dir/mktemp"
printf 'line one\n' > "$TMP/fresh-log-callsite.out"
out="$(
PATH="$bin_dir:$PATH"
set -euo pipefail
FRESH_LOG=""
trap 'rm -f "$FRESH_LOG"' EXIT
if ! FRESH_LOG="$(capture_fresh_log_region "$TMP/fresh-log-callsite.out" 0)"; then
echo "WARN skipped"
fi
echo "SUBSHELL_REACHED_END"
)" && rc=0 || rc=$?
[ "$rc" -eq 0 ] \
|| fail "the guarded call-site shape aborted under set -e when mktemp failed (rc=$rc) — a successful restart must not be reported as a failure"
printf '%s' "$out" | grep -qF 'SUBSHELL_REACHED_END' \
|| fail "the subshell aborted before reaching the end — a post-restart mktemp failure must not abort the script"
printf '%s' "$out" | grep -qF 'WARN skipped' \
|| fail "the guarded call site did not report the post-restart check as skipped when mktemp failed"
}
# fleetd #552 item 4 (the fourth reader): the result section must check REDEPLOY_AMQP_CHECK_SKIPPED
# and say so BEFORE it ever consults REDEPLOY_ERROR_COUNT — otherwise a skipped capture (which
# leaves REDEPLOY_ERROR_COUNT at its untouched 0) falls through the same "no ERROR lines" gate a
# genuinely clean restart uses, and (because REDEPLOY_DRAIN_STATE is "skipped", not "complete"/"n/a")
# prints nothing at all: a silence indistinguishable from the clean bill of health this ticket exists
# to prevent. Sourcing stops before the main flow runs, so this is a source-text ordering check, the
# same shape as test_report_shutdown_drain_ordered_after_classify_and_before_result above.
# fleetd #552 — closes the gap test_no_unguarded_mktemp_assignments_after_restart cannot: after
# extracting the mktemp into capture_fresh_log_region, the SAME class of hazard (an unguarded
# command-substitution assignment that can abort the script under set -e, because capture_fresh_
# log_region still returns non-zero on its own mktemp failure) would reappear if the main flow's
# call to it ever lost its own "if !" guard — and that line no longer contains the literal text
# "mktemp", so the sweep above cannot see it. Pin the real call site directly instead, the same
# shape test_swap_ordered_after_wait_and_before_start above uses for its own call site.
test_capture_fresh_log_region_call_site_is_guarded() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'if ! FRESH_LOG="$(capture_fresh_log_region "$OUT" "$RESTART_MARK")"; then' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's guarded capture_fresh_log_region call site in redeploy-fleetd.sh"
}
test_result_section_checks_amqp_skip_before_no_error_lines() {
local src="$ROOT/scripts/redeploy-fleetd.sh" result_line skip_line ok_line
result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)"
skip_line="$(grep -Fn '"$REDEPLOY_AMQP_CHECK_SKIPPED" -eq 1' "$src" | tail -1 | cut -d: -f1 || true)"
ok_line="$(grep -Fn 'ok "no ERROR lines since restart"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$result_line" ] || fail "could not find the result section in redeploy-fleetd.sh"
[ -n "$skip_line" ] || fail "could not find a REDEPLOY_AMQP_CHECK_SKIPPED check in redeploy-fleetd.sh"
[ -n "$ok_line" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh"
[ "$skip_line" -gt "$result_line" ] \
|| fail "the REDEPLOY_AMQP_CHECK_SKIPPED check (line $skip_line) is not inside the result section (starts line $result_line)"
[ "$skip_line" -lt "$ok_line" ] \
|| fail "the REDEPLOY_AMQP_CHECK_SKIPPED check (line $skip_line) is not checked before 'no ERROR lines since restart' (line $ok_line)"
}
# fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a
# hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not
# the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real
# caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout
# survives that boundary, so kind AND detail must both cross on it — this test proves they do.
test_require_drivable_supervisor_refuses_unclear() {
launchd_installed() { return 1; }
launchd_loaded() { return 1; }
systemd_installed() { return 0; }
systemd_loaded() { return 1; }
local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0
# Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand.
SUPERVISOR_RAW="$(detect_supervisor)"
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal"
[ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \
|| fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell"
printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \
|| fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded"
output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
# Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate
# recovery instructions name the unit unconditionally either way ("systemctl --user status
# $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only
# the specific detail string proves the crossed value, not the boilerplate, reached the message.
printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \
|| fail "refusal message does not contain the specific detail that crossed the subshell boundary"
}
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
# `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that
# forks a subshell, so its exit() only ends the subshell and this test script keeps running under
# `set -e`.
test_require_drivable_supervisor_refuses_ambiguous() {
local output rc=0
output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor"
printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \
|| fail "refusal message does not name the launchd label it found"
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|| fail "refusal message does not name the systemd unit it found"
}
test_require_drivable_supervisor_accepts_known_kinds() {
require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor"
require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor"
require_drivable_supervisor "none" || fail "refused the unsupervised case"
}
# fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor
# produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see
# it because either daemon alone satisfies them.
test_count_daemon_pids() {
assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list"
assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid"
assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids"
}
test_assert_single_daemon_accepts_one_pid() {
assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid"
}
test_assert_single_daemon_rejects_two_pids() {
local output rc=0
output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids"
printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found"
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
}
# fleetd #593 instance 2 — `running_pid()` used to be a bare `pgrep -f "$PATTERN"`, which matches
# ANY process whose full command line contains the pattern TEXT, including a shell that merely
# embeds it as literal text rather than being the daemon. Measured live on this Mac: `pgrep -c`
# (a one-call count) does not exist on BSD at all, and `bash -c "<single command>"` execs in place
# so no parent shell survives to hold the pattern — which is exactly why the defect did not
# reproduce from a plain script and needs a wrapper shaped like this instead. A `sh -c '...; ...'`
# with MORE THAN ONE statement does not get that exec-in-place treatment: the shell forks a child
# for the second statement and stays alive itself, holding the whole `-c` string — pattern text
# included — in its own `ps -o args`, for as long as it runs. That is the same shape an
# `ssh host "…; …"` wrapper or a hand-typed pipeline leaves behind. Before the fix this test would
# have found the wrapper's pid in running_pid()'s output; it must not.
test_running_pid_excludes_self_matching_wrapper_shell() {
local before after wrapper_pid
before="$(running_pid)"
sh -c 'echo "target/fleetd.jar" >/dev/null; sleep 20' &
wrapper_pid=$!
sleep 0.3
after="$(running_pid)"
kill "$wrapper_pid" 2>/dev/null || true
wait "$wrapper_pid" 2>/dev/null || true
[ "$after" = "$before" ] \
|| fail "running_pid() counted a self-matching wrapper shell (pid $wrapper_pid, holding the pattern as literal text in its own argv, not the daemon): before=[$before] after=[$after]"
}
# fleetd #593 CORRECTION 1 — the round-1 version of this test gave its standin an argv[0]
# containing the pattern text (via `exec -a`) and left `comm` as whatever that override produced,
# which was never `java`. That was fine for a denylist-of-shells filter, but the allowlist below
# now requires `comm = java` specifically, so the standin here must actually carry that comm, not
# just avoid being a shell. `exec -a java` overrides argv[0] to `java` while the process itself
# stays a genuine, harmless `sh`; combining it with the same non-exec'ing multi-statement shape
# the wrapper-shell test above uses keeps the pattern text in the process's own `ps -o args` for
# as long as it runs. Measured live on this Mac (BSD/macOS: `ps -o comm=` here reflects argv[0]):
# `comm=java`, `args` contains the pattern, `pgrep -f "$PATTERN"` finds it. Copying a real system
# binary into a scratch path and executing it from there was tried first, for a more literal
# stand-in daemon, and the OS killed it outright (SIGKILL, exit 137 — almost certainly a
# code-signing check on a relocated binary); `exec -a` needs no binary of its own and nothing
# under a scratch directory, and it is the technique CORRECTION 1 names as the right one.
#
# This is the one live-process test in this file whose result could differ on Linux: Linux sets
# `comm` from the actually-executed binary's own path, not from `exec -a`'s argv[0] override (BSD
# ties `comm` to argv[0], which is what makes this technique work here) — so on Linux this
# specific fixture might report `comm=sh`, not `comm=java`, even though the REAL daemon (a literal
# `java -jar target/fleetd.jar` process, never fabricated) is unaffected either way. I could not
# verify this fixture's behavior on Linux, so test_running_pid_counts_a_pid_whose_comm_is_java
# below backstops the same claim (the allowlist admits a pid whose comm is `java`) with a stubbed
# `ps`, which is identical bash on every platform and carries no such platform question.
test_running_pid_finds_a_real_java_named_second_process() {
local before after standin_pid
before="$(running_pid)"
( exec -a java sh -c 'echo "target/fleetd.jar" >/dev/null; sleep 20' ) &
standin_pid=$!
sleep 0.3
after="$(running_pid)"
kill "$standin_pid" 2>/dev/null || true
wait "$standin_pid" 2>/dev/null || true
printf '%s\n' "$after" | grep -qxF "$standin_pid" \
|| fail "running_pid() did not find a real second process (pid $standin_pid, comm forced to 'java' via exec -a) whose own argv holds the pattern: before=[$before] after=[$after]"
}
# fleetd #593 CORRECTION 1, hole 2 — the round-1 filter denied known shell names (sh/bash/zsh/
# dash/ksh) and counted everything else. `ssh`, `perl`, `python3`, `ruby`, `tail` — anything not on
# that list, carrying the pattern in its own argv — was still counted right alongside the real
# daemon, and the ticket names `ssh` as a live route. Stubbing `pgrep`/`ps` (rather than spawning a
# real perl/ssh process) pins the exact discriminator this correction is about — comm, not the
# caller's shape — deterministically on every platform, with no dependency on perl/python3/ruby
# being installed in whatever environment runs this suite, and no dependency on how a given OS
# derives `comm` for a fabricated process (see the comment above
# test_running_pid_finds_a_real_java_named_second_process for why that matters here).
test_running_pid_drops_a_pid_whose_comm_is_not_java() {
pgrep() { printf '4242\n'; }
ps() { printf 'perl\n'; }
local found
found="$(running_pid)"
unset -f pgrep ps
[ -z "$found" ] \
|| fail "running_pid() counted pid 4242 whose comm is 'perl', not 'java' — denying known shell names does not exclude a non-shell wrapper such as ssh or perl (fleetd #593 CORRECTION 1): found=[$found]"
}
# fleetd #593 CORRECTION 1, hole 1 — pgrep can list a pid that exits before the following
# `ps -o comm=` lookup runs; on a gone pid `ps` prints nothing, so `comm` comes back empty. Under
# the round-1 denylist an empty string matched none of the denied shell names, so the dead pid was
# still counted — the exact false-positive shape the ticket exists to remove, just rarer. The
# allowlist fixes this for free: an empty comm is not `java` either.
test_running_pid_drops_a_pid_that_exited_before_the_comm_lookup() {
pgrep() { printf '4242\n'; }
ps() { :; } # a pid that no longer exists: the real `ps -p <gone>` prints nothing and this mirrors that
local found
found="$(running_pid)"
unset -f pgrep ps
[ -z "$found" ] \
|| fail "running_pid() counted pid 4242 whose comm lookup came back empty (the pid had already exited before the lookup ran) — an empty comm must not pass the allowlist (fleetd #593 CORRECTION 1): found=[$found]"
}
# The positive backstop for both stubbed tests above, and for
# test_running_pid_finds_a_real_java_named_second_process on whatever platform that live fixture
# does not itself carry comm=java: the allowlist must still ADMIT the one comm value the real
# daemon actually has. Measured on the real, currently-running daemon on this Mac: `comm=java`.
test_running_pid_counts_a_pid_whose_comm_is_java() {
pgrep() { printf '4242\n'; }
ps() { printf 'java\n'; }
local found
found="$(running_pid)"
unset -f pgrep ps
printf '%s\n' "$found" | grep -qxF '4242' \
|| fail "running_pid() did not count pid 4242 whose comm is 'java' — the daemon's own name must pass the allowlist: found=[$found]"
}
# fleetd #593 instance 3 — assert_single_daemon's refusal message used to tell the operator to
# "Investigate with 'pgrep -f \"\$PATTERN\"'", which — typed by hand or over ssh — is precisely the
# self-matching invocation instance 2 above fixes. A source-text check, the same technique
# test_no_error_lines_message_gated_by_drain_state uses: this is prose inside a die() call, never
# reached by sourcing (the SOURCED guard stops before the main flow, and this text only prints
# from inside a call assert_single_daemon makes when it is already refusing).
test_die_message_does_not_recommend_bare_pgrep_as_remediation() {
local src="$ROOT/scripts/redeploy-fleetd.sh" block bad
block="$(grep -A6 -F 'racing supervisor produces' "$src" || true)"
[ -n "$block" ] || fail "could not find the assert_single_daemon refusal message in redeploy-fleetd.sh"
bad="$(printf '%s' "$block" | grep -F "Investigate with 'pgrep -f" || true)"
[ -z "$bad" ] \
|| fail "assert_single_daemon's die message still hands the operator a bare 'pgrep -f \"\$PATTERN\"' as remediation (fleetd #593) — that is exactly the self-matching invocation"
printf '%s' "$block" | grep -qF 'fleetd #593' \
|| fail "assert_single_daemon's die message does not say in words that a pattern can match the caller (fleetd #593)"
}
# fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports
# $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call
# must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR.
# Two files with different content, so a default pointed at the wrong one reports the wrong hash
# rather than accidentally matching.
test_jar_id_defaults_to_live_and_reports_explicit_path() {
local dir saved_jar="$JAR" saved_staged="$JAR_STAGED"
local live_hash staged_hash default_result explicit_result
dir="$TMP/jar-id"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
printf 'live jar bytes' > "$JAR"
printf 'staged jar bytes, not the same content' > "$JAR_STAGED"
# fleetd #550: this reference hash must be computed the same portable way jar_id() itself now
# computes one — a bare, unguarded call to the macOS-only hasher here was exactly the item-2
# defect, dying with "command not found" on any Linux runner that has no such hasher at all.
live_hash="$(hash256 "$JAR")"
staged_hash="$(hash256 "$JAR_STAGED")"
default_result="$(jar_id)"
explicit_result="$(jar_id "$JAR_STAGED")"
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
[ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same"
assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR"
assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR"
}
# fleetd #550 — closes a gap the test above leaves open. That test's own reference hash is now ALSO
# computed by calling hash256 (needed for item 2: the old bare macOS-only-hasher call there was the
# Linux crash), so its subject (jar_id, via hash256) and its reference (also hash256) share one
# instrument — they agree no matter which algorithm hash256 actually runs, so a mutation that swaps
# BOTH of hash256's arms for the wrong algorithm is invisible to it. This test's expected value
# comes from neither hasher: it is the published SHA-256 test vector for the 3-byte input "abc"
# (no trailing newline), written here as a literal constant, so it can still tell "hashed
# correctly" from "hashed, just with the wrong algorithm" — which is what this whole ticket is
# about.
test_hash256_computes_a_real_sha256() {
local dir f result
dir="$TMP/hash256-known-vector"; mkdir -p "$dir"
f="$dir/abc.txt"
printf 'abc' > "$f"
result="$(hash256 "$f")"
# SHA-256("abc") = ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad, the standard
# FIPS 180 test vector — first 12 hex chars, matching hash256's own `cut -c1-12`.
assert_equals "ba7816bf8f01" "$result" "hash256 of the literal 3-byte input 'abc' must be the known SHA-256 prefix, not some other algorithm's"
}
# fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511)
# proves both halves of the present-file contract but never exercises the missing-file path. This
# word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean`
# deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points
# operators at `--check` for exactly this. Covers both the no-argument default and an explicit path,
# since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both.
test_jar_id_reports_absent_for_missing_file() {
local saved_jar="$JAR" dir default_result explicit_result
dir="$TMP/jar-id-absent"; mkdir -p "$dir"
JAR="$dir/does-not-exist.jar"
[ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR"
default_result="$(jar_id)"
explicit_result="$(jar_id "$dir/also-does-not-exist.jar")"
JAR="$saved_jar"
assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist"
assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent"
}
# fleetd #550 — the whole point of this ticket: a jar that IS there but could not be hashed must
# never read the same as a jar that is not there at all. Drives the REAL hash256/jar_id bodies
# (never stubbed) through a stub PATH that contains neither of the two hashers this script knows —
# same technique test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses for `mktemp`,
# except here the stub directory is used to REPLACE PATH rather than prepend to it, because the
# point is to make BOTH hashers unreachable, not to intercept one specific command while leaving
# everything else on the real PATH reachable. `[ -f ... ]` and the shell's own `command`/`echo`
# builtins need no PATH at all, so this is safe even with PATH reduced to an empty directory.
test_jar_id_reports_unhashable_when_no_hasher_on_path() {
local bin_dir dir saved_jar="$JAR" default_result explicit_result
bin_dir="$TMP/stub-bin-no-hasher"; mkdir -p "$bin_dir"
dir="$TMP/jar-id-no-hasher"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"
printf 'a real jar that exists but nothing here can hash' > "$JAR"
[ -f "$JAR" ] || fail "test fixture error: \$JAR does not exist at $JAR"
default_result="$(PATH="$bin_dir" jar_id)"
explicit_result="$(PATH="$bin_dir" jar_id "$JAR")"
JAR="$saved_jar"
[ "$default_result" != "absent" ] \
|| fail "jar_id reported absent for a file that exists, only because no hasher was on PATH"
[ "$explicit_result" != "absent" ] \
|| fail "jar_id (explicit path) reported absent for a file that exists, only because no hasher was on PATH"
# A 12-char hex hash is the OTHER wrong answer here: with no hasher at all, nothing could have
# produced one, so a value that merely happens to look like one would mean the stub failed to
# hide the real hashers rather than that jar_id degraded correctly.
printf '%s' "$default_result" | grep -Eq '^[0-9a-f]{12}$' \
&& fail "test fixture error: PATH stub did not actually hide the real hasher(s) — got what looks like a real hash"
assert_equals "unhashable" "$default_result" "jar_id with no hasher on PATH must report a third, distinct state — never absent, never a hash"
assert_equals "unhashable" "$explicit_result" "jar_id (explicit path) with no hasher on PATH must report the same third state"
}
# fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar
# are exercised directly against real files on disk (not stubs), because the whole point is file
# behavior (does the content move, does the source disappear, does a failure leave both sides
# intact) that a stubbed function cannot prove.
test_stage_built_jar_moves_off_live_path() {
local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED"
dir="$TMP/stage-ok"; mkdir -p "$dir"
jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar"
printf 'built jar bytes' > "$jar"
JAR="$jar"; JAR_STAGED="$staged"
stage_built_jar || fail "stage_built_jar rejected a real build output"
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
[ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar"
[ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged"
grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content"
}
test_stage_built_jar_dies_when_build_produced_nothing() {
local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED"
dir="$TMP/stage-missing"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
output="$(stage_built_jar 2>&1)" || rc=$?
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
[ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output"
printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \
|| fail "refusal message does not name the missing jar path"
}
test_swap_staged_jar_moves_staged_onto_live() {
local dir staged live
dir="$TMP/swap-ok"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar"
printf 'swapped jar bytes' > "$staged"
swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar"
[ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged"
[ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live"
grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content"
}
# The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on
# failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes
# inside a command substitution to contain that exit to a subshell.
test_swap_staged_jar_dies_without_staged_file() {
local dir output rc=0
dir="$TMP/swap-missing"; mkdir -p "$dir"
output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar"
[ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged"
printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \
|| fail "refusal message does not name the missing staged path"
}
test_swap_staged_jar_dies_when_mv_fails() {
local dir staged live output rc=0
dir="$TMP/swap-fail"; mkdir -p "$dir/src"
staged="$dir/src/fleetd-new.jar"
printf 'fake jar bytes' > "$staged"
live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails
output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$?
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv"
[ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails"
[ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed"
printf '%s' "$output" | grep -qF "$staged" \
|| fail "refusal message does not name the staged path that could not be moved"
}
# --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this
# path) and must still die with the exact wording documented in the script's own header comment.
test_require_no_build_jar_dies_when_absent() {
local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar"
JAR="$missing"
output="$(require_no_build_jar 2>&1)" || rc=$?
JAR="$saved_jar"
[ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar"
printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \
|| fail "refusal message does not match the documented --no-build wording"
}
test_require_no_build_jar_accepts_present_jar() {
local saved_jar="$JAR" dir
dir="$TMP/no-build-present"; mkdir -p "$dir"
JAR="$dir/fleetd.jar"
printf 'existing jar' > "$JAR"
require_no_build_jar || fail "require_no_build_jar rejected an existing jar"
JAR="$saved_jar"
}
# wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while
# running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so
# the timeout-loop test does not actually wait out its budget.
test_wait_for_daemon_exit_returns_true_once_pid_clears() {
# running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so
# a plain shell variable it increments would reset on each call instead of accumulating. Count in
# a file instead, which is the one thing that actually survives across those subshells.
local counter_file="$TMP/wait-exit-calls" final_calls
printf '0' > "$counter_file"
running_pid() {
local n
n="$(cat "$counter_file")"
n=$((n + 1))
printf '%s' "$n" > "$counter_file"
if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi
}
sleep() { :; }
wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared"
final_calls="$(cat "$counter_file")"
[ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid"
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
}
test_wait_for_daemon_exit_times_out_if_pid_never_clears() {
local rc=0
running_pid() { printf '4242'; }
sleep() { :; }
wait_for_daemon_exit 3 || rc=$?
[ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared"
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
}
# fleetd #603 — wait_for_new_pid is the dual of wait_for_daemon_exit above: it must not report
# success while running_pid() still answers empty, and must report success the moment a pid
# appears. Same counter-file idiom as test_wait_for_daemon_exit_returns_true_once_pid_clears above,
# for the same reason (running_pid() runs inside a `$(...)` subshell on every call).
test_wait_for_new_pid_returns_true_once_pid_appears() {
local counter_file="$TMP/wait-new-pid-calls" final_calls
printf '0' > "$counter_file"
running_pid() {
local n
n="$(cat "$counter_file")"
n=$((n + 1))
printf '%s' "$n" > "$counter_file"
if [ "$n" -lt 3 ]; then printf ''; else printf '4242'; fi
}
sleep() { :; }
wait_for_new_pid 10 || fail "wait_for_new_pid did not report success once the pid appeared"
final_calls="$(cat "$counter_file")"
[ "$final_calls" -ge 3 ] || fail "wait_for_new_pid returned before actually re-checking running_pid"
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
}
test_wait_for_new_pid_times_out_if_pid_never_appears() {
local rc=0
running_pid() { printf ''; }
sleep() { :; }
wait_for_new_pid 3 || rc=$?
[ "$rc" -ne 0 ] || fail "wait_for_new_pid reported success while the pid never appeared"
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
}
# fleetd #603 — await_daemon_started folds the pid-appeared check and the healthz check into one
# decision (see its own comment in redeploy-fleetd.sh for why a source-text grep cannot tell the two
# possible behaviors apart here). These two tests are the ticket's own acceptance criteria, run
# together in this one suite invocation so neither can be satisfied by code that never fails at all:
#
# 1. a slow start must still succeed — running_pid mimics a process that does not appear until
# well after the OLD, buggy 10-second budget, but does appear, and healthz answers.
# 2. a genuine failure must still fail, and still print the log tail — running_pid and the health
# check both report nothing at all, ever.
test_await_daemon_started_slow_pid_then_healthy_succeeds() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
local counter_file="$TMP/await-slow-pid-calls" output
printf '0' > "$counter_file"
running_pid() {
local n
n="$(cat "$counter_file")"
n=$((n + 1))
printf '%s' "$n" > "$counter_file"
# Stays empty well past the old 10-second budget, then appears — the exact shape #603 reports.
if [ "$n" -lt 12 ]; then printf ''; else printf '4242'; fi
}
sleep() { :; }
poll_health_body() { printf '{"status":"ok"}'; return 0; }
# NOT `output="$(await_daemon_started ...)"`: that would run the call in a subshell, and
# NEW_PID — a plain global assignment inside the function, by design (see its own comment) — would
# die with that subshell instead of reaching this test's own shell. Redirect to a file instead, the
# same hazard await_daemon_started's own comment warns die() itself is subject to.
await_daemon_started 60 "" "http://ignored/healthz" "$TMP/await-slow-pid.out" \
> "$TMP/await-slow-pid-output.log" 2>&1
output="$(cat "$TMP/await-slow-pid-output.log")"
[ "$DIED_CALLED" = 0 ] \
|| fail "await_daemon_started must not die on a slow-but-real start: $DIED_MESSAGE"
assert_equals "4242" "$NEW_PID" "await_daemon_started NEW_PID after a slow-but-real start"
printf '%s' "$output" | grep -qF 'started, pid 4242' \
|| fail "await_daemon_started did not report the pid once it finally appeared"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_await_daemon_started_never_appears_dies_with_log_tail() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
local out_file="$TMP/await-never-appears.out"
printf 'boot line one\nboot line two\n' > "$out_file"
running_pid() { printf ''; }
sleep() { :; }
poll_health_body() { return 1; }
await_daemon_started 2 "" "http://127.0.0.1:1/healthz" "$out_file" > /dev/null 2>&1
[ "$DIED_CALLED" = 1 ] \
|| fail "await_daemon_started must die when the daemon never appears and never becomes healthy"
printf '%s' "$DIED_MESSAGE" | grep -qF 'never answered' \
|| fail "await_daemon_started die message does not say healthz never answered"
printf '%s' "$DIED_MESSAGE" | grep -qF 'boot line two' \
|| fail "await_daemon_started die message does not include the log tail"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
# fleetd #603 review — the path the fall-through actually exists for, and the one gap a lead
# mutation found in the first version of this test file: running_pid() NEVER finds anything (as its
# own doc comment says it eventually will, once the daemon stops being launched as a plain
# `java -jar` its allowlist recognises), while /healthz answers anyway. Neither of the two tests
# above drives this: the slow-pid test has the pid appear, so the `else` branch never runs, and the
# never-appears test fails BOTH checks, so it dies either way and cannot tell which branch fired.
# This must not die, must warn (so the operator is told the pid could not be identified), and must
# leave NEW_PID empty — the honest "could not establish this" answer, never a guessed pid, which is
# what the final result line's `${NEW_PID:-unknown}` fallback exists to print truthfully.
#
# Proof this actually pins the behavior, not just the source text (paste from a real run, not
# claimed): reverting the `warn` below back to `die "no process appeared"` (the old fleetd #603
# defect, reintroduced) turns this one test red —
# FAIL: await_daemon_started must not die when the pid is never found but healthz answers
# — and restoring `warn` turns the whole suite green again. Both halves observed, not asserted.
test_await_daemon_started_pid_never_found_but_healthy_warns_and_survives() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
local out_file="$TMP/await-pid-never-found.out" output
printf 'boot line\n' > "$out_file"
running_pid() { printf ''; }
sleep() { :; }
poll_health_body() { printf '{"status":"ok"}'; return 0; }
# NOT `output="$(await_daemon_started ...)"` — see the slow-pid test above for why that would
# drop NEW_PID's assignment in a subshell instead of reaching this test's own shell.
await_daemon_started 3 "" "http://ignored/healthz" "$out_file" \
> "$TMP/await-pid-never-found-output.log" 2>&1
output="$(cat "$TMP/await-pid-never-found-output.log")"
[ "$DIED_CALLED" = 0 ] \
|| fail "await_daemon_started must not die when the pid is never found but healthz answers: $DIED_MESSAGE"
printf '%s' "$output" | grep -qF 'falling through to the health check' \
|| fail "await_daemon_started did not warn that the pid could not be identified"
assert_equals "" "$NEW_PID" \
"await_daemon_started NEW_PID when the pid is never found but healthz answers — must stay empty, never a guessed pid"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_await_daemon_started_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'await_daemon_started "$HEALTH_WAIT" "$OLD_PID" "$HEALTH" "$OUT"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's await_daemon_started call site in redeploy-fleetd.sh"
}
# fleetd #521 — the swap step's guard, at two levels.
#
# The first two tests call the predicate should_swap() directly. They pin its logic, and that is all
# they pin. On their own they did NOT close #521, and this was measured rather than argued: with the
# main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left
# this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs
# the swap consult the predicate at all. Extracting the decision had moved the untested decision up
# a level, not removed it.
#
# So the last two tests call swap_if_built() — the function the main flow actually calls, holding the
# guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the
# guard is removed, inverted, or stops being consulted.
#
# What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow
# altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing
# stops before the main flow runs, so no test in this file can invoke it.
test_should_swap_true_when_build_ran() {
should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true"
}
test_should_swap_false_when_build_skipped() {
if should_swap 0; then
fail "should_swap 0 (--no-build; nothing was staged this run) must return false"
fi
}
# Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is
# global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id
# overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear),
# and again at the END, so their own stubs do not leak into every test that runs after them.
test_swap_if_built_performs_the_swap_when_build_ran() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local marker="$TMP/swap-if-built-ran"
rm -f "$marker"
swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; }
jar_id() { printf 'stubbed\n'; }
swap_if_built 1 > /dev/null
[ -f "$marker" ] \
|| fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_swap_if_built_skips_the_swap_when_build_skipped() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local marker="$TMP/swap-if-built-skipped"
rm -f "$marker"
swap_staged_jar() { printf 'swapped\n' > "$marker"; }
jar_id() { printf 'stubbed\n'; }
swap_if_built 0 > /dev/null
if [ -f "$marker" ]; then
fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did"
fi
source "$ROOT/scripts/redeploy-fleetd.sh"
}
# fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the
# main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee
# itself — as opposed to the pure functions it's built from — can only be checked by reading the
# script's own call sites, the same way test_recovery_patterns_match_source below checks Java
# source shape instead of behavior it cannot invoke directly.
#
# Two details about the three greps below, both of which have already gone wrong here.
#
# The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not
# `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside
# swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using
# it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)"
# — a true statement about a function definition, and nothing at all about the order of the steps.
#
# Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the
# pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set
# -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to
# report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no
# name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak.
test_swap_ordered_after_wait_and_before_start() {
local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line
wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)"
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh"
[ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh"
[ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh"
[ "$swap_line" -gt "$wait_line" ] \
|| fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)"
[ "$swap_line" -lt "$start_line" ] \
|| fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)"
}
# fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator
# declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)"
# to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has
# already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal
# ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start
# above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow),
# so the only way to pin its exact wording is to read the source.
test_drain_gate_abort_message_says_no_no_build() {
local src="$ROOT/scripts/redeploy-fleetd.sh" msg
msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")"
[ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh"
if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then
fail "abort message still claims a rerun WITH --no-build can finish the restart"
fi
printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \
|| fail "abort message does not tell the operator to rerun without --no-build"
printf '%s' "$msg" | grep -qF 'no longer at the live path' \
|| fail "abort message does not say why --no-build cannot finish the restart"
}
# fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was
# a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's
# own file for the wording, which stays in the file even if the `if` guarding it is mutated to
# `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead,
# so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT
# alongside these, not replaced, because it catches a different regression (a re-wording that still
# reaches the right branch would not change which case fires here, but would still be worth pinning).
test_drain_gate_refusal_build_ran_staged_present() {
local dir staged result
dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'staged jar bytes' > "$staged"
result="$(drain_gate_refusal 1 "$staged")"
printf '%s' "$result" | grep -qF "$staged" \
|| fail "build-ran+staged-present refusal does not name the staged jar path"
printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \
|| fail "build-ran+staged-present refusal does not tell the operator how to finish the restart"
if printf '%s' "$result" | grep -qF 'nothing changed'; then
fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved"
fi
}
test_drain_gate_refusal_build_ran_staged_absent() {
local dir result
dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir"
result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")"
assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording"
}
# --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar
# found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the
# comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning.
test_drain_gate_refusal_no_build_staged_present() {
local dir staged result
dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'leftover staged jar bytes' > "$staged"
result="$(drain_gate_refusal 0 "$staged")"
assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed"
}
test_drain_gate_refusal_no_build_staged_absent() {
local dir result
dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir"
result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")"
assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording"
}
# fleetd #528 — the four tests above pin drain_gate_refusal(), and that is ALL they pin: they call
# the predicate directly and never touch the main flow's call site. That was measured to be not
# enough, the same way test_should_swap_true_when_build_ran/test_should_swap_false_when_build_skipped
# were not enough for #521: with the main flow reading `die "$(drain_gate_refusal "$DO_BUILD"
# "$JAR_STAGED")"`, replacing that whole line with a flat `die "aborted — nothing changed"` left this
# suite at exit 0 with zero FAIL lines and byte-identical output to a clean run. Nothing above could
# tell the difference, because none of it calls anything at or above the call site itself.
#
# So these four call refuse_drain_gate() — the function the main flow actually calls, holding the
# composed message and the die() together — with die() stubbed to RECORD whether it was called and
# with what message, instead of exiting the process. That fails if refuse_drain_gate stops consulting
# drain_gate_refusal, mangles what it passes it, or simply never calls die.
#
# What none of these four can catch: deleting the `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` line
# from the main flow altogether — see the comment above refuse_drain_gate in redeploy-fleetd.sh for
# why no test in this file can do better than that (sourcing stops before the main flow runs).
DIED_CALLED=0
DIED_MESSAGE=""
stub_die_recorder() {
DIED_CALLED=0
DIED_MESSAGE=""
die() { DIED_CALLED=1; DIED_MESSAGE="$*"; }
}
test_refuse_drain_gate_build_ran_staged_present() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir staged
dir="$TMP/refuse-drain-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'staged jar bytes' > "$staged"
stub_die_recorder
refuse_drain_gate 1 "$staged"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate build-ran+staged-present must call die, and did not"
printf '%s' "$DIED_MESSAGE" | grep -qF "$staged" \
|| fail "refuse_drain_gate build-ran+staged-present die message does not name the staged jar"
printf '%s' "$DIED_MESSAGE" | grep -qF 'Rerun WITHOUT --no-build' \
|| fail "refuse_drain_gate build-ran+staged-present die message is missing the rerun instruction"
if printf '%s' "$DIED_MESSAGE" | grep -qF 'nothing changed'; then
fail "refuse_drain_gate build-ran+staged-present must not claim nothing changed — the jar already moved"
fi
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_refuse_drain_gate_build_ran_staged_absent() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir
dir="$TMP/refuse-drain-build-no-staged"; mkdir -p "$dir"
stub_die_recorder
refuse_drain_gate 1 "$dir/fleetd-new.jar"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate build-ran+staged-absent must call die, and did not"
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate build-ran+staged-absent die message"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_refuse_drain_gate_no_build_staged_present() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir staged
dir="$TMP/refuse-drain-no-build-staged"; mkdir -p "$dir"
staged="$dir/fleetd-new.jar"
printf 'leftover staged jar bytes' > "$staged"
stub_die_recorder
refuse_drain_gate 0 "$staged"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate no-build+staged-present must call die, and did not"
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-present die message"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_refuse_drain_gate_no_build_staged_absent() {
source "$ROOT/scripts/redeploy-fleetd.sh"
local dir
dir="$TMP/refuse-drain-no-build-no-staged"; mkdir -p "$dir"
stub_die_recorder
refuse_drain_gate 0 "$dir/fleetd-new.jar"
[ "$DIED_CALLED" = 1 ] \
|| fail "refuse_drain_gate no-build+staged-absent must call die, and did not"
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-absent die message"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
# fleetd #528 — closes the one gap the four behavioural tests above cannot: they call
# refuse_drain_gate directly, and sourcing stops before the main flow ever runs (the SOURCED guard),
# so none of them can prove the main flow still CALLS refuse_drain_gate at all. Same shape as
# test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site. This is
# what actually kills the item-1 mutation from the ticket — replacing the main flow's call with a
# flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural
# tests above would even notice.
#
# fleetd #555 — the main flow's own call site moved: it used to read
# `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` directly; it now reads
# `run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"`, and run_drain_gate (tested
# directly below by test_run_drain_gate_*) is what calls refuse_drain_gate with its own local names.
# This grep now pins THAT call site — the thing that would go missing if a future edit deleted the
# main flow's call to run_drain_gate altogether, the same residual gap #521/#528 already accepted for
# swap_if_built/refuse_drain_gate (sourcing stops before the main flow runs, so no test in this file
# can do better than reading the source for this one specific gap).
#
# The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail
# the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below
# ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body).
test_run_drain_gate_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's run_drain_gate call site in redeploy-fleetd.sh"
}
# fleetd #555 item 1 — `--check` must stay genuinely read-only: should_stop_for_check is the
# predicate, stop_if_check_only is the only thing the main flow calls.
test_should_stop_for_check_true_when_check_only() {
should_stop_for_check 1 || fail "should_stop_for_check must be true when CHECK_ONLY=1"
}
test_should_stop_for_check_false_when_not_check_only() {
should_stop_for_check 0 && fail "should_stop_for_check must be false when CHECK_ONLY=0"
return 0
}
# Captured via $(...): stop_if_check_only's own exit only ends this subshell (same reason the
# unload/stop "tolerates clean negative" tests above capture this way), so this test's own message
# is what reaches the report if the exit ever stops happening.
test_stop_if_check_only_exits_zero_and_says_nothing_changed() {
local output rc=0
output="$(stop_if_check_only 1)" || rc=$?
[ "$rc" -eq 0 ] || fail "stop_if_check_only 1 must exit 0, got $rc"
printf '%s' "$output" | grep -qF -- '--check: nothing changed' \
|| fail "stop_if_check_only 1 did not print the --check message"
}
test_stop_if_check_only_returns_without_exiting_when_not_check_only() {
local output
output="$(stop_if_check_only 0; echo REACHED)"
printf '%s' "$output" | grep -qF 'REACHED' \
|| fail "stop_if_check_only 0 must return, not exit — the caller's next line never ran"
}
test_stop_if_check_only_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'stop_if_check_only "$CHECK_ONLY"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's stop_if_check_only call site in redeploy-fleetd.sh"
}
# fleetd #555 items 2 and 3 — the drain-gate entry (`-n "$OLD_PID" && "$ASSUME_YES" = 0`) and the
# reply comparison (`"$reply" != "yes"`).
test_drain_gate_required_true_when_pid_and_not_assume_yes() {
drain_gate_required "123" 0 || fail "drain_gate_required must be true with an old pid and ASSUME_YES=0"
}
test_drain_gate_required_false_when_no_pid() {
drain_gate_required "" 0 && fail "drain_gate_required must be false with no old pid"
return 0
}
test_drain_gate_required_false_when_assume_yes() {
drain_gate_required "123" 1 && fail "drain_gate_required must be false when ASSUME_YES=1"
return 0
}
test_drain_confirmed_true_on_yes() {
drain_confirmed "yes" || fail "drain_confirmed must be true on exactly 'yes'"
}
test_drain_confirmed_false_on_anything_else() {
drain_confirmed "no" && fail "drain_confirmed must be false on 'no'"
drain_confirmed "" && fail "drain_confirmed must be false on empty input"
return 0
}
# run_drain_gate end-to-end, driving real stdin through `read` the same way the real prompt does.
# Redirected from a FILE, never a pipe: `printf ... | run_drain_gate ...` would run the function as
# the last stage of a pipeline, which bash runs in a subshell by default (no `lastpipe`), so any
# global this function set (DIED_CALLED/DIED_MESSAGE below) would be lost the instant the pipe
# closed. `< file` redirection on a plain function call carries no such subshell.
test_run_drain_gate_skips_prompt_when_not_required() {
local output
output="$(run_drain_gate "" 0 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)"
if printf '%s' "$output" | grep -qF 'Fleet drained?'; then
fail "run_drain_gate prompted even though drain_gate_required should have been false (no old pid)"
fi
output="$(run_drain_gate "123" 1 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)"
if printf '%s' "$output" | grep -qF 'Fleet drained?'; then
fail "run_drain_gate prompted even though ASSUME_YES=1 should have skipped the gate"
fi
}
test_run_drain_gate_confirmed_reply_does_not_refuse() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
printf 'yes\n' > "$TMP/drain-reply-yes.txt"
run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-yes.txt" \
> "$TMP/drain-gate-yes-output" 2>&1
[ "$DIED_CALLED" = 0 ] \
|| fail "run_drain_gate must not refuse when the reply confirms ('yes')"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_run_drain_gate_declined_reply_refuses() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
printf 'no\n' > "$TMP/drain-reply-no.txt"
run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-no.txt" \
> "$TMP/drain-gate-no-output" 2>&1
[ "$DIED_CALLED" = 1 ] \
|| fail "run_drain_gate must refuse (call die via refuse_drain_gate) when the reply does not confirm"
printf '%s' "$DIED_MESSAGE" | grep -qF 'aborted' \
|| fail "run_drain_gate's refusal message does not read as an abort: $DIED_MESSAGE"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
# fleetd #555 item 4 — the report-state dispatch on $SUPERVISOR_KIND. Inverting this used to report
# the wrong supervisor and, for the launchd arm specifically, skip check_log_path_matches_plist.
CHECK_LOG_PATH_CALLED=0
stub_check_log_path_recorder() {
CHECK_LOG_PATH_CALLED=0
check_log_path_matches_plist() { CHECK_LOG_PATH_CALLED=1; }
}
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_check_log_path_recorder
SUPERVISED=0
report_supervisor_state "launchd" > "$TMP/report-supervisor-launchd-output" 2>&1
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state launchd must set SUPERVISED=1"
[ "$CHECK_LOG_PATH_CALLED" = 1 ] \
|| fail "report_supervisor_state launchd must call check_log_path_matches_plist"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_supervisor_state_systemd_sets_supervised_without_log_path_check() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_check_log_path_recorder
SUPERVISED=0
report_supervisor_state "systemd" > "$TMP/report-supervisor-systemd-output" 2>&1
[ "$SUPERVISED" = 1 ] || fail "report_supervisor_state systemd must set SUPERVISED=1"
[ "$CHECK_LOG_PATH_CALLED" = 0 ] \
|| fail "report_supervisor_state systemd must NOT call check_log_path_matches_plist"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_supervisor_state_none_leaves_supervised_zero() {
SUPERVISED=1
report_supervisor_state "none" > "$TMP/report-supervisor-none-output" 2>&1
[ "$SUPERVISED" = 0 ] || fail "report_supervisor_state none must leave SUPERVISED=0"
grep -qF 'no supervisor loaded' "$TMP/report-supervisor-none-output" \
|| fail "report_supervisor_state none did not print the unsupervised message"
}
test_report_supervisor_state_unrecognized_kind_warns_and_continues() {
SUPERVISED=1
report_supervisor_state "bogus-kind" > "$TMP/report-supervisor-bogus-output" 2>&1
[ "$SUPERVISED" = 0 ] || fail "report_supervisor_state must reset SUPERVISED for an unrecognized kind"
grep -qF "unrecognised supervisor kind: 'bogus-kind'" "$TMP/report-supervisor-bogus-output" \
|| fail "report_supervisor_state did not warn about the unrecognized kind"
}
test_report_supervisor_state_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'report_supervisor_state "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's report_supervisor_state call site in redeploy-fleetd.sh"
}
# fleetd #555 item 5 — the stop dispatch on $SUPERVISOR_KIND. Inverting this kills a supervised
# daemon with a raw kill instead of launchctl/systemctl, reviving the OLD jar (CB-594/#492). Each
# mechanism is stubbed to record which one ran, so a swapped or dropped arm shows up directly.
STOP_DISPATCH_CALLED=""
stub_stop_dispatch_recorders() {
STOP_DISPATCH_CALLED=""
stop_via_launchd() { STOP_DISPATCH_CALLED="launchd"; }
stop_via_systemd() { STOP_DISPATCH_CALLED="systemd"; }
stop_via_kill() { STOP_DISPATCH_CALLED="kill:$1"; }
}
test_dispatch_stop_launchd_calls_stop_via_launchd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
dispatch_stop "launchd" "123"
assert_equals "launchd" "$STOP_DISPATCH_CALLED" "dispatch_stop launchd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_systemd_calls_stop_via_systemd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
dispatch_stop "systemd" "123"
assert_equals "systemd" "$STOP_DISPATCH_CALLED" "dispatch_stop systemd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_none_calls_stop_via_kill_with_pid() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
dispatch_stop "none" "456"
assert_equals "kill:456" "$STOP_DISPATCH_CALLED" "dispatch_stop none"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_unrecognized_kind_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_stop_dispatch_recorders
stub_die_recorder
dispatch_stop "bogus-kind" "789"
[ "$DIED_CALLED" = 1 ] || fail "dispatch_stop must die on an unrecognized kind"
[ -z "$STOP_DISPATCH_CALLED" ] \
|| fail "dispatch_stop must not call any stop mechanism on an unrecognized kind"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_stop_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'dispatch_stop "$SUPERVISOR_KIND" "$OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's dispatch_stop call site in redeploy-fleetd.sh"
}
# fleetd #555 item 6 — the start dispatch on $SUPERVISOR_KIND. Same shape as the stop dispatch.
START_DISPATCH_CALLED=""
stub_start_dispatch_recorders() {
START_DISPATCH_CALLED=""
start_via_launchd() { START_DISPATCH_CALLED="launchd"; }
start_via_systemd() { START_DISPATCH_CALLED="systemd"; }
start_via_none() { START_DISPATCH_CALLED="none"; }
}
test_dispatch_start_launchd_calls_start_via_launchd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
dispatch_start "launchd"
assert_equals "launchd" "$START_DISPATCH_CALLED" "dispatch_start launchd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_systemd_calls_start_via_systemd() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
dispatch_start "systemd"
assert_equals "systemd" "$START_DISPATCH_CALLED" "dispatch_start systemd"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_none_calls_start_via_none() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
dispatch_start "none"
assert_equals "none" "$START_DISPATCH_CALLED" "dispatch_start none"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_unrecognized_kind_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_start_dispatch_recorders
stub_die_recorder
dispatch_start "bogus-kind"
[ "$DIED_CALLED" = 1 ] || fail "dispatch_start must die on an unrecognized kind"
[ -z "$START_DISPATCH_CALLED" ] \
|| fail "dispatch_start must not call any start mechanism on an unrecognized kind"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_dispatch_start_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'dispatch_start "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's dispatch_start call site in redeploy-fleetd.sh"
}
# fleetd #555 item 7 — the HEALTH_BODY poll loop and the `[ -z "$HEALTH_BODY" ]` branch. Inverting
# health_is_up makes a dead daemon report /healthz 200 or a live one report failure.
test_health_is_up_true_with_body() {
health_is_up "some body" || fail "health_is_up must be true with a non-empty body"
}
test_health_is_up_false_with_empty_body() {
health_is_up "" && fail "health_is_up must be false with an empty body"
return 0
}
# fleetd #555: stub_die_recorder replaces die() globally for the rest of the process, the same as
# every other user of it in this file — re-source before AND after so a later test (e.g.
# test_unload_launchd_if_loaded_dies_on_real_failure, which needs the REAL die() to actually exit)
# never inherits a stub left over from here.
test_report_health_up_reports_ok_and_never_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
local output
output="$(report_health '{"status":"ok"}' "200" "$TMP/fake.out" 60)"
[ "$DIED_CALLED" = 0 ] || fail "report_health must not die when the body is non-empty"
printf '%s' "$output" | grep -qF '/healthz 200' \
|| fail "report_health did not report the healthy body"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_health_degraded_503_warns_and_never_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
local output
output="$(report_health "" "503" "$TMP/fake.out" 60 2>&1)"
[ "$DIED_CALLED" = 0 ] || fail "report_health must not die on a 503 (degraded, not dead)"
printf '%s' "$output" | grep -qF '503 degraded' \
|| fail "report_health did not report the 503-degraded case"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_report_health_dead_dies() {
source "$ROOT/scripts/redeploy-fleetd.sh"
stub_die_recorder
report_health "" "000" "$TMP/fake.out" 60 > "$TMP/report-health-dead-output" 2>&1
[ "$DIED_CALLED" = 1 ] || fail "report_health must die when the body is empty and the code is not 503"
printf '%s' "$DIED_MESSAGE" | grep -qF 'never answered' \
|| fail "report_health die message does not say healthz never answered"
source "$ROOT/scripts/redeploy-fleetd.sh"
}
test_poll_health_body_returns_nonzero_when_unreachable() {
local rc=0
poll_health_body "http://127.0.0.1:1/healthz" 1 > /dev/null || rc=$?
[ "$rc" -ne 0 ] || fail "poll_health_body must return non-zero when the health endpoint is unreachable"
}
test_report_health_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
# fleetd #603 — moved from a literal main-flow call into await_daemon_started (see its own
# comment above for why); this now finds the call inside that function instead.
call_line="$(grep -Fn 'report_health "$HEALTH_BODY" "$HEALTH_CODE" "$out_file" "$health_wait"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find await_daemon_started's report_health call site in redeploy-fleetd.sh"
}
# fleetd #555 item 8 — `HAD_OLD_PID=0; [ -n "$OLD_PID" ] && HAD_OLD_PID=1`. Measured safe under
# `set -e` at both bash 3.2.57 and 5.x (the ticket's own header discussion); still an untested
# computation feeding report_shutdown_drain's four-way decision until now.
test_compute_had_old_pid_zero_when_empty() {
assert_equals "0" "$(compute_had_old_pid "")" "compute_had_old_pid empty pid"
}
test_compute_had_old_pid_one_when_present() {
assert_equals "1" "$(compute_had_old_pid "12345")" "compute_had_old_pid present pid"
}
test_compute_had_old_pid_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'HAD_OLD_PID="$(compute_had_old_pid "$OLD_PID")"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's compute_had_old_pid call site in redeploy-fleetd.sh"
}
# fleetd #555 — the structural guard the ticket actually asks for: "pick the shape, not the eight
# sites." A future ninth bare main-flow conditional must fail THIS test on sight, not slip through
# with every function-level test (67, now many more) still green.
#
# Scope: from the SOURCED guard onward — the ticket's own "main flow" (report state / build / drain
# / stop / swap / start / verify). The `for arg in "$@"; do case "$arg" in ... esac; done` option
# parser above the guard is out of scope, the same way it sits outside the ticket's own eight-item
# table: it runs even when this file is sourced for tests, is not part of the mutating procedure,
# and none of the eight decisions the ticket names live there.
#
# For every line whose first non-blank token is `if`, `elif`, or `case`, and which sits OUTSIDE any
# function body, this demands an EXACT match — full trimmed line, never a regex anchor (a regex can
# match inside unrelated text this same mutation could introduce, e.g. a comment) — against
# MAIN_FLOW_ALLOWED_CONDITIONALS below. That allowlist is a closed set of the report-only/display
# conditionals already in the file (bare state prints: is the daemon running, is X installed, does
# this secret resolve — none of them gate a build/stop/start/swap decision) plus the two
# loaded-but-not-running elif branches that are the SAME shape as fleetd #504 items 2/3/4, one
# function over, and explicitly out of THIS ticket's scope per its own "Related" section.
#
# Every one of the eight decisions the ticket names has been lifted into its own predicate+action
# function above this point in the file (should_stop_for_check, drain_gate_required/run_drain_gate,
# report_supervisor_state, dispatch_stop, dispatch_start, health_is_up/report_health,
# compute_had_old_pid) — none of their guards appear in this list any more, which is what makes
# their specific inversions untestable-by-construction now: there is no bare guard left in the main
# flow for a future edit to invert silently. A NEW bare conditional (the ninth) is, by construction,
# not in this list, so this test goes red the instant one is added, naming the exact line — passing
# requires either lifting the decision into a function (with its own predicate test, same shape as
# above) or explicitly extending the allowlist, which is a diff a reviewer sees and can question.
#
# Function-body detection relies on this file's one consistent convention: a function opens as
# `name() {` alone on its own line and closes as a bare `}` alone on its own line — verified: every
# multi-line function in this file follows it, and the handful of one-liners like `say() { ...; }`
# sit above the SOURCED guard, outside the region this scans. Indentation is NOT used to decide
# "inside a function": a bare `if` inside a top-level `for`/`while` loop body (indented, but not
# inside any function) would be exactly as reportable as one at column 0 — this file currently has
# none, but a future one that hid inside a loop must not get a free pass for being indented.
MAIN_FLOW_ALLOWED_CONDITIONALS=(
'if [ -n "$OLD_PID" ]; then'
'if launchd_installed; then'
'if systemd_installed; then'
'if zsh -lc '\''[ -n "${WORKER_GITEA_TOKEN:-}" ]'\'' 2>/dev/null; then'
'if zsh -lc '\''[ -n "${AI_GATEWAY_TOKEN:-}" ]'\'' 2>/dev/null; then'
'if [ -z "$BROKER_URI_ENV" ]; then'
'elif zsh -lc "[ -n \"\${$BROKER_URI_ENV:-}\" ]" 2>/dev/null; then'
'if [ "$DO_BUILD" = 1 ]; then'
'if ! mvn -f "$MODULE/pom.xml" clean install > "$BUILD_LOG" 2>&1; then'
'if ! wait_for_daemon_exit "$STOP_WAIT"; then'
'elif [ "$SUPERVISOR_KIND" = "launchd" ]; then'
'elif [ "$SUPERVISOR_KIND" = "systemd" ]; then'
'if tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -q '\''fleetd listening'\''; then'
'if ! FRESH_LOG="$(capture_fresh_log_region "$OUT" "$RESTART_MARK")"; then'
'if [ "$REDEPLOY_AMQP_CHECK_SKIPPED" -eq 1 ]; then'
'elif [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then'
'if [ "$REDEPLOY_DRAIN_STATE" = "complete" ] || [ "$REDEPLOY_DRAIN_STATE" = "n/a" ]; then'
'elif [ "$REDEPLOY_UNEXPLAINED_ERRORS" -eq 0 ]; then'
)
# Emits "<lineno>\tCOND\t<trimmed line>" for every if/elif/case found outside a function, and
# "<lineno>\tFUNC\t<line>" for every function DEFINITION found after guard_line — from guard_line
# onward. A separate function (rather than inlined into the test) so the deliberate-mutation proof
# in the PR description can call it directly against a scratch copy of the script.
#
# fleetd #555 rework (comment 17012): a function defined after the SOURCED guard can never be
# reached by `source`-ing this script (sourcing returns before the main flow, and before any code
# below the guard runs), so anything inside such a function is untestable by construction — the
# depth tracker below would otherwise read it as "inside a function, therefore fine" and wave every
# conditional in it through unexamined. Emitting FUNC records lets the caller flag the function
# itself as the violation, independently of whether its body happens to contain a conditional.
mainflow_bare_conditionals() {
local src="$1" guard_line="$2" in_func=0 lineno=0 line trimmed
while IFS= read -r line || [ -n "$line" ]; do
lineno=$((lineno + 1))
[ "$lineno" -le "$guard_line" ] && continue
if [[ "$line" =~ ^[A-Za-z_][A-Za-z0-9_]*\(\)[[:space:]]*\{[[:space:]]*$ ]]; then
in_func=1
printf '%d\tFUNC\t%s\n' "$lineno" "$line"
continue
fi
if [ "$in_func" = 1 ] && [[ "$line" =~ ^\}[[:space:]]*$ ]]; then
in_func=0
continue
fi
if [ "$in_func" = 0 ]; then
trimmed="$(printf '%s' "$line" | sed -E 's/^[[:space:]]+//')"
if [[ "$trimmed" =~ ^(if|elif|case)[[:space:]] ]]; then
printf '%d\tCOND\t%s\n' "$lineno" "$trimmed"
fi
fi
done < "$src"
}
test_no_untested_main_flow_conditionals() {
local src="$ROOT/scripts/redeploy-fleetd.sh" guard_line
guard_line="$(grep -Fn 'if (return 0 2>/dev/null); then' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$guard_line" ] || { fail "could not find the SOURCED guard in redeploy-fleetd.sh"; return 1; }
local violations=0 report="" found_line found_kind found_text allowed candidate
while IFS=$'\t' read -r found_line found_kind found_text; do
[ -n "$found_line" ] || continue
if [ "$found_kind" = "FUNC" ]; then
# fleetd #555 rework: a function defined after the SOURCED guard line can never be sourced
# by this suite, so it can never be tested — that is a violation on its own, regardless of
# what its body contains or whether the allowlist would otherwise excuse a bare conditional
# inside it.
violations=$((violations + 1))
report="$report
line $found_line: function defined after the SOURCED guard (line $guard_line) — it cannot be sourced, so it cannot be tested: $found_text"
continue
fi
allowed=0
for candidate in "${MAIN_FLOW_ALLOWED_CONDITIONALS[@]}"; do
if [ "$found_text" = "$candidate" ]; then
allowed=1
break
fi
done
if [ "$allowed" = 0 ]; then
violations=$((violations + 1))
report="$report
line $found_line: $found_text"
fi
done < <(mainflow_bare_conditionals "$src" "$guard_line")
if [ "$violations" -gt 0 ]; then
fail "found $violations untested main-flow if/elif/case line(s) or function definition(s) after the SOURCED guard, not lifted into a tested predicate function and not in MAIN_FLOW_ALLOWED_CONDITIONALS:$report"
fi
}
# fleetd #504 — the "loaded but not currently running" branches for launchd/systemd used to run
# `launchctl unload`/`systemctl --user stop` with `2>/dev/null || true` and print `ok`
# unconditionally, so a real supervisor failure (e.g. it cannot reach launchd/the systemd user bus)
# read exactly like a harmless already-stopped answer. unload_launchd_if_loaded/
# stop_systemd_if_loaded (redeploy-fleetd.sh, right after systemd_loaded) apply systemd_loaded's own
# "capture stderr separately — only a non-zero exit WITH stderr is a real failure" pattern to the
# WRITE side. Both call the real `launchctl`/`systemctl` binaries directly (they are not overridable
# wrapper functions the way launchd_loaded/systemd_loaded are), so these tests put a stub binary
# first on PATH — the same technique test_detect_supervisor_systemd_probe_error_is_unclear above
# already uses for `systemctl`.
test_unload_launchd_if_loaded_dies_on_real_failure() {
local bin_dir output rc=0 saved_plist="$LAUNCHD_PLIST"
bin_dir="$TMP/stub-bin-launchctl-error"; mkdir -p "$bin_dir"
cat > "$bin_dir/launchctl" <<'STUB'
#!/usr/bin/env bash
echo "Could not find specified service" >&2
exit 1
STUB
chmod +x "$bin_dir/launchctl"
LAUNCHD_PLIST="$TMP/fake-fail.plist"
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
LAUNCHD_PLIST="$saved_plist"
[ "$rc" -ne 0 ] \
|| fail "unload_launchd_if_loaded must die when launchctl exits non-zero AND writes to stderr"
printf '%s' "$output" | grep -qF 'launchctl unload' \
|| fail "die message does not name the failing launchctl unload command"
}
# Captured via $(...) rather than called bare: unload_launchd_if_loaded's own die() does a hard
# `exit`, and calling it directly at this level would let a regression that makes it die on this
# clean-negative case kill the WHOLE suite before the `|| fail` below ever ran — printing die's own
# message instead of this test's. Inside a command substitution, that `exit` only ends the subshell
# (a-guard-is-defeated-by-its-calling-context: the same reason the *_dies_on_real_failure tests
# above capture this way), so this test's own message is what actually reaches the report.
test_unload_launchd_if_loaded_tolerates_clean_negative() {
local bin_dir saved_plist="$LAUNCHD_PLIST" output rc=0
bin_dir="$TMP/stub-bin-launchctl-noop"; mkdir -p "$bin_dir"
cat > "$bin_dir/launchctl" <<'STUB'
#!/usr/bin/env bash
exit 1
STUB
chmod +x "$bin_dir/launchctl"
LAUNCHD_PLIST="$TMP/fake-noop.plist"
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
LAUNCHD_PLIST="$saved_plist"
[ "$rc" -eq 0 ] \
|| fail "unload_launchd_if_loaded must tolerate a clean already-unloaded answer (non-zero exit, empty stderr): $output"
}
test_stop_systemd_if_loaded_dies_on_real_failure() {
local bin_dir output rc=0
bin_dir="$TMP/stub-bin-systemctl-stop-error"; mkdir -p "$bin_dir"
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
echo "Failed to connect to bus: No such file or directory" >&2
exit 1
STUB
chmod +x "$bin_dir/systemctl"
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
[ "$rc" -ne 0 ] \
|| fail "stop_systemd_if_loaded must die when systemctl exits non-zero AND writes to stderr"
printf '%s' "$output" | grep -qF 'systemctl --user stop' \
|| fail "die message does not name the failing systemctl --user stop command"
}
# Same subshell-capture reasoning as test_unload_launchd_if_loaded_tolerates_clean_negative above:
# stop_systemd_if_loaded's own die() does a hard `exit`, so this must run inside $(...) or a
# regression here would kill the whole suite with die's message instead of this test's.
test_stop_systemd_if_loaded_tolerates_clean_negative() {
local bin_dir output rc=0
bin_dir="$TMP/stub-bin-systemctl-stop-noop"; mkdir -p "$bin_dir"
cat > "$bin_dir/systemctl" <<'STUB'
#!/usr/bin/env bash
exit 1
STUB
chmod +x "$bin_dir/systemctl"
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
[ "$rc" -eq 0 ] \
|| fail "stop_systemd_if_loaded must tolerate a clean already-stopped answer (non-zero exit, empty stderr): $output"
}
# Closes the same gap test_run_drain_gate_call_site_present closes for the drain gate: the four
# tests above call unload_launchd_if_loaded/stop_systemd_if_loaded directly, and sourcing stops
# before the main flow ever runs (the SOURCED guard), so none of them can prove the main flow still
# CALLS these two functions instead of the original bare `2>/dev/null || true`. A source-text check,
# like test_swap_ordered_after_wait_and_before_start. The call-site needle is anchored (`^ name$`)
# so it cannot be satisfied by the comment lines above each call site that merely mention the
# function by name.
test_stop_branches_call_tolerant_helpers_not_bare_or_true() {
local src="$ROOT/scripts/redeploy-fleetd.sh" unload_call_line stop_call_line
unload_call_line="$(grep -n '^ unload_launchd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
stop_call_line="$(grep -n '^ stop_systemd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$unload_call_line" ] \
|| fail "could not find the main flow's call to unload_launchd_if_loaded in redeploy-fleetd.sh"
[ -n "$stop_call_line" ] \
|| fail "could not find the main flow's call to stop_systemd_if_loaded in redeploy-fleetd.sh"
if grep -qF 'launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true' "$src"; then
fail "the bare 'launchctl unload ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
fi
if grep -qF 'systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true' "$src"; then
fail "the bare 'systemctl --user stop ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
fi
}
test_no_errors() {
cat > "$TMP/no-errors.log" <<'LOG'
2026-09-05 12:00:00 INFO fleetd listening
LOG
classify_fixture no-errors.log
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total"
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained"
# fleetd #552 control: a REAL, readable log region with nothing in it must never look like a
# region that could not be captured at all.
assert_equals 0 "$REDEPLOY_AMQP_CHECK_SKIPPED" "no-errors must not report the check as skipped"
}
# fleetd #552 — the third state: "" (capture_fresh_log_region's own sentinel for "the post-restart
# region could never be captured") must never look like "read a file with nothing in it", which is
# exactly what REDEPLOY_ERROR_COUNT staying at 0 already means on its own (see test_no_errors above).
test_classify_amqp_connection_errors_reports_skipped_when_log_missing() {
classify_amqp_connection_errors "" > "$TMP/classify-skipped-output" 2>&1
assert_equals 1 "$REDEPLOY_AMQP_CHECK_SKIPPED" "an empty log_file must set REDEPLOY_AMQP_CHECK_SKIPPED=1"
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "a skipped check must still leave REDEPLOY_ERROR_COUNT at its initialized 0 (the flag, not the count, carries the distinction)"
grep -qF 'could not be captured' "$TMP/classify-skipped-output" \
|| fail "classify_amqp_connection_errors did not say the post-restart log region could not be captured"
}
test_recovery_patterns_match_source() {
grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|| fail "AMQP failure pattern no longer matches source"
grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|| fail "reply-inbox recovery pattern no longer matches source"
grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \
|| fail "lead-mailbox recovery pattern no longer matches source"
}
test_attributed_recovered_connection_error() {
cat > "$TMP/attributed-recovered.log" <<'LOG'
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
LOG
classify_fixture attributed-recovered.log
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total"
assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors"
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained"
}
test_source_derived_error_shapes_recover_by_connection() {
# These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check
# after redeploy because the new code has not yet written a production line.
cat > "$TMP/source-derived.log" <<'LOG'
17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery!
17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!
17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
LOG
classify_fixture source-derived.log
assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total"
assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered"
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained"
}
test_cross_connection_unattributable_errors_stay_loud() {
# This candidate has neither stable connection name, so LeadMailbox recovery must not consume it.
cat > "$TMP/cross-unattributable.log" <<'LOG'
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
LOG
classify_fixture cross-unattributable.log
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total"
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered"
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained"
}
test_attributed_cross_connection_errors_stay_loud() {
# LeadMailbox recovery cannot heal AmqpReplyInbox errors.
cat > "$TMP/cross-attributed.log" <<'LOG'
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
LOG
classify_fixture cross-attributed.log
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total"
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered"
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained"
}
test_attributed_unrecovered_connection_error() {
cat > "$TMP/unrecovered.log" <<'LOG'
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
LOG
classify_fixture unrecovered.log
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total"
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors"
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained"
}
test_other_error_is_unexplained() {
cat > "$TMP/other-error.log" <<'LOG'
2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed
2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
LOG
classify_fixture other-error.log
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total"
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained"
}
test_recovery_requirement_mutation_is_caught() {
classify_amqp_connection_errors() {
local log_file="$1" line
REDEPLOY_ERROR_COUNT=0
REDEPLOY_RECOVERED_AMQP_ERRORS=0
REDEPLOY_UNEXPLAINED_ERRORS=0
while IFS= read -r line || [ -n "$line" ]; do
case "$line" in
*' ERROR '*|*' SEVERE '*)
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
case "$line" in
*'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*)
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
esac
done < "$log_file"
}
if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then
fail "mutation accepted an unrecovered connection error"
fi
grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \
|| fail "mutation failed without the expected assertion"
printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n'
}
test_shared_counter_mutation_is_caught() {
classify_amqp_connection_errors() {
local log_file="$1" line pending=0
REDEPLOY_ERROR_COUNT=0
REDEPLOY_RECOVERED_AMQP_ERRORS=0
REDEPLOY_UNEXPLAINED_ERRORS=0
while IFS= read -r line || [ -n "$line" ]; do
case "$line" in
*' ERROR '*|*' SEVERE '*)
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
case "$line" in
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
case "$line" in
*'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
*'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*)
if [ "$pending" -gt 0 ]; then
pending=$((pending - 1))
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
fi
;;
esac
done < "$log_file"
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending))
}
if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then
fail "shared counter mutation accepted cross-connection recovery"
fi
grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \
|| fail "shared counter mutation failed without the expected assertion"
printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n'
}
test_unattributable_quiet_mutation_is_caught() {
classify_amqp_connection_errors() {
local log_file="$1" line
REDEPLOY_ERROR_COUNT=0
REDEPLOY_RECOVERED_AMQP_ERRORS=0
REDEPLOY_UNEXPLAINED_ERRORS=0
while IFS= read -r line || [ -n "$line" ]; do
case "$line" in
*' ERROR '*|*' SEVERE '*)
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
case "$line" in
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
;;
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
esac
;;
esac
done < "$log_file"
}
if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then
fail "unattributable mutation accepted an unknown connection"
fi
grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \
|| fail "unattributable mutation failed without the expected assertion"
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
}
# fleetd #512 part 2 — the negative check (scan_uncaught_exceptions). The heart of this half of the
# ticket: a fixture with the uncaught-exception shape and NO line carrying an ERROR token at all,
# proving the scan finds it without one. A fixture that also carried an ERROR line would pass for
# the wrong reason.
test_scan_uncaught_exceptions_finds_shape_without_error_token() {
cat > "$TMP/scan-died.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
LOG
local error_count
error_count="$(grep -c ' ERROR ' "$TMP/scan-died.log" || true)"
[ "$error_count" = "0" ] \
|| fail "test fixture error: scan-died.log unexpectedly carries an ERROR token"
scan_uncaught_exceptions "$TMP/scan-died.log"
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "scan must find the exception without an ERROR token"
printf '%s' "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" | grep -qF 'NoClassDefFoundError' \
|| fail "scan did not capture the matching line as the sample"
}
test_scan_uncaught_exceptions_clean_control() {
cat > "$TMP/scan-clean.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=0 abandoned=0 (still BUSY at the shutdown deadline)
LOG
scan_uncaught_exceptions "$TMP/scan-clean.log"
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "clean control must find no uncaught exception"
assert_equals "" "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" "clean control sample must be empty"
}
# fleetd #512 part 2 — the positive check (find_drain_complete_line). Both halves of #522's line:
# present, and absent.
test_find_drain_complete_line_present() {
cat > "$TMP/drain-line-present.log" <<'LOG'
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=2 abandoned=1 (still BUSY at the shutdown deadline)
LOG
find_drain_complete_line "$TMP/drain-line-present.log"
printf '%s' "$REDEPLOY_DRAIN_COMPLETE_LINE" | grep -qF 'released=2 abandoned=1' \
|| fail "find_drain_complete_line did not capture the present line"
}
test_find_drain_complete_line_absent() {
cat > "$TMP/drain-line-absent.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
LOG
find_drain_complete_line "$TMP/drain-line-absent.log"
assert_equals "" "$REDEPLOY_DRAIN_COMPLETE_LINE" "find_drain_complete_line must report empty when absent"
}
# fleetd #512 part 2 — report_shutdown_drain, the composite decision+action function the main flow
# calls unconditionally (same shape as swap_if_built/refuse_drain_gate, #521/#528). These four cover
# the four outcomes named in the ticket's "trap": complete, died, unknown ("cannot tell" — neither a
# pass nor a failure), and n/a (no previous daemon was actually stopped this run).
#
# Deliberately NOT run inside `$(...)`: report_shutdown_drain sets REDEPLOY_DRAIN_STATE as a global
# side effect that these tests need to read back afterward, and a command substitution forks a
# subshell that global assignment would not survive (the exact trap documented above
# detect_supervisor in redeploy-fleetd.sh, for the same reason). Plain output redirection to a file
# does not fork a subshell, so it is used to capture what was printed instead.
test_report_shutdown_drain_died_without_error_token() {
cat > "$TMP/drain-died.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
LOG
local error_count
error_count="$(grep -c ' ERROR ' "$TMP/drain-died.log" || true)"
[ "$error_count" = "0" ] \
|| fail "test fixture error: drain-died.log unexpectedly carries an ERROR token"
report_shutdown_drain "$TMP/drain-died.log" 1 > "$TMP/drain-died-output" 2>&1
assert_equals "died" "$REDEPLOY_DRAIN_STATE" "died fixture must set REDEPLOY_DRAIN_STATE=died"
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "died fixture uncaught-exception count"
grep -qF 'NoClassDefFoundError' "$TMP/drain-died-output" \
|| fail "report_shutdown_drain did not report the uncaught-exception shape it found"
grep -qF 'DIED' "$TMP/drain-died-output" \
|| fail "report_shutdown_drain did not report the drain as DIED"
}
test_report_shutdown_drain_complete_control() {
cat > "$TMP/drain-complete.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=3 abandoned=0 (still BUSY at the shutdown deadline)
LOG
report_shutdown_drain "$TMP/drain-complete.log" 1 > "$TMP/drain-complete-output" 2>&1
assert_equals "complete" "$REDEPLOY_DRAIN_STATE" "complete-control fixture must set REDEPLOY_DRAIN_STATE=complete"
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "complete-control fixture must find no uncaught exception"
grep -qF 'released=3 abandoned=0' "$TMP/drain-complete-output" \
|| fail "report_shutdown_drain did not report the drain-complete counts"
}
test_report_shutdown_drain_unknown_cannot_tell() {
cat > "$TMP/drain-unknown.log" <<'LOG'
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
LOG
report_shutdown_drain "$TMP/drain-unknown.log" 1 > "$TMP/drain-unknown-output" 2>&1
assert_equals "unknown" "$REDEPLOY_DRAIN_STATE" "cannot-tell fixture must set REDEPLOY_DRAIN_STATE=unknown"
grep -qF 'cannot tell' "$TMP/drain-unknown-output" \
|| fail "report_shutdown_drain did not say it could not tell"
if grep -qF ' ok' "$TMP/drain-unknown-output"; then
fail "cannot-tell outcome must not be printed via ok() — it is neither a pass nor a failure"
fi
}
# A cold start (or a restart where nothing was actually stopped) has no previous-daemon shutdown
# window to have an opinion about at all. This fixture's log content looks exactly like a died drain
# — proving the had_previous_daemon=0 gate is actually consulted, not merely documented: without it,
# this would misreport "died" or "unknown" on every clean cold start.
test_report_shutdown_drain_no_previous_daemon_is_na() {
cat > "$TMP/drain-na.log" <<'LOG'
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
LOG
report_shutdown_drain "$TMP/drain-na.log" 0 > "$TMP/drain-na-output" 2>&1
assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon fixture must set REDEPLOY_DRAIN_STATE=n/a even though the log content looks like a died drain"
grep -qF 'nothing to check' "$TMP/drain-na-output" \
|| fail "report_shutdown_drain did not report that there was nothing to check"
}
# fleetd #552 — a fifth outcome: log_file="" (capture_fresh_log_region's own sentinel) with a
# previous daemon that WAS stopped this run. Distinct from "unknown" (a log was read and neither
# signal was found in it) — here there was no log to read at all.
test_report_shutdown_drain_skipped_when_log_missing() {
report_shutdown_drain "" 1 > "$TMP/drain-skipped-output" 2>&1
assert_equals "skipped" "$REDEPLOY_DRAIN_STATE" "an empty log_file with a previous daemon must set REDEPLOY_DRAIN_STATE=skipped"
grep -qF 'could not be captured' "$TMP/drain-skipped-output" \
|| fail "report_shutdown_drain did not say the post-restart log region could not be captured"
if grep -qF ' ok' "$TMP/drain-skipped-output"; then
fail "the skipped outcome must not be printed via ok() — it is neither a pass nor a failure"
fi
}
# fleetd #552 — proves the had_previous_daemon=0 check really does run BEFORE the missing-log check:
# with BOTH conditions true at once, n/a must win, because a cold start has nothing to check
# regardless of whether the log region could be captured.
test_report_shutdown_drain_na_wins_over_missing_log() {
report_shutdown_drain "" 0 > "$TMP/drain-na-and-missing-output" 2>&1
assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon must win over a missing log region"
}
# fleetd #512 — closes the gap none of the seven tests above can: they call report_shutdown_drain
# directly, and sourcing stops before the main flow ever runs (the SOURCED guard), so none of them
# can prove the main flow still calls it at all. Same shape as test_run_drain_gate_call_site_present
# and test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site, plus
# an ordering check against its neighbours in the verify/result flow.
test_report_shutdown_drain_call_site_present() {
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
call_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$call_line" ] \
|| fail "could not find the main flow's report_shutdown_drain call site in redeploy-fleetd.sh"
}
test_report_shutdown_drain_ordered_after_classify_and_before_result() {
local src="$ROOT/scripts/redeploy-fleetd.sh" classify_line drain_line result_line
classify_line="$(grep -Fn 'classify_amqp_connection_errors "$FRESH_LOG"' "$src" | tail -1 | cut -d: -f1 || true)"
drain_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)"
[ -n "$classify_line" ] || fail "could not find the classify_amqp_connection_errors call site"
[ -n "$drain_line" ] || fail "could not find the report_shutdown_drain call site"
[ -n "$result_line" ] || fail "could not find the result section"
[ "$drain_line" -gt "$classify_line" ] \
|| fail "report_shutdown_drain (line $drain_line) is not after classify_amqp_connection_errors (line $classify_line)"
[ "$drain_line" -lt "$result_line" ] \
|| fail "report_shutdown_drain (line $drain_line) is not before the result section (line $result_line)"
}
# fleetd #512 item 4 — the summary line must not read as reassurance when the shutdown-drain check
# found something wrong (or could not tell). Sourcing stops before the main flow runs, so this is a
# source-text check like test_drain_gate_abort_message_says_no_no_build above.
test_no_error_lines_message_gated_by_drain_state() {
local src="$ROOT/scripts/redeploy-fleetd.sh" block
block="$(grep -B2 -F 'ok "no ERROR lines since restart"' "$src")"
[ -n "$block" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh"
printf '%s' "$block" | grep -qF 'REDEPLOY_DRAIN_STATE' \
|| fail "'no ERROR lines since restart' is not guarded by the shutdown-drain outcome (fleetd #512 item 4)"
}
test_detect_supervisor_launchd_only
test_detect_supervisor_systemd_only
test_detect_supervisor_none
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
test_detect_supervisor_systemd_probe_error_is_unclear
test_detect_supervisor_systemd_probe_setup_failure_is_unclear
test_mktemp_dash_t_templates_have_x_placeholders
test_no_unguarded_macos_only_hasher_calls
test_no_unguarded_mktemp_assignments_after_restart
test_capture_fresh_log_region_control
test_capture_fresh_log_region_mktemp_failure_returns_nonzero_and_prints_nothing
test_fresh_log_capture_failure_does_not_abort_under_sete
test_capture_fresh_log_region_call_site_is_guarded
test_result_section_checks_amqp_skip_before_no_error_lines
test_require_drivable_supervisor_refuses_ambiguous
test_require_drivable_supervisor_refuses_unclear
test_require_drivable_supervisor_accepts_known_kinds
test_count_daemon_pids
test_assert_single_daemon_accepts_one_pid
test_assert_single_daemon_rejects_two_pids
test_running_pid_excludes_self_matching_wrapper_shell
test_running_pid_finds_a_real_java_named_second_process
test_running_pid_drops_a_pid_whose_comm_is_not_java
test_running_pid_drops_a_pid_that_exited_before_the_comm_lookup
test_running_pid_counts_a_pid_whose_comm_is_java
test_die_message_does_not_recommend_bare_pgrep_as_remediation
test_jar_id_defaults_to_live_and_reports_explicit_path
test_hash256_computes_a_real_sha256
test_jar_id_reports_absent_for_missing_file
test_jar_id_reports_unhashable_when_no_hasher_on_path
test_stage_built_jar_moves_off_live_path
test_stage_built_jar_dies_when_build_produced_nothing
test_swap_staged_jar_moves_staged_onto_live
test_swap_staged_jar_dies_without_staged_file
test_swap_staged_jar_dies_when_mv_fails
test_should_swap_true_when_build_ran
test_should_swap_false_when_build_skipped
test_swap_if_built_performs_the_swap_when_build_ran
test_swap_if_built_skips_the_swap_when_build_skipped
test_require_no_build_jar_dies_when_absent
test_require_no_build_jar_accepts_present_jar
test_wait_for_daemon_exit_returns_true_once_pid_clears
test_wait_for_daemon_exit_times_out_if_pid_never_clears
test_wait_for_new_pid_returns_true_once_pid_appears
test_wait_for_new_pid_times_out_if_pid_never_appears
test_await_daemon_started_slow_pid_then_healthy_succeeds
test_await_daemon_started_never_appears_dies_with_log_tail
test_await_daemon_started_pid_never_found_but_healthy_warns_and_survives
test_await_daemon_started_call_site_present
test_swap_ordered_after_wait_and_before_start
test_drain_gate_abort_message_says_no_no_build
test_drain_gate_refusal_build_ran_staged_present
test_drain_gate_refusal_build_ran_staged_absent
test_drain_gate_refusal_no_build_staged_present
test_drain_gate_refusal_no_build_staged_absent
test_refuse_drain_gate_build_ran_staged_present
test_refuse_drain_gate_build_ran_staged_absent
test_refuse_drain_gate_no_build_staged_present
test_refuse_drain_gate_no_build_staged_absent
test_run_drain_gate_call_site_present
test_should_stop_for_check_true_when_check_only
test_should_stop_for_check_false_when_not_check_only
test_stop_if_check_only_exits_zero_and_says_nothing_changed
test_stop_if_check_only_returns_without_exiting_when_not_check_only
test_stop_if_check_only_call_site_present
test_drain_gate_required_true_when_pid_and_not_assume_yes
test_drain_gate_required_false_when_no_pid
test_drain_gate_required_false_when_assume_yes
test_drain_confirmed_true_on_yes
test_drain_confirmed_false_on_anything_else
test_run_drain_gate_skips_prompt_when_not_required
test_run_drain_gate_confirmed_reply_does_not_refuse
test_run_drain_gate_declined_reply_refuses
test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path
test_report_supervisor_state_systemd_sets_supervised_without_log_path_check
test_report_supervisor_state_none_leaves_supervised_zero
test_report_supervisor_state_unrecognized_kind_warns_and_continues
test_report_supervisor_state_call_site_present
test_dispatch_stop_launchd_calls_stop_via_launchd
test_dispatch_stop_systemd_calls_stop_via_systemd
test_dispatch_stop_none_calls_stop_via_kill_with_pid
test_dispatch_stop_unrecognized_kind_dies
test_dispatch_stop_call_site_present
test_dispatch_start_launchd_calls_start_via_launchd
test_dispatch_start_systemd_calls_start_via_systemd
test_dispatch_start_none_calls_start_via_none
test_dispatch_start_unrecognized_kind_dies
test_dispatch_start_call_site_present
test_health_is_up_true_with_body
test_health_is_up_false_with_empty_body
test_report_health_up_reports_ok_and_never_dies
test_report_health_degraded_503_warns_and_never_dies
test_report_health_dead_dies
test_poll_health_body_returns_nonzero_when_unreachable
test_report_health_call_site_present
test_compute_had_old_pid_zero_when_empty
test_compute_had_old_pid_one_when_present
test_compute_had_old_pid_call_site_present
test_no_untested_main_flow_conditionals
test_unload_launchd_if_loaded_dies_on_real_failure
test_unload_launchd_if_loaded_tolerates_clean_negative
test_stop_systemd_if_loaded_dies_on_real_failure
test_stop_systemd_if_loaded_tolerates_clean_negative
test_stop_branches_call_tolerant_helpers_not_bare_or_true
test_no_errors
test_classify_amqp_connection_errors_reports_skipped_when_log_missing
test_recovery_patterns_match_source
test_attributed_recovered_connection_error
test_source_derived_error_shapes_recover_by_connection
test_cross_connection_unattributable_errors_stay_loud
test_attributed_cross_connection_errors_stay_loud
test_attributed_unrecovered_connection_error
test_other_error_is_unexplained
test_recovery_requirement_mutation_is_caught
test_shared_counter_mutation_is_caught
test_unattributable_quiet_mutation_is_caught
test_scan_uncaught_exceptions_finds_shape_without_error_token
test_scan_uncaught_exceptions_clean_control
test_find_drain_complete_line_present
test_find_drain_complete_line_absent
test_report_shutdown_drain_died_without_error_token
test_report_shutdown_drain_complete_control
test_report_shutdown_drain_unknown_cannot_tell
test_report_shutdown_drain_no_previous_daemon_is_na
test_report_shutdown_drain_skipped_when_log_missing
test_report_shutdown_drain_na_wins_over_missing_log
test_report_shutdown_drain_call_site_present
test_report_shutdown_drain_ordered_after_classify_and_before_result
test_no_error_lines_message_gated_by_drain_state
printf 'PASS: redeploy log classifier\n'