#!/usr/bin/env bash # Self-contained checks for the pure log classifier in redeploy-fleetd.sh. set -euo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")" trap 'rm -rf "$TMP"' EXIT # Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon. source "$ROOT/scripts/redeploy-fleetd.sh" fail() { printf 'FAIL: %s\n' "$*" >&2 return 1 } assert_equals() { local expected="$1" actual="$2" description="$3" [ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual" } classify_fixture() { local name="$1" classify_amqp_connection_errors "$TMP/$name" } # fleetd #492 follow-up: detect_supervisor's stdout is now "kinddetail" (see the constraints # comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about # the kind must split it out with the SAME in-shell parameter expansion the real call site (:438) # uses, never a bare string comparison against the raw output. supervisor_kind_of() { printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}" } supervisor_detail_of() { printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}" } # fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so # each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either # being naturally absent: this machine may itself be running a real fleetd under launchd right now # (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving # launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state # instead of the fixture. test_detect_supervisor_launchd_only() { launchd_installed() { return 0; } launchd_loaded() { return 0; } systemd_installed() { return 1; } systemd_loaded() { return 1; } assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection" } test_detect_supervisor_systemd_only() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } systemd_loaded() { return 0; } assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection" } test_detect_supervisor_none() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 1; } systemd_loaded() { return 1; } assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection" } # fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not # tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that # fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit # reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending # auto-restart too, and every one of those is a host that IS under systemd. test_detect_supervisor_systemd_installed_not_loaded_is_unclear() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } # the unit file IS there systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart local raw raw="$(detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail does not name the systemd unit it found installed-but-not-loaded" } # Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being # removed, or about to be reloaded) must not read as "no supervisor" either. test_detect_supervisor_launchd_installed_not_loaded_is_unclear() { launchd_installed() { return 0; } # the plist IS there launchd_loaded() { return 1; } # launchctl list says not loaded systemd_installed() { return 1; } systemd_loaded() { return 1; } local raw raw="$(detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \ || fail "detail does not name the launchd label it found installed-but-not-loaded" } # Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl` # stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl # that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no # lingering). This must read as unclear, never none: a probe that could not answer at all is not # the same fact as "no supervisor is loaded". test_detect_supervisor_systemd_probe_error_is_unclear() { # Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this # file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point # of those tests), and a bash function definition is global for the rest of the process — without # this, systemd_loaded here would still be whatever the previous test left it as, never touching # a real `systemctl` call at all. source "$ROOT/scripts/redeploy-fleetd.sh" local bin_dir result rc=0 bin_dir="$TMP/stub-bin-systemctl-errors" mkdir -p "$bin_dir" cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash echo "Failed to connect to bus: No such file or directory" >&2 exit 1 STUB chmod +x "$bin_dir/systemctl" PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$? [ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored" assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative" launchd_installed() { return 1; } launchd_loaded() { return 1; } result="$(PATH="$bin_dir:$PATH" detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$result")" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail does not name the systemd unit whose probe errored" } # fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a # hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not # the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real # caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout # survives that boundary, so kind AND detail must both cross on it — this test proves they do. test_require_drivable_supervisor_refuses_unclear() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } systemd_loaded() { return 1; } local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0 # Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand. SUPERVISOR_RAW="$(detect_supervisor)" SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}" SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}" assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal" [ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \ || fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell" printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded" output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor" # Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate # recovery instructions name the unit unconditionally either way ("systemctl --user status # $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only # the specific detail string proves the crossed value, not the boilerplate, reached the message. printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \ || fail "refusal message does not contain the specific detail that crossed the subshell boundary" } # The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to # `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that # forks a subshell, so its exit() only ends the subshell and this test script keeps running under # `set -e`. test_require_drivable_supervisor_refuses_ambiguous() { local output rc=0 output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor" printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \ || fail "refusal message does not name the launchd label it found" printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \ || fail "refusal message does not name the systemd unit it found" } test_require_drivable_supervisor_accepts_known_kinds() { require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor" require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor" require_drivable_supervisor "none" || fail "refused the unsupervised case" } # fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor # produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see # it because either daemon alone satisfies them. test_count_daemon_pids() { assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list" assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid" assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids" } test_assert_single_daemon_accepts_one_pid() { assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid" } test_assert_single_daemon_rejects_two_pids() { local output rc=0 output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids" printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found" printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found" } # fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports # $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call # must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR. # Two files with different content, so a default pointed at the wrong one reports the wrong hash # rather than accidentally matching. test_jar_id_defaults_to_live_and_reports_explicit_path() { local dir saved_jar="$JAR" saved_staged="$JAR_STAGED" local live_hash staged_hash default_result explicit_result dir="$TMP/jar-id"; mkdir -p "$dir" JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar" printf 'live jar bytes' > "$JAR" printf 'staged jar bytes, not the same content' > "$JAR_STAGED" live_hash="$(shasum -a 256 "$JAR" | cut -c1-12)" staged_hash="$(shasum -a 256 "$JAR_STAGED" | cut -c1-12)" default_result="$(jar_id)" explicit_result="$(jar_id "$JAR_STAGED")" JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same" assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR" assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR" } # fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511) # proves both halves of the present-file contract but never exercises the missing-file path. This # word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean` # deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points # operators at `--check` for exactly this. Covers both the no-argument default and an explicit path, # since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both. test_jar_id_reports_absent_for_missing_file() { local saved_jar="$JAR" dir default_result explicit_result dir="$TMP/jar-id-absent"; mkdir -p "$dir" JAR="$dir/does-not-exist.jar" [ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR" default_result="$(jar_id)" explicit_result="$(jar_id "$dir/also-does-not-exist.jar")" JAR="$saved_jar" assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist" assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent" } # fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar # are exercised directly against real files on disk (not stubs), because the whole point is file # behavior (does the content move, does the source disappear, does a failure leave both sides # intact) that a stubbed function cannot prove. test_stage_built_jar_moves_off_live_path() { local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED" dir="$TMP/stage-ok"; mkdir -p "$dir" jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar" printf 'built jar bytes' > "$jar" JAR="$jar"; JAR_STAGED="$staged" stage_built_jar || fail "stage_built_jar rejected a real build output" JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar" [ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged" grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content" } test_stage_built_jar_dies_when_build_produced_nothing() { local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED" dir="$TMP/stage-missing"; mkdir -p "$dir" JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar" output="$(stage_built_jar 2>&1)" || rc=$? JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output" printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \ || fail "refusal message does not name the missing jar path" } test_swap_staged_jar_moves_staged_onto_live() { local dir staged live dir="$TMP/swap-ok"; mkdir -p "$dir" staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar" printf 'swapped jar bytes' > "$staged" swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar" [ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged" [ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live" grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content" } # The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on # failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes # inside a command substitution to contain that exit to a subshell. test_swap_staged_jar_dies_without_staged_file() { local dir output rc=0 dir="$TMP/swap-missing"; mkdir -p "$dir" output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar" [ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged" printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \ || fail "refusal message does not name the missing staged path" } test_swap_staged_jar_dies_when_mv_fails() { local dir staged live output rc=0 dir="$TMP/swap-fail"; mkdir -p "$dir/src" staged="$dir/src/fleetd-new.jar" printf 'fake jar bytes' > "$staged" live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv" [ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails" [ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed" printf '%s' "$output" | grep -qF "$staged" \ || fail "refusal message does not name the staged path that could not be moved" } # --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this # path) and must still die with the exact wording documented in the script's own header comment. test_require_no_build_jar_dies_when_absent() { local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar" JAR="$missing" output="$(require_no_build_jar 2>&1)" || rc=$? JAR="$saved_jar" [ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar" printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \ || fail "refusal message does not match the documented --no-build wording" } test_require_no_build_jar_accepts_present_jar() { local saved_jar="$JAR" dir dir="$TMP/no-build-present"; mkdir -p "$dir" JAR="$dir/fleetd.jar" printf 'existing jar' > "$JAR" require_no_build_jar || fail "require_no_build_jar rejected an existing jar" JAR="$saved_jar" } # wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while # running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so # the timeout-loop test does not actually wait out its budget. test_wait_for_daemon_exit_returns_true_once_pid_clears() { # running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so # a plain shell variable it increments would reset on each call instead of accumulating. Count in # a file instead, which is the one thing that actually survives across those subshells. local counter_file="$TMP/wait-exit-calls" final_calls printf '0' > "$counter_file" running_pid() { local n n="$(cat "$counter_file")" n=$((n + 1)) printf '%s' "$n" > "$counter_file" if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi } sleep() { :; } wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared" final_calls="$(cat "$counter_file")" [ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } test_wait_for_daemon_exit_times_out_if_pid_never_clears() { local rc=0 running_pid() { printf '4242'; } sleep() { :; } wait_for_daemon_exit 3 || rc=$? [ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } # fleetd #521 — the swap step's guard, at two levels. # # The first two tests call the predicate should_swap() directly. They pin its logic, and that is all # they pin. On their own they did NOT close #521, and this was measured rather than argued: with the # main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left # this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs # the swap consult the predicate at all. Extracting the decision had moved the untested decision up # a level, not removed it. # # So the last two tests call swap_if_built() — the function the main flow actually calls, holding the # guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the # guard is removed, inverted, or stops being consulted. # # What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow # altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing # stops before the main flow runs, so no test in this file can invoke it. test_should_swap_true_when_build_ran() { should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true" } test_should_swap_false_when_build_skipped() { if should_swap 0; then fail "should_swap 0 (--no-build; nothing was staged this run) must return false" fi } # Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is # global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id # overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear), # and again at the END, so their own stubs do not leak into every test that runs after them. test_swap_if_built_performs_the_swap_when_build_ran() { source "$ROOT/scripts/redeploy-fleetd.sh" local marker="$TMP/swap-if-built-ran" rm -f "$marker" swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; } jar_id() { printf 'stubbed\n'; } swap_if_built 1 > /dev/null [ -f "$marker" ] \ || fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not" source "$ROOT/scripts/redeploy-fleetd.sh" } test_swap_if_built_skips_the_swap_when_build_skipped() { source "$ROOT/scripts/redeploy-fleetd.sh" local marker="$TMP/swap-if-built-skipped" rm -f "$marker" swap_staged_jar() { printf 'swapped\n' > "$marker"; } jar_id() { printf 'stubbed\n'; } swap_if_built 0 > /dev/null if [ -f "$marker" ]; then fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did" fi source "$ROOT/scripts/redeploy-fleetd.sh" } # fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the # main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee # itself — as opposed to the pure functions it's built from — can only be checked by reading the # script's own call sites, the same way test_recovery_patterns_match_source below checks Java # source shape instead of behavior it cannot invoke directly. # # Two details about the three greps below, both of which have already gone wrong here. # # The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not # `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside # swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using # it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)" # — a true statement about a function definition, and nothing at all about the order of the steps. # # Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the # pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set # -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to # report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no # name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak. test_swap_ordered_after_wait_and_before_start() { local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)" swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)" start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh" [ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh" [ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh" [ "$swap_line" -gt "$wait_line" ] \ || fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)" [ "$swap_line" -lt "$start_line" ] \ || fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)" } # fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator # declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)" # to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has # already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal # ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start # above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow), # so the only way to pin its exact wording is to read the source. test_drain_gate_abort_message_says_no_no_build() { local src="$ROOT/scripts/redeploy-fleetd.sh" msg msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")" [ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh" if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then fail "abort message still claims a rerun WITH --no-build can finish the restart" fi printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \ || fail "abort message does not tell the operator to rerun without --no-build" printf '%s' "$msg" | grep -qF 'no longer at the live path' \ || fail "abort message does not say why --no-build cannot finish the restart" } # fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was # a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's # own file for the wording, which stays in the file even if the `if` guarding it is mutated to # `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead, # so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT # alongside these, not replaced, because it catches a different regression (a re-wording that still # reaches the right branch would not change which case fires here, but would still be worth pinning). test_drain_gate_refusal_build_ran_staged_present() { local dir staged result dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'staged jar bytes' > "$staged" result="$(drain_gate_refusal 1 "$staged")" printf '%s' "$result" | grep -qF "$staged" \ || fail "build-ran+staged-present refusal does not name the staged jar path" printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \ || fail "build-ran+staged-present refusal does not tell the operator how to finish the restart" if printf '%s' "$result" | grep -qF 'nothing changed'; then fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved" fi } test_drain_gate_refusal_build_ran_staged_absent() { local dir result dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir" result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")" assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording" } # --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar # found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the # comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning. test_drain_gate_refusal_no_build_staged_present() { local dir staged result dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'leftover staged jar bytes' > "$staged" result="$(drain_gate_refusal 0 "$staged")" assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed" } test_drain_gate_refusal_no_build_staged_absent() { local dir result dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir" result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")" assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording" } # fleetd #528 — the four tests above pin drain_gate_refusal(), and that is ALL they pin: they call # the predicate directly and never touch the main flow's call site. That was measured to be not # enough, the same way test_should_swap_true_when_build_ran/test_should_swap_false_when_build_skipped # were not enough for #521: with the main flow reading `die "$(drain_gate_refusal "$DO_BUILD" # "$JAR_STAGED")"`, replacing that whole line with a flat `die "aborted — nothing changed"` left this # suite at exit 0 with zero FAIL lines and byte-identical output to a clean run. Nothing above could # tell the difference, because none of it calls anything at or above the call site itself. # # So these four call refuse_drain_gate() — the function the main flow actually calls, holding the # composed message and the die() together — with die() stubbed to RECORD whether it was called and # with what message, instead of exiting the process. That fails if refuse_drain_gate stops consulting # drain_gate_refusal, mangles what it passes it, or simply never calls die. # # What none of these four can catch: deleting the `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` line # from the main flow altogether — see the comment above refuse_drain_gate in redeploy-fleetd.sh for # why no test in this file can do better than that (sourcing stops before the main flow runs). DIED_CALLED=0 DIED_MESSAGE="" stub_die_recorder() { DIED_CALLED=0 DIED_MESSAGE="" die() { DIED_CALLED=1; DIED_MESSAGE="$*"; } } test_refuse_drain_gate_build_ran_staged_present() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir staged dir="$TMP/refuse-drain-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'staged jar bytes' > "$staged" stub_die_recorder refuse_drain_gate 1 "$staged" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate build-ran+staged-present must call die, and did not" printf '%s' "$DIED_MESSAGE" | grep -qF "$staged" \ || fail "refuse_drain_gate build-ran+staged-present die message does not name the staged jar" printf '%s' "$DIED_MESSAGE" | grep -qF 'Rerun WITHOUT --no-build' \ || fail "refuse_drain_gate build-ran+staged-present die message is missing the rerun instruction" if printf '%s' "$DIED_MESSAGE" | grep -qF 'nothing changed'; then fail "refuse_drain_gate build-ran+staged-present must not claim nothing changed — the jar already moved" fi source "$ROOT/scripts/redeploy-fleetd.sh" } test_refuse_drain_gate_build_ran_staged_absent() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir dir="$TMP/refuse-drain-build-no-staged"; mkdir -p "$dir" stub_die_recorder refuse_drain_gate 1 "$dir/fleetd-new.jar" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate build-ran+staged-absent must call die, and did not" assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate build-ran+staged-absent die message" source "$ROOT/scripts/redeploy-fleetd.sh" } test_refuse_drain_gate_no_build_staged_present() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir staged dir="$TMP/refuse-drain-no-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'leftover staged jar bytes' > "$staged" stub_die_recorder refuse_drain_gate 0 "$staged" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate no-build+staged-present must call die, and did not" assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-present die message" source "$ROOT/scripts/redeploy-fleetd.sh" } test_refuse_drain_gate_no_build_staged_absent() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir dir="$TMP/refuse-drain-no-build-no-staged"; mkdir -p "$dir" stub_die_recorder refuse_drain_gate 0 "$dir/fleetd-new.jar" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate no-build+staged-absent must call die, and did not" assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-absent die message" source "$ROOT/scripts/redeploy-fleetd.sh" } # fleetd #528 — closes the one gap the four behavioural tests above cannot: they call # refuse_drain_gate directly, and sourcing stops before the main flow ever runs (the SOURCED guard), # so none of them can prove the main flow still CALLS refuse_drain_gate at all. Same shape as # test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site. This is # what actually kills the item-1 mutation from the ticket — replacing the main flow's call with a # flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural # tests above would even notice. # # The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail # the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below # ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body). test_refuse_drain_gate_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's refuse_drain_gate call site in redeploy-fleetd.sh" } # fleetd #504 — the "loaded but not currently running" branches for launchd/systemd used to run # `launchctl unload`/`systemctl --user stop` with `2>/dev/null || true` and print `ok` # unconditionally, so a real supervisor failure (e.g. it cannot reach launchd/the systemd user bus) # read exactly like a harmless already-stopped answer. unload_launchd_if_loaded/ # stop_systemd_if_loaded (redeploy-fleetd.sh, right after systemd_loaded) apply systemd_loaded's own # "capture stderr separately — only a non-zero exit WITH stderr is a real failure" pattern to the # WRITE side. Both call the real `launchctl`/`systemctl` binaries directly (they are not overridable # wrapper functions the way launchd_loaded/systemd_loaded are), so these tests put a stub binary # first on PATH — the same technique test_detect_supervisor_systemd_probe_error_is_unclear above # already uses for `systemctl`. test_unload_launchd_if_loaded_dies_on_real_failure() { local bin_dir output rc=0 saved_plist="$LAUNCHD_PLIST" bin_dir="$TMP/stub-bin-launchctl-error"; mkdir -p "$bin_dir" cat > "$bin_dir/launchctl" <<'STUB' #!/usr/bin/env bash echo "Could not find specified service" >&2 exit 1 STUB chmod +x "$bin_dir/launchctl" LAUNCHD_PLIST="$TMP/fake-fail.plist" output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$? LAUNCHD_PLIST="$saved_plist" [ "$rc" -ne 0 ] \ || fail "unload_launchd_if_loaded must die when launchctl exits non-zero AND writes to stderr" printf '%s' "$output" | grep -qF 'launchctl unload' \ || fail "die message does not name the failing launchctl unload command" } # Captured via $(...) rather than called bare: unload_launchd_if_loaded's own die() does a hard # `exit`, and calling it directly at this level would let a regression that makes it die on this # clean-negative case kill the WHOLE suite before the `|| fail` below ever ran — printing die's own # message instead of this test's. Inside a command substitution, that `exit` only ends the subshell # (a-guard-is-defeated-by-its-calling-context: the same reason the *_dies_on_real_failure tests # above capture this way), so this test's own message is what actually reaches the report. test_unload_launchd_if_loaded_tolerates_clean_negative() { local bin_dir saved_plist="$LAUNCHD_PLIST" output rc=0 bin_dir="$TMP/stub-bin-launchctl-noop"; mkdir -p "$bin_dir" cat > "$bin_dir/launchctl" <<'STUB' #!/usr/bin/env bash exit 1 STUB chmod +x "$bin_dir/launchctl" LAUNCHD_PLIST="$TMP/fake-noop.plist" output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$? LAUNCHD_PLIST="$saved_plist" [ "$rc" -eq 0 ] \ || fail "unload_launchd_if_loaded must tolerate a clean already-unloaded answer (non-zero exit, empty stderr): $output" } test_stop_systemd_if_loaded_dies_on_real_failure() { local bin_dir output rc=0 bin_dir="$TMP/stub-bin-systemctl-stop-error"; mkdir -p "$bin_dir" cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash echo "Failed to connect to bus: No such file or directory" >&2 exit 1 STUB chmod +x "$bin_dir/systemctl" output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$? [ "$rc" -ne 0 ] \ || fail "stop_systemd_if_loaded must die when systemctl exits non-zero AND writes to stderr" printf '%s' "$output" | grep -qF 'systemctl --user stop' \ || fail "die message does not name the failing systemctl --user stop command" } # Same subshell-capture reasoning as test_unload_launchd_if_loaded_tolerates_clean_negative above: # stop_systemd_if_loaded's own die() does a hard `exit`, so this must run inside $(...) or a # regression here would kill the whole suite with die's message instead of this test's. test_stop_systemd_if_loaded_tolerates_clean_negative() { local bin_dir output rc=0 bin_dir="$TMP/stub-bin-systemctl-stop-noop"; mkdir -p "$bin_dir" cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash exit 1 STUB chmod +x "$bin_dir/systemctl" output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$? [ "$rc" -eq 0 ] \ || fail "stop_systemd_if_loaded must tolerate a clean already-stopped answer (non-zero exit, empty stderr): $output" } # Closes the same gap test_refuse_drain_gate_call_site_present closes for the drain gate: the four # tests above call unload_launchd_if_loaded/stop_systemd_if_loaded directly, and sourcing stops # before the main flow ever runs (the SOURCED guard), so none of them can prove the main flow still # CALLS these two functions instead of the original bare `2>/dev/null || true`. A source-text check, # like test_swap_ordered_after_wait_and_before_start. The call-site needle is anchored (`^ name$`) # so it cannot be satisfied by the comment lines above each call site that merely mention the # function by name. test_stop_branches_call_tolerant_helpers_not_bare_or_true() { local src="$ROOT/scripts/redeploy-fleetd.sh" unload_call_line stop_call_line unload_call_line="$(grep -n '^ unload_launchd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)" stop_call_line="$(grep -n '^ stop_systemd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$unload_call_line" ] \ || fail "could not find the main flow's call to unload_launchd_if_loaded in redeploy-fleetd.sh" [ -n "$stop_call_line" ] \ || fail "could not find the main flow's call to stop_systemd_if_loaded in redeploy-fleetd.sh" if grep -qF 'launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true' "$src"; then fail "the bare 'launchctl unload ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh" fi if grep -qF 'systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true' "$src"; then fail "the bare 'systemctl --user stop ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh" fi } test_no_errors() { cat > "$TMP/no-errors.log" <<'LOG' 2026-09-05 12:00:00 INFO fleetd listening LOG classify_fixture no-errors.log assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained" } test_recovery_patterns_match_source() { grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \ || fail "AMQP failure pattern no longer matches source" grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \ "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \ || fail "reply-inbox recovery pattern no longer matches source" grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \ "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \ || fail "lead-mailbox recovery pattern no longer matches source" } test_attributed_recovered_connection_error() { cat > "$TMP/attributed-recovered.log" <<'LOG' 2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery LOG classify_fixture attributed-recovered.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total" assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained" } test_source_derived_error_shapes_recover_by_connection() { # These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check # after redeploy because the new code has not yet written a production line. cat > "$TMP/source-derived.log" <<'LOG' 17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery! 17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred 17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery! 17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred 17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture source-derived.log assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total" assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained" } test_cross_connection_unattributable_errors_stay_loud() { # This candidate has neither stable connection name, so LeadMailbox recovery must not consume it. cat > "$TMP/cross-unattributable.log" <<'LOG' 2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred 2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred 2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture cross-unattributable.log assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered" assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained" } test_attributed_cross_connection_errors_stay_loud() { # LeadMailbox recovery cannot heal AmqpReplyInbox errors. cat > "$TMP/cross-attributed.log" <<'LOG' 2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture cross-attributed.log assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered" assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained" } test_attributed_unrecovered_connection_error() { cat > "$TMP/unrecovered.log" <<'LOG' 2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred LOG classify_fixture unrecovered.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors" assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained" } test_other_error_is_unexplained() { cat > "$TMP/other-error.log" <<'LOG' 2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed 2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery LOG classify_fixture other-error.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total" assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained" } test_recovery_requirement_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; esac done < "$log_file" } if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then fail "mutation accepted an unrecovered connection error" fi grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \ || fail "mutation failed without the expected assertion" printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n' } test_shared_counter_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line pending=0 REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*) case "$line" in *'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; *'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*) if [ "$pending" -gt 0 ]; then pending=$((pending - 1)) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) fi ;; esac done < "$log_file" REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending)) } if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then fail "shared counter mutation accepted cross-connection recovery" fi grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \ || fail "shared counter mutation failed without the expected assertion" printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n' } test_unattributable_quiet_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; esac done < "$log_file" } if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then fail "unattributable mutation accepted an unknown connection" fi grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \ || fail "unattributable mutation failed without the expected assertion" printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n' } # fleetd #512 part 2 — the negative check (scan_uncaught_exceptions). The heart of this half of the # ticket: a fixture with the uncaught-exception shape and NO line carrying an ERROR token at all, # proving the scan finds it without one. A fixture that also carried an ERROR line would pass for # the wrong reason. test_scan_uncaught_exceptions_finds_shape_without_error_token() { cat > "$TMP/scan-died.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081) LOG local error_count error_count="$(grep -c ' ERROR ' "$TMP/scan-died.log" || true)" [ "$error_count" = "0" ] \ || fail "test fixture error: scan-died.log unexpectedly carries an ERROR token" scan_uncaught_exceptions "$TMP/scan-died.log" assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "scan must find the exception without an ERROR token" printf '%s' "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" | grep -qF 'NoClassDefFoundError' \ || fail "scan did not capture the matching line as the sample" } test_scan_uncaught_exceptions_clean_control() { cat > "$TMP/scan-clean.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=0 abandoned=0 (still BUSY at the shutdown deadline) LOG scan_uncaught_exceptions "$TMP/scan-clean.log" assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "clean control must find no uncaught exception" assert_equals "" "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" "clean control sample must be empty" } # fleetd #512 part 2 — the positive check (find_drain_complete_line). Both halves of #522's line: # present, and absent. test_find_drain_complete_line_present() { cat > "$TMP/drain-line-present.log" <<'LOG' 2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=2 abandoned=1 (still BUSY at the shutdown deadline) LOG find_drain_complete_line "$TMP/drain-line-present.log" printf '%s' "$REDEPLOY_DRAIN_COMPLETE_LINE" | grep -qF 'released=2 abandoned=1' \ || fail "find_drain_complete_line did not capture the present line" } test_find_drain_complete_line_absent() { cat > "$TMP/drain-line-absent.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 LOG find_drain_complete_line "$TMP/drain-line-absent.log" assert_equals "" "$REDEPLOY_DRAIN_COMPLETE_LINE" "find_drain_complete_line must report empty when absent" } # fleetd #512 part 2 — report_shutdown_drain, the composite decision+action function the main flow # calls unconditionally (same shape as swap_if_built/refuse_drain_gate, #521/#528). These four cover # the four outcomes named in the ticket's "trap": complete, died, unknown ("cannot tell" — neither a # pass nor a failure), and n/a (no previous daemon was actually stopped this run). # # Deliberately NOT run inside `$(...)`: report_shutdown_drain sets REDEPLOY_DRAIN_STATE as a global # side effect that these tests need to read back afterward, and a command substitution forks a # subshell that global assignment would not survive (the exact trap documented above # detect_supervisor in redeploy-fleetd.sh, for the same reason). Plain output redirection to a file # does not fork a subshell, so it is used to capture what was printed instead. test_report_shutdown_drain_died_without_error_token() { cat > "$TMP/drain-died.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081) LOG local error_count error_count="$(grep -c ' ERROR ' "$TMP/drain-died.log" || true)" [ "$error_count" = "0" ] \ || fail "test fixture error: drain-died.log unexpectedly carries an ERROR token" report_shutdown_drain "$TMP/drain-died.log" 1 > "$TMP/drain-died-output" 2>&1 assert_equals "died" "$REDEPLOY_DRAIN_STATE" "died fixture must set REDEPLOY_DRAIN_STATE=died" assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "died fixture uncaught-exception count" grep -qF 'NoClassDefFoundError' "$TMP/drain-died-output" \ || fail "report_shutdown_drain did not report the uncaught-exception shape it found" grep -qF 'DIED' "$TMP/drain-died-output" \ || fail "report_shutdown_drain did not report the drain as DIED" } test_report_shutdown_drain_complete_control() { cat > "$TMP/drain-complete.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down 2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=3 abandoned=0 (still BUSY at the shutdown deadline) LOG report_shutdown_drain "$TMP/drain-complete.log" 1 > "$TMP/drain-complete-output" 2>&1 assert_equals "complete" "$REDEPLOY_DRAIN_STATE" "complete-control fixture must set REDEPLOY_DRAIN_STATE=complete" assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "complete-control fixture must find no uncaught exception" grep -qF 'released=3 abandoned=0' "$TMP/drain-complete-output" \ || fail "report_shutdown_drain did not report the drain-complete counts" } test_report_shutdown_drain_unknown_cannot_tell() { cat > "$TMP/drain-unknown.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down LOG report_shutdown_drain "$TMP/drain-unknown.log" 1 > "$TMP/drain-unknown-output" 2>&1 assert_equals "unknown" "$REDEPLOY_DRAIN_STATE" "cannot-tell fixture must set REDEPLOY_DRAIN_STATE=unknown" grep -qF 'cannot tell' "$TMP/drain-unknown-output" \ || fail "report_shutdown_drain did not say it could not tell" if grep -qF ' ok' "$TMP/drain-unknown-output"; then fail "cannot-tell outcome must not be printed via ok() — it is neither a pass nor a failure" fi } # A cold start (or a restart where nothing was actually stopped) has no previous-daemon shutdown # window to have an opinion about at all. This fixture's log content looks exactly like a died drain # — proving the had_previous_daemon=0 gate is actually consulted, not merely documented: without it, # this would misreport "died" or "unknown" on every clean cold start. test_report_shutdown_drain_no_previous_daemon_is_na() { cat > "$TMP/drain-na.log" <<'LOG' Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions LOG report_shutdown_drain "$TMP/drain-na.log" 0 > "$TMP/drain-na-output" 2>&1 assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon fixture must set REDEPLOY_DRAIN_STATE=n/a even though the log content looks like a died drain" grep -qF 'nothing to check' "$TMP/drain-na-output" \ || fail "report_shutdown_drain did not report that there was nothing to check" } # fleetd #512 — closes the gap none of the seven tests above can: they call report_shutdown_drain # directly, and sourcing stops before the main flow ever runs (the SOURCED guard), so none of them # can prove the main flow still calls it at all. Same shape as test_refuse_drain_gate_call_site_present # and test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site, plus # an ordering check against its neighbours in the verify/result flow. test_report_shutdown_drain_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's report_shutdown_drain call site in redeploy-fleetd.sh" } test_report_shutdown_drain_ordered_after_classify_and_before_result() { local src="$ROOT/scripts/redeploy-fleetd.sh" classify_line drain_line result_line classify_line="$(grep -Fn 'classify_amqp_connection_errors "$FRESH_LOG"' "$src" | tail -1 | cut -d: -f1 || true)" drain_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)" result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$classify_line" ] || fail "could not find the classify_amqp_connection_errors call site" [ -n "$drain_line" ] || fail "could not find the report_shutdown_drain call site" [ -n "$result_line" ] || fail "could not find the result section" [ "$drain_line" -gt "$classify_line" ] \ || fail "report_shutdown_drain (line $drain_line) is not after classify_amqp_connection_errors (line $classify_line)" [ "$drain_line" -lt "$result_line" ] \ || fail "report_shutdown_drain (line $drain_line) is not before the result section (line $result_line)" } # fleetd #512 item 4 — the summary line must not read as reassurance when the shutdown-drain check # found something wrong (or could not tell). Sourcing stops before the main flow runs, so this is a # source-text check like test_drain_gate_abort_message_says_no_no_build above. test_no_error_lines_message_gated_by_drain_state() { local src="$ROOT/scripts/redeploy-fleetd.sh" block block="$(grep -B2 -F 'ok "no ERROR lines since restart"' "$src")" [ -n "$block" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh" printf '%s' "$block" | grep -qF 'REDEPLOY_DRAIN_STATE' \ || fail "'no ERROR lines since restart' is not guarded by the shutdown-drain outcome (fleetd #512 item 4)" } test_detect_supervisor_launchd_only test_detect_supervisor_systemd_only test_detect_supervisor_none test_detect_supervisor_systemd_installed_not_loaded_is_unclear test_detect_supervisor_launchd_installed_not_loaded_is_unclear test_detect_supervisor_systemd_probe_error_is_unclear test_require_drivable_supervisor_refuses_ambiguous test_require_drivable_supervisor_refuses_unclear test_require_drivable_supervisor_accepts_known_kinds test_count_daemon_pids test_assert_single_daemon_accepts_one_pid test_assert_single_daemon_rejects_two_pids test_jar_id_defaults_to_live_and_reports_explicit_path test_jar_id_reports_absent_for_missing_file test_stage_built_jar_moves_off_live_path test_stage_built_jar_dies_when_build_produced_nothing test_swap_staged_jar_moves_staged_onto_live test_swap_staged_jar_dies_without_staged_file test_swap_staged_jar_dies_when_mv_fails test_should_swap_true_when_build_ran test_should_swap_false_when_build_skipped test_swap_if_built_performs_the_swap_when_build_ran test_swap_if_built_skips_the_swap_when_build_skipped test_require_no_build_jar_dies_when_absent test_require_no_build_jar_accepts_present_jar test_wait_for_daemon_exit_returns_true_once_pid_clears test_wait_for_daemon_exit_times_out_if_pid_never_clears test_swap_ordered_after_wait_and_before_start test_drain_gate_abort_message_says_no_no_build test_drain_gate_refusal_build_ran_staged_present test_drain_gate_refusal_build_ran_staged_absent test_drain_gate_refusal_no_build_staged_present test_drain_gate_refusal_no_build_staged_absent test_refuse_drain_gate_build_ran_staged_present test_refuse_drain_gate_build_ran_staged_absent test_refuse_drain_gate_no_build_staged_present test_refuse_drain_gate_no_build_staged_absent test_refuse_drain_gate_call_site_present test_unload_launchd_if_loaded_dies_on_real_failure test_unload_launchd_if_loaded_tolerates_clean_negative test_stop_systemd_if_loaded_dies_on_real_failure test_stop_systemd_if_loaded_tolerates_clean_negative test_stop_branches_call_tolerant_helpers_not_bare_or_true test_no_errors test_recovery_patterns_match_source test_attributed_recovered_connection_error test_source_derived_error_shapes_recover_by_connection test_cross_connection_unattributable_errors_stay_loud test_attributed_cross_connection_errors_stay_loud test_attributed_unrecovered_connection_error test_other_error_is_unexplained test_recovery_requirement_mutation_is_caught test_shared_counter_mutation_is_caught test_unattributable_quiet_mutation_is_caught test_scan_uncaught_exceptions_finds_shape_without_error_token test_scan_uncaught_exceptions_clean_control test_find_drain_complete_line_present test_find_drain_complete_line_absent test_report_shutdown_drain_died_without_error_token test_report_shutdown_drain_complete_control test_report_shutdown_drain_unknown_cannot_tell test_report_shutdown_drain_no_previous_daemon_is_na test_report_shutdown_drain_call_site_present test_report_shutdown_drain_ordered_after_classify_and_before_result test_no_error_lines_message_gated_by_drain_state printf 'PASS: redeploy log classifier\n'