#!/usr/bin/env bash # Self-contained checks for the pure log classifier in redeploy-fleetd.sh. set -euo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")" trap 'rm -rf "$TMP"' EXIT # Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon. source "$ROOT/scripts/redeploy-fleetd.sh" fail() { printf 'FAIL: %s\n' "$*" >&2 return 1 } assert_equals() { local expected="$1" actual="$2" description="$3" [ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual" } classify_fixture() { local name="$1" classify_amqp_connection_errors "$TMP/$name" } # fleetd #492 follow-up: detect_supervisor's stdout is now "kinddetail" (see the constraints # comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about # the kind must split it out with the SAME in-shell parameter expansion the real call site (:438) # uses, never a bare string comparison against the raw output. supervisor_kind_of() { printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}" } supervisor_detail_of() { printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}" } # fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so # each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either # being naturally absent: this machine may itself be running a real fleetd under launchd right now # (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving # launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state # instead of the fixture. test_detect_supervisor_launchd_only() { launchd_installed() { return 0; } launchd_loaded() { return 0; } systemd_installed() { return 1; } systemd_loaded() { return 1; } assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection" } test_detect_supervisor_systemd_only() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } systemd_loaded() { return 0; } assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection" } test_detect_supervisor_none() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 1; } systemd_loaded() { return 1; } assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection" } # fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not # tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that # fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit # reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending # auto-restart too, and every one of those is a host that IS under systemd. test_detect_supervisor_systemd_installed_not_loaded_is_unclear() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } # the unit file IS there systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart local raw raw="$(detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail does not name the systemd unit it found installed-but-not-loaded" } # Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being # removed, or about to be reloaded) must not read as "no supervisor" either. test_detect_supervisor_launchd_installed_not_loaded_is_unclear() { launchd_installed() { return 0; } # the plist IS there launchd_loaded() { return 1; } # launchctl list says not loaded systemd_installed() { return 1; } systemd_loaded() { return 1; } local raw raw="$(detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \ || fail "detail does not name the launchd label it found installed-but-not-loaded" } # Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl` # stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl # that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no # lingering). This must read as unclear, never none: a probe that could not answer at all is not # the same fact as "no supervisor is loaded". test_detect_supervisor_systemd_probe_error_is_unclear() { # Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this # file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point # of those tests), and a bash function definition is global for the rest of the process — without # this, systemd_loaded here would still be whatever the previous test left it as, never touching # a real `systemctl` call at all. source "$ROOT/scripts/redeploy-fleetd.sh" local bin_dir result rc=0 bin_dir="$TMP/stub-bin-systemctl-errors" mkdir -p "$bin_dir" cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash echo "Failed to connect to bus: No such file or directory" >&2 exit 1 STUB chmod +x "$bin_dir/systemctl" PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$? [ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored" assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative" launchd_installed() { return 1; } launchd_loaded() { return 1; } result="$(PATH="$bin_dir:$PATH" detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none" local detail detail="$(supervisor_detail_of "$result")" printf '%s' "$detail" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail does not name the systemd unit whose probe errored" # fleetd #545: this is the PROBE-RAN-AND-ANSWERED-BADLY case (systemctl actually executed and # wrote to stderr) — it must carry that story and never the SET-UP-FAILED story (mktemp never # even ran here), or the two "unclear" causes have collapsed back into one message that asserts a # cause it did not measure, which is the exact defect this ticket exists to fix. printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \ || fail "detail does not say systemctl ran and answered with stderr: $detail" printf '%s' "$detail" | grep -qF "could not even be set up" \ && fail "detail wrongly claims the probe could not be set up, but systemctl actually ran and answered on stderr: $detail" return 0 } # fleetd #545 — the companion case to the probe-error test above: here `mktemp` itself fails # (whatever the reason — the historical bug was a GNU-mktemp-rejects-a-template-with-no-Xs case, # but this stub simulates ANY reason the probe's own stderr-capture temp file cannot be created, # e.g. a full or unwritable temp dir) and `systemctl` is never invoked at all. Before this ticket, # this collapsed into the SAME "systemctl exited non-zero and reported an error on stderr" detail # as the sibling test above, which asserts a cause (systemctl ran and answered badly) that was # never measured, because systemctl never ran. This proves the SET-UP-FAILED detail is distinct and # does not claim systemctl said anything. test_detect_supervisor_systemd_probe_setup_failure_is_unclear() { # Re-source first for the same reason test_detect_supervisor_systemd_probe_error_is_unclear does: # restore the REAL probe bodies before driving them through a stub PATH. source "$ROOT/scripts/redeploy-fleetd.sh" local bin_dir result rc=0 bin_dir="$TMP/stub-bin-mktemp-fails" mkdir -p "$bin_dir" # A systemctl stub that would fail loudly if it were ever actually invoked — proves the mktemp # failure short-circuits the probe before systemctl runs, not merely that this test forgot to # supply a working systemctl. cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash echo "systemctl must never run when mktemp already failed" >&2 exit 1 STUB chmod +x "$bin_dir/systemctl" cat > "$bin_dir/mktemp" <<'STUB' #!/usr/bin/env bash echo "mktemp: cannot create temp file" >&2 exit 1 STUB chmod +x "$bin_dir/mktemp" PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$? [ "$rc" -ne 0 ] \ || fail "systemd_loaded must not report loaded=true when its own mktemp setup failed" assert_equals "2" "$SYSTEMD_LOADED_ERRORED" \ "systemd_loaded must flag a SETUP failure (2), distinct from a probe-answered-with-stderr failure (1)" launchd_installed() { return 1; } launchd_loaded() { return 1; } result="$(PATH="$bin_dir:$PATH" detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe setup failure must read as unclear, not none" local detail detail="$(supervisor_detail_of "$result")" printf '%s' "$detail" | grep -qF "could not even be set up" \ || fail "detail does not say the probe could not be SET UP: $detail" printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \ && fail "detail wrongly asserts systemctl exited non-zero and reported an error on stderr, but systemctl was never run: $detail" return 0 } # fleetd #545 — source-text check: every `mktemp -t` template in redeploy-fleetd.sh must contain an # `X` placeholder. BSD mktemp (macOS) tolerates a bare template with no `X`s and just appends its # own random suffix, which is exactly why six such sites survived undetected here — GNU mktemp # (every Linux distribution) refuses a template with fewer than three `X`s and exits non-zero. There # is no BSD-vs-GNU seam to stub on this Mac, so this is a source-text check rather than a # behavioural one, the same shape as test_run_drain_gate_call_site_present above. Anchored on # `mktemp -t ` (with the trailing space) so it inspects only the `-t`-style templates this ticket is # about, never the `mktemp -d` calls this file and test-probe-member-credentials.sh already use # (both already carry their own `XXXXXX` and are a different mktemp mode entirely). test_mktemp_dash_t_templates_have_x_placeholders() { local src="$ROOT/scripts/redeploy-fleetd.sh" bad bad="$(grep -n 'mktemp -t ' "$src" | grep -v 'XXX' || true)" [ -z "$bad" ] \ || fail "mktemp -t template(s) with no X placeholder (fails under GNU coreutils): $bad" } # fleetd #550 — the shape, not the named lines: #545 showed the exact same failure mode (a # macOS-only idiom used with no portable fallback) spread from two sites to six across 91 commits # before anyone tested the SHAPE rather than specific lines. This is the shasum sibling: any script # under scripts/ that actually INVOKES the macOS-only hasher to compute a hash (as opposed to # merely probing whether it exists with `command -v`, or mentioning it in prose) must also check # for the portable one first in that same file — the prefer-portable-fall-back-to-macOS-only idiom # probe-member-credentials.sh:273-279 and this ticket's own hash256 helper both follow. # # The needle is built from two concatenated pieces, deliberately never written as one literal # string in this file: written whole, it would match THIS CHECK'S OWN source line once the loop # below reaches this very file, and the check would then "pass" by matching itself rather than any # real invocation elsewhere — a zero-findings result indistinguishable from a clean file. test_no_unguarded_macos_only_hasher_calls() { local needle f bad="" usage needle='shasum'; needle="$needle -a" for f in "$ROOT"/scripts/*.sh; do [ -f "$f" ] || continue usage="$(grep -Fn "$needle" "$f" || true)" if [ -n "$usage" ]; then grep -q 'command -v sha256sum' "$f" \ || bad="$bad$(basename "$f") " fi done [ -z "$bad" ] \ || fail "script(s) invoke the macOS-only hasher with no portable-hasher-first fallback guard in the same file: $bad" } # fleetd #552 — the shape, not the one line just fixed: a `mktemp` failure AFTER the daemon has # already been restarted must never be a bare, unguarded assignment again, anywhere in the script, # so the next one added is caught too — not just line 1113 as it stood at 26f380a. Anchored on the # real "start" section (the restart call itself), the same anchor test_swap_ordered_after_wait_and_ # before_start above already uses for "before the start section". Needle built from two concatenated # pieces, the same trick test_no_unguarded_macos_only_hasher_calls uses above, so this check's own # description of the pattern it looks for can never become a match for itself. test_no_unguarded_mktemp_assignments_after_restart() { local src="$ROOT/scripts/redeploy-fleetd.sh" start_line needle bad start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$start_line" ] || fail "could not find the restart call's start section in redeploy-fleetd.sh" needle='^[[:space:]]*[A-Za-z_][A-Za-z0-9_]*="' needle="$needle"'\$\(mktemp' bad="$(grep -nE "$needle" "$src" | awk -F: -v start="$start_line" '$1+0>start')" [ -z "$bad" ] \ || fail "unguarded 'VAR=\"\$(mktemp ...)\"' assignment(s) after the restart call (fleetd #552): $bad" } # fleetd #552 — capture_fresh_log_region itself. Sourcing stops before the main flow can be driven # directly (the SOURCED guard below), so this exercises the extracted function the real call site # now uses, the same stub-mktemp-on-PATH technique # test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses above. test_capture_fresh_log_region_control() { printf 'line one\nline two\nline three\n' > "$TMP/fresh-log-src.out" local result result="$(capture_fresh_log_region "$TMP/fresh-log-src.out" 0)" [ -n "$result" ] || fail "capture_fresh_log_region returned nothing on a working mktemp" [ -f "$result" ] || fail "capture_fresh_log_region did not create the file it named" assert_equals "$(printf 'line one\nline two\nline three')" "$(cat "$result")" \ "captured region did not carry the whole source file (restart_mark=0)" rm -f "$result" } test_capture_fresh_log_region_mktemp_failure_returns_nonzero_and_prints_nothing() { local bin_dir rc=0 out bin_dir="$TMP/stub-bin-fresh-log-mktemp-fails"; mkdir -p "$bin_dir" cat > "$bin_dir/mktemp" <<'STUB' #!/usr/bin/env bash echo "mktemp: cannot create temp file" >&2 exit 1 STUB chmod +x "$bin_dir/mktemp" printf 'line one\n' > "$TMP/fresh-log-src2.out" out="$(PATH="$bin_dir:$PATH" capture_fresh_log_region "$TMP/fresh-log-src2.out" 0)" && rc=0 || rc=$? [ "$rc" -ne 0 ] \ || fail "capture_fresh_log_region must return non-zero when its own mktemp fails" [ -z "$out" ] \ || fail "capture_fresh_log_region printed a path even though mktemp failed: $out" } # fleetd #552 acceptance: "the exit status after a stubbed mktemp failure must still report the # restart's own outcome". Sourcing stops before the main flow runs, so this drives the REAL # call-site shape (the `if ! FRESH_LOG="$(capture_fresh_log_region ...)"; then warn; fi` guard that # now stands where the old unguarded assignment did) in a subshell, under the SAME `set -euo # pipefail` the real script runs under, with `mktemp` stubbed to fail on PATH. A regression back to # the old unguarded `FRESH_LOG="$(mktemp ...)"` shape would abort this subshell before it reaches # SUBSHELL_REACHED_END, and this test would go red with a non-zero rc. test_fresh_log_capture_failure_does_not_abort_under_sete() { local bin_dir rc=0 out bin_dir="$TMP/stub-bin-call-site-mktemp-fails"; mkdir -p "$bin_dir" cat > "$bin_dir/mktemp" <<'STUB' #!/usr/bin/env bash echo "mktemp: cannot create temp file" >&2 exit 1 STUB chmod +x "$bin_dir/mktemp" printf 'line one\n' > "$TMP/fresh-log-callsite.out" out="$( PATH="$bin_dir:$PATH" set -euo pipefail FRESH_LOG="" trap 'rm -f "$FRESH_LOG"' EXIT if ! FRESH_LOG="$(capture_fresh_log_region "$TMP/fresh-log-callsite.out" 0)"; then echo "WARN skipped" fi echo "SUBSHELL_REACHED_END" )" && rc=0 || rc=$? [ "$rc" -eq 0 ] \ || fail "the guarded call-site shape aborted under set -e when mktemp failed (rc=$rc) — a successful restart must not be reported as a failure" printf '%s' "$out" | grep -qF 'SUBSHELL_REACHED_END' \ || fail "the subshell aborted before reaching the end — a post-restart mktemp failure must not abort the script" printf '%s' "$out" | grep -qF 'WARN skipped' \ || fail "the guarded call site did not report the post-restart check as skipped when mktemp failed" } # fleetd #552 item 4 (the fourth reader): the result section must check REDEPLOY_AMQP_CHECK_SKIPPED # and say so BEFORE it ever consults REDEPLOY_ERROR_COUNT — otherwise a skipped capture (which # leaves REDEPLOY_ERROR_COUNT at its untouched 0) falls through the same "no ERROR lines" gate a # genuinely clean restart uses, and (because REDEPLOY_DRAIN_STATE is "skipped", not "complete"/"n/a") # prints nothing at all: a silence indistinguishable from the clean bill of health this ticket exists # to prevent. Sourcing stops before the main flow runs, so this is a source-text ordering check, the # same shape as test_report_shutdown_drain_ordered_after_classify_and_before_result above. # fleetd #552 — closes the gap test_no_unguarded_mktemp_assignments_after_restart cannot: after # extracting the mktemp into capture_fresh_log_region, the SAME class of hazard (an unguarded # command-substitution assignment that can abort the script under set -e, because capture_fresh_ # log_region still returns non-zero on its own mktemp failure) would reappear if the main flow's # call to it ever lost its own "if !" guard — and that line no longer contains the literal text # "mktemp", so the sweep above cannot see it. Pin the real call site directly instead, the same # shape test_swap_ordered_after_wait_and_before_start above uses for its own call site. test_capture_fresh_log_region_call_site_is_guarded() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'if ! FRESH_LOG="$(capture_fresh_log_region "$OUT" "$RESTART_MARK")"; then' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's guarded capture_fresh_log_region call site in redeploy-fleetd.sh" } test_result_section_checks_amqp_skip_before_no_error_lines() { local src="$ROOT/scripts/redeploy-fleetd.sh" result_line skip_line ok_line result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)" skip_line="$(grep -Fn '"$REDEPLOY_AMQP_CHECK_SKIPPED" -eq 1' "$src" | tail -1 | cut -d: -f1 || true)" ok_line="$(grep -Fn 'ok "no ERROR lines since restart"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$result_line" ] || fail "could not find the result section in redeploy-fleetd.sh" [ -n "$skip_line" ] || fail "could not find a REDEPLOY_AMQP_CHECK_SKIPPED check in redeploy-fleetd.sh" [ -n "$ok_line" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh" [ "$skip_line" -gt "$result_line" ] \ || fail "the REDEPLOY_AMQP_CHECK_SKIPPED check (line $skip_line) is not inside the result section (starts line $result_line)" [ "$skip_line" -lt "$ok_line" ] \ || fail "the REDEPLOY_AMQP_CHECK_SKIPPED check (line $skip_line) is not checked before 'no ERROR lines since restart' (line $ok_line)" } # fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a # hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not # the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real # caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout # survives that boundary, so kind AND detail must both cross on it — this test proves they do. test_require_drivable_supervisor_refuses_unclear() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } systemd_loaded() { return 1; } local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0 # Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand. SUPERVISOR_RAW="$(detect_supervisor)" SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}" SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}" assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal" [ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \ || fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell" printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded" output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor" # Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate # recovery instructions name the unit unconditionally either way ("systemctl --user status # $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only # the specific detail string proves the crossed value, not the boilerplate, reached the message. printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \ || fail "refusal message does not contain the specific detail that crossed the subshell boundary" } # The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to # `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that # forks a subshell, so its exit() only ends the subshell and this test script keeps running under # `set -e`. test_require_drivable_supervisor_refuses_ambiguous() { local output rc=0 output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor" printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \ || fail "refusal message does not name the launchd label it found" printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \ || fail "refusal message does not name the systemd unit it found" } test_require_drivable_supervisor_accepts_known_kinds() { require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor" require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor" require_drivable_supervisor "none" || fail "refused the unsupervised case" } # fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor # produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see # it because either daemon alone satisfies them. test_count_daemon_pids() { assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list" assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid" assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids" } test_assert_single_daemon_accepts_one_pid() { assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid" } test_assert_single_daemon_rejects_two_pids() { local output rc=0 output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids" printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found" printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found" } # fleetd #593 instance 2 — `running_pid()` used to be a bare `pgrep -f "$PATTERN"`, which matches # ANY process whose full command line contains the pattern TEXT, including a shell that merely # embeds it as literal text rather than being the daemon. Measured live on this Mac: `pgrep -c` # (a one-call count) does not exist on BSD at all, and `bash -c ""` execs in place # so no parent shell survives to hold the pattern — which is exactly why the defect did not # reproduce from a plain script and needs a wrapper shaped like this instead. A `sh -c '...; ...'` # with MORE THAN ONE statement does not get that exec-in-place treatment: the shell forks a child # for the second statement and stays alive itself, holding the whole `-c` string — pattern text # included — in its own `ps -o args`, for as long as it runs. That is the same shape an # `ssh host "…; …"` wrapper or a hand-typed pipeline leaves behind. Before the fix this test would # have found the wrapper's pid in running_pid()'s output; it must not. test_running_pid_excludes_self_matching_wrapper_shell() { local before after wrapper_pid before="$(running_pid)" sh -c 'echo "target/fleetd.jar" >/dev/null; sleep 20' & wrapper_pid=$! sleep 0.3 after="$(running_pid)" kill "$wrapper_pid" 2>/dev/null || true wait "$wrapper_pid" 2>/dev/null || true [ "$after" = "$before" ] \ || fail "running_pid() counted a self-matching wrapper shell (pid $wrapper_pid, holding the pattern as literal text in its own argv, not the daemon): before=[$before] after=[$after]" } # fleetd #593 CORRECTION 1 — the round-1 version of this test gave its standin an argv[0] # containing the pattern text (via `exec -a`) and left `comm` as whatever that override produced, # which was never `java`. That was fine for a denylist-of-shells filter, but the allowlist below # now requires `comm = java` specifically, so the standin here must actually carry that comm, not # just avoid being a shell. `exec -a java` overrides argv[0] to `java` while the process itself # stays a genuine, harmless `sh`; combining it with the same non-exec'ing multi-statement shape # the wrapper-shell test above uses keeps the pattern text in the process's own `ps -o args` for # as long as it runs. Measured live on this Mac (BSD/macOS: `ps -o comm=` here reflects argv[0]): # `comm=java`, `args` contains the pattern, `pgrep -f "$PATTERN"` finds it. Copying a real system # binary into a scratch path and executing it from there was tried first, for a more literal # stand-in daemon, and the OS killed it outright (SIGKILL, exit 137 — almost certainly a # code-signing check on a relocated binary); `exec -a` needs no binary of its own and nothing # under a scratch directory, and it is the technique CORRECTION 1 names as the right one. # # This is the one live-process test in this file whose result could differ on Linux: Linux sets # `comm` from the actually-executed binary's own path, not from `exec -a`'s argv[0] override (BSD # ties `comm` to argv[0], which is what makes this technique work here) — so on Linux this # specific fixture might report `comm=sh`, not `comm=java`, even though the REAL daemon (a literal # `java -jar target/fleetd.jar` process, never fabricated) is unaffected either way. I could not # verify this fixture's behavior on Linux, so test_running_pid_counts_a_pid_whose_comm_is_java # below backstops the same claim (the allowlist admits a pid whose comm is `java`) with a stubbed # `ps`, which is identical bash on every platform and carries no such platform question. test_running_pid_finds_a_real_java_named_second_process() { local before after standin_pid before="$(running_pid)" ( exec -a java sh -c 'echo "target/fleetd.jar" >/dev/null; sleep 20' ) & standin_pid=$! sleep 0.3 after="$(running_pid)" kill "$standin_pid" 2>/dev/null || true wait "$standin_pid" 2>/dev/null || true printf '%s\n' "$after" | grep -qxF "$standin_pid" \ || fail "running_pid() did not find a real second process (pid $standin_pid, comm forced to 'java' via exec -a) whose own argv holds the pattern: before=[$before] after=[$after]" } # fleetd #593 CORRECTION 1, hole 2 — the round-1 filter denied known shell names (sh/bash/zsh/ # dash/ksh) and counted everything else. `ssh`, `perl`, `python3`, `ruby`, `tail` — anything not on # that list, carrying the pattern in its own argv — was still counted right alongside the real # daemon, and the ticket names `ssh` as a live route. Stubbing `pgrep`/`ps` (rather than spawning a # real perl/ssh process) pins the exact discriminator this correction is about — comm, not the # caller's shape — deterministically on every platform, with no dependency on perl/python3/ruby # being installed in whatever environment runs this suite, and no dependency on how a given OS # derives `comm` for a fabricated process (see the comment above # test_running_pid_finds_a_real_java_named_second_process for why that matters here). test_running_pid_drops_a_pid_whose_comm_is_not_java() { pgrep() { printf '4242\n'; } ps() { printf 'perl\n'; } local found found="$(running_pid)" unset -f pgrep ps [ -z "$found" ] \ || fail "running_pid() counted pid 4242 whose comm is 'perl', not 'java' — denying known shell names does not exclude a non-shell wrapper such as ssh or perl (fleetd #593 CORRECTION 1): found=[$found]" } # fleetd #593 CORRECTION 1, hole 1 — pgrep can list a pid that exits before the following # `ps -o comm=` lookup runs; on a gone pid `ps` prints nothing, so `comm` comes back empty. Under # the round-1 denylist an empty string matched none of the denied shell names, so the dead pid was # still counted — the exact false-positive shape the ticket exists to remove, just rarer. The # allowlist fixes this for free: an empty comm is not `java` either. test_running_pid_drops_a_pid_that_exited_before_the_comm_lookup() { pgrep() { printf '4242\n'; } ps() { :; } # a pid that no longer exists: the real `ps -p ` prints nothing and this mirrors that local found found="$(running_pid)" unset -f pgrep ps [ -z "$found" ] \ || fail "running_pid() counted pid 4242 whose comm lookup came back empty (the pid had already exited before the lookup ran) — an empty comm must not pass the allowlist (fleetd #593 CORRECTION 1): found=[$found]" } # The positive backstop for both stubbed tests above, and for # test_running_pid_finds_a_real_java_named_second_process on whatever platform that live fixture # does not itself carry comm=java: the allowlist must still ADMIT the one comm value the real # daemon actually has. Measured on the real, currently-running daemon on this Mac: `comm=java`. test_running_pid_counts_a_pid_whose_comm_is_java() { pgrep() { printf '4242\n'; } ps() { printf 'java\n'; } local found found="$(running_pid)" unset -f pgrep ps printf '%s\n' "$found" | grep -qxF '4242' \ || fail "running_pid() did not count pid 4242 whose comm is 'java' — the daemon's own name must pass the allowlist: found=[$found]" } # fleetd #593 instance 3 — assert_single_daemon's refusal message used to tell the operator to # "Investigate with 'pgrep -f \"\$PATTERN\"'", which — typed by hand or over ssh — is precisely the # self-matching invocation instance 2 above fixes. A source-text check, the same technique # test_no_error_lines_message_gated_by_drain_state uses: this is prose inside a die() call, never # reached by sourcing (the SOURCED guard stops before the main flow, and this text only prints # from inside a call assert_single_daemon makes when it is already refusing). test_die_message_does_not_recommend_bare_pgrep_as_remediation() { local src="$ROOT/scripts/redeploy-fleetd.sh" block bad block="$(grep -A6 -F 'racing supervisor produces' "$src" || true)" [ -n "$block" ] || fail "could not find the assert_single_daemon refusal message in redeploy-fleetd.sh" bad="$(printf '%s' "$block" | grep -F "Investigate with 'pgrep -f" || true)" [ -z "$bad" ] \ || fail "assert_single_daemon's die message still hands the operator a bare 'pgrep -f \"\$PATTERN\"' as remediation (fleetd #593) — that is exactly the self-matching invocation" printf '%s' "$block" | grep -qF 'fleetd #593' \ || fail "assert_single_daemon's die message does not say in words that a pattern can match the caller (fleetd #593)" } # fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports # $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call # must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR. # Two files with different content, so a default pointed at the wrong one reports the wrong hash # rather than accidentally matching. test_jar_id_defaults_to_live_and_reports_explicit_path() { local dir saved_jar="$JAR" saved_staged="$JAR_STAGED" local live_hash staged_hash default_result explicit_result dir="$TMP/jar-id"; mkdir -p "$dir" JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar" printf 'live jar bytes' > "$JAR" printf 'staged jar bytes, not the same content' > "$JAR_STAGED" # fleetd #550: this reference hash must be computed the same portable way jar_id() itself now # computes one — a bare, unguarded call to the macOS-only hasher here was exactly the item-2 # defect, dying with "command not found" on any Linux runner that has no such hasher at all. live_hash="$(hash256 "$JAR")" staged_hash="$(hash256 "$JAR_STAGED")" default_result="$(jar_id)" explicit_result="$(jar_id "$JAR_STAGED")" JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same" assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR" assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR" } # fleetd #550 — closes a gap the test above leaves open. That test's own reference hash is now ALSO # computed by calling hash256 (needed for item 2: the old bare macOS-only-hasher call there was the # Linux crash), so its subject (jar_id, via hash256) and its reference (also hash256) share one # instrument — they agree no matter which algorithm hash256 actually runs, so a mutation that swaps # BOTH of hash256's arms for the wrong algorithm is invisible to it. This test's expected value # comes from neither hasher: it is the published SHA-256 test vector for the 3-byte input "abc" # (no trailing newline), written here as a literal constant, so it can still tell "hashed # correctly" from "hashed, just with the wrong algorithm" — which is what this whole ticket is # about. test_hash256_computes_a_real_sha256() { local dir f result dir="$TMP/hash256-known-vector"; mkdir -p "$dir" f="$dir/abc.txt" printf 'abc' > "$f" result="$(hash256 "$f")" # SHA-256("abc") = ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad, the standard # FIPS 180 test vector — first 12 hex chars, matching hash256's own `cut -c1-12`. assert_equals "ba7816bf8f01" "$result" "hash256 of the literal 3-byte input 'abc' must be the known SHA-256 prefix, not some other algorithm's" } # fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511) # proves both halves of the present-file contract but never exercises the missing-file path. This # word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean` # deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points # operators at `--check` for exactly this. Covers both the no-argument default and an explicit path, # since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both. test_jar_id_reports_absent_for_missing_file() { local saved_jar="$JAR" dir default_result explicit_result dir="$TMP/jar-id-absent"; mkdir -p "$dir" JAR="$dir/does-not-exist.jar" [ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR" default_result="$(jar_id)" explicit_result="$(jar_id "$dir/also-does-not-exist.jar")" JAR="$saved_jar" assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist" assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent" } # fleetd #550 — the whole point of this ticket: a jar that IS there but could not be hashed must # never read the same as a jar that is not there at all. Drives the REAL hash256/jar_id bodies # (never stubbed) through a stub PATH that contains neither of the two hashers this script knows — # same technique test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses for `mktemp`, # except here the stub directory is used to REPLACE PATH rather than prepend to it, because the # point is to make BOTH hashers unreachable, not to intercept one specific command while leaving # everything else on the real PATH reachable. `[ -f ... ]` and the shell's own `command`/`echo` # builtins need no PATH at all, so this is safe even with PATH reduced to an empty directory. test_jar_id_reports_unhashable_when_no_hasher_on_path() { local bin_dir dir saved_jar="$JAR" default_result explicit_result bin_dir="$TMP/stub-bin-no-hasher"; mkdir -p "$bin_dir" dir="$TMP/jar-id-no-hasher"; mkdir -p "$dir" JAR="$dir/fleetd.jar" printf 'a real jar that exists but nothing here can hash' > "$JAR" [ -f "$JAR" ] || fail "test fixture error: \$JAR does not exist at $JAR" default_result="$(PATH="$bin_dir" jar_id)" explicit_result="$(PATH="$bin_dir" jar_id "$JAR")" JAR="$saved_jar" [ "$default_result" != "absent" ] \ || fail "jar_id reported absent for a file that exists, only because no hasher was on PATH" [ "$explicit_result" != "absent" ] \ || fail "jar_id (explicit path) reported absent for a file that exists, only because no hasher was on PATH" # A 12-char hex hash is the OTHER wrong answer here: with no hasher at all, nothing could have # produced one, so a value that merely happens to look like one would mean the stub failed to # hide the real hashers rather than that jar_id degraded correctly. printf '%s' "$default_result" | grep -Eq '^[0-9a-f]{12}$' \ && fail "test fixture error: PATH stub did not actually hide the real hasher(s) — got what looks like a real hash" assert_equals "unhashable" "$default_result" "jar_id with no hasher on PATH must report a third, distinct state — never absent, never a hash" assert_equals "unhashable" "$explicit_result" "jar_id (explicit path) with no hasher on PATH must report the same third state" } # fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar # are exercised directly against real files on disk (not stubs), because the whole point is file # behavior (does the content move, does the source disappear, does a failure leave both sides # intact) that a stubbed function cannot prove. test_stage_built_jar_moves_off_live_path() { local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED" dir="$TMP/stage-ok"; mkdir -p "$dir" jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar" printf 'built jar bytes' > "$jar" JAR="$jar"; JAR_STAGED="$staged" stage_built_jar || fail "stage_built_jar rejected a real build output" JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar" [ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged" grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content" } test_stage_built_jar_dies_when_build_produced_nothing() { local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED" dir="$TMP/stage-missing"; mkdir -p "$dir" JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar" output="$(stage_built_jar 2>&1)" || rc=$? JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output" printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \ || fail "refusal message does not name the missing jar path" } test_swap_staged_jar_moves_staged_onto_live() { local dir staged live dir="$TMP/swap-ok"; mkdir -p "$dir" staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar" printf 'swapped jar bytes' > "$staged" swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar" [ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged" [ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live" grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content" } # The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on # failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes # inside a command substitution to contain that exit to a subshell. test_swap_staged_jar_dies_without_staged_file() { local dir output rc=0 dir="$TMP/swap-missing"; mkdir -p "$dir" output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar" [ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged" printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \ || fail "refusal message does not name the missing staged path" } test_swap_staged_jar_dies_when_mv_fails() { local dir staged live output rc=0 dir="$TMP/swap-fail"; mkdir -p "$dir/src" staged="$dir/src/fleetd-new.jar" printf 'fake jar bytes' > "$staged" live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv" [ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails" [ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed" printf '%s' "$output" | grep -qF "$staged" \ || fail "refusal message does not name the staged path that could not be moved" } # --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this # path) and must still die with the exact wording documented in the script's own header comment. test_require_no_build_jar_dies_when_absent() { local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar" JAR="$missing" output="$(require_no_build_jar 2>&1)" || rc=$? JAR="$saved_jar" [ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar" printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \ || fail "refusal message does not match the documented --no-build wording" } test_require_no_build_jar_accepts_present_jar() { local saved_jar="$JAR" dir dir="$TMP/no-build-present"; mkdir -p "$dir" JAR="$dir/fleetd.jar" printf 'existing jar' > "$JAR" require_no_build_jar || fail "require_no_build_jar rejected an existing jar" JAR="$saved_jar" } # wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while # running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so # the timeout-loop test does not actually wait out its budget. test_wait_for_daemon_exit_returns_true_once_pid_clears() { # running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so # a plain shell variable it increments would reset on each call instead of accumulating. Count in # a file instead, which is the one thing that actually survives across those subshells. local counter_file="$TMP/wait-exit-calls" final_calls printf '0' > "$counter_file" running_pid() { local n n="$(cat "$counter_file")" n=$((n + 1)) printf '%s' "$n" > "$counter_file" if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi } sleep() { :; } wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared" final_calls="$(cat "$counter_file")" [ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } test_wait_for_daemon_exit_times_out_if_pid_never_clears() { local rc=0 running_pid() { printf '4242'; } sleep() { :; } wait_for_daemon_exit 3 || rc=$? [ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } # fleetd #603 — wait_for_new_pid is the dual of wait_for_daemon_exit above: it must not report # success while running_pid() still answers empty, and must report success the moment a pid # appears. Same counter-file idiom as test_wait_for_daemon_exit_returns_true_once_pid_clears above, # for the same reason (running_pid() runs inside a `$(...)` subshell on every call). test_wait_for_new_pid_returns_true_once_pid_appears() { local counter_file="$TMP/wait-new-pid-calls" final_calls printf '0' > "$counter_file" running_pid() { local n n="$(cat "$counter_file")" n=$((n + 1)) printf '%s' "$n" > "$counter_file" if [ "$n" -lt 3 ]; then printf ''; else printf '4242'; fi } sleep() { :; } wait_for_new_pid 10 || fail "wait_for_new_pid did not report success once the pid appeared" final_calls="$(cat "$counter_file")" [ "$final_calls" -ge 3 ] || fail "wait_for_new_pid returned before actually re-checking running_pid" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } test_wait_for_new_pid_times_out_if_pid_never_appears() { local rc=0 running_pid() { printf ''; } sleep() { :; } wait_for_new_pid 3 || rc=$? [ "$rc" -ne 0 ] || fail "wait_for_new_pid reported success while the pid never appeared" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } # fleetd #603 — await_daemon_started folds the pid-appeared check and the healthz check into one # decision (see its own comment in redeploy-fleetd.sh for why a source-text grep cannot tell the two # possible behaviors apart here). These two tests are the ticket's own acceptance criteria, run # together in this one suite invocation so neither can be satisfied by code that never fails at all: # # 1. a slow start must still succeed — running_pid mimics a process that does not appear until # well after the OLD, buggy 10-second budget, but does appear, and healthz answers. # 2. a genuine failure must still fail, and still print the log tail — running_pid and the health # check both report nothing at all, ever. test_await_daemon_started_slow_pid_then_healthy_succeeds() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder local counter_file="$TMP/await-slow-pid-calls" output printf '0' > "$counter_file" running_pid() { local n n="$(cat "$counter_file")" n=$((n + 1)) printf '%s' "$n" > "$counter_file" # Stays empty well past the old 10-second budget, then appears — the exact shape #603 reports. if [ "$n" -lt 12 ]; then printf ''; else printf '4242'; fi } sleep() { :; } poll_health_body() { printf '{"status":"ok"}'; return 0; } # NOT `output="$(await_daemon_started ...)"`: that would run the call in a subshell, and # NEW_PID — a plain global assignment inside the function, by design (see its own comment) — would # die with that subshell instead of reaching this test's own shell. Redirect to a file instead, the # same hazard await_daemon_started's own comment warns die() itself is subject to. await_daemon_started 60 "" "http://ignored/healthz" "$TMP/await-slow-pid.out" \ > "$TMP/await-slow-pid-output.log" 2>&1 output="$(cat "$TMP/await-slow-pid-output.log")" [ "$DIED_CALLED" = 0 ] \ || fail "await_daemon_started must not die on a slow-but-real start: $DIED_MESSAGE" assert_equals "4242" "$NEW_PID" "await_daemon_started NEW_PID after a slow-but-real start" printf '%s' "$output" | grep -qF 'started, pid 4242' \ || fail "await_daemon_started did not report the pid once it finally appeared" source "$ROOT/scripts/redeploy-fleetd.sh" } test_await_daemon_started_never_appears_dies_with_log_tail() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder local out_file="$TMP/await-never-appears.out" printf 'boot line one\nboot line two\n' > "$out_file" running_pid() { printf ''; } sleep() { :; } poll_health_body() { return 1; } await_daemon_started 2 "" "http://127.0.0.1:1/healthz" "$out_file" > /dev/null 2>&1 [ "$DIED_CALLED" = 1 ] \ || fail "await_daemon_started must die when the daemon never appears and never becomes healthy" printf '%s' "$DIED_MESSAGE" | grep -qF 'never answered' \ || fail "await_daemon_started die message does not say healthz never answered" printf '%s' "$DIED_MESSAGE" | grep -qF 'boot line two' \ || fail "await_daemon_started die message does not include the log tail" source "$ROOT/scripts/redeploy-fleetd.sh" } # fleetd #603 review — the path the fall-through actually exists for, and the one gap a lead # mutation found in the first version of this test file: running_pid() NEVER finds anything (as its # own doc comment says it eventually will, once the daemon stops being launched as a plain # `java -jar` its allowlist recognises), while /healthz answers anyway. Neither of the two tests # above drives this: the slow-pid test has the pid appear, so the `else` branch never runs, and the # never-appears test fails BOTH checks, so it dies either way and cannot tell which branch fired. # This must not die, must warn (so the operator is told the pid could not be identified), and must # leave NEW_PID empty — the honest "could not establish this" answer, never a guessed pid, which is # what the final result line's `${NEW_PID:-unknown}` fallback exists to print truthfully. # # Proof this actually pins the behavior, not just the source text (paste from a real run, not # claimed): reverting the `warn` below back to `die "no process appeared"` (the old fleetd #603 # defect, reintroduced) turns this one test red — # FAIL: await_daemon_started must not die when the pid is never found but healthz answers # — and restoring `warn` turns the whole suite green again. Both halves observed, not asserted. test_await_daemon_started_pid_never_found_but_healthy_warns_and_survives() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder local out_file="$TMP/await-pid-never-found.out" output printf 'boot line\n' > "$out_file" running_pid() { printf ''; } sleep() { :; } poll_health_body() { printf '{"status":"ok"}'; return 0; } # NOT `output="$(await_daemon_started ...)"` — see the slow-pid test above for why that would # drop NEW_PID's assignment in a subshell instead of reaching this test's own shell. await_daemon_started 3 "" "http://ignored/healthz" "$out_file" \ > "$TMP/await-pid-never-found-output.log" 2>&1 output="$(cat "$TMP/await-pid-never-found-output.log")" [ "$DIED_CALLED" = 0 ] \ || fail "await_daemon_started must not die when the pid is never found but healthz answers: $DIED_MESSAGE" printf '%s' "$output" | grep -qF 'falling through to the health check' \ || fail "await_daemon_started did not warn that the pid could not be identified" assert_equals "" "$NEW_PID" \ "await_daemon_started NEW_PID when the pid is never found but healthz answers — must stay empty, never a guessed pid" source "$ROOT/scripts/redeploy-fleetd.sh" } test_await_daemon_started_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'await_daemon_started "$HEALTH_WAIT" "$OLD_PID" "$HEALTH" "$OUT"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's await_daemon_started call site in redeploy-fleetd.sh" } # fleetd #521 — the swap step's guard, at two levels. # # The first two tests call the predicate should_swap() directly. They pin its logic, and that is all # they pin. On their own they did NOT close #521, and this was measured rather than argued: with the # main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left # this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs # the swap consult the predicate at all. Extracting the decision had moved the untested decision up # a level, not removed it. # # So the last two tests call swap_if_built() — the function the main flow actually calls, holding the # guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the # guard is removed, inverted, or stops being consulted. # # What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow # altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing # stops before the main flow runs, so no test in this file can invoke it. test_should_swap_true_when_build_ran() { should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true" } test_should_swap_false_when_build_skipped() { if should_swap 0; then fail "should_swap 0 (--no-build; nothing was staged this run) must return false" fi } # Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is # global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id # overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear), # and again at the END, so their own stubs do not leak into every test that runs after them. test_swap_if_built_performs_the_swap_when_build_ran() { source "$ROOT/scripts/redeploy-fleetd.sh" local marker="$TMP/swap-if-built-ran" rm -f "$marker" swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; } jar_id() { printf 'stubbed\n'; } swap_if_built 1 > /dev/null [ -f "$marker" ] \ || fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not" source "$ROOT/scripts/redeploy-fleetd.sh" } test_swap_if_built_skips_the_swap_when_build_skipped() { source "$ROOT/scripts/redeploy-fleetd.sh" local marker="$TMP/swap-if-built-skipped" rm -f "$marker" swap_staged_jar() { printf 'swapped\n' > "$marker"; } jar_id() { printf 'stubbed\n'; } swap_if_built 0 > /dev/null if [ -f "$marker" ]; then fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did" fi source "$ROOT/scripts/redeploy-fleetd.sh" } # fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the # main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee # itself — as opposed to the pure functions it's built from — can only be checked by reading the # script's own call sites, the same way test_recovery_patterns_match_source below checks Java # source shape instead of behavior it cannot invoke directly. # # Two details about the three greps below, both of which have already gone wrong here. # # The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not # `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside # swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using # it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)" # — a true statement about a function definition, and nothing at all about the order of the steps. # # Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the # pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set # -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to # report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no # name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak. test_swap_ordered_after_wait_and_before_start() { local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)" swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)" start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh" [ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh" [ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh" [ "$swap_line" -gt "$wait_line" ] \ || fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)" [ "$swap_line" -lt "$start_line" ] \ || fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)" } # fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator # declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)" # to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has # already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal # ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start # above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow), # so the only way to pin its exact wording is to read the source. test_drain_gate_abort_message_says_no_no_build() { local src="$ROOT/scripts/redeploy-fleetd.sh" msg msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")" [ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh" if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then fail "abort message still claims a rerun WITH --no-build can finish the restart" fi printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \ || fail "abort message does not tell the operator to rerun without --no-build" printf '%s' "$msg" | grep -qF 'no longer at the live path' \ || fail "abort message does not say why --no-build cannot finish the restart" } # fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was # a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's # own file for the wording, which stays in the file even if the `if` guarding it is mutated to # `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead, # so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT # alongside these, not replaced, because it catches a different regression (a re-wording that still # reaches the right branch would not change which case fires here, but would still be worth pinning). test_drain_gate_refusal_build_ran_staged_present() { local dir staged result dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'staged jar bytes' > "$staged" result="$(drain_gate_refusal 1 "$staged")" printf '%s' "$result" | grep -qF "$staged" \ || fail "build-ran+staged-present refusal does not name the staged jar path" printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \ || fail "build-ran+staged-present refusal does not tell the operator how to finish the restart" if printf '%s' "$result" | grep -qF 'nothing changed'; then fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved" fi } test_drain_gate_refusal_build_ran_staged_absent() { local dir result dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir" result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")" assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording" } # --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar # found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the # comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning. test_drain_gate_refusal_no_build_staged_present() { local dir staged result dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'leftover staged jar bytes' > "$staged" result="$(drain_gate_refusal 0 "$staged")" assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed" } test_drain_gate_refusal_no_build_staged_absent() { local dir result dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir" result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")" assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording" } # fleetd #528 — the four tests above pin drain_gate_refusal(), and that is ALL they pin: they call # the predicate directly and never touch the main flow's call site. That was measured to be not # enough, the same way test_should_swap_true_when_build_ran/test_should_swap_false_when_build_skipped # were not enough for #521: with the main flow reading `die "$(drain_gate_refusal "$DO_BUILD" # "$JAR_STAGED")"`, replacing that whole line with a flat `die "aborted — nothing changed"` left this # suite at exit 0 with zero FAIL lines and byte-identical output to a clean run. Nothing above could # tell the difference, because none of it calls anything at or above the call site itself. # # So these four call refuse_drain_gate() — the function the main flow actually calls, holding the # composed message and the die() together — with die() stubbed to RECORD whether it was called and # with what message, instead of exiting the process. That fails if refuse_drain_gate stops consulting # drain_gate_refusal, mangles what it passes it, or simply never calls die. # # What none of these four can catch: deleting the `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` line # from the main flow altogether — see the comment above refuse_drain_gate in redeploy-fleetd.sh for # why no test in this file can do better than that (sourcing stops before the main flow runs). DIED_CALLED=0 DIED_MESSAGE="" stub_die_recorder() { DIED_CALLED=0 DIED_MESSAGE="" die() { DIED_CALLED=1; DIED_MESSAGE="$*"; } } test_refuse_drain_gate_build_ran_staged_present() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir staged dir="$TMP/refuse-drain-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'staged jar bytes' > "$staged" stub_die_recorder refuse_drain_gate 1 "$staged" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate build-ran+staged-present must call die, and did not" printf '%s' "$DIED_MESSAGE" | grep -qF "$staged" \ || fail "refuse_drain_gate build-ran+staged-present die message does not name the staged jar" printf '%s' "$DIED_MESSAGE" | grep -qF 'Rerun WITHOUT --no-build' \ || fail "refuse_drain_gate build-ran+staged-present die message is missing the rerun instruction" if printf '%s' "$DIED_MESSAGE" | grep -qF 'nothing changed'; then fail "refuse_drain_gate build-ran+staged-present must not claim nothing changed — the jar already moved" fi source "$ROOT/scripts/redeploy-fleetd.sh" } test_refuse_drain_gate_build_ran_staged_absent() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir dir="$TMP/refuse-drain-build-no-staged"; mkdir -p "$dir" stub_die_recorder refuse_drain_gate 1 "$dir/fleetd-new.jar" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate build-ran+staged-absent must call die, and did not" assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate build-ran+staged-absent die message" source "$ROOT/scripts/redeploy-fleetd.sh" } test_refuse_drain_gate_no_build_staged_present() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir staged dir="$TMP/refuse-drain-no-build-staged"; mkdir -p "$dir" staged="$dir/fleetd-new.jar" printf 'leftover staged jar bytes' > "$staged" stub_die_recorder refuse_drain_gate 0 "$staged" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate no-build+staged-present must call die, and did not" assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-present die message" source "$ROOT/scripts/redeploy-fleetd.sh" } test_refuse_drain_gate_no_build_staged_absent() { source "$ROOT/scripts/redeploy-fleetd.sh" local dir dir="$TMP/refuse-drain-no-build-no-staged"; mkdir -p "$dir" stub_die_recorder refuse_drain_gate 0 "$dir/fleetd-new.jar" [ "$DIED_CALLED" = 1 ] \ || fail "refuse_drain_gate no-build+staged-absent must call die, and did not" assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-absent die message" source "$ROOT/scripts/redeploy-fleetd.sh" } # fleetd #528 — closes the one gap the four behavioural tests above cannot: they call # refuse_drain_gate directly, and sourcing stops before the main flow ever runs (the SOURCED guard), # so none of them can prove the main flow still CALLS refuse_drain_gate at all. Same shape as # test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site. This is # what actually kills the item-1 mutation from the ticket — replacing the main flow's call with a # flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural # tests above would even notice. # # fleetd #555 — the main flow's own call site moved: it used to read # `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` directly; it now reads # `run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"`, and run_drain_gate (tested # directly below by test_run_drain_gate_*) is what calls refuse_drain_gate with its own local names. # This grep now pins THAT call site — the thing that would go missing if a future edit deleted the # main flow's call to run_drain_gate altogether, the same residual gap #521/#528 already accepted for # swap_if_built/refuse_drain_gate (sourcing stops before the main flow runs, so no test in this file # can do better than reading the source for this one specific gap). # # The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail # the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below # ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body). test_run_drain_gate_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'run_drain_gate "$OLD_PID" "$ASSUME_YES" "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's run_drain_gate call site in redeploy-fleetd.sh" } # fleetd #555 item 1 — `--check` must stay genuinely read-only: should_stop_for_check is the # predicate, stop_if_check_only is the only thing the main flow calls. test_should_stop_for_check_true_when_check_only() { should_stop_for_check 1 || fail "should_stop_for_check must be true when CHECK_ONLY=1" } test_should_stop_for_check_false_when_not_check_only() { should_stop_for_check 0 && fail "should_stop_for_check must be false when CHECK_ONLY=0" return 0 } # Captured via $(...): stop_if_check_only's own exit only ends this subshell (same reason the # unload/stop "tolerates clean negative" tests above capture this way), so this test's own message # is what reaches the report if the exit ever stops happening. test_stop_if_check_only_exits_zero_and_says_nothing_changed() { local output rc=0 output="$(stop_if_check_only 1)" || rc=$? [ "$rc" -eq 0 ] || fail "stop_if_check_only 1 must exit 0, got $rc" printf '%s' "$output" | grep -qF -- '--check: nothing changed' \ || fail "stop_if_check_only 1 did not print the --check message" } test_stop_if_check_only_returns_without_exiting_when_not_check_only() { local output output="$(stop_if_check_only 0; echo REACHED)" printf '%s' "$output" | grep -qF 'REACHED' \ || fail "stop_if_check_only 0 must return, not exit — the caller's next line never ran" } test_stop_if_check_only_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'stop_if_check_only "$CHECK_ONLY"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's stop_if_check_only call site in redeploy-fleetd.sh" } # fleetd #555 items 2 and 3 — the drain-gate entry (`-n "$OLD_PID" && "$ASSUME_YES" = 0`) and the # reply comparison (`"$reply" != "yes"`). test_drain_gate_required_true_when_pid_and_not_assume_yes() { drain_gate_required "123" 0 || fail "drain_gate_required must be true with an old pid and ASSUME_YES=0" } test_drain_gate_required_false_when_no_pid() { drain_gate_required "" 0 && fail "drain_gate_required must be false with no old pid" return 0 } test_drain_gate_required_false_when_assume_yes() { drain_gate_required "123" 1 && fail "drain_gate_required must be false when ASSUME_YES=1" return 0 } test_drain_confirmed_true_on_yes() { drain_confirmed "yes" || fail "drain_confirmed must be true on exactly 'yes'" } test_drain_confirmed_false_on_anything_else() { drain_confirmed "no" && fail "drain_confirmed must be false on 'no'" drain_confirmed "" && fail "drain_confirmed must be false on empty input" return 0 } # run_drain_gate end-to-end, driving real stdin through `read` the same way the real prompt does. # Redirected from a FILE, never a pipe: `printf ... | run_drain_gate ...` would run the function as # the last stage of a pipeline, which bash runs in a subshell by default (no `lastpipe`), so any # global this function set (DIED_CALLED/DIED_MESSAGE below) would be lost the instant the pipe # closed. `< file` redirection on a plain function call carries no such subshell. test_run_drain_gate_skips_prompt_when_not_required() { local output output="$(run_drain_gate "" 0 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)" if printf '%s' "$output" | grep -qF 'Fleet drained?'; then fail "run_drain_gate prompted even though drain_gate_required should have been false (no old pid)" fi output="$(run_drain_gate "123" 1 1 "$TMP/nonexistent-staged.jar" < /dev/null 2>&1)" if printf '%s' "$output" | grep -qF 'Fleet drained?'; then fail "run_drain_gate prompted even though ASSUME_YES=1 should have skipped the gate" fi } test_run_drain_gate_confirmed_reply_does_not_refuse() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder printf 'yes\n' > "$TMP/drain-reply-yes.txt" run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-yes.txt" \ > "$TMP/drain-gate-yes-output" 2>&1 [ "$DIED_CALLED" = 0 ] \ || fail "run_drain_gate must not refuse when the reply confirms ('yes')" source "$ROOT/scripts/redeploy-fleetd.sh" } test_run_drain_gate_declined_reply_refuses() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder printf 'no\n' > "$TMP/drain-reply-no.txt" run_drain_gate "123" 0 1 "$TMP/nonexistent-staged.jar" < "$TMP/drain-reply-no.txt" \ > "$TMP/drain-gate-no-output" 2>&1 [ "$DIED_CALLED" = 1 ] \ || fail "run_drain_gate must refuse (call die via refuse_drain_gate) when the reply does not confirm" printf '%s' "$DIED_MESSAGE" | grep -qF 'aborted' \ || fail "run_drain_gate's refusal message does not read as an abort: $DIED_MESSAGE" source "$ROOT/scripts/redeploy-fleetd.sh" } # fleetd #555 item 4 — the report-state dispatch on $SUPERVISOR_KIND. Inverting this used to report # the wrong supervisor and, for the launchd arm specifically, skip check_log_path_matches_plist. CHECK_LOG_PATH_CALLED=0 stub_check_log_path_recorder() { CHECK_LOG_PATH_CALLED=0 check_log_path_matches_plist() { CHECK_LOG_PATH_CALLED=1; } } test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_check_log_path_recorder SUPERVISED=0 report_supervisor_state "launchd" > "$TMP/report-supervisor-launchd-output" 2>&1 [ "$SUPERVISED" = 1 ] || fail "report_supervisor_state launchd must set SUPERVISED=1" [ "$CHECK_LOG_PATH_CALLED" = 1 ] \ || fail "report_supervisor_state launchd must call check_log_path_matches_plist" source "$ROOT/scripts/redeploy-fleetd.sh" } test_report_supervisor_state_systemd_sets_supervised_without_log_path_check() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_check_log_path_recorder SUPERVISED=0 report_supervisor_state "systemd" > "$TMP/report-supervisor-systemd-output" 2>&1 [ "$SUPERVISED" = 1 ] || fail "report_supervisor_state systemd must set SUPERVISED=1" [ "$CHECK_LOG_PATH_CALLED" = 0 ] \ || fail "report_supervisor_state systemd must NOT call check_log_path_matches_plist" source "$ROOT/scripts/redeploy-fleetd.sh" } test_report_supervisor_state_none_leaves_supervised_zero() { SUPERVISED=1 report_supervisor_state "none" > "$TMP/report-supervisor-none-output" 2>&1 [ "$SUPERVISED" = 0 ] || fail "report_supervisor_state none must leave SUPERVISED=0" grep -qF 'no supervisor loaded' "$TMP/report-supervisor-none-output" \ || fail "report_supervisor_state none did not print the unsupervised message" } test_report_supervisor_state_unrecognized_kind_warns_and_continues() { SUPERVISED=1 report_supervisor_state "bogus-kind" > "$TMP/report-supervisor-bogus-output" 2>&1 [ "$SUPERVISED" = 0 ] || fail "report_supervisor_state must reset SUPERVISED for an unrecognized kind" grep -qF "unrecognised supervisor kind: 'bogus-kind'" "$TMP/report-supervisor-bogus-output" \ || fail "report_supervisor_state did not warn about the unrecognized kind" } test_report_supervisor_state_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'report_supervisor_state "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's report_supervisor_state call site in redeploy-fleetd.sh" } # fleetd #555 item 5 — the stop dispatch on $SUPERVISOR_KIND. Inverting this kills a supervised # daemon with a raw kill instead of launchctl/systemctl, reviving the OLD jar (CB-594/#492). Each # mechanism is stubbed to record which one ran, so a swapped or dropped arm shows up directly. STOP_DISPATCH_CALLED="" stub_stop_dispatch_recorders() { STOP_DISPATCH_CALLED="" stop_via_launchd() { STOP_DISPATCH_CALLED="launchd"; } stop_via_systemd() { STOP_DISPATCH_CALLED="systemd"; } stop_via_kill() { STOP_DISPATCH_CALLED="kill:$1"; } } test_dispatch_stop_launchd_calls_stop_via_launchd() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_stop_dispatch_recorders dispatch_stop "launchd" "123" assert_equals "launchd" "$STOP_DISPATCH_CALLED" "dispatch_stop launchd" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_stop_systemd_calls_stop_via_systemd() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_stop_dispatch_recorders dispatch_stop "systemd" "123" assert_equals "systemd" "$STOP_DISPATCH_CALLED" "dispatch_stop systemd" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_stop_none_calls_stop_via_kill_with_pid() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_stop_dispatch_recorders dispatch_stop "none" "456" assert_equals "kill:456" "$STOP_DISPATCH_CALLED" "dispatch_stop none" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_stop_unrecognized_kind_dies() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_stop_dispatch_recorders stub_die_recorder dispatch_stop "bogus-kind" "789" [ "$DIED_CALLED" = 1 ] || fail "dispatch_stop must die on an unrecognized kind" [ -z "$STOP_DISPATCH_CALLED" ] \ || fail "dispatch_stop must not call any stop mechanism on an unrecognized kind" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_stop_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'dispatch_stop "$SUPERVISOR_KIND" "$OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's dispatch_stop call site in redeploy-fleetd.sh" } # fleetd #555 item 6 — the start dispatch on $SUPERVISOR_KIND. Same shape as the stop dispatch. START_DISPATCH_CALLED="" stub_start_dispatch_recorders() { START_DISPATCH_CALLED="" start_via_launchd() { START_DISPATCH_CALLED="launchd"; } start_via_systemd() { START_DISPATCH_CALLED="systemd"; } start_via_none() { START_DISPATCH_CALLED="none"; } } test_dispatch_start_launchd_calls_start_via_launchd() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_start_dispatch_recorders dispatch_start "launchd" assert_equals "launchd" "$START_DISPATCH_CALLED" "dispatch_start launchd" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_start_systemd_calls_start_via_systemd() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_start_dispatch_recorders dispatch_start "systemd" assert_equals "systemd" "$START_DISPATCH_CALLED" "dispatch_start systemd" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_start_none_calls_start_via_none() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_start_dispatch_recorders dispatch_start "none" assert_equals "none" "$START_DISPATCH_CALLED" "dispatch_start none" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_start_unrecognized_kind_dies() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_start_dispatch_recorders stub_die_recorder dispatch_start "bogus-kind" [ "$DIED_CALLED" = 1 ] || fail "dispatch_start must die on an unrecognized kind" [ -z "$START_DISPATCH_CALLED" ] \ || fail "dispatch_start must not call any start mechanism on an unrecognized kind" source "$ROOT/scripts/redeploy-fleetd.sh" } test_dispatch_start_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'dispatch_start "$SUPERVISOR_KIND"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's dispatch_start call site in redeploy-fleetd.sh" } # fleetd #555 item 7 — the HEALTH_BODY poll loop and the `[ -z "$HEALTH_BODY" ]` branch. Inverting # health_is_up makes a dead daemon report /healthz 200 or a live one report failure. test_health_is_up_true_with_body() { health_is_up "some body" || fail "health_is_up must be true with a non-empty body" } test_health_is_up_false_with_empty_body() { health_is_up "" && fail "health_is_up must be false with an empty body" return 0 } # fleetd #555: stub_die_recorder replaces die() globally for the rest of the process, the same as # every other user of it in this file — re-source before AND after so a later test (e.g. # test_unload_launchd_if_loaded_dies_on_real_failure, which needs the REAL die() to actually exit) # never inherits a stub left over from here. test_report_health_up_reports_ok_and_never_dies() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder local output output="$(report_health '{"status":"ok"}' "200" "$TMP/fake.out" 60)" [ "$DIED_CALLED" = 0 ] || fail "report_health must not die when the body is non-empty" printf '%s' "$output" | grep -qF '/healthz 200' \ || fail "report_health did not report the healthy body" source "$ROOT/scripts/redeploy-fleetd.sh" } test_report_health_degraded_503_warns_and_never_dies() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder local output output="$(report_health "" "503" "$TMP/fake.out" 60 2>&1)" [ "$DIED_CALLED" = 0 ] || fail "report_health must not die on a 503 (degraded, not dead)" printf '%s' "$output" | grep -qF '503 degraded' \ || fail "report_health did not report the 503-degraded case" source "$ROOT/scripts/redeploy-fleetd.sh" } test_report_health_dead_dies() { source "$ROOT/scripts/redeploy-fleetd.sh" stub_die_recorder report_health "" "000" "$TMP/fake.out" 60 > "$TMP/report-health-dead-output" 2>&1 [ "$DIED_CALLED" = 1 ] || fail "report_health must die when the body is empty and the code is not 503" printf '%s' "$DIED_MESSAGE" | grep -qF 'never answered' \ || fail "report_health die message does not say healthz never answered" source "$ROOT/scripts/redeploy-fleetd.sh" } test_poll_health_body_returns_nonzero_when_unreachable() { local rc=0 poll_health_body "http://127.0.0.1:1/healthz" 1 > /dev/null || rc=$? [ "$rc" -ne 0 ] || fail "poll_health_body must return non-zero when the health endpoint is unreachable" } test_report_health_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line # fleetd #603 — moved from a literal main-flow call into await_daemon_started (see its own # comment above for why); this now finds the call inside that function instead. call_line="$(grep -Fn 'report_health "$HEALTH_BODY" "$HEALTH_CODE" "$out_file" "$health_wait"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find await_daemon_started's report_health call site in redeploy-fleetd.sh" } # fleetd #555 item 8 — `HAD_OLD_PID=0; [ -n "$OLD_PID" ] && HAD_OLD_PID=1`. Measured safe under # `set -e` at both bash 3.2.57 and 5.x (the ticket's own header discussion); still an untested # computation feeding report_shutdown_drain's four-way decision until now. test_compute_had_old_pid_zero_when_empty() { assert_equals "0" "$(compute_had_old_pid "")" "compute_had_old_pid empty pid" } test_compute_had_old_pid_one_when_present() { assert_equals "1" "$(compute_had_old_pid "12345")" "compute_had_old_pid present pid" } test_compute_had_old_pid_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'HAD_OLD_PID="$(compute_had_old_pid "$OLD_PID")"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's compute_had_old_pid call site in redeploy-fleetd.sh" } # fleetd #555 — the structural guard the ticket actually asks for: "pick the shape, not the eight # sites." A future ninth bare main-flow conditional must fail THIS test on sight, not slip through # with every function-level test (67, now many more) still green. # # Scope: from the SOURCED guard onward — the ticket's own "main flow" (report state / build / drain # / stop / swap / start / verify). The `for arg in "$@"; do case "$arg" in ... esac; done` option # parser above the guard is out of scope, the same way it sits outside the ticket's own eight-item # table: it runs even when this file is sourced for tests, is not part of the mutating procedure, # and none of the eight decisions the ticket names live there. # # For every line whose first non-blank token is `if`, `elif`, or `case`, and which sits OUTSIDE any # function body, this demands an EXACT match — full trimmed line, never a regex anchor (a regex can # match inside unrelated text this same mutation could introduce, e.g. a comment) — against # MAIN_FLOW_ALLOWED_CONDITIONALS below. That allowlist is a closed set of the report-only/display # conditionals already in the file (bare state prints: is the daemon running, is X installed, does # this secret resolve — none of them gate a build/stop/start/swap decision) plus the two # loaded-but-not-running elif branches that are the SAME shape as fleetd #504 items 2/3/4, one # function over, and explicitly out of THIS ticket's scope per its own "Related" section. # # Every one of the eight decisions the ticket names has been lifted into its own predicate+action # function above this point in the file (should_stop_for_check, drain_gate_required/run_drain_gate, # report_supervisor_state, dispatch_stop, dispatch_start, health_is_up/report_health, # compute_had_old_pid) — none of their guards appear in this list any more, which is what makes # their specific inversions untestable-by-construction now: there is no bare guard left in the main # flow for a future edit to invert silently. A NEW bare conditional (the ninth) is, by construction, # not in this list, so this test goes red the instant one is added, naming the exact line — passing # requires either lifting the decision into a function (with its own predicate test, same shape as # above) or explicitly extending the allowlist, which is a diff a reviewer sees and can question. # # Function-body detection relies on this file's one consistent convention: a function opens as # `name() {` alone on its own line and closes as a bare `}` alone on its own line — verified: every # multi-line function in this file follows it, and the handful of one-liners like `say() { ...; }` # sit above the SOURCED guard, outside the region this scans. Indentation is NOT used to decide # "inside a function": a bare `if` inside a top-level `for`/`while` loop body (indented, but not # inside any function) would be exactly as reportable as one at column 0 — this file currently has # none, but a future one that hid inside a loop must not get a free pass for being indented. MAIN_FLOW_ALLOWED_CONDITIONALS=( 'if [ -n "$OLD_PID" ]; then' 'if launchd_installed; then' 'if systemd_installed; then' 'if zsh -lc '\''[ -n "${WORKER_GITEA_TOKEN:-}" ]'\'' 2>/dev/null; then' 'if zsh -lc '\''[ -n "${AI_GATEWAY_TOKEN:-}" ]'\'' 2>/dev/null; then' 'if [ -z "$BROKER_URI_ENV" ]; then' 'elif zsh -lc "[ -n \"\${$BROKER_URI_ENV:-}\" ]" 2>/dev/null; then' 'if [ "$DO_BUILD" = 1 ]; then' 'if ! mvn -f "$MODULE/pom.xml" clean install > "$BUILD_LOG" 2>&1; then' 'if ! wait_for_daemon_exit "$STOP_WAIT"; then' 'elif [ "$SUPERVISOR_KIND" = "launchd" ]; then' 'elif [ "$SUPERVISOR_KIND" = "systemd" ]; then' 'if tail -n "+$((RESTART_MARK + 1))" "$OUT" 2>/dev/null | grep -q '\''fleetd listening'\''; then' 'if ! FRESH_LOG="$(capture_fresh_log_region "$OUT" "$RESTART_MARK")"; then' 'if [ "$REDEPLOY_AMQP_CHECK_SKIPPED" -eq 1 ]; then' 'elif [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then' 'if [ "$REDEPLOY_DRAIN_STATE" = "complete" ] || [ "$REDEPLOY_DRAIN_STATE" = "n/a" ]; then' 'elif [ "$REDEPLOY_UNEXPLAINED_ERRORS" -eq 0 ]; then' ) # Emits "\tCOND\t" for every if/elif/case found outside a function, and # "\tFUNC\t" for every function DEFINITION found after guard_line — from guard_line # onward. A separate function (rather than inlined into the test) so the deliberate-mutation proof # in the PR description can call it directly against a scratch copy of the script. # # fleetd #555 rework (comment 17012): a function defined after the SOURCED guard can never be # reached by `source`-ing this script (sourcing returns before the main flow, and before any code # below the guard runs), so anything inside such a function is untestable by construction — the # depth tracker below would otherwise read it as "inside a function, therefore fine" and wave every # conditional in it through unexamined. Emitting FUNC records lets the caller flag the function # itself as the violation, independently of whether its body happens to contain a conditional. mainflow_bare_conditionals() { local src="$1" guard_line="$2" in_func=0 lineno=0 line trimmed while IFS= read -r line || [ -n "$line" ]; do lineno=$((lineno + 1)) [ "$lineno" -le "$guard_line" ] && continue if [[ "$line" =~ ^[A-Za-z_][A-Za-z0-9_]*\(\)[[:space:]]*\{[[:space:]]*$ ]]; then in_func=1 printf '%d\tFUNC\t%s\n' "$lineno" "$line" continue fi if [ "$in_func" = 1 ] && [[ "$line" =~ ^\}[[:space:]]*$ ]]; then in_func=0 continue fi if [ "$in_func" = 0 ]; then trimmed="$(printf '%s' "$line" | sed -E 's/^[[:space:]]+//')" if [[ "$trimmed" =~ ^(if|elif|case)[[:space:]] ]]; then printf '%d\tCOND\t%s\n' "$lineno" "$trimmed" fi fi done < "$src" } test_no_untested_main_flow_conditionals() { local src="$ROOT/scripts/redeploy-fleetd.sh" guard_line guard_line="$(grep -Fn 'if (return 0 2>/dev/null); then' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$guard_line" ] || { fail "could not find the SOURCED guard in redeploy-fleetd.sh"; return 1; } local violations=0 report="" found_line found_kind found_text allowed candidate while IFS=$'\t' read -r found_line found_kind found_text; do [ -n "$found_line" ] || continue if [ "$found_kind" = "FUNC" ]; then # fleetd #555 rework: a function defined after the SOURCED guard line can never be sourced # by this suite, so it can never be tested — that is a violation on its own, regardless of # what its body contains or whether the allowlist would otherwise excuse a bare conditional # inside it. violations=$((violations + 1)) report="$report line $found_line: function defined after the SOURCED guard (line $guard_line) — it cannot be sourced, so it cannot be tested: $found_text" continue fi allowed=0 for candidate in "${MAIN_FLOW_ALLOWED_CONDITIONALS[@]}"; do if [ "$found_text" = "$candidate" ]; then allowed=1 break fi done if [ "$allowed" = 0 ]; then violations=$((violations + 1)) report="$report line $found_line: $found_text" fi done < <(mainflow_bare_conditionals "$src" "$guard_line") if [ "$violations" -gt 0 ]; then fail "found $violations untested main-flow if/elif/case line(s) or function definition(s) after the SOURCED guard, not lifted into a tested predicate function and not in MAIN_FLOW_ALLOWED_CONDITIONALS:$report" fi } # fleetd #504 — the "loaded but not currently running" branches for launchd/systemd used to run # `launchctl unload`/`systemctl --user stop` with `2>/dev/null || true` and print `ok` # unconditionally, so a real supervisor failure (e.g. it cannot reach launchd/the systemd user bus) # read exactly like a harmless already-stopped answer. unload_launchd_if_loaded/ # stop_systemd_if_loaded (redeploy-fleetd.sh, right after systemd_loaded) apply systemd_loaded's own # "capture stderr separately — only a non-zero exit WITH stderr is a real failure" pattern to the # WRITE side. Both call the real `launchctl`/`systemctl` binaries directly (they are not overridable # wrapper functions the way launchd_loaded/systemd_loaded are), so these tests put a stub binary # first on PATH — the same technique test_detect_supervisor_systemd_probe_error_is_unclear above # already uses for `systemctl`. test_unload_launchd_if_loaded_dies_on_real_failure() { local bin_dir output rc=0 saved_plist="$LAUNCHD_PLIST" bin_dir="$TMP/stub-bin-launchctl-error"; mkdir -p "$bin_dir" cat > "$bin_dir/launchctl" <<'STUB' #!/usr/bin/env bash echo "Could not find specified service" >&2 exit 1 STUB chmod +x "$bin_dir/launchctl" LAUNCHD_PLIST="$TMP/fake-fail.plist" output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$? LAUNCHD_PLIST="$saved_plist" [ "$rc" -ne 0 ] \ || fail "unload_launchd_if_loaded must die when launchctl exits non-zero AND writes to stderr" printf '%s' "$output" | grep -qF 'launchctl unload' \ || fail "die message does not name the failing launchctl unload command" } # Captured via $(...) rather than called bare: unload_launchd_if_loaded's own die() does a hard # `exit`, and calling it directly at this level would let a regression that makes it die on this # clean-negative case kill the WHOLE suite before the `|| fail` below ever ran — printing die's own # message instead of this test's. Inside a command substitution, that `exit` only ends the subshell # (a-guard-is-defeated-by-its-calling-context: the same reason the *_dies_on_real_failure tests # above capture this way), so this test's own message is what actually reaches the report. test_unload_launchd_if_loaded_tolerates_clean_negative() { local bin_dir saved_plist="$LAUNCHD_PLIST" output rc=0 bin_dir="$TMP/stub-bin-launchctl-noop"; mkdir -p "$bin_dir" cat > "$bin_dir/launchctl" <<'STUB' #!/usr/bin/env bash exit 1 STUB chmod +x "$bin_dir/launchctl" LAUNCHD_PLIST="$TMP/fake-noop.plist" output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$? LAUNCHD_PLIST="$saved_plist" [ "$rc" -eq 0 ] \ || fail "unload_launchd_if_loaded must tolerate a clean already-unloaded answer (non-zero exit, empty stderr): $output" } test_stop_systemd_if_loaded_dies_on_real_failure() { local bin_dir output rc=0 bin_dir="$TMP/stub-bin-systemctl-stop-error"; mkdir -p "$bin_dir" cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash echo "Failed to connect to bus: No such file or directory" >&2 exit 1 STUB chmod +x "$bin_dir/systemctl" output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$? [ "$rc" -ne 0 ] \ || fail "stop_systemd_if_loaded must die when systemctl exits non-zero AND writes to stderr" printf '%s' "$output" | grep -qF 'systemctl --user stop' \ || fail "die message does not name the failing systemctl --user stop command" } # Same subshell-capture reasoning as test_unload_launchd_if_loaded_tolerates_clean_negative above: # stop_systemd_if_loaded's own die() does a hard `exit`, so this must run inside $(...) or a # regression here would kill the whole suite with die's message instead of this test's. test_stop_systemd_if_loaded_tolerates_clean_negative() { local bin_dir output rc=0 bin_dir="$TMP/stub-bin-systemctl-stop-noop"; mkdir -p "$bin_dir" cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash exit 1 STUB chmod +x "$bin_dir/systemctl" output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$? [ "$rc" -eq 0 ] \ || fail "stop_systemd_if_loaded must tolerate a clean already-stopped answer (non-zero exit, empty stderr): $output" } # Closes the same gap test_run_drain_gate_call_site_present closes for the drain gate: the four # tests above call unload_launchd_if_loaded/stop_systemd_if_loaded directly, and sourcing stops # before the main flow ever runs (the SOURCED guard), so none of them can prove the main flow still # CALLS these two functions instead of the original bare `2>/dev/null || true`. A source-text check, # like test_swap_ordered_after_wait_and_before_start. The call-site needle is anchored (`^ name$`) # so it cannot be satisfied by the comment lines above each call site that merely mention the # function by name. test_stop_branches_call_tolerant_helpers_not_bare_or_true() { local src="$ROOT/scripts/redeploy-fleetd.sh" unload_call_line stop_call_line unload_call_line="$(grep -n '^ unload_launchd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)" stop_call_line="$(grep -n '^ stop_systemd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$unload_call_line" ] \ || fail "could not find the main flow's call to unload_launchd_if_loaded in redeploy-fleetd.sh" [ -n "$stop_call_line" ] \ || fail "could not find the main flow's call to stop_systemd_if_loaded in redeploy-fleetd.sh" if grep -qF 'launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true' "$src"; then fail "the bare 'launchctl unload ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh" fi if grep -qF 'systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true' "$src"; then fail "the bare 'systemctl --user stop ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh" fi } test_no_errors() { cat > "$TMP/no-errors.log" <<'LOG' 2026-09-05 12:00:00 INFO fleetd listening LOG classify_fixture no-errors.log assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained" # fleetd #552 control: a REAL, readable log region with nothing in it must never look like a # region that could not be captured at all. assert_equals 0 "$REDEPLOY_AMQP_CHECK_SKIPPED" "no-errors must not report the check as skipped" } # fleetd #552 — the third state: "" (capture_fresh_log_region's own sentinel for "the post-restart # region could never be captured") must never look like "read a file with nothing in it", which is # exactly what REDEPLOY_ERROR_COUNT staying at 0 already means on its own (see test_no_errors above). test_classify_amqp_connection_errors_reports_skipped_when_log_missing() { classify_amqp_connection_errors "" > "$TMP/classify-skipped-output" 2>&1 assert_equals 1 "$REDEPLOY_AMQP_CHECK_SKIPPED" "an empty log_file must set REDEPLOY_AMQP_CHECK_SKIPPED=1" assert_equals 0 "$REDEPLOY_ERROR_COUNT" "a skipped check must still leave REDEPLOY_ERROR_COUNT at its initialized 0 (the flag, not the count, carries the distinction)" grep -qF 'could not be captured' "$TMP/classify-skipped-output" \ || fail "classify_amqp_connection_errors did not say the post-restart log region could not be captured" } test_recovery_patterns_match_source() { grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \ || fail "AMQP failure pattern no longer matches source" grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \ "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \ || fail "reply-inbox recovery pattern no longer matches source" grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \ "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \ || fail "lead-mailbox recovery pattern no longer matches source" } test_attributed_recovered_connection_error() { cat > "$TMP/attributed-recovered.log" <<'LOG' 2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery LOG classify_fixture attributed-recovered.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total" assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained" } test_source_derived_error_shapes_recover_by_connection() { # These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check # after redeploy because the new code has not yet written a production line. cat > "$TMP/source-derived.log" <<'LOG' 17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery! 17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred 17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery! 17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred 17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture source-derived.log assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total" assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained" } test_cross_connection_unattributable_errors_stay_loud() { # This candidate has neither stable connection name, so LeadMailbox recovery must not consume it. cat > "$TMP/cross-unattributable.log" <<'LOG' 2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred 2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred 2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture cross-unattributable.log assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered" assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained" } test_attributed_cross_connection_errors_stay_loud() { # LeadMailbox recovery cannot heal AmqpReplyInbox errors. cat > "$TMP/cross-attributed.log" <<'LOG' 2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture cross-attributed.log assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered" assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained" } test_attributed_unrecovered_connection_error() { cat > "$TMP/unrecovered.log" <<'LOG' 2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred LOG classify_fixture unrecovered.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors" assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained" } test_other_error_is_unexplained() { cat > "$TMP/other-error.log" <<'LOG' 2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed 2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery LOG classify_fixture other-error.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total" assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained" } test_recovery_requirement_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; esac done < "$log_file" } if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then fail "mutation accepted an unrecovered connection error" fi grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \ || fail "mutation failed without the expected assertion" printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n' } test_shared_counter_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line pending=0 REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*) case "$line" in *'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; *'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*) if [ "$pending" -gt 0 ]; then pending=$((pending - 1)) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) fi ;; esac done < "$log_file" REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending)) } if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then fail "shared counter mutation accepted cross-connection recovery" fi grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \ || fail "shared counter mutation failed without the expected assertion" printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n' } test_unattributable_quiet_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; esac done < "$log_file" } if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then fail "unattributable mutation accepted an unknown connection" fi grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \ || fail "unattributable mutation failed without the expected assertion" printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n' } # fleetd #512 part 2 — the negative check (scan_uncaught_exceptions). The heart of this half of the # ticket: a fixture with the uncaught-exception shape and NO line carrying an ERROR token at all, # proving the scan finds it without one. A fixture that also carried an ERROR line would pass for # the wrong reason. test_scan_uncaught_exceptions_finds_shape_without_error_token() { cat > "$TMP/scan-died.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081) LOG local error_count error_count="$(grep -c ' ERROR ' "$TMP/scan-died.log" || true)" [ "$error_count" = "0" ] \ || fail "test fixture error: scan-died.log unexpectedly carries an ERROR token" scan_uncaught_exceptions "$TMP/scan-died.log" assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "scan must find the exception without an ERROR token" printf '%s' "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" | grep -qF 'NoClassDefFoundError' \ || fail "scan did not capture the matching line as the sample" } test_scan_uncaught_exceptions_clean_control() { cat > "$TMP/scan-clean.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=0 abandoned=0 (still BUSY at the shutdown deadline) LOG scan_uncaught_exceptions "$TMP/scan-clean.log" assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "clean control must find no uncaught exception" assert_equals "" "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" "clean control sample must be empty" } # fleetd #512 part 2 — the positive check (find_drain_complete_line). Both halves of #522's line: # present, and absent. test_find_drain_complete_line_present() { cat > "$TMP/drain-line-present.log" <<'LOG' 2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=2 abandoned=1 (still BUSY at the shutdown deadline) LOG find_drain_complete_line "$TMP/drain-line-present.log" printf '%s' "$REDEPLOY_DRAIN_COMPLETE_LINE" | grep -qF 'released=2 abandoned=1' \ || fail "find_drain_complete_line did not capture the present line" } test_find_drain_complete_line_absent() { cat > "$TMP/drain-line-absent.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 LOG find_drain_complete_line "$TMP/drain-line-absent.log" assert_equals "" "$REDEPLOY_DRAIN_COMPLETE_LINE" "find_drain_complete_line must report empty when absent" } # fleetd #512 part 2 — report_shutdown_drain, the composite decision+action function the main flow # calls unconditionally (same shape as swap_if_built/refuse_drain_gate, #521/#528). These four cover # the four outcomes named in the ticket's "trap": complete, died, unknown ("cannot tell" — neither a # pass nor a failure), and n/a (no previous daemon was actually stopped this run). # # Deliberately NOT run inside `$(...)`: report_shutdown_drain sets REDEPLOY_DRAIN_STATE as a global # side effect that these tests need to read back afterward, and a command substitution forks a # subshell that global assignment would not survive (the exact trap documented above # detect_supervisor in redeploy-fleetd.sh, for the same reason). Plain output redirection to a file # does not fork a subshell, so it is used to capture what was printed instead. test_report_shutdown_drain_died_without_error_token() { cat > "$TMP/drain-died.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081) LOG local error_count error_count="$(grep -c ' ERROR ' "$TMP/drain-died.log" || true)" [ "$error_count" = "0" ] \ || fail "test fixture error: drain-died.log unexpectedly carries an ERROR token" report_shutdown_drain "$TMP/drain-died.log" 1 > "$TMP/drain-died-output" 2>&1 assert_equals "died" "$REDEPLOY_DRAIN_STATE" "died fixture must set REDEPLOY_DRAIN_STATE=died" assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "died fixture uncaught-exception count" grep -qF 'NoClassDefFoundError' "$TMP/drain-died-output" \ || fail "report_shutdown_drain did not report the uncaught-exception shape it found" grep -qF 'DIED' "$TMP/drain-died-output" \ || fail "report_shutdown_drain did not report the drain as DIED" } test_report_shutdown_drain_complete_control() { cat > "$TMP/drain-complete.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down 2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=3 abandoned=0 (still BUSY at the shutdown deadline) LOG report_shutdown_drain "$TMP/drain-complete.log" 1 > "$TMP/drain-complete-output" 2>&1 assert_equals "complete" "$REDEPLOY_DRAIN_STATE" "complete-control fixture must set REDEPLOY_DRAIN_STATE=complete" assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "complete-control fixture must find no uncaught exception" grep -qF 'released=3 abandoned=0' "$TMP/drain-complete-output" \ || fail "report_shutdown_drain did not report the drain-complete counts" } test_report_shutdown_drain_unknown_cannot_tell() { cat > "$TMP/drain-unknown.log" <<'LOG' 2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765 2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down LOG report_shutdown_drain "$TMP/drain-unknown.log" 1 > "$TMP/drain-unknown-output" 2>&1 assert_equals "unknown" "$REDEPLOY_DRAIN_STATE" "cannot-tell fixture must set REDEPLOY_DRAIN_STATE=unknown" grep -qF 'cannot tell' "$TMP/drain-unknown-output" \ || fail "report_shutdown_drain did not say it could not tell" if grep -qF ' ok' "$TMP/drain-unknown-output"; then fail "cannot-tell outcome must not be printed via ok() — it is neither a pass nor a failure" fi } # A cold start (or a restart where nothing was actually stopped) has no previous-daemon shutdown # window to have an opinion about at all. This fixture's log content looks exactly like a died drain # — proving the had_previous_daemon=0 gate is actually consulted, not merely documented: without it, # this would misreport "died" or "unknown" on every clean cold start. test_report_shutdown_drain_no_previous_daemon_is_na() { cat > "$TMP/drain-na.log" <<'LOG' Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions LOG report_shutdown_drain "$TMP/drain-na.log" 0 > "$TMP/drain-na-output" 2>&1 assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon fixture must set REDEPLOY_DRAIN_STATE=n/a even though the log content looks like a died drain" grep -qF 'nothing to check' "$TMP/drain-na-output" \ || fail "report_shutdown_drain did not report that there was nothing to check" } # fleetd #552 — a fifth outcome: log_file="" (capture_fresh_log_region's own sentinel) with a # previous daemon that WAS stopped this run. Distinct from "unknown" (a log was read and neither # signal was found in it) — here there was no log to read at all. test_report_shutdown_drain_skipped_when_log_missing() { report_shutdown_drain "" 1 > "$TMP/drain-skipped-output" 2>&1 assert_equals "skipped" "$REDEPLOY_DRAIN_STATE" "an empty log_file with a previous daemon must set REDEPLOY_DRAIN_STATE=skipped" grep -qF 'could not be captured' "$TMP/drain-skipped-output" \ || fail "report_shutdown_drain did not say the post-restart log region could not be captured" if grep -qF ' ok' "$TMP/drain-skipped-output"; then fail "the skipped outcome must not be printed via ok() — it is neither a pass nor a failure" fi } # fleetd #552 — proves the had_previous_daemon=0 check really does run BEFORE the missing-log check: # with BOTH conditions true at once, n/a must win, because a cold start has nothing to check # regardless of whether the log region could be captured. test_report_shutdown_drain_na_wins_over_missing_log() { report_shutdown_drain "" 0 > "$TMP/drain-na-and-missing-output" 2>&1 assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon must win over a missing log region" } # fleetd #512 — closes the gap none of the seven tests above can: they call report_shutdown_drain # directly, and sourcing stops before the main flow ever runs (the SOURCED guard), so none of them # can prove the main flow still calls it at all. Same shape as test_run_drain_gate_call_site_present # and test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site, plus # an ordering check against its neighbours in the verify/result flow. test_report_shutdown_drain_call_site_present() { local src="$ROOT/scripts/redeploy-fleetd.sh" call_line call_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$call_line" ] \ || fail "could not find the main flow's report_shutdown_drain call site in redeploy-fleetd.sh" } test_report_shutdown_drain_ordered_after_classify_and_before_result() { local src="$ROOT/scripts/redeploy-fleetd.sh" classify_line drain_line result_line classify_line="$(grep -Fn 'classify_amqp_connection_errors "$FRESH_LOG"' "$src" | tail -1 | cut -d: -f1 || true)" drain_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)" result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)" [ -n "$classify_line" ] || fail "could not find the classify_amqp_connection_errors call site" [ -n "$drain_line" ] || fail "could not find the report_shutdown_drain call site" [ -n "$result_line" ] || fail "could not find the result section" [ "$drain_line" -gt "$classify_line" ] \ || fail "report_shutdown_drain (line $drain_line) is not after classify_amqp_connection_errors (line $classify_line)" [ "$drain_line" -lt "$result_line" ] \ || fail "report_shutdown_drain (line $drain_line) is not before the result section (line $result_line)" } # fleetd #512 item 4 — the summary line must not read as reassurance when the shutdown-drain check # found something wrong (or could not tell). Sourcing stops before the main flow runs, so this is a # source-text check like test_drain_gate_abort_message_says_no_no_build above. test_no_error_lines_message_gated_by_drain_state() { local src="$ROOT/scripts/redeploy-fleetd.sh" block block="$(grep -B2 -F 'ok "no ERROR lines since restart"' "$src")" [ -n "$block" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh" printf '%s' "$block" | grep -qF 'REDEPLOY_DRAIN_STATE' \ || fail "'no ERROR lines since restart' is not guarded by the shutdown-drain outcome (fleetd #512 item 4)" } test_detect_supervisor_launchd_only test_detect_supervisor_systemd_only test_detect_supervisor_none test_detect_supervisor_systemd_installed_not_loaded_is_unclear test_detect_supervisor_launchd_installed_not_loaded_is_unclear test_detect_supervisor_systemd_probe_error_is_unclear test_detect_supervisor_systemd_probe_setup_failure_is_unclear test_mktemp_dash_t_templates_have_x_placeholders test_no_unguarded_macos_only_hasher_calls test_no_unguarded_mktemp_assignments_after_restart test_capture_fresh_log_region_control test_capture_fresh_log_region_mktemp_failure_returns_nonzero_and_prints_nothing test_fresh_log_capture_failure_does_not_abort_under_sete test_capture_fresh_log_region_call_site_is_guarded test_result_section_checks_amqp_skip_before_no_error_lines test_require_drivable_supervisor_refuses_ambiguous test_require_drivable_supervisor_refuses_unclear test_require_drivable_supervisor_accepts_known_kinds test_count_daemon_pids test_assert_single_daemon_accepts_one_pid test_assert_single_daemon_rejects_two_pids test_running_pid_excludes_self_matching_wrapper_shell test_running_pid_finds_a_real_java_named_second_process test_running_pid_drops_a_pid_whose_comm_is_not_java test_running_pid_drops_a_pid_that_exited_before_the_comm_lookup test_running_pid_counts_a_pid_whose_comm_is_java test_die_message_does_not_recommend_bare_pgrep_as_remediation test_jar_id_defaults_to_live_and_reports_explicit_path test_hash256_computes_a_real_sha256 test_jar_id_reports_absent_for_missing_file test_jar_id_reports_unhashable_when_no_hasher_on_path test_stage_built_jar_moves_off_live_path test_stage_built_jar_dies_when_build_produced_nothing test_swap_staged_jar_moves_staged_onto_live test_swap_staged_jar_dies_without_staged_file test_swap_staged_jar_dies_when_mv_fails test_should_swap_true_when_build_ran test_should_swap_false_when_build_skipped test_swap_if_built_performs_the_swap_when_build_ran test_swap_if_built_skips_the_swap_when_build_skipped test_require_no_build_jar_dies_when_absent test_require_no_build_jar_accepts_present_jar test_wait_for_daemon_exit_returns_true_once_pid_clears test_wait_for_daemon_exit_times_out_if_pid_never_clears test_wait_for_new_pid_returns_true_once_pid_appears test_wait_for_new_pid_times_out_if_pid_never_appears test_await_daemon_started_slow_pid_then_healthy_succeeds test_await_daemon_started_never_appears_dies_with_log_tail test_await_daemon_started_pid_never_found_but_healthy_warns_and_survives test_await_daemon_started_call_site_present test_swap_ordered_after_wait_and_before_start test_drain_gate_abort_message_says_no_no_build test_drain_gate_refusal_build_ran_staged_present test_drain_gate_refusal_build_ran_staged_absent test_drain_gate_refusal_no_build_staged_present test_drain_gate_refusal_no_build_staged_absent test_refuse_drain_gate_build_ran_staged_present test_refuse_drain_gate_build_ran_staged_absent test_refuse_drain_gate_no_build_staged_present test_refuse_drain_gate_no_build_staged_absent test_run_drain_gate_call_site_present test_should_stop_for_check_true_when_check_only test_should_stop_for_check_false_when_not_check_only test_stop_if_check_only_exits_zero_and_says_nothing_changed test_stop_if_check_only_returns_without_exiting_when_not_check_only test_stop_if_check_only_call_site_present test_drain_gate_required_true_when_pid_and_not_assume_yes test_drain_gate_required_false_when_no_pid test_drain_gate_required_false_when_assume_yes test_drain_confirmed_true_on_yes test_drain_confirmed_false_on_anything_else test_run_drain_gate_skips_prompt_when_not_required test_run_drain_gate_confirmed_reply_does_not_refuse test_run_drain_gate_declined_reply_refuses test_report_supervisor_state_launchd_sets_supervised_and_checks_log_path test_report_supervisor_state_systemd_sets_supervised_without_log_path_check test_report_supervisor_state_none_leaves_supervised_zero test_report_supervisor_state_unrecognized_kind_warns_and_continues test_report_supervisor_state_call_site_present test_dispatch_stop_launchd_calls_stop_via_launchd test_dispatch_stop_systemd_calls_stop_via_systemd test_dispatch_stop_none_calls_stop_via_kill_with_pid test_dispatch_stop_unrecognized_kind_dies test_dispatch_stop_call_site_present test_dispatch_start_launchd_calls_start_via_launchd test_dispatch_start_systemd_calls_start_via_systemd test_dispatch_start_none_calls_start_via_none test_dispatch_start_unrecognized_kind_dies test_dispatch_start_call_site_present test_health_is_up_true_with_body test_health_is_up_false_with_empty_body test_report_health_up_reports_ok_and_never_dies test_report_health_degraded_503_warns_and_never_dies test_report_health_dead_dies test_poll_health_body_returns_nonzero_when_unreachable test_report_health_call_site_present test_compute_had_old_pid_zero_when_empty test_compute_had_old_pid_one_when_present test_compute_had_old_pid_call_site_present test_no_untested_main_flow_conditionals test_unload_launchd_if_loaded_dies_on_real_failure test_unload_launchd_if_loaded_tolerates_clean_negative test_stop_systemd_if_loaded_dies_on_real_failure test_stop_systemd_if_loaded_tolerates_clean_negative test_stop_branches_call_tolerant_helpers_not_bare_or_true test_no_errors test_classify_amqp_connection_errors_reports_skipped_when_log_missing test_recovery_patterns_match_source test_attributed_recovered_connection_error test_source_derived_error_shapes_recover_by_connection test_cross_connection_unattributable_errors_stay_loud test_attributed_cross_connection_errors_stay_loud test_attributed_unrecovered_connection_error test_other_error_is_unexplained test_recovery_requirement_mutation_is_caught test_shared_counter_mutation_is_caught test_unattributable_quiet_mutation_is_caught test_scan_uncaught_exceptions_finds_shape_without_error_token test_scan_uncaught_exceptions_clean_control test_find_drain_complete_line_present test_find_drain_complete_line_absent test_report_shutdown_drain_died_without_error_token test_report_shutdown_drain_complete_control test_report_shutdown_drain_unknown_cannot_tell test_report_shutdown_drain_no_previous_daemon_is_na test_report_shutdown_drain_skipped_when_log_missing test_report_shutdown_drain_na_wins_over_missing_log test_report_shutdown_drain_call_site_present test_report_shutdown_drain_ordered_after_classify_and_before_result test_no_error_lines_message_gated_by_drain_state printf 'PASS: redeploy log classifier\n'