#!/usr/bin/env bash # Self-contained checks for the pure log classifier in redeploy-fleetd.sh. set -euo pipefail ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)" TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")" trap 'rm -rf "$TMP"' EXIT # Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon. source "$ROOT/scripts/redeploy-fleetd.sh" fail() { printf 'FAIL: %s\n' "$*" >&2 return 1 } assert_equals() { local expected="$1" actual="$2" description="$3" [ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual" } classify_fixture() { local name="$1" classify_amqp_connection_errors "$TMP/$name" } # fleetd #492 follow-up: detect_supervisor's stdout is now "kinddetail" (see the constraints # comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about # the kind must split it out with the SAME in-shell parameter expansion the real call site (:438) # uses, never a bare string comparison against the raw output. supervisor_kind_of() { printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}" } supervisor_detail_of() { printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}" } # fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so # each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either # being naturally absent: this machine may itself be running a real fleetd under launchd right now # (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving # launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state # instead of the fixture. test_detect_supervisor_launchd_only() { launchd_installed() { return 0; } launchd_loaded() { return 0; } systemd_installed() { return 1; } systemd_loaded() { return 1; } assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection" } test_detect_supervisor_systemd_only() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } systemd_loaded() { return 0; } assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection" } test_detect_supervisor_none() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 1; } systemd_loaded() { return 1; } assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection" } # fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not # tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that # fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit # reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending # auto-restart too, and every one of those is a host that IS under systemd. test_detect_supervisor_systemd_installed_not_loaded_is_unclear() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } # the unit file IS there systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart local raw raw="$(detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail does not name the systemd unit it found installed-but-not-loaded" } # Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being # removed, or about to be reloaded) must not read as "no supervisor" either. test_detect_supervisor_launchd_installed_not_loaded_is_unclear() { launchd_installed() { return 0; } # the plist IS there launchd_loaded() { return 1; } # launchctl list says not loaded systemd_installed() { return 1; } systemd_loaded() { return 1; } local raw raw="$(detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \ || fail "detail does not name the launchd label it found installed-but-not-loaded" } # Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl` # stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl # that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no # lingering). This must read as unclear, never none: a probe that could not answer at all is not # the same fact as "no supervisor is loaded". test_detect_supervisor_systemd_probe_error_is_unclear() { # Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this # file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point # of those tests), and a bash function definition is global for the rest of the process — without # this, systemd_loaded here would still be whatever the previous test left it as, never touching # a real `systemctl` call at all. source "$ROOT/scripts/redeploy-fleetd.sh" local bin_dir result rc=0 bin_dir="$TMP/stub-bin-systemctl-errors" mkdir -p "$bin_dir" cat > "$bin_dir/systemctl" <<'STUB' #!/usr/bin/env bash echo "Failed to connect to bus: No such file or directory" >&2 exit 1 STUB chmod +x "$bin_dir/systemctl" PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$? [ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored" assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative" launchd_installed() { return 1; } launchd_loaded() { return 1; } result="$(PATH="$bin_dir:$PATH" detect_supervisor)" assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none" printf '%s' "$(supervisor_detail_of "$result")" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail does not name the systemd unit whose probe errored" } # fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a # hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not # the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real # caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout # survives that boundary, so kind AND detail must both cross on it — this test proves they do. test_require_drivable_supervisor_refuses_unclear() { launchd_installed() { return 1; } launchd_loaded() { return 1; } systemd_installed() { return 0; } systemd_loaded() { return 1; } local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0 # Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand. SUPERVISOR_RAW="$(detect_supervisor)" SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}" SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}" assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal" [ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \ || fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell" printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \ || fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded" output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor" # Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate # recovery instructions name the unit unconditionally either way ("systemctl --user status # $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only # the specific detail string proves the crossed value, not the boilerplate, reached the message. printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \ || fail "refusal message does not contain the specific detail that crossed the subshell boundary" } # The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to # `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that # forks a subshell, so its exit() only ends the subshell and this test script keeps running under # `set -e`. test_require_drivable_supervisor_refuses_ambiguous() { local output rc=0 output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor" printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \ || fail "refusal message does not name the launchd label it found" printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \ || fail "refusal message does not name the systemd unit it found" } test_require_drivable_supervisor_accepts_known_kinds() { require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor" require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor" require_drivable_supervisor "none" || fail "refused the unsupervised case" } # fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor # produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see # it because either daemon alone satisfies them. test_count_daemon_pids() { assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list" assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid" assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids" } test_assert_single_daemon_accepts_one_pid() { assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid" } test_assert_single_daemon_rejects_two_pids() { local output rc=0 output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids" printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found" printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found" } # fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports # $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call # must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR. # Two files with different content, so a default pointed at the wrong one reports the wrong hash # rather than accidentally matching. test_jar_id_defaults_to_live_and_reports_explicit_path() { local dir saved_jar="$JAR" saved_staged="$JAR_STAGED" local live_hash staged_hash default_result explicit_result dir="$TMP/jar-id"; mkdir -p "$dir" JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar" printf 'live jar bytes' > "$JAR" printf 'staged jar bytes, not the same content' > "$JAR_STAGED" live_hash="$(shasum -a 256 "$JAR" | cut -c1-12)" staged_hash="$(shasum -a 256 "$JAR_STAGED" | cut -c1-12)" default_result="$(jar_id)" explicit_result="$(jar_id "$JAR_STAGED")" JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same" assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR" assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR" } # fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar # are exercised directly against real files on disk (not stubs), because the whole point is file # behavior (does the content move, does the source disappear, does a failure leave both sides # intact) that a stubbed function cannot prove. test_stage_built_jar_moves_off_live_path() { local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED" dir="$TMP/stage-ok"; mkdir -p "$dir" jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar" printf 'built jar bytes' > "$jar" JAR="$jar"; JAR_STAGED="$staged" stage_built_jar || fail "stage_built_jar rejected a real build output" JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar" [ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged" grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content" } test_stage_built_jar_dies_when_build_produced_nothing() { local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED" dir="$TMP/stage-missing"; mkdir -p "$dir" JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar" output="$(stage_built_jar 2>&1)" || rc=$? JAR="$saved_jar"; JAR_STAGED="$saved_staged" [ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output" printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \ || fail "refusal message does not name the missing jar path" } test_swap_staged_jar_moves_staged_onto_live() { local dir staged live dir="$TMP/swap-ok"; mkdir -p "$dir" staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar" printf 'swapped jar bytes' > "$staged" swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar" [ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged" [ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live" grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content" } # The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on # failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes # inside a command substitution to contain that exit to a subshell. test_swap_staged_jar_dies_without_staged_file() { local dir output rc=0 dir="$TMP/swap-missing"; mkdir -p "$dir" output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar" [ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged" printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \ || fail "refusal message does not name the missing staged path" } test_swap_staged_jar_dies_when_mv_fails() { local dir staged live output rc=0 dir="$TMP/swap-fail"; mkdir -p "$dir/src" staged="$dir/src/fleetd-new.jar" printf 'fake jar bytes' > "$staged" live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$? [ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv" [ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails" [ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed" printf '%s' "$output" | grep -qF "$staged" \ || fail "refusal message does not name the staged path that could not be moved" } # --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this # path) and must still die with the exact wording documented in the script's own header comment. test_require_no_build_jar_dies_when_absent() { local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar" JAR="$missing" output="$(require_no_build_jar 2>&1)" || rc=$? JAR="$saved_jar" [ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar" printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \ || fail "refusal message does not match the documented --no-build wording" } test_require_no_build_jar_accepts_present_jar() { local saved_jar="$JAR" dir dir="$TMP/no-build-present"; mkdir -p "$dir" JAR="$dir/fleetd.jar" printf 'existing jar' > "$JAR" require_no_build_jar || fail "require_no_build_jar rejected an existing jar" JAR="$saved_jar" } # wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while # running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so # the timeout-loop test does not actually wait out its budget. test_wait_for_daemon_exit_returns_true_once_pid_clears() { # running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so # a plain shell variable it increments would reset on each call instead of accumulating. Count in # a file instead, which is the one thing that actually survives across those subshells. local counter_file="$TMP/wait-exit-calls" final_calls printf '0' > "$counter_file" running_pid() { local n n="$(cat "$counter_file")" n=$((n + 1)) printf '%s' "$n" > "$counter_file" if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi } sleep() { :; } wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared" final_calls="$(cat "$counter_file")" [ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } test_wait_for_daemon_exit_times_out_if_pid_never_clears() { local rc=0 running_pid() { printf '4242'; } sleep() { :; } wait_for_daemon_exit 3 || rc=$? [ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared" source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests } # fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the # main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee # itself — as opposed to the pure functions it's built from — can only be checked by reading the # script's own call sites, the same way test_recovery_patterns_match_source below checks Java # source shape instead of behavior it cannot invoke directly. test_swap_ordered_after_wait_and_before_start() { local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1)" swap_line="$(grep -Fn 'swap_staged_jar "$JAR_STAGED" "$JAR"' "$src" | head -1 | cut -d: -f1)" start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1)" [ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh" [ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh" [ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh" [ "$swap_line" -gt "$wait_line" ] \ || fail "swap_staged_jar (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)" [ "$swap_line" -lt "$start_line" ] \ || fail "swap_staged_jar (line $swap_line) is not before the start section (line $start_line)" } # fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator # declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)" # to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has # already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal # ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start # above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow), # so the only way to pin its exact wording is to read the source. test_drain_gate_abort_message_says_no_no_build() { local src="$ROOT/scripts/redeploy-fleetd.sh" msg msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")" [ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh" if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then fail "abort message still claims a rerun WITH --no-build can finish the restart" fi printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \ || fail "abort message does not tell the operator to rerun without --no-build" printf '%s' "$msg" | grep -qF 'no longer at the live path' \ || fail "abort message does not say why --no-build cannot finish the restart" } test_no_errors() { cat > "$TMP/no-errors.log" <<'LOG' 2026-09-05 12:00:00 INFO fleetd listening LOG classify_fixture no-errors.log assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained" } test_recovery_patterns_match_source() { grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \ || fail "AMQP failure pattern no longer matches source" grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \ "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \ || fail "reply-inbox recovery pattern no longer matches source" grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \ "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \ || fail "lead-mailbox recovery pattern no longer matches source" } test_attributed_recovered_connection_error() { cat > "$TMP/attributed-recovered.log" <<'LOG' 2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery LOG classify_fixture attributed-recovered.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total" assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained" } test_source_derived_error_shapes_recover_by_connection() { # These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check # after redeploy because the new code has not yet written a production line. cat > "$TMP/source-derived.log" <<'LOG' 17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery! 17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred 17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery! 17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred 17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery 17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture source-derived.log assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total" assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered" assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained" } test_cross_connection_unattributable_errors_stay_loud() { # This candidate has neither stable connection name, so LeadMailbox recovery must not consume it. cat > "$TMP/cross-unattributable.log" <<'LOG' 2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred 2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred 2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture cross-unattributable.log assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered" assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained" } test_attributed_cross_connection_errors_stay_loud() { # LeadMailbox recovery cannot heal AmqpReplyInbox errors. cat > "$TMP/cross-attributed.log" <<'LOG' 2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred 2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery 2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery LOG classify_fixture cross-attributed.log assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered" assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained" } test_attributed_unrecovered_connection_error() { cat > "$TMP/unrecovered.log" <<'LOG' 2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred LOG classify_fixture unrecovered.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total" assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors" assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained" } test_other_error_is_unexplained() { cat > "$TMP/other-error.log" <<'LOG' 2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed 2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery LOG classify_fixture other-error.log assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total" assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained" } test_recovery_requirement_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; esac done < "$log_file" } if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then fail "mutation accepted an unrecovered connection error" fi grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \ || fail "mutation failed without the expected assertion" printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n' } test_shared_counter_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line pending=0 REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*) case "$line" in *'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; *'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*) if [ "$pending" -gt 0 ]; then pending=$((pending - 1)) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) fi ;; esac done < "$log_file" REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending)) } if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then fail "shared counter mutation accepted cross-connection recovery" fi grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \ || fail "shared counter mutation failed without the expected assertion" printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n' } test_unattributable_quiet_mutation_is_caught() { classify_amqp_connection_errors() { local log_file="$1" line REDEPLOY_ERROR_COUNT=0 REDEPLOY_RECOVERED_AMQP_ERRORS=0 REDEPLOY_UNEXPLAINED_ERRORS=0 while IFS= read -r line || [ -n "$line" ]; do case "$line" in *' ERROR '*|*' SEVERE '*) REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1)) case "$line" in *'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*) REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1)) ;; *) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;; esac ;; esac done < "$log_file" } if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then fail "unattributable mutation accepted an unknown connection" fi grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \ || fail "unattributable mutation failed without the expected assertion" printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n' } test_detect_supervisor_launchd_only test_detect_supervisor_systemd_only test_detect_supervisor_none test_detect_supervisor_systemd_installed_not_loaded_is_unclear test_detect_supervisor_launchd_installed_not_loaded_is_unclear test_detect_supervisor_systemd_probe_error_is_unclear test_require_drivable_supervisor_refuses_ambiguous test_require_drivable_supervisor_refuses_unclear test_require_drivable_supervisor_accepts_known_kinds test_count_daemon_pids test_assert_single_daemon_accepts_one_pid test_assert_single_daemon_rejects_two_pids test_jar_id_defaults_to_live_and_reports_explicit_path test_stage_built_jar_moves_off_live_path test_stage_built_jar_dies_when_build_produced_nothing test_swap_staged_jar_moves_staged_onto_live test_swap_staged_jar_dies_without_staged_file test_swap_staged_jar_dies_when_mv_fails test_require_no_build_jar_dies_when_absent test_require_no_build_jar_accepts_present_jar test_wait_for_daemon_exit_returns_true_once_pid_clears test_wait_for_daemon_exit_times_out_if_pid_never_clears test_swap_ordered_after_wait_and_before_start test_drain_gate_abort_message_says_no_no_build test_no_errors test_recovery_patterns_match_source test_attributed_recovered_connection_error test_source_derived_error_shapes_recover_by_connection test_cross_connection_unattributable_errors_stay_loud test_attributed_cross_connection_errors_stay_loud test_attributed_unrecovered_connection_error test_other_error_is_unexplained test_recovery_requirement_mutation_is_caught test_shared_counter_mutation_is_caught test_unattributable_quiet_mutation_is_caught printf 'PASS: redeploy log classifier\n'