7c34e8f4f9
drain_gate_refusal composes the correct abort message and is well tested, but the main flow built its own `die "$(drain_gate_refusal ...)"` call — nothing proved that call site was ever consulted. Mutating it to a flat `die "aborted -- nothing changed"` left the whole suite green, silently reinstating the exact defect #517 was filed to fix. Same shape as #521/#526's should_swap/swap_if_built: the decision and the die() now live together in refuse_drain_gate, which the main flow calls unconditionally. drain_gate_refusal stays separate and separately tested for the message logic; four new behavioural tests stub die() to prove refuse_drain_gate calls it correctly for all four cases, and a fifth source-text test pins the main flow's call site itself (the only thing that can catch deleting the call, since sourcing stops before the main flow runs).
873 lines
47 KiB
Bash
Executable File
873 lines
47 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Self-contained checks for the pure log classifier in redeploy-fleetd.sh.
|
|
|
|
set -euo pipefail
|
|
|
|
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")"
|
|
trap 'rm -rf "$TMP"' EXIT
|
|
|
|
# Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
|
|
fail() {
|
|
printf 'FAIL: %s\n' "$*" >&2
|
|
return 1
|
|
}
|
|
|
|
assert_equals() {
|
|
local expected="$1" actual="$2" description="$3"
|
|
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
|
|
}
|
|
|
|
classify_fixture() {
|
|
local name="$1"
|
|
classify_amqp_connection_errors "$TMP/$name"
|
|
}
|
|
|
|
# fleetd #492 follow-up: detect_supervisor's stdout is now "kind<SEP>detail" (see the constraints
|
|
# comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about
|
|
# the kind must split it out with the SAME in-shell parameter expansion the real call site (:438)
|
|
# uses, never a bare string comparison against the raw output.
|
|
supervisor_kind_of() {
|
|
printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}"
|
|
}
|
|
supervisor_detail_of() {
|
|
printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}"
|
|
}
|
|
|
|
|
|
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
|
|
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
|
|
# being naturally absent: this machine may itself be running a real fleetd under launchd right now
|
|
# (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving
|
|
# launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state
|
|
# instead of the fixture.
|
|
test_detect_supervisor_launchd_only() {
|
|
launchd_installed() { return 0; }
|
|
launchd_loaded() { return 0; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_systemd_only() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 0; }
|
|
assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_none() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection"
|
|
}
|
|
|
|
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
|
|
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
|
|
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
|
|
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
|
|
# auto-restart too, and every one of those is a host that IS under systemd.
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; } # the unit file IS there
|
|
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
|
|
local raw
|
|
raw="$(detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail does not name the systemd unit it found installed-but-not-loaded"
|
|
}
|
|
|
|
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
|
|
# removed, or about to be reloaded) must not read as "no supervisor" either.
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 0; } # the plist IS there
|
|
launchd_loaded() { return 1; } # launchctl list says not loaded
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
local raw
|
|
raw="$(detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \
|
|
|| fail "detail does not name the launchd label it found installed-but-not-loaded"
|
|
}
|
|
|
|
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
|
|
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
|
|
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
|
|
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
|
|
# the same fact as "no supervisor is loaded".
|
|
test_detect_supervisor_systemd_probe_error_is_unclear() {
|
|
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
|
|
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
|
|
# of those tests), and a bash function definition is global for the rest of the process — without
|
|
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
|
|
# a real `systemctl` call at all.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local bin_dir result rc=0
|
|
bin_dir="$TMP/stub-bin-systemctl-errors"
|
|
mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/systemctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "Failed to connect to bus: No such file or directory" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/systemctl"
|
|
|
|
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
|
|
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
|
|
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$result")" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail does not name the systemd unit whose probe errored"
|
|
}
|
|
|
|
# fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a
|
|
# hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not
|
|
# the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real
|
|
# caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout
|
|
# survives that boundary, so kind AND detail must both cross on it — this test proves they do.
|
|
test_require_drivable_supervisor_refuses_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 1; }
|
|
|
|
local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0
|
|
# Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand.
|
|
SUPERVISOR_RAW="$(detect_supervisor)"
|
|
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
|
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
|
|
|
assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal"
|
|
[ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \
|
|
|| fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell"
|
|
printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded"
|
|
|
|
output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
|
|
# Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate
|
|
# recovery instructions name the unit unconditionally either way ("systemctl --user status
|
|
# $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only
|
|
# the specific detail string proves the crossed value, not the boilerplate, reached the message.
|
|
printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \
|
|
|| fail "refusal message does not contain the specific detail that crossed the subshell boundary"
|
|
}
|
|
|
|
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
|
|
# `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that
|
|
# forks a subshell, so its exit() only ends the subshell and this test script keeps running under
|
|
# `set -e`.
|
|
test_require_drivable_supervisor_refuses_ambiguous() {
|
|
local output rc=0
|
|
output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor"
|
|
printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \
|
|
|| fail "refusal message does not name the launchd label it found"
|
|
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "refusal message does not name the systemd unit it found"
|
|
}
|
|
|
|
test_require_drivable_supervisor_accepts_known_kinds() {
|
|
require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor"
|
|
require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor"
|
|
require_drivable_supervisor "none" || fail "refused the unsupervised case"
|
|
}
|
|
|
|
# fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor
|
|
# produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see
|
|
# it because either daemon alone satisfies them.
|
|
test_count_daemon_pids() {
|
|
assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list"
|
|
assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid"
|
|
assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids"
|
|
}
|
|
|
|
test_assert_single_daemon_accepts_one_pid() {
|
|
assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid"
|
|
}
|
|
|
|
test_assert_single_daemon_rejects_two_pids() {
|
|
local output rc=0
|
|
output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids"
|
|
printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found"
|
|
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
|
|
}
|
|
|
|
# fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports
|
|
# $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call
|
|
# must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR.
|
|
# Two files with different content, so a default pointed at the wrong one reports the wrong hash
|
|
# rather than accidentally matching.
|
|
test_jar_id_defaults_to_live_and_reports_explicit_path() {
|
|
local dir saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
local live_hash staged_hash default_result explicit_result
|
|
dir="$TMP/jar-id"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
|
printf 'live jar bytes' > "$JAR"
|
|
printf 'staged jar bytes, not the same content' > "$JAR_STAGED"
|
|
live_hash="$(shasum -a 256 "$JAR" | cut -c1-12)"
|
|
staged_hash="$(shasum -a 256 "$JAR_STAGED" | cut -c1-12)"
|
|
default_result="$(jar_id)"
|
|
explicit_result="$(jar_id "$JAR_STAGED")"
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same"
|
|
assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR"
|
|
assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR"
|
|
}
|
|
|
|
# fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511)
|
|
# proves both halves of the present-file contract but never exercises the missing-file path. This
|
|
# word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean`
|
|
# deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points
|
|
# operators at `--check` for exactly this. Covers both the no-argument default and an explicit path,
|
|
# since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both.
|
|
test_jar_id_reports_absent_for_missing_file() {
|
|
local saved_jar="$JAR" dir default_result explicit_result
|
|
dir="$TMP/jar-id-absent"; mkdir -p "$dir"
|
|
JAR="$dir/does-not-exist.jar"
|
|
[ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR"
|
|
default_result="$(jar_id)"
|
|
explicit_result="$(jar_id "$dir/also-does-not-exist.jar")"
|
|
JAR="$saved_jar"
|
|
assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist"
|
|
assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent"
|
|
}
|
|
|
|
# fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar
|
|
# are exercised directly against real files on disk (not stubs), because the whole point is file
|
|
# behavior (does the content move, does the source disappear, does a failure leave both sides
|
|
# intact) that a stubbed function cannot prove.
|
|
test_stage_built_jar_moves_off_live_path() {
|
|
local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
dir="$TMP/stage-ok"; mkdir -p "$dir"
|
|
jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar"
|
|
printf 'built jar bytes' > "$jar"
|
|
JAR="$jar"; JAR_STAGED="$staged"
|
|
stage_built_jar || fail "stage_built_jar rejected a real build output"
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar"
|
|
[ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged"
|
|
grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content"
|
|
}
|
|
|
|
test_stage_built_jar_dies_when_build_produced_nothing() {
|
|
local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
dir="$TMP/stage-missing"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
|
output="$(stage_built_jar 2>&1)" || rc=$?
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output"
|
|
printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \
|
|
|| fail "refusal message does not name the missing jar path"
|
|
}
|
|
|
|
test_swap_staged_jar_moves_staged_onto_live() {
|
|
local dir staged live
|
|
dir="$TMP/swap-ok"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar"
|
|
printf 'swapped jar bytes' > "$staged"
|
|
swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar"
|
|
[ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged"
|
|
[ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live"
|
|
grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content"
|
|
}
|
|
|
|
# The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on
|
|
# failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes
|
|
# inside a command substitution to contain that exit to a subshell.
|
|
test_swap_staged_jar_dies_without_staged_file() {
|
|
local dir output rc=0
|
|
dir="$TMP/swap-missing"; mkdir -p "$dir"
|
|
output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar"
|
|
[ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged"
|
|
printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \
|
|
|| fail "refusal message does not name the missing staged path"
|
|
}
|
|
|
|
test_swap_staged_jar_dies_when_mv_fails() {
|
|
local dir staged live output rc=0
|
|
dir="$TMP/swap-fail"; mkdir -p "$dir/src"
|
|
staged="$dir/src/fleetd-new.jar"
|
|
printf 'fake jar bytes' > "$staged"
|
|
live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails
|
|
output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv"
|
|
[ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails"
|
|
[ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed"
|
|
printf '%s' "$output" | grep -qF "$staged" \
|
|
|| fail "refusal message does not name the staged path that could not be moved"
|
|
}
|
|
|
|
# --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this
|
|
# path) and must still die with the exact wording documented in the script's own header comment.
|
|
test_require_no_build_jar_dies_when_absent() {
|
|
local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar"
|
|
JAR="$missing"
|
|
output="$(require_no_build_jar 2>&1)" || rc=$?
|
|
JAR="$saved_jar"
|
|
[ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar"
|
|
printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \
|
|
|| fail "refusal message does not match the documented --no-build wording"
|
|
}
|
|
|
|
test_require_no_build_jar_accepts_present_jar() {
|
|
local saved_jar="$JAR" dir
|
|
dir="$TMP/no-build-present"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"
|
|
printf 'existing jar' > "$JAR"
|
|
require_no_build_jar || fail "require_no_build_jar rejected an existing jar"
|
|
JAR="$saved_jar"
|
|
}
|
|
|
|
# wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while
|
|
# running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so
|
|
# the timeout-loop test does not actually wait out its budget.
|
|
test_wait_for_daemon_exit_returns_true_once_pid_clears() {
|
|
# running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so
|
|
# a plain shell variable it increments would reset on each call instead of accumulating. Count in
|
|
# a file instead, which is the one thing that actually survives across those subshells.
|
|
local counter_file="$TMP/wait-exit-calls" final_calls
|
|
printf '0' > "$counter_file"
|
|
running_pid() {
|
|
local n
|
|
n="$(cat "$counter_file")"
|
|
n=$((n + 1))
|
|
printf '%s' "$n" > "$counter_file"
|
|
if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi
|
|
}
|
|
sleep() { :; }
|
|
wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared"
|
|
final_calls="$(cat "$counter_file")"
|
|
[ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
|
}
|
|
|
|
test_wait_for_daemon_exit_times_out_if_pid_never_clears() {
|
|
local rc=0
|
|
running_pid() { printf '4242'; }
|
|
sleep() { :; }
|
|
wait_for_daemon_exit 3 || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
|
}
|
|
|
|
# fleetd #521 — the swap step's guard, at two levels.
|
|
#
|
|
# The first two tests call the predicate should_swap() directly. They pin its logic, and that is all
|
|
# they pin. On their own they did NOT close #521, and this was measured rather than argued: with the
|
|
# main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left
|
|
# this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs
|
|
# the swap consult the predicate at all. Extracting the decision had moved the untested decision up
|
|
# a level, not removed it.
|
|
#
|
|
# So the last two tests call swap_if_built() — the function the main flow actually calls, holding the
|
|
# guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the
|
|
# guard is removed, inverted, or stops being consulted.
|
|
#
|
|
# What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow
|
|
# altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing
|
|
# stops before the main flow runs, so no test in this file can invoke it.
|
|
test_should_swap_true_when_build_ran() {
|
|
should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true"
|
|
}
|
|
|
|
test_should_swap_false_when_build_skipped() {
|
|
if should_swap 0; then
|
|
fail "should_swap 0 (--no-build; nothing was staged this run) must return false"
|
|
fi
|
|
}
|
|
|
|
# Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is
|
|
# global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id
|
|
# overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear),
|
|
# and again at the END, so their own stubs do not leak into every test that runs after them.
|
|
test_swap_if_built_performs_the_swap_when_build_ran() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local marker="$TMP/swap-if-built-ran"
|
|
rm -f "$marker"
|
|
swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; }
|
|
jar_id() { printf 'stubbed\n'; }
|
|
swap_if_built 1 > /dev/null
|
|
[ -f "$marker" ] \
|
|
|| fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_swap_if_built_skips_the_swap_when_build_skipped() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local marker="$TMP/swap-if-built-skipped"
|
|
rm -f "$marker"
|
|
swap_staged_jar() { printf 'swapped\n' > "$marker"; }
|
|
jar_id() { printf 'stubbed\n'; }
|
|
swap_if_built 0 > /dev/null
|
|
if [ -f "$marker" ]; then
|
|
fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did"
|
|
fi
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
# fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the
|
|
# main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee
|
|
# itself — as opposed to the pure functions it's built from — can only be checked by reading the
|
|
# script's own call sites, the same way test_recovery_patterns_match_source below checks Java
|
|
# source shape instead of behavior it cannot invoke directly.
|
|
#
|
|
# Two details about the three greps below, both of which have already gone wrong here.
|
|
#
|
|
# The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not
|
|
# `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside
|
|
# swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using
|
|
# it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)"
|
|
# — a true statement about a function definition, and nothing at all about the order of the steps.
|
|
#
|
|
# Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the
|
|
# pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set
|
|
# -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to
|
|
# report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no
|
|
# name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak.
|
|
test_swap_ordered_after_wait_and_before_start() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line
|
|
wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh"
|
|
[ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh"
|
|
[ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh"
|
|
[ "$swap_line" -gt "$wait_line" ] \
|
|
|| fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)"
|
|
[ "$swap_line" -lt "$start_line" ] \
|
|
|| fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)"
|
|
}
|
|
|
|
# fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator
|
|
# declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)"
|
|
# to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has
|
|
# already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal
|
|
# ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start
|
|
# above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow),
|
|
# so the only way to pin its exact wording is to read the source.
|
|
test_drain_gate_abort_message_says_no_no_build() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" msg
|
|
msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")"
|
|
[ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh"
|
|
if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then
|
|
fail "abort message still claims a rerun WITH --no-build can finish the restart"
|
|
fi
|
|
printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \
|
|
|| fail "abort message does not tell the operator to rerun without --no-build"
|
|
printf '%s' "$msg" | grep -qF 'no longer at the live path' \
|
|
|| fail "abort message does not say why --no-build cannot finish the restart"
|
|
}
|
|
|
|
# fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was
|
|
# a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's
|
|
# own file for the wording, which stays in the file even if the `if` guarding it is mutated to
|
|
# `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead,
|
|
# so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT
|
|
# alongside these, not replaced, because it catches a different regression (a re-wording that still
|
|
# reaches the right branch would not change which case fires here, but would still be worth pinning).
|
|
test_drain_gate_refusal_build_ran_staged_present() {
|
|
local dir staged result
|
|
dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'staged jar bytes' > "$staged"
|
|
result="$(drain_gate_refusal 1 "$staged")"
|
|
printf '%s' "$result" | grep -qF "$staged" \
|
|
|| fail "build-ran+staged-present refusal does not name the staged jar path"
|
|
printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \
|
|
|| fail "build-ran+staged-present refusal does not tell the operator how to finish the restart"
|
|
if printf '%s' "$result" | grep -qF 'nothing changed'; then
|
|
fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved"
|
|
fi
|
|
}
|
|
|
|
test_drain_gate_refusal_build_ran_staged_absent() {
|
|
local dir result
|
|
dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir"
|
|
result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")"
|
|
assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording"
|
|
}
|
|
|
|
# --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar
|
|
# found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the
|
|
# comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning.
|
|
test_drain_gate_refusal_no_build_staged_present() {
|
|
local dir staged result
|
|
dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'leftover staged jar bytes' > "$staged"
|
|
result="$(drain_gate_refusal 0 "$staged")"
|
|
assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed"
|
|
}
|
|
|
|
test_drain_gate_refusal_no_build_staged_absent() {
|
|
local dir result
|
|
dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir"
|
|
result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")"
|
|
assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording"
|
|
}
|
|
|
|
# fleetd #528 — the four tests above pin drain_gate_refusal(), and that is ALL they pin: they call
|
|
# the predicate directly and never touch the main flow's call site. That was measured to be not
|
|
# enough, the same way test_should_swap_true_when_build_ran/test_should_swap_false_when_build_skipped
|
|
# were not enough for #521: with the main flow reading `die "$(drain_gate_refusal "$DO_BUILD"
|
|
# "$JAR_STAGED")"`, replacing that whole line with a flat `die "aborted — nothing changed"` left this
|
|
# suite at exit 0 with zero FAIL lines and byte-identical output to a clean run. Nothing above could
|
|
# tell the difference, because none of it calls anything at or above the call site itself.
|
|
#
|
|
# So these four call refuse_drain_gate() — the function the main flow actually calls, holding the
|
|
# composed message and the die() together — with die() stubbed to RECORD whether it was called and
|
|
# with what message, instead of exiting the process. That fails if refuse_drain_gate stops consulting
|
|
# drain_gate_refusal, mangles what it passes it, or simply never calls die.
|
|
#
|
|
# What none of these four can catch: deleting the `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` line
|
|
# from the main flow altogether — see the comment above refuse_drain_gate in redeploy-fleetd.sh for
|
|
# why no test in this file can do better than that (sourcing stops before the main flow runs).
|
|
DIED_CALLED=0
|
|
DIED_MESSAGE=""
|
|
stub_die_recorder() {
|
|
DIED_CALLED=0
|
|
DIED_MESSAGE=""
|
|
die() { DIED_CALLED=1; DIED_MESSAGE="$*"; }
|
|
}
|
|
|
|
test_refuse_drain_gate_build_ran_staged_present() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir staged
|
|
dir="$TMP/refuse-drain-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'staged jar bytes' > "$staged"
|
|
stub_die_recorder
|
|
refuse_drain_gate 1 "$staged"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate build-ran+staged-present must call die, and did not"
|
|
printf '%s' "$DIED_MESSAGE" | grep -qF "$staged" \
|
|
|| fail "refuse_drain_gate build-ran+staged-present die message does not name the staged jar"
|
|
printf '%s' "$DIED_MESSAGE" | grep -qF 'Rerun WITHOUT --no-build' \
|
|
|| fail "refuse_drain_gate build-ran+staged-present die message is missing the rerun instruction"
|
|
if printf '%s' "$DIED_MESSAGE" | grep -qF 'nothing changed'; then
|
|
fail "refuse_drain_gate build-ran+staged-present must not claim nothing changed — the jar already moved"
|
|
fi
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_refuse_drain_gate_build_ran_staged_absent() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir
|
|
dir="$TMP/refuse-drain-build-no-staged"; mkdir -p "$dir"
|
|
stub_die_recorder
|
|
refuse_drain_gate 1 "$dir/fleetd-new.jar"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate build-ran+staged-absent must call die, and did not"
|
|
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate build-ran+staged-absent die message"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_refuse_drain_gate_no_build_staged_present() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir staged
|
|
dir="$TMP/refuse-drain-no-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'leftover staged jar bytes' > "$staged"
|
|
stub_die_recorder
|
|
refuse_drain_gate 0 "$staged"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate no-build+staged-present must call die, and did not"
|
|
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-present die message"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_refuse_drain_gate_no_build_staged_absent() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir
|
|
dir="$TMP/refuse-drain-no-build-no-staged"; mkdir -p "$dir"
|
|
stub_die_recorder
|
|
refuse_drain_gate 0 "$dir/fleetd-new.jar"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate no-build+staged-absent must call die, and did not"
|
|
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-absent die message"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
# fleetd #528 — closes the one gap the four behavioural tests above cannot: they call
|
|
# refuse_drain_gate directly, and sourcing stops before the main flow ever runs (the SOURCED guard),
|
|
# so none of them can prove the main flow still CALLS refuse_drain_gate at all. Same shape as
|
|
# test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site. This is
|
|
# what actually kills the item-1 mutation from the ticket — replacing the main flow's call with a
|
|
# flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural
|
|
# tests above would even notice.
|
|
#
|
|
# The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail
|
|
# the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below
|
|
# ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body).
|
|
test_refuse_drain_gate_call_site_present() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
|
call_line="$(grep -Fn 'refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$call_line" ] \
|
|
|| fail "could not find the main flow's refuse_drain_gate call site in redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_no_errors() {
|
|
cat > "$TMP/no-errors.log" <<'LOG'
|
|
2026-09-05 12:00:00 INFO fleetd listening
|
|
LOG
|
|
classify_fixture no-errors.log
|
|
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained"
|
|
}
|
|
|
|
test_recovery_patterns_match_source() {
|
|
grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "AMQP failure pattern no longer matches source"
|
|
grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "reply-inbox recovery pattern no longer matches source"
|
|
grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \
|
|
|| fail "lead-mailbox recovery pattern no longer matches source"
|
|
}
|
|
|
|
test_attributed_recovered_connection_error() {
|
|
cat > "$TMP/attributed-recovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture attributed-recovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total"
|
|
assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained"
|
|
}
|
|
|
|
test_source_derived_error_shapes_recover_by_connection() {
|
|
# These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check
|
|
# after redeploy because the new code has not yet written a production line.
|
|
cat > "$TMP/source-derived.log" <<'LOG'
|
|
17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery!
|
|
17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!
|
|
17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture source-derived.log
|
|
assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total"
|
|
assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained"
|
|
}
|
|
|
|
test_cross_connection_unattributable_errors_stay_loud() {
|
|
# This candidate has neither stable connection name, so LeadMailbox recovery must not consume it.
|
|
cat > "$TMP/cross-unattributable.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-unattributable.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained"
|
|
}
|
|
|
|
test_attributed_cross_connection_errors_stay_loud() {
|
|
# LeadMailbox recovery cannot heal AmqpReplyInbox errors.
|
|
cat > "$TMP/cross-attributed.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-attributed.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained"
|
|
}
|
|
|
|
test_attributed_unrecovered_connection_error() {
|
|
cat > "$TMP/unrecovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
LOG
|
|
classify_fixture unrecovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained"
|
|
}
|
|
|
|
test_other_error_is_unexplained() {
|
|
cat > "$TMP/other-error.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed
|
|
2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture other-error.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained"
|
|
}
|
|
|
|
test_recovery_requirement_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then
|
|
fail "mutation accepted an unrecovered connection error"
|
|
fi
|
|
grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \
|
|
|| fail "mutation failed without the expected assertion"
|
|
printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n'
|
|
}
|
|
|
|
test_shared_counter_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line pending=0
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
case "$line" in
|
|
*'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*)
|
|
if [ "$pending" -gt 0 ]; then
|
|
pending=$((pending - 1))
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
fi
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending))
|
|
}
|
|
|
|
if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then
|
|
fail "shared counter mutation accepted cross-connection recovery"
|
|
fi
|
|
grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \
|
|
|| fail "shared counter mutation failed without the expected assertion"
|
|
printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
test_unattributable_quiet_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then
|
|
fail "unattributable mutation accepted an unknown connection"
|
|
fi
|
|
grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \
|
|
|| fail "unattributable mutation failed without the expected assertion"
|
|
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
test_detect_supervisor_launchd_only
|
|
test_detect_supervisor_systemd_only
|
|
test_detect_supervisor_none
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_systemd_probe_error_is_unclear
|
|
test_require_drivable_supervisor_refuses_ambiguous
|
|
test_require_drivable_supervisor_refuses_unclear
|
|
test_require_drivable_supervisor_accepts_known_kinds
|
|
test_count_daemon_pids
|
|
test_assert_single_daemon_accepts_one_pid
|
|
test_assert_single_daemon_rejects_two_pids
|
|
test_jar_id_defaults_to_live_and_reports_explicit_path
|
|
test_jar_id_reports_absent_for_missing_file
|
|
test_stage_built_jar_moves_off_live_path
|
|
test_stage_built_jar_dies_when_build_produced_nothing
|
|
test_swap_staged_jar_moves_staged_onto_live
|
|
test_swap_staged_jar_dies_without_staged_file
|
|
test_swap_staged_jar_dies_when_mv_fails
|
|
test_should_swap_true_when_build_ran
|
|
test_should_swap_false_when_build_skipped
|
|
test_swap_if_built_performs_the_swap_when_build_ran
|
|
test_swap_if_built_skips_the_swap_when_build_skipped
|
|
test_require_no_build_jar_dies_when_absent
|
|
test_require_no_build_jar_accepts_present_jar
|
|
test_wait_for_daemon_exit_returns_true_once_pid_clears
|
|
test_wait_for_daemon_exit_times_out_if_pid_never_clears
|
|
test_swap_ordered_after_wait_and_before_start
|
|
test_drain_gate_abort_message_says_no_no_build
|
|
test_drain_gate_refusal_build_ran_staged_present
|
|
test_drain_gate_refusal_build_ran_staged_absent
|
|
test_drain_gate_refusal_no_build_staged_present
|
|
test_drain_gate_refusal_no_build_staged_absent
|
|
test_refuse_drain_gate_build_ran_staged_present
|
|
test_refuse_drain_gate_build_ran_staged_absent
|
|
test_refuse_drain_gate_no_build_staged_present
|
|
test_refuse_drain_gate_no_build_staged_absent
|
|
test_refuse_drain_gate_call_site_present
|
|
test_no_errors
|
|
test_recovery_patterns_match_source
|
|
test_attributed_recovered_connection_error
|
|
test_source_derived_error_shapes_recover_by_connection
|
|
test_cross_connection_unattributable_errors_stay_loud
|
|
test_attributed_cross_connection_errors_stay_loud
|
|
test_attributed_unrecovered_connection_error
|
|
test_other_error_is_unexplained
|
|
test_recovery_requirement_mutation_is_caught
|
|
test_shared_counter_mutation_is_caught
|
|
test_unattributable_quiet_mutation_is_caught
|
|
printf 'PASS: redeploy log classifier\n'
|