08771e270b
The extraction in the previous commit did what #521 asked for — a should_swap() predicate with a test for each value — and I measured that it does not close the defect. With the main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left the whole suite at exit 0 with zero FAIL lines. The swap still never ran, and a redeploy would still report success while starting on no jar. That is my ticket's fault, not the implementer's: "extract the decision so the suite can call it" pins the decision and never the wiring. Extraction moved the untested decision up one level instead of removing it. Fix: the decision and the action now live together in swap_if_built(), which the main flow calls unconditionally — there is no guard left in the main flow to get wrong. should_swap() stays, because it is the decision and is worth naming and testing on its own. Two new tests call swap_if_built() with a recording stub in place of the real mv, so they fail if the guard is removed, inverted, or stops being consulted. Also fixed, found while verifying this: * test_swap_ordered_after_wait_and_before_start had to follow the call site to `swap_if_built "$DO_BUILD"`. Left on the old needle it reported "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)" — true of a function definition, and nothing about step order. * That test's three `[ -n ... ] || fail "could not find ... call site"` guards were dead code. Under `set -euo pipefail` an absent needle fails the assignment and `set -e` kills the suite before the guard runs. Measured: deleting the swap call gave exit 1 with ZERO bytes of output, no FAIL line, nothing naming what was missing. Each grep now ends in `|| true` so the assignment succeeds empty and the guard can speak. Verified by me on this revision: * suite exit 0, 0 `^FAIL:` lines, 44 tests defined and 44 invoked * bash -n rc=0 on both scripts under /bin/bash 3.2.57 and bash 5.3.9 * four mutations, each killed with its own named FAIL line, each restored byte-identical, green control after the battery: - guard removed inside swap_if_built -> "must not swap, but it did" - guard inverted -> "must perform the swap, and did not" - should_swap's comparison changed -> "must return true" - main-flow call deleted -> "could not find the swap call site in redeploy-fleetd.sh" (this one printed 0 bytes before the dead-guard fix, which is the before/after proof for it) Not fixed here, filed separately: drain_gate_refusal has the same shape. Replacing `die "$(drain_gate_refusal ...)"` with `die "aborted — nothing changed"` leaves the suite at exit 0 with output byte-identical to a clean run, which reinstates the exact wrong message #517 was filed to fix.
768 lines
41 KiB
Bash
Executable File
768 lines
41 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Self-contained checks for the pure log classifier in redeploy-fleetd.sh.
|
|
|
|
set -euo pipefail
|
|
|
|
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")"
|
|
trap 'rm -rf "$TMP"' EXIT
|
|
|
|
# Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
|
|
fail() {
|
|
printf 'FAIL: %s\n' "$*" >&2
|
|
return 1
|
|
}
|
|
|
|
assert_equals() {
|
|
local expected="$1" actual="$2" description="$3"
|
|
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
|
|
}
|
|
|
|
classify_fixture() {
|
|
local name="$1"
|
|
classify_amqp_connection_errors "$TMP/$name"
|
|
}
|
|
|
|
# fleetd #492 follow-up: detect_supervisor's stdout is now "kind<SEP>detail" (see the constraints
|
|
# comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about
|
|
# the kind must split it out with the SAME in-shell parameter expansion the real call site (:438)
|
|
# uses, never a bare string comparison against the raw output.
|
|
supervisor_kind_of() {
|
|
printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}"
|
|
}
|
|
supervisor_detail_of() {
|
|
printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}"
|
|
}
|
|
|
|
|
|
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
|
|
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
|
|
# being naturally absent: this machine may itself be running a real fleetd under launchd right now
|
|
# (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving
|
|
# launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state
|
|
# instead of the fixture.
|
|
test_detect_supervisor_launchd_only() {
|
|
launchd_installed() { return 0; }
|
|
launchd_loaded() { return 0; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_systemd_only() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 0; }
|
|
assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_none() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection"
|
|
}
|
|
|
|
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
|
|
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
|
|
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
|
|
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
|
|
# auto-restart too, and every one of those is a host that IS under systemd.
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; } # the unit file IS there
|
|
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
|
|
local raw
|
|
raw="$(detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail does not name the systemd unit it found installed-but-not-loaded"
|
|
}
|
|
|
|
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
|
|
# removed, or about to be reloaded) must not read as "no supervisor" either.
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 0; } # the plist IS there
|
|
launchd_loaded() { return 1; } # launchctl list says not loaded
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
local raw
|
|
raw="$(detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \
|
|
|| fail "detail does not name the launchd label it found installed-but-not-loaded"
|
|
}
|
|
|
|
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
|
|
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
|
|
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
|
|
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
|
|
# the same fact as "no supervisor is loaded".
|
|
test_detect_supervisor_systemd_probe_error_is_unclear() {
|
|
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
|
|
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
|
|
# of those tests), and a bash function definition is global for the rest of the process — without
|
|
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
|
|
# a real `systemctl` call at all.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local bin_dir result rc=0
|
|
bin_dir="$TMP/stub-bin-systemctl-errors"
|
|
mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/systemctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "Failed to connect to bus: No such file or directory" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/systemctl"
|
|
|
|
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
|
|
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
|
|
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$result")" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail does not name the systemd unit whose probe errored"
|
|
}
|
|
|
|
# fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a
|
|
# hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not
|
|
# the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real
|
|
# caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout
|
|
# survives that boundary, so kind AND detail must both cross on it — this test proves they do.
|
|
test_require_drivable_supervisor_refuses_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 1; }
|
|
|
|
local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0
|
|
# Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand.
|
|
SUPERVISOR_RAW="$(detect_supervisor)"
|
|
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
|
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
|
|
|
assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal"
|
|
[ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \
|
|
|| fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell"
|
|
printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded"
|
|
|
|
output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
|
|
# Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate
|
|
# recovery instructions name the unit unconditionally either way ("systemctl --user status
|
|
# $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only
|
|
# the specific detail string proves the crossed value, not the boilerplate, reached the message.
|
|
printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \
|
|
|| fail "refusal message does not contain the specific detail that crossed the subshell boundary"
|
|
}
|
|
|
|
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
|
|
# `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that
|
|
# forks a subshell, so its exit() only ends the subshell and this test script keeps running under
|
|
# `set -e`.
|
|
test_require_drivable_supervisor_refuses_ambiguous() {
|
|
local output rc=0
|
|
output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor"
|
|
printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \
|
|
|| fail "refusal message does not name the launchd label it found"
|
|
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "refusal message does not name the systemd unit it found"
|
|
}
|
|
|
|
test_require_drivable_supervisor_accepts_known_kinds() {
|
|
require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor"
|
|
require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor"
|
|
require_drivable_supervisor "none" || fail "refused the unsupervised case"
|
|
}
|
|
|
|
# fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor
|
|
# produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see
|
|
# it because either daemon alone satisfies them.
|
|
test_count_daemon_pids() {
|
|
assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list"
|
|
assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid"
|
|
assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids"
|
|
}
|
|
|
|
test_assert_single_daemon_accepts_one_pid() {
|
|
assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid"
|
|
}
|
|
|
|
test_assert_single_daemon_rejects_two_pids() {
|
|
local output rc=0
|
|
output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids"
|
|
printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found"
|
|
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
|
|
}
|
|
|
|
# fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports
|
|
# $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call
|
|
# must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR.
|
|
# Two files with different content, so a default pointed at the wrong one reports the wrong hash
|
|
# rather than accidentally matching.
|
|
test_jar_id_defaults_to_live_and_reports_explicit_path() {
|
|
local dir saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
local live_hash staged_hash default_result explicit_result
|
|
dir="$TMP/jar-id"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
|
printf 'live jar bytes' > "$JAR"
|
|
printf 'staged jar bytes, not the same content' > "$JAR_STAGED"
|
|
live_hash="$(shasum -a 256 "$JAR" | cut -c1-12)"
|
|
staged_hash="$(shasum -a 256 "$JAR_STAGED" | cut -c1-12)"
|
|
default_result="$(jar_id)"
|
|
explicit_result="$(jar_id "$JAR_STAGED")"
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same"
|
|
assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR"
|
|
assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR"
|
|
}
|
|
|
|
# fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511)
|
|
# proves both halves of the present-file contract but never exercises the missing-file path. This
|
|
# word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean`
|
|
# deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points
|
|
# operators at `--check` for exactly this. Covers both the no-argument default and an explicit path,
|
|
# since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both.
|
|
test_jar_id_reports_absent_for_missing_file() {
|
|
local saved_jar="$JAR" dir default_result explicit_result
|
|
dir="$TMP/jar-id-absent"; mkdir -p "$dir"
|
|
JAR="$dir/does-not-exist.jar"
|
|
[ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR"
|
|
default_result="$(jar_id)"
|
|
explicit_result="$(jar_id "$dir/also-does-not-exist.jar")"
|
|
JAR="$saved_jar"
|
|
assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist"
|
|
assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent"
|
|
}
|
|
|
|
# fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar
|
|
# are exercised directly against real files on disk (not stubs), because the whole point is file
|
|
# behavior (does the content move, does the source disappear, does a failure leave both sides
|
|
# intact) that a stubbed function cannot prove.
|
|
test_stage_built_jar_moves_off_live_path() {
|
|
local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
dir="$TMP/stage-ok"; mkdir -p "$dir"
|
|
jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar"
|
|
printf 'built jar bytes' > "$jar"
|
|
JAR="$jar"; JAR_STAGED="$staged"
|
|
stage_built_jar || fail "stage_built_jar rejected a real build output"
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar"
|
|
[ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged"
|
|
grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content"
|
|
}
|
|
|
|
test_stage_built_jar_dies_when_build_produced_nothing() {
|
|
local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
dir="$TMP/stage-missing"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
|
output="$(stage_built_jar 2>&1)" || rc=$?
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output"
|
|
printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \
|
|
|| fail "refusal message does not name the missing jar path"
|
|
}
|
|
|
|
test_swap_staged_jar_moves_staged_onto_live() {
|
|
local dir staged live
|
|
dir="$TMP/swap-ok"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar"
|
|
printf 'swapped jar bytes' > "$staged"
|
|
swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar"
|
|
[ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged"
|
|
[ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live"
|
|
grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content"
|
|
}
|
|
|
|
# The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on
|
|
# failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes
|
|
# inside a command substitution to contain that exit to a subshell.
|
|
test_swap_staged_jar_dies_without_staged_file() {
|
|
local dir output rc=0
|
|
dir="$TMP/swap-missing"; mkdir -p "$dir"
|
|
output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar"
|
|
[ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged"
|
|
printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \
|
|
|| fail "refusal message does not name the missing staged path"
|
|
}
|
|
|
|
test_swap_staged_jar_dies_when_mv_fails() {
|
|
local dir staged live output rc=0
|
|
dir="$TMP/swap-fail"; mkdir -p "$dir/src"
|
|
staged="$dir/src/fleetd-new.jar"
|
|
printf 'fake jar bytes' > "$staged"
|
|
live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails
|
|
output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv"
|
|
[ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails"
|
|
[ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed"
|
|
printf '%s' "$output" | grep -qF "$staged" \
|
|
|| fail "refusal message does not name the staged path that could not be moved"
|
|
}
|
|
|
|
# --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this
|
|
# path) and must still die with the exact wording documented in the script's own header comment.
|
|
test_require_no_build_jar_dies_when_absent() {
|
|
local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar"
|
|
JAR="$missing"
|
|
output="$(require_no_build_jar 2>&1)" || rc=$?
|
|
JAR="$saved_jar"
|
|
[ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar"
|
|
printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \
|
|
|| fail "refusal message does not match the documented --no-build wording"
|
|
}
|
|
|
|
test_require_no_build_jar_accepts_present_jar() {
|
|
local saved_jar="$JAR" dir
|
|
dir="$TMP/no-build-present"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"
|
|
printf 'existing jar' > "$JAR"
|
|
require_no_build_jar || fail "require_no_build_jar rejected an existing jar"
|
|
JAR="$saved_jar"
|
|
}
|
|
|
|
# wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while
|
|
# running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so
|
|
# the timeout-loop test does not actually wait out its budget.
|
|
test_wait_for_daemon_exit_returns_true_once_pid_clears() {
|
|
# running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so
|
|
# a plain shell variable it increments would reset on each call instead of accumulating. Count in
|
|
# a file instead, which is the one thing that actually survives across those subshells.
|
|
local counter_file="$TMP/wait-exit-calls" final_calls
|
|
printf '0' > "$counter_file"
|
|
running_pid() {
|
|
local n
|
|
n="$(cat "$counter_file")"
|
|
n=$((n + 1))
|
|
printf '%s' "$n" > "$counter_file"
|
|
if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi
|
|
}
|
|
sleep() { :; }
|
|
wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared"
|
|
final_calls="$(cat "$counter_file")"
|
|
[ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
|
}
|
|
|
|
test_wait_for_daemon_exit_times_out_if_pid_never_clears() {
|
|
local rc=0
|
|
running_pid() { printf '4242'; }
|
|
sleep() { :; }
|
|
wait_for_daemon_exit 3 || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
|
}
|
|
|
|
# fleetd #521 — the swap step's guard, at two levels.
|
|
#
|
|
# The first two tests call the predicate should_swap() directly. They pin its logic, and that is all
|
|
# they pin. On their own they did NOT close #521, and this was measured rather than argued: with the
|
|
# main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left
|
|
# this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs
|
|
# the swap consult the predicate at all. Extracting the decision had moved the untested decision up
|
|
# a level, not removed it.
|
|
#
|
|
# So the last two tests call swap_if_built() — the function the main flow actually calls, holding the
|
|
# guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the
|
|
# guard is removed, inverted, or stops being consulted.
|
|
#
|
|
# What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow
|
|
# altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing
|
|
# stops before the main flow runs, so no test in this file can invoke it.
|
|
test_should_swap_true_when_build_ran() {
|
|
should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true"
|
|
}
|
|
|
|
test_should_swap_false_when_build_skipped() {
|
|
if should_swap 0; then
|
|
fail "should_swap 0 (--no-build; nothing was staged this run) must return false"
|
|
fi
|
|
}
|
|
|
|
# Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is
|
|
# global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id
|
|
# overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear),
|
|
# and again at the END, so their own stubs do not leak into every test that runs after them.
|
|
test_swap_if_built_performs_the_swap_when_build_ran() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local marker="$TMP/swap-if-built-ran"
|
|
rm -f "$marker"
|
|
swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; }
|
|
jar_id() { printf 'stubbed\n'; }
|
|
swap_if_built 1 > /dev/null
|
|
[ -f "$marker" ] \
|
|
|| fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_swap_if_built_skips_the_swap_when_build_skipped() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local marker="$TMP/swap-if-built-skipped"
|
|
rm -f "$marker"
|
|
swap_staged_jar() { printf 'swapped\n' > "$marker"; }
|
|
jar_id() { printf 'stubbed\n'; }
|
|
swap_if_built 0 > /dev/null
|
|
if [ -f "$marker" ]; then
|
|
fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did"
|
|
fi
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
# fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the
|
|
# main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee
|
|
# itself — as opposed to the pure functions it's built from — can only be checked by reading the
|
|
# script's own call sites, the same way test_recovery_patterns_match_source below checks Java
|
|
# source shape instead of behavior it cannot invoke directly.
|
|
#
|
|
# Two details about the three greps below, both of which have already gone wrong here.
|
|
#
|
|
# The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not
|
|
# `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside
|
|
# swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using
|
|
# it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)"
|
|
# — a true statement about a function definition, and nothing at all about the order of the steps.
|
|
#
|
|
# Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the
|
|
# pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set
|
|
# -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to
|
|
# report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no
|
|
# name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak.
|
|
test_swap_ordered_after_wait_and_before_start() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line
|
|
wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh"
|
|
[ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh"
|
|
[ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh"
|
|
[ "$swap_line" -gt "$wait_line" ] \
|
|
|| fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)"
|
|
[ "$swap_line" -lt "$start_line" ] \
|
|
|| fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)"
|
|
}
|
|
|
|
# fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator
|
|
# declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)"
|
|
# to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has
|
|
# already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal
|
|
# ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start
|
|
# above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow),
|
|
# so the only way to pin its exact wording is to read the source.
|
|
test_drain_gate_abort_message_says_no_no_build() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" msg
|
|
msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")"
|
|
[ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh"
|
|
if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then
|
|
fail "abort message still claims a rerun WITH --no-build can finish the restart"
|
|
fi
|
|
printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \
|
|
|| fail "abort message does not tell the operator to rerun without --no-build"
|
|
printf '%s' "$msg" | grep -qF 'no longer at the live path' \
|
|
|| fail "abort message does not say why --no-build cannot finish the restart"
|
|
}
|
|
|
|
# fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was
|
|
# a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's
|
|
# own file for the wording, which stays in the file even if the `if` guarding it is mutated to
|
|
# `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead,
|
|
# so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT
|
|
# alongside these, not replaced, because it catches a different regression (a re-wording that still
|
|
# reaches the right branch would not change which case fires here, but would still be worth pinning).
|
|
test_drain_gate_refusal_build_ran_staged_present() {
|
|
local dir staged result
|
|
dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'staged jar bytes' > "$staged"
|
|
result="$(drain_gate_refusal 1 "$staged")"
|
|
printf '%s' "$result" | grep -qF "$staged" \
|
|
|| fail "build-ran+staged-present refusal does not name the staged jar path"
|
|
printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \
|
|
|| fail "build-ran+staged-present refusal does not tell the operator how to finish the restart"
|
|
if printf '%s' "$result" | grep -qF 'nothing changed'; then
|
|
fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved"
|
|
fi
|
|
}
|
|
|
|
test_drain_gate_refusal_build_ran_staged_absent() {
|
|
local dir result
|
|
dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir"
|
|
result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")"
|
|
assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording"
|
|
}
|
|
|
|
# --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar
|
|
# found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the
|
|
# comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning.
|
|
test_drain_gate_refusal_no_build_staged_present() {
|
|
local dir staged result
|
|
dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'leftover staged jar bytes' > "$staged"
|
|
result="$(drain_gate_refusal 0 "$staged")"
|
|
assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed"
|
|
}
|
|
|
|
test_drain_gate_refusal_no_build_staged_absent() {
|
|
local dir result
|
|
dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir"
|
|
result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")"
|
|
assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording"
|
|
}
|
|
|
|
test_no_errors() {
|
|
cat > "$TMP/no-errors.log" <<'LOG'
|
|
2026-09-05 12:00:00 INFO fleetd listening
|
|
LOG
|
|
classify_fixture no-errors.log
|
|
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained"
|
|
}
|
|
|
|
test_recovery_patterns_match_source() {
|
|
grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "AMQP failure pattern no longer matches source"
|
|
grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "reply-inbox recovery pattern no longer matches source"
|
|
grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \
|
|
|| fail "lead-mailbox recovery pattern no longer matches source"
|
|
}
|
|
|
|
test_attributed_recovered_connection_error() {
|
|
cat > "$TMP/attributed-recovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture attributed-recovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total"
|
|
assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained"
|
|
}
|
|
|
|
test_source_derived_error_shapes_recover_by_connection() {
|
|
# These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check
|
|
# after redeploy because the new code has not yet written a production line.
|
|
cat > "$TMP/source-derived.log" <<'LOG'
|
|
17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery!
|
|
17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!
|
|
17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture source-derived.log
|
|
assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total"
|
|
assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained"
|
|
}
|
|
|
|
test_cross_connection_unattributable_errors_stay_loud() {
|
|
# This candidate has neither stable connection name, so LeadMailbox recovery must not consume it.
|
|
cat > "$TMP/cross-unattributable.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-unattributable.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained"
|
|
}
|
|
|
|
test_attributed_cross_connection_errors_stay_loud() {
|
|
# LeadMailbox recovery cannot heal AmqpReplyInbox errors.
|
|
cat > "$TMP/cross-attributed.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-attributed.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained"
|
|
}
|
|
|
|
test_attributed_unrecovered_connection_error() {
|
|
cat > "$TMP/unrecovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
LOG
|
|
classify_fixture unrecovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained"
|
|
}
|
|
|
|
test_other_error_is_unexplained() {
|
|
cat > "$TMP/other-error.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed
|
|
2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture other-error.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained"
|
|
}
|
|
|
|
test_recovery_requirement_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then
|
|
fail "mutation accepted an unrecovered connection error"
|
|
fi
|
|
grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \
|
|
|| fail "mutation failed without the expected assertion"
|
|
printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n'
|
|
}
|
|
|
|
test_shared_counter_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line pending=0
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
case "$line" in
|
|
*'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*)
|
|
if [ "$pending" -gt 0 ]; then
|
|
pending=$((pending - 1))
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
fi
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending))
|
|
}
|
|
|
|
if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then
|
|
fail "shared counter mutation accepted cross-connection recovery"
|
|
fi
|
|
grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \
|
|
|| fail "shared counter mutation failed without the expected assertion"
|
|
printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
test_unattributable_quiet_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then
|
|
fail "unattributable mutation accepted an unknown connection"
|
|
fi
|
|
grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \
|
|
|| fail "unattributable mutation failed without the expected assertion"
|
|
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
test_detect_supervisor_launchd_only
|
|
test_detect_supervisor_systemd_only
|
|
test_detect_supervisor_none
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_systemd_probe_error_is_unclear
|
|
test_require_drivable_supervisor_refuses_ambiguous
|
|
test_require_drivable_supervisor_refuses_unclear
|
|
test_require_drivable_supervisor_accepts_known_kinds
|
|
test_count_daemon_pids
|
|
test_assert_single_daemon_accepts_one_pid
|
|
test_assert_single_daemon_rejects_two_pids
|
|
test_jar_id_defaults_to_live_and_reports_explicit_path
|
|
test_jar_id_reports_absent_for_missing_file
|
|
test_stage_built_jar_moves_off_live_path
|
|
test_stage_built_jar_dies_when_build_produced_nothing
|
|
test_swap_staged_jar_moves_staged_onto_live
|
|
test_swap_staged_jar_dies_without_staged_file
|
|
test_swap_staged_jar_dies_when_mv_fails
|
|
test_should_swap_true_when_build_ran
|
|
test_should_swap_false_when_build_skipped
|
|
test_swap_if_built_performs_the_swap_when_build_ran
|
|
test_swap_if_built_skips_the_swap_when_build_skipped
|
|
test_require_no_build_jar_dies_when_absent
|
|
test_require_no_build_jar_accepts_present_jar
|
|
test_wait_for_daemon_exit_returns_true_once_pid_clears
|
|
test_wait_for_daemon_exit_times_out_if_pid_never_clears
|
|
test_swap_ordered_after_wait_and_before_start
|
|
test_drain_gate_abort_message_says_no_no_build
|
|
test_drain_gate_refusal_build_ran_staged_present
|
|
test_drain_gate_refusal_build_ran_staged_absent
|
|
test_drain_gate_refusal_no_build_staged_present
|
|
test_drain_gate_refusal_no_build_staged_absent
|
|
test_no_errors
|
|
test_recovery_patterns_match_source
|
|
test_attributed_recovered_connection_error
|
|
test_source_derived_error_shapes_recover_by_connection
|
|
test_cross_connection_unattributable_errors_stay_loud
|
|
test_attributed_cross_connection_errors_stay_loud
|
|
test_attributed_unrecovered_connection_error
|
|
test_other_error_is_unexplained
|
|
test_recovery_requirement_mutation_is_caught
|
|
test_shared_counter_mutation_is_caught
|
|
test_unattributable_quiet_mutation_is_caught
|
|
printf 'PASS: redeploy log classifier\n'
|