b8182c96c2
test_jar_id_defaults_to_live_and_reports_explicit_path's reference hash is computed by calling hash256 itself (needed so it doesn't call the Linux-crashing bare shasum directly). That made subject and reference the same instrument: they agree no matter which algorithm hash256 actually runs, so a mutation swapping both of hash256's arms for the wrong algorithm was invisible to the suite. Adds test_hash256_computes_a_real_sha256, pinned against the published SHA-256 test vector for the 3-byte input "abc" (ba7816bf8f01...), written as a literal constant rather than computed by any hasher at test time. Verified the constant myself both ways (sha256sum and shasum -a 256) before writing it in.
1314 lines
72 KiB
Bash
Executable File
1314 lines
72 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Self-contained checks for the pure log classifier in redeploy-fleetd.sh.
|
|
|
|
set -euo pipefail
|
|
|
|
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")"
|
|
trap 'rm -rf "$TMP"' EXIT
|
|
|
|
# Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
|
|
fail() {
|
|
printf 'FAIL: %s\n' "$*" >&2
|
|
return 1
|
|
}
|
|
|
|
assert_equals() {
|
|
local expected="$1" actual="$2" description="$3"
|
|
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
|
|
}
|
|
|
|
classify_fixture() {
|
|
local name="$1"
|
|
classify_amqp_connection_errors "$TMP/$name"
|
|
}
|
|
|
|
# fleetd #492 follow-up: detect_supervisor's stdout is now "kind<SEP>detail" (see the constraints
|
|
# comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about
|
|
# the kind must split it out with the SAME in-shell parameter expansion the real call site (:438)
|
|
# uses, never a bare string comparison against the raw output.
|
|
supervisor_kind_of() {
|
|
printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}"
|
|
}
|
|
supervisor_detail_of() {
|
|
printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}"
|
|
}
|
|
|
|
|
|
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
|
|
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
|
|
# being naturally absent: this machine may itself be running a real fleetd under launchd right now
|
|
# (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving
|
|
# launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state
|
|
# instead of the fixture.
|
|
test_detect_supervisor_launchd_only() {
|
|
launchd_installed() { return 0; }
|
|
launchd_loaded() { return 0; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_systemd_only() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 0; }
|
|
assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_none() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection"
|
|
}
|
|
|
|
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
|
|
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
|
|
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
|
|
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
|
|
# auto-restart too, and every one of those is a host that IS under systemd.
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; } # the unit file IS there
|
|
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
|
|
local raw
|
|
raw="$(detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail does not name the systemd unit it found installed-but-not-loaded"
|
|
}
|
|
|
|
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
|
|
# removed, or about to be reloaded) must not read as "no supervisor" either.
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 0; } # the plist IS there
|
|
launchd_loaded() { return 1; } # launchctl list says not loaded
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
local raw
|
|
raw="$(detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none"
|
|
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \
|
|
|| fail "detail does not name the launchd label it found installed-but-not-loaded"
|
|
}
|
|
|
|
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
|
|
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
|
|
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
|
|
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
|
|
# the same fact as "no supervisor is loaded".
|
|
test_detect_supervisor_systemd_probe_error_is_unclear() {
|
|
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
|
|
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
|
|
# of those tests), and a bash function definition is global for the rest of the process — without
|
|
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
|
|
# a real `systemctl` call at all.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local bin_dir result rc=0
|
|
bin_dir="$TMP/stub-bin-systemctl-errors"
|
|
mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/systemctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "Failed to connect to bus: No such file or directory" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/systemctl"
|
|
|
|
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
|
|
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
|
|
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none"
|
|
local detail
|
|
detail="$(supervisor_detail_of "$result")"
|
|
printf '%s' "$detail" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail does not name the systemd unit whose probe errored"
|
|
# fleetd #545: this is the PROBE-RAN-AND-ANSWERED-BADLY case (systemctl actually executed and
|
|
# wrote to stderr) — it must carry that story and never the SET-UP-FAILED story (mktemp never
|
|
# even ran here), or the two "unclear" causes have collapsed back into one message that asserts a
|
|
# cause it did not measure, which is the exact defect this ticket exists to fix.
|
|
printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \
|
|
|| fail "detail does not say systemctl ran and answered with stderr: $detail"
|
|
printf '%s' "$detail" | grep -qF "could not even be set up" \
|
|
&& fail "detail wrongly claims the probe could not be set up, but systemctl actually ran and answered on stderr: $detail"
|
|
return 0
|
|
}
|
|
|
|
# fleetd #545 — the companion case to the probe-error test above: here `mktemp` itself fails
|
|
# (whatever the reason — the historical bug was a GNU-mktemp-rejects-a-template-with-no-Xs case,
|
|
# but this stub simulates ANY reason the probe's own stderr-capture temp file cannot be created,
|
|
# e.g. a full or unwritable temp dir) and `systemctl` is never invoked at all. Before this ticket,
|
|
# this collapsed into the SAME "systemctl exited non-zero and reported an error on stderr" detail
|
|
# as the sibling test above, which asserts a cause (systemctl ran and answered badly) that was
|
|
# never measured, because systemctl never ran. This proves the SET-UP-FAILED detail is distinct and
|
|
# does not claim systemctl said anything.
|
|
test_detect_supervisor_systemd_probe_setup_failure_is_unclear() {
|
|
# Re-source first for the same reason test_detect_supervisor_systemd_probe_error_is_unclear does:
|
|
# restore the REAL probe bodies before driving them through a stub PATH.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local bin_dir result rc=0
|
|
bin_dir="$TMP/stub-bin-mktemp-fails"
|
|
mkdir -p "$bin_dir"
|
|
# A systemctl stub that would fail loudly if it were ever actually invoked — proves the mktemp
|
|
# failure short-circuits the probe before systemctl runs, not merely that this test forgot to
|
|
# supply a working systemctl.
|
|
cat > "$bin_dir/systemctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "systemctl must never run when mktemp already failed" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/systemctl"
|
|
cat > "$bin_dir/mktemp" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "mktemp: cannot create temp file" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/mktemp"
|
|
|
|
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
|
|
[ "$rc" -ne 0 ] \
|
|
|| fail "systemd_loaded must not report loaded=true when its own mktemp setup failed"
|
|
assert_equals "2" "$SYSTEMD_LOADED_ERRORED" \
|
|
"systemd_loaded must flag a SETUP failure (2), distinct from a probe-answered-with-stderr failure (1)"
|
|
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
|
|
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe setup failure must read as unclear, not none"
|
|
local detail
|
|
detail="$(supervisor_detail_of "$result")"
|
|
printf '%s' "$detail" | grep -qF "could not even be set up" \
|
|
|| fail "detail does not say the probe could not be SET UP: $detail"
|
|
printf '%s' "$detail" | grep -qF "systemctl exited non-zero and reported an error on stderr" \
|
|
&& fail "detail wrongly asserts systemctl exited non-zero and reported an error on stderr, but systemctl was never run: $detail"
|
|
return 0
|
|
}
|
|
|
|
# fleetd #545 — source-text check: every `mktemp -t` template in redeploy-fleetd.sh must contain an
|
|
# `X` placeholder. BSD mktemp (macOS) tolerates a bare template with no `X`s and just appends its
|
|
# own random suffix, which is exactly why six such sites survived undetected here — GNU mktemp
|
|
# (every Linux distribution) refuses a template with fewer than three `X`s and exits non-zero. There
|
|
# is no BSD-vs-GNU seam to stub on this Mac, so this is a source-text check rather than a
|
|
# behavioural one, the same shape as test_refuse_drain_gate_call_site_present above. Anchored on
|
|
# `mktemp -t ` (with the trailing space) so it inspects only the `-t`-style templates this ticket is
|
|
# about, never the `mktemp -d` calls this file and test-probe-member-credentials.sh already use
|
|
# (both already carry their own `XXXXXX` and are a different mktemp mode entirely).
|
|
test_mktemp_dash_t_templates_have_x_placeholders() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" bad
|
|
bad="$(grep -n 'mktemp -t ' "$src" | grep -v 'XXX' || true)"
|
|
[ -z "$bad" ] \
|
|
|| fail "mktemp -t template(s) with no X placeholder (fails under GNU coreutils): $bad"
|
|
}
|
|
|
|
# fleetd #550 — the shape, not the named lines: #545 showed the exact same failure mode (a
|
|
# macOS-only idiom used with no portable fallback) spread from two sites to six across 91 commits
|
|
# before anyone tested the SHAPE rather than specific lines. This is the shasum sibling: any script
|
|
# under scripts/ that actually INVOKES the macOS-only hasher to compute a hash (as opposed to
|
|
# merely probing whether it exists with `command -v`, or mentioning it in prose) must also check
|
|
# for the portable one first in that same file — the prefer-portable-fall-back-to-macOS-only idiom
|
|
# probe-member-credentials.sh:273-279 and this ticket's own hash256 helper both follow.
|
|
#
|
|
# The needle is built from two concatenated pieces, deliberately never written as one literal
|
|
# string in this file: written whole, it would match THIS CHECK'S OWN source line once the loop
|
|
# below reaches this very file, and the check would then "pass" by matching itself rather than any
|
|
# real invocation elsewhere — a zero-findings result indistinguishable from a clean file.
|
|
test_no_unguarded_macos_only_hasher_calls() {
|
|
local needle f bad="" usage
|
|
needle='shasum'; needle="$needle -a"
|
|
for f in "$ROOT"/scripts/*.sh; do
|
|
[ -f "$f" ] || continue
|
|
usage="$(grep -Fn "$needle" "$f" || true)"
|
|
if [ -n "$usage" ]; then
|
|
grep -q 'command -v sha256sum' "$f" \
|
|
|| bad="$bad$(basename "$f") "
|
|
fi
|
|
done
|
|
[ -z "$bad" ] \
|
|
|| fail "script(s) invoke the macOS-only hasher with no portable-hasher-first fallback guard in the same file: $bad"
|
|
}
|
|
|
|
# fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a
|
|
# hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not
|
|
# the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real
|
|
# caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout
|
|
# survives that boundary, so kind AND detail must both cross on it — this test proves they do.
|
|
test_require_drivable_supervisor_refuses_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 1; }
|
|
|
|
local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0
|
|
# Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand.
|
|
SUPERVISOR_RAW="$(detect_supervisor)"
|
|
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
|
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
|
|
|
assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal"
|
|
[ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \
|
|
|| fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell"
|
|
printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded"
|
|
|
|
output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
|
|
# Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate
|
|
# recovery instructions name the unit unconditionally either way ("systemctl --user status
|
|
# $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only
|
|
# the specific detail string proves the crossed value, not the boilerplate, reached the message.
|
|
printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \
|
|
|| fail "refusal message does not contain the specific detail that crossed the subshell boundary"
|
|
}
|
|
|
|
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
|
|
# `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that
|
|
# forks a subshell, so its exit() only ends the subshell and this test script keeps running under
|
|
# `set -e`.
|
|
test_require_drivable_supervisor_refuses_ambiguous() {
|
|
local output rc=0
|
|
output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor"
|
|
printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \
|
|
|| fail "refusal message does not name the launchd label it found"
|
|
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "refusal message does not name the systemd unit it found"
|
|
}
|
|
|
|
test_require_drivable_supervisor_accepts_known_kinds() {
|
|
require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor"
|
|
require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor"
|
|
require_drivable_supervisor "none" || fail "refused the unsupervised case"
|
|
}
|
|
|
|
# fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor
|
|
# produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see
|
|
# it because either daemon alone satisfies them.
|
|
test_count_daemon_pids() {
|
|
assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list"
|
|
assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid"
|
|
assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids"
|
|
}
|
|
|
|
test_assert_single_daemon_accepts_one_pid() {
|
|
assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid"
|
|
}
|
|
|
|
test_assert_single_daemon_rejects_two_pids() {
|
|
local output rc=0
|
|
output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids"
|
|
printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found"
|
|
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
|
|
}
|
|
|
|
# fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports
|
|
# $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call
|
|
# must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR.
|
|
# Two files with different content, so a default pointed at the wrong one reports the wrong hash
|
|
# rather than accidentally matching.
|
|
test_jar_id_defaults_to_live_and_reports_explicit_path() {
|
|
local dir saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
local live_hash staged_hash default_result explicit_result
|
|
dir="$TMP/jar-id"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
|
printf 'live jar bytes' > "$JAR"
|
|
printf 'staged jar bytes, not the same content' > "$JAR_STAGED"
|
|
# fleetd #550: this reference hash must be computed the same portable way jar_id() itself now
|
|
# computes one — a bare, unguarded call to the macOS-only hasher here was exactly the item-2
|
|
# defect, dying with "command not found" on any Linux runner that has no such hasher at all.
|
|
live_hash="$(hash256 "$JAR")"
|
|
staged_hash="$(hash256 "$JAR_STAGED")"
|
|
default_result="$(jar_id)"
|
|
explicit_result="$(jar_id "$JAR_STAGED")"
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same"
|
|
assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR"
|
|
assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR"
|
|
}
|
|
|
|
# fleetd #550 — closes a gap the test above leaves open. That test's own reference hash is now ALSO
|
|
# computed by calling hash256 (needed for item 2: the old bare macOS-only-hasher call there was the
|
|
# Linux crash), so its subject (jar_id, via hash256) and its reference (also hash256) share one
|
|
# instrument — they agree no matter which algorithm hash256 actually runs, so a mutation that swaps
|
|
# BOTH of hash256's arms for the wrong algorithm is invisible to it. This test's expected value
|
|
# comes from neither hasher: it is the published SHA-256 test vector for the 3-byte input "abc"
|
|
# (no trailing newline), written here as a literal constant, so it can still tell "hashed
|
|
# correctly" from "hashed, just with the wrong algorithm" — which is what this whole ticket is
|
|
# about.
|
|
test_hash256_computes_a_real_sha256() {
|
|
local dir f result
|
|
dir="$TMP/hash256-known-vector"; mkdir -p "$dir"
|
|
f="$dir/abc.txt"
|
|
printf 'abc' > "$f"
|
|
result="$(hash256 "$f")"
|
|
# SHA-256("abc") = ba7816bf8f01cfea414140de5dae2223b00361a396177a9cb410ff61f20015ad, the standard
|
|
# FIPS 180 test vector — first 12 hex chars, matching hash256's own `cut -c1-12`.
|
|
assert_equals "ba7816bf8f01" "$result" "hash256 of the literal 3-byte input 'abc' must be the known SHA-256 prefix, not some other algorithm's"
|
|
}
|
|
|
|
# fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511)
|
|
# proves both halves of the present-file contract but never exercises the missing-file path. This
|
|
# word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean`
|
|
# deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points
|
|
# operators at `--check` for exactly this. Covers both the no-argument default and an explicit path,
|
|
# since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both.
|
|
test_jar_id_reports_absent_for_missing_file() {
|
|
local saved_jar="$JAR" dir default_result explicit_result
|
|
dir="$TMP/jar-id-absent"; mkdir -p "$dir"
|
|
JAR="$dir/does-not-exist.jar"
|
|
[ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR"
|
|
default_result="$(jar_id)"
|
|
explicit_result="$(jar_id "$dir/also-does-not-exist.jar")"
|
|
JAR="$saved_jar"
|
|
assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist"
|
|
assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent"
|
|
}
|
|
|
|
# fleetd #550 — the whole point of this ticket: a jar that IS there but could not be hashed must
|
|
# never read the same as a jar that is not there at all. Drives the REAL hash256/jar_id bodies
|
|
# (never stubbed) through a stub PATH that contains neither of the two hashers this script knows —
|
|
# same technique test_detect_supervisor_systemd_probe_setup_failure_is_unclear uses for `mktemp`,
|
|
# except here the stub directory is used to REPLACE PATH rather than prepend to it, because the
|
|
# point is to make BOTH hashers unreachable, not to intercept one specific command while leaving
|
|
# everything else on the real PATH reachable. `[ -f ... ]` and the shell's own `command`/`echo`
|
|
# builtins need no PATH at all, so this is safe even with PATH reduced to an empty directory.
|
|
test_jar_id_reports_unhashable_when_no_hasher_on_path() {
|
|
local bin_dir dir saved_jar="$JAR" default_result explicit_result
|
|
bin_dir="$TMP/stub-bin-no-hasher"; mkdir -p "$bin_dir"
|
|
dir="$TMP/jar-id-no-hasher"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"
|
|
printf 'a real jar that exists but nothing here can hash' > "$JAR"
|
|
[ -f "$JAR" ] || fail "test fixture error: \$JAR does not exist at $JAR"
|
|
default_result="$(PATH="$bin_dir" jar_id)"
|
|
explicit_result="$(PATH="$bin_dir" jar_id "$JAR")"
|
|
JAR="$saved_jar"
|
|
[ "$default_result" != "absent" ] \
|
|
|| fail "jar_id reported absent for a file that exists, only because no hasher was on PATH"
|
|
[ "$explicit_result" != "absent" ] \
|
|
|| fail "jar_id (explicit path) reported absent for a file that exists, only because no hasher was on PATH"
|
|
# A 12-char hex hash is the OTHER wrong answer here: with no hasher at all, nothing could have
|
|
# produced one, so a value that merely happens to look like one would mean the stub failed to
|
|
# hide the real hashers rather than that jar_id degraded correctly.
|
|
printf '%s' "$default_result" | grep -Eq '^[0-9a-f]{12}$' \
|
|
&& fail "test fixture error: PATH stub did not actually hide the real hasher(s) — got what looks like a real hash"
|
|
assert_equals "unhashable" "$default_result" "jar_id with no hasher on PATH must report a third, distinct state — never absent, never a hash"
|
|
assert_equals "unhashable" "$explicit_result" "jar_id (explicit path) with no hasher on PATH must report the same third state"
|
|
}
|
|
|
|
# fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar
|
|
# are exercised directly against real files on disk (not stubs), because the whole point is file
|
|
# behavior (does the content move, does the source disappear, does a failure leave both sides
|
|
# intact) that a stubbed function cannot prove.
|
|
test_stage_built_jar_moves_off_live_path() {
|
|
local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
dir="$TMP/stage-ok"; mkdir -p "$dir"
|
|
jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar"
|
|
printf 'built jar bytes' > "$jar"
|
|
JAR="$jar"; JAR_STAGED="$staged"
|
|
stage_built_jar || fail "stage_built_jar rejected a real build output"
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar"
|
|
[ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged"
|
|
grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content"
|
|
}
|
|
|
|
test_stage_built_jar_dies_when_build_produced_nothing() {
|
|
local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
|
dir="$TMP/stage-missing"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
|
output="$(stage_built_jar 2>&1)" || rc=$?
|
|
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
|
[ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output"
|
|
printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \
|
|
|| fail "refusal message does not name the missing jar path"
|
|
}
|
|
|
|
test_swap_staged_jar_moves_staged_onto_live() {
|
|
local dir staged live
|
|
dir="$TMP/swap-ok"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar"
|
|
printf 'swapped jar bytes' > "$staged"
|
|
swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar"
|
|
[ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged"
|
|
[ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live"
|
|
grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content"
|
|
}
|
|
|
|
# The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on
|
|
# failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes
|
|
# inside a command substitution to contain that exit to a subshell.
|
|
test_swap_staged_jar_dies_without_staged_file() {
|
|
local dir output rc=0
|
|
dir="$TMP/swap-missing"; mkdir -p "$dir"
|
|
output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar"
|
|
[ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged"
|
|
printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \
|
|
|| fail "refusal message does not name the missing staged path"
|
|
}
|
|
|
|
test_swap_staged_jar_dies_when_mv_fails() {
|
|
local dir staged live output rc=0
|
|
dir="$TMP/swap-fail"; mkdir -p "$dir/src"
|
|
staged="$dir/src/fleetd-new.jar"
|
|
printf 'fake jar bytes' > "$staged"
|
|
live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails
|
|
output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv"
|
|
[ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails"
|
|
[ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed"
|
|
printf '%s' "$output" | grep -qF "$staged" \
|
|
|| fail "refusal message does not name the staged path that could not be moved"
|
|
}
|
|
|
|
# --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this
|
|
# path) and must still die with the exact wording documented in the script's own header comment.
|
|
test_require_no_build_jar_dies_when_absent() {
|
|
local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar"
|
|
JAR="$missing"
|
|
output="$(require_no_build_jar 2>&1)" || rc=$?
|
|
JAR="$saved_jar"
|
|
[ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar"
|
|
printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \
|
|
|| fail "refusal message does not match the documented --no-build wording"
|
|
}
|
|
|
|
test_require_no_build_jar_accepts_present_jar() {
|
|
local saved_jar="$JAR" dir
|
|
dir="$TMP/no-build-present"; mkdir -p "$dir"
|
|
JAR="$dir/fleetd.jar"
|
|
printf 'existing jar' > "$JAR"
|
|
require_no_build_jar || fail "require_no_build_jar rejected an existing jar"
|
|
JAR="$saved_jar"
|
|
}
|
|
|
|
# wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while
|
|
# running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so
|
|
# the timeout-loop test does not actually wait out its budget.
|
|
test_wait_for_daemon_exit_returns_true_once_pid_clears() {
|
|
# running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so
|
|
# a plain shell variable it increments would reset on each call instead of accumulating. Count in
|
|
# a file instead, which is the one thing that actually survives across those subshells.
|
|
local counter_file="$TMP/wait-exit-calls" final_calls
|
|
printf '0' > "$counter_file"
|
|
running_pid() {
|
|
local n
|
|
n="$(cat "$counter_file")"
|
|
n=$((n + 1))
|
|
printf '%s' "$n" > "$counter_file"
|
|
if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi
|
|
}
|
|
sleep() { :; }
|
|
wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared"
|
|
final_calls="$(cat "$counter_file")"
|
|
[ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
|
}
|
|
|
|
test_wait_for_daemon_exit_times_out_if_pid_never_clears() {
|
|
local rc=0
|
|
running_pid() { printf '4242'; }
|
|
sleep() { :; }
|
|
wait_for_daemon_exit 3 || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
|
}
|
|
|
|
# fleetd #521 — the swap step's guard, at two levels.
|
|
#
|
|
# The first two tests call the predicate should_swap() directly. They pin its logic, and that is all
|
|
# they pin. On their own they did NOT close #521, and this was measured rather than argued: with the
|
|
# main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left
|
|
# this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs
|
|
# the swap consult the predicate at all. Extracting the decision had moved the untested decision up
|
|
# a level, not removed it.
|
|
#
|
|
# So the last two tests call swap_if_built() — the function the main flow actually calls, holding the
|
|
# guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the
|
|
# guard is removed, inverted, or stops being consulted.
|
|
#
|
|
# What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow
|
|
# altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing
|
|
# stops before the main flow runs, so no test in this file can invoke it.
|
|
test_should_swap_true_when_build_ran() {
|
|
should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true"
|
|
}
|
|
|
|
test_should_swap_false_when_build_skipped() {
|
|
if should_swap 0; then
|
|
fail "should_swap 0 (--no-build; nothing was staged this run) must return false"
|
|
fi
|
|
}
|
|
|
|
# Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is
|
|
# global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id
|
|
# overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear),
|
|
# and again at the END, so their own stubs do not leak into every test that runs after them.
|
|
test_swap_if_built_performs_the_swap_when_build_ran() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local marker="$TMP/swap-if-built-ran"
|
|
rm -f "$marker"
|
|
swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; }
|
|
jar_id() { printf 'stubbed\n'; }
|
|
swap_if_built 1 > /dev/null
|
|
[ -f "$marker" ] \
|
|
|| fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_swap_if_built_skips_the_swap_when_build_skipped() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local marker="$TMP/swap-if-built-skipped"
|
|
rm -f "$marker"
|
|
swap_staged_jar() { printf 'swapped\n' > "$marker"; }
|
|
jar_id() { printf 'stubbed\n'; }
|
|
swap_if_built 0 > /dev/null
|
|
if [ -f "$marker" ]; then
|
|
fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did"
|
|
fi
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
# fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the
|
|
# main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee
|
|
# itself — as opposed to the pure functions it's built from — can only be checked by reading the
|
|
# script's own call sites, the same way test_recovery_patterns_match_source below checks Java
|
|
# source shape instead of behavior it cannot invoke directly.
|
|
#
|
|
# Two details about the three greps below, both of which have already gone wrong here.
|
|
#
|
|
# The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not
|
|
# `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside
|
|
# swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using
|
|
# it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)"
|
|
# — a true statement about a function definition, and nothing at all about the order of the steps.
|
|
#
|
|
# Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the
|
|
# pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set
|
|
# -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to
|
|
# report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no
|
|
# name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak.
|
|
test_swap_ordered_after_wait_and_before_start() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line
|
|
wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh"
|
|
[ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh"
|
|
[ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh"
|
|
[ "$swap_line" -gt "$wait_line" ] \
|
|
|| fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)"
|
|
[ "$swap_line" -lt "$start_line" ] \
|
|
|| fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)"
|
|
}
|
|
|
|
# fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator
|
|
# declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)"
|
|
# to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has
|
|
# already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal
|
|
# ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start
|
|
# above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow),
|
|
# so the only way to pin its exact wording is to read the source.
|
|
test_drain_gate_abort_message_says_no_no_build() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" msg
|
|
msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")"
|
|
[ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh"
|
|
if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then
|
|
fail "abort message still claims a rerun WITH --no-build can finish the restart"
|
|
fi
|
|
printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \
|
|
|| fail "abort message does not tell the operator to rerun without --no-build"
|
|
printf '%s' "$msg" | grep -qF 'no longer at the live path' \
|
|
|| fail "abort message does not say why --no-build cannot finish the restart"
|
|
}
|
|
|
|
# fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was
|
|
# a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's
|
|
# own file for the wording, which stays in the file even if the `if` guarding it is mutated to
|
|
# `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead,
|
|
# so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT
|
|
# alongside these, not replaced, because it catches a different regression (a re-wording that still
|
|
# reaches the right branch would not change which case fires here, but would still be worth pinning).
|
|
test_drain_gate_refusal_build_ran_staged_present() {
|
|
local dir staged result
|
|
dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'staged jar bytes' > "$staged"
|
|
result="$(drain_gate_refusal 1 "$staged")"
|
|
printf '%s' "$result" | grep -qF "$staged" \
|
|
|| fail "build-ran+staged-present refusal does not name the staged jar path"
|
|
printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \
|
|
|| fail "build-ran+staged-present refusal does not tell the operator how to finish the restart"
|
|
if printf '%s' "$result" | grep -qF 'nothing changed'; then
|
|
fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved"
|
|
fi
|
|
}
|
|
|
|
test_drain_gate_refusal_build_ran_staged_absent() {
|
|
local dir result
|
|
dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir"
|
|
result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")"
|
|
assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording"
|
|
}
|
|
|
|
# --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar
|
|
# found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the
|
|
# comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning.
|
|
test_drain_gate_refusal_no_build_staged_present() {
|
|
local dir staged result
|
|
dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'leftover staged jar bytes' > "$staged"
|
|
result="$(drain_gate_refusal 0 "$staged")"
|
|
assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed"
|
|
}
|
|
|
|
test_drain_gate_refusal_no_build_staged_absent() {
|
|
local dir result
|
|
dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir"
|
|
result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")"
|
|
assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording"
|
|
}
|
|
|
|
# fleetd #528 — the four tests above pin drain_gate_refusal(), and that is ALL they pin: they call
|
|
# the predicate directly and never touch the main flow's call site. That was measured to be not
|
|
# enough, the same way test_should_swap_true_when_build_ran/test_should_swap_false_when_build_skipped
|
|
# were not enough for #521: with the main flow reading `die "$(drain_gate_refusal "$DO_BUILD"
|
|
# "$JAR_STAGED")"`, replacing that whole line with a flat `die "aborted — nothing changed"` left this
|
|
# suite at exit 0 with zero FAIL lines and byte-identical output to a clean run. Nothing above could
|
|
# tell the difference, because none of it calls anything at or above the call site itself.
|
|
#
|
|
# So these four call refuse_drain_gate() — the function the main flow actually calls, holding the
|
|
# composed message and the die() together — with die() stubbed to RECORD whether it was called and
|
|
# with what message, instead of exiting the process. That fails if refuse_drain_gate stops consulting
|
|
# drain_gate_refusal, mangles what it passes it, or simply never calls die.
|
|
#
|
|
# What none of these four can catch: deleting the `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` line
|
|
# from the main flow altogether — see the comment above refuse_drain_gate in redeploy-fleetd.sh for
|
|
# why no test in this file can do better than that (sourcing stops before the main flow runs).
|
|
DIED_CALLED=0
|
|
DIED_MESSAGE=""
|
|
stub_die_recorder() {
|
|
DIED_CALLED=0
|
|
DIED_MESSAGE=""
|
|
die() { DIED_CALLED=1; DIED_MESSAGE="$*"; }
|
|
}
|
|
|
|
test_refuse_drain_gate_build_ran_staged_present() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir staged
|
|
dir="$TMP/refuse-drain-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'staged jar bytes' > "$staged"
|
|
stub_die_recorder
|
|
refuse_drain_gate 1 "$staged"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate build-ran+staged-present must call die, and did not"
|
|
printf '%s' "$DIED_MESSAGE" | grep -qF "$staged" \
|
|
|| fail "refuse_drain_gate build-ran+staged-present die message does not name the staged jar"
|
|
printf '%s' "$DIED_MESSAGE" | grep -qF 'Rerun WITHOUT --no-build' \
|
|
|| fail "refuse_drain_gate build-ran+staged-present die message is missing the rerun instruction"
|
|
if printf '%s' "$DIED_MESSAGE" | grep -qF 'nothing changed'; then
|
|
fail "refuse_drain_gate build-ran+staged-present must not claim nothing changed — the jar already moved"
|
|
fi
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_refuse_drain_gate_build_ran_staged_absent() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir
|
|
dir="$TMP/refuse-drain-build-no-staged"; mkdir -p "$dir"
|
|
stub_die_recorder
|
|
refuse_drain_gate 1 "$dir/fleetd-new.jar"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate build-ran+staged-absent must call die, and did not"
|
|
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate build-ran+staged-absent die message"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_refuse_drain_gate_no_build_staged_present() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir staged
|
|
dir="$TMP/refuse-drain-no-build-staged"; mkdir -p "$dir"
|
|
staged="$dir/fleetd-new.jar"
|
|
printf 'leftover staged jar bytes' > "$staged"
|
|
stub_die_recorder
|
|
refuse_drain_gate 0 "$staged"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate no-build+staged-present must call die, and did not"
|
|
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-present die message"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_refuse_drain_gate_no_build_staged_absent() {
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local dir
|
|
dir="$TMP/refuse-drain-no-build-no-staged"; mkdir -p "$dir"
|
|
stub_die_recorder
|
|
refuse_drain_gate 0 "$dir/fleetd-new.jar"
|
|
[ "$DIED_CALLED" = 1 ] \
|
|
|| fail "refuse_drain_gate no-build+staged-absent must call die, and did not"
|
|
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-absent die message"
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
}
|
|
|
|
# fleetd #528 — closes the one gap the four behavioural tests above cannot: they call
|
|
# refuse_drain_gate directly, and sourcing stops before the main flow ever runs (the SOURCED guard),
|
|
# so none of them can prove the main flow still CALLS refuse_drain_gate at all. Same shape as
|
|
# test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site. This is
|
|
# what actually kills the item-1 mutation from the ticket — replacing the main flow's call with a
|
|
# flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural
|
|
# tests above would even notice.
|
|
#
|
|
# The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail
|
|
# the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below
|
|
# ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body).
|
|
test_refuse_drain_gate_call_site_present() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
|
call_line="$(grep -Fn 'refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$call_line" ] \
|
|
|| fail "could not find the main flow's refuse_drain_gate call site in redeploy-fleetd.sh"
|
|
}
|
|
|
|
# fleetd #504 — the "loaded but not currently running" branches for launchd/systemd used to run
|
|
# `launchctl unload`/`systemctl --user stop` with `2>/dev/null || true` and print `ok`
|
|
# unconditionally, so a real supervisor failure (e.g. it cannot reach launchd/the systemd user bus)
|
|
# read exactly like a harmless already-stopped answer. unload_launchd_if_loaded/
|
|
# stop_systemd_if_loaded (redeploy-fleetd.sh, right after systemd_loaded) apply systemd_loaded's own
|
|
# "capture stderr separately — only a non-zero exit WITH stderr is a real failure" pattern to the
|
|
# WRITE side. Both call the real `launchctl`/`systemctl` binaries directly (they are not overridable
|
|
# wrapper functions the way launchd_loaded/systemd_loaded are), so these tests put a stub binary
|
|
# first on PATH — the same technique test_detect_supervisor_systemd_probe_error_is_unclear above
|
|
# already uses for `systemctl`.
|
|
test_unload_launchd_if_loaded_dies_on_real_failure() {
|
|
local bin_dir output rc=0 saved_plist="$LAUNCHD_PLIST"
|
|
bin_dir="$TMP/stub-bin-launchctl-error"; mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/launchctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "Could not find specified service" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/launchctl"
|
|
LAUNCHD_PLIST="$TMP/fake-fail.plist"
|
|
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
|
|
LAUNCHD_PLIST="$saved_plist"
|
|
[ "$rc" -ne 0 ] \
|
|
|| fail "unload_launchd_if_loaded must die when launchctl exits non-zero AND writes to stderr"
|
|
printf '%s' "$output" | grep -qF 'launchctl unload' \
|
|
|| fail "die message does not name the failing launchctl unload command"
|
|
}
|
|
|
|
# Captured via $(...) rather than called bare: unload_launchd_if_loaded's own die() does a hard
|
|
# `exit`, and calling it directly at this level would let a regression that makes it die on this
|
|
# clean-negative case kill the WHOLE suite before the `|| fail` below ever ran — printing die's own
|
|
# message instead of this test's. Inside a command substitution, that `exit` only ends the subshell
|
|
# (a-guard-is-defeated-by-its-calling-context: the same reason the *_dies_on_real_failure tests
|
|
# above capture this way), so this test's own message is what actually reaches the report.
|
|
test_unload_launchd_if_loaded_tolerates_clean_negative() {
|
|
local bin_dir saved_plist="$LAUNCHD_PLIST" output rc=0
|
|
bin_dir="$TMP/stub-bin-launchctl-noop"; mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/launchctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/launchctl"
|
|
LAUNCHD_PLIST="$TMP/fake-noop.plist"
|
|
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
|
|
LAUNCHD_PLIST="$saved_plist"
|
|
[ "$rc" -eq 0 ] \
|
|
|| fail "unload_launchd_if_loaded must tolerate a clean already-unloaded answer (non-zero exit, empty stderr): $output"
|
|
}
|
|
|
|
test_stop_systemd_if_loaded_dies_on_real_failure() {
|
|
local bin_dir output rc=0
|
|
bin_dir="$TMP/stub-bin-systemctl-stop-error"; mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/systemctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "Failed to connect to bus: No such file or directory" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/systemctl"
|
|
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] \
|
|
|| fail "stop_systemd_if_loaded must die when systemctl exits non-zero AND writes to stderr"
|
|
printf '%s' "$output" | grep -qF 'systemctl --user stop' \
|
|
|| fail "die message does not name the failing systemctl --user stop command"
|
|
}
|
|
|
|
# Same subshell-capture reasoning as test_unload_launchd_if_loaded_tolerates_clean_negative above:
|
|
# stop_systemd_if_loaded's own die() does a hard `exit`, so this must run inside $(...) or a
|
|
# regression here would kill the whole suite with die's message instead of this test's.
|
|
test_stop_systemd_if_loaded_tolerates_clean_negative() {
|
|
local bin_dir output rc=0
|
|
bin_dir="$TMP/stub-bin-systemctl-stop-noop"; mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/systemctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/systemctl"
|
|
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
|
|
[ "$rc" -eq 0 ] \
|
|
|| fail "stop_systemd_if_loaded must tolerate a clean already-stopped answer (non-zero exit, empty stderr): $output"
|
|
}
|
|
|
|
# Closes the same gap test_refuse_drain_gate_call_site_present closes for the drain gate: the four
|
|
# tests above call unload_launchd_if_loaded/stop_systemd_if_loaded directly, and sourcing stops
|
|
# before the main flow ever runs (the SOURCED guard), so none of them can prove the main flow still
|
|
# CALLS these two functions instead of the original bare `2>/dev/null || true`. A source-text check,
|
|
# like test_swap_ordered_after_wait_and_before_start. The call-site needle is anchored (`^ name$`)
|
|
# so it cannot be satisfied by the comment lines above each call site that merely mention the
|
|
# function by name.
|
|
test_stop_branches_call_tolerant_helpers_not_bare_or_true() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" unload_call_line stop_call_line
|
|
unload_call_line="$(grep -n '^ unload_launchd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
|
|
stop_call_line="$(grep -n '^ stop_systemd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$unload_call_line" ] \
|
|
|| fail "could not find the main flow's call to unload_launchd_if_loaded in redeploy-fleetd.sh"
|
|
[ -n "$stop_call_line" ] \
|
|
|| fail "could not find the main flow's call to stop_systemd_if_loaded in redeploy-fleetd.sh"
|
|
if grep -qF 'launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true' "$src"; then
|
|
fail "the bare 'launchctl unload ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
|
|
fi
|
|
if grep -qF 'systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true' "$src"; then
|
|
fail "the bare 'systemctl --user stop ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
|
|
fi
|
|
}
|
|
|
|
test_no_errors() {
|
|
cat > "$TMP/no-errors.log" <<'LOG'
|
|
2026-09-05 12:00:00 INFO fleetd listening
|
|
LOG
|
|
classify_fixture no-errors.log
|
|
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained"
|
|
}
|
|
|
|
test_recovery_patterns_match_source() {
|
|
grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "AMQP failure pattern no longer matches source"
|
|
grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "reply-inbox recovery pattern no longer matches source"
|
|
grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \
|
|
|| fail "lead-mailbox recovery pattern no longer matches source"
|
|
}
|
|
|
|
test_attributed_recovered_connection_error() {
|
|
cat > "$TMP/attributed-recovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture attributed-recovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total"
|
|
assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained"
|
|
}
|
|
|
|
test_source_derived_error_shapes_recover_by_connection() {
|
|
# These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check
|
|
# after redeploy because the new code has not yet written a production line.
|
|
cat > "$TMP/source-derived.log" <<'LOG'
|
|
17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery!
|
|
17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!
|
|
17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture source-derived.log
|
|
assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total"
|
|
assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained"
|
|
}
|
|
|
|
test_cross_connection_unattributable_errors_stay_loud() {
|
|
# This candidate has neither stable connection name, so LeadMailbox recovery must not consume it.
|
|
cat > "$TMP/cross-unattributable.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-unattributable.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained"
|
|
}
|
|
|
|
test_attributed_cross_connection_errors_stay_loud() {
|
|
# LeadMailbox recovery cannot heal AmqpReplyInbox errors.
|
|
cat > "$TMP/cross-attributed.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-attributed.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained"
|
|
}
|
|
|
|
test_attributed_unrecovered_connection_error() {
|
|
cat > "$TMP/unrecovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
LOG
|
|
classify_fixture unrecovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained"
|
|
}
|
|
|
|
test_other_error_is_unexplained() {
|
|
cat > "$TMP/other-error.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed
|
|
2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture other-error.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained"
|
|
}
|
|
|
|
test_recovery_requirement_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then
|
|
fail "mutation accepted an unrecovered connection error"
|
|
fi
|
|
grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \
|
|
|| fail "mutation failed without the expected assertion"
|
|
printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n'
|
|
}
|
|
|
|
test_shared_counter_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line pending=0
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
case "$line" in
|
|
*'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*)
|
|
if [ "$pending" -gt 0 ]; then
|
|
pending=$((pending - 1))
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
fi
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending))
|
|
}
|
|
|
|
if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then
|
|
fail "shared counter mutation accepted cross-connection recovery"
|
|
fi
|
|
grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \
|
|
|| fail "shared counter mutation failed without the expected assertion"
|
|
printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
test_unattributable_quiet_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then
|
|
fail "unattributable mutation accepted an unknown connection"
|
|
fi
|
|
grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \
|
|
|| fail "unattributable mutation failed without the expected assertion"
|
|
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
# fleetd #512 part 2 — the negative check (scan_uncaught_exceptions). The heart of this half of the
|
|
# ticket: a fixture with the uncaught-exception shape and NO line carrying an ERROR token at all,
|
|
# proving the scan finds it without one. A fixture that also carried an ERROR line would pass for
|
|
# the wrong reason.
|
|
test_scan_uncaught_exceptions_finds_shape_without_error_token() {
|
|
cat > "$TMP/scan-died.log" <<'LOG'
|
|
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
|
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
|
|
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
|
|
LOG
|
|
local error_count
|
|
error_count="$(grep -c ' ERROR ' "$TMP/scan-died.log" || true)"
|
|
[ "$error_count" = "0" ] \
|
|
|| fail "test fixture error: scan-died.log unexpectedly carries an ERROR token"
|
|
scan_uncaught_exceptions "$TMP/scan-died.log"
|
|
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "scan must find the exception without an ERROR token"
|
|
printf '%s' "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" | grep -qF 'NoClassDefFoundError' \
|
|
|| fail "scan did not capture the matching line as the sample"
|
|
}
|
|
|
|
test_scan_uncaught_exceptions_clean_control() {
|
|
cat > "$TMP/scan-clean.log" <<'LOG'
|
|
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
|
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=0 abandoned=0 (still BUSY at the shutdown deadline)
|
|
LOG
|
|
scan_uncaught_exceptions "$TMP/scan-clean.log"
|
|
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "clean control must find no uncaught exception"
|
|
assert_equals "" "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" "clean control sample must be empty"
|
|
}
|
|
|
|
# fleetd #512 part 2 — the positive check (find_drain_complete_line). Both halves of #522's line:
|
|
# present, and absent.
|
|
test_find_drain_complete_line_present() {
|
|
cat > "$TMP/drain-line-present.log" <<'LOG'
|
|
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=2 abandoned=1 (still BUSY at the shutdown deadline)
|
|
LOG
|
|
find_drain_complete_line "$TMP/drain-line-present.log"
|
|
printf '%s' "$REDEPLOY_DRAIN_COMPLETE_LINE" | grep -qF 'released=2 abandoned=1' \
|
|
|| fail "find_drain_complete_line did not capture the present line"
|
|
}
|
|
|
|
test_find_drain_complete_line_absent() {
|
|
cat > "$TMP/drain-line-absent.log" <<'LOG'
|
|
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
|
LOG
|
|
find_drain_complete_line "$TMP/drain-line-absent.log"
|
|
assert_equals "" "$REDEPLOY_DRAIN_COMPLETE_LINE" "find_drain_complete_line must report empty when absent"
|
|
}
|
|
|
|
# fleetd #512 part 2 — report_shutdown_drain, the composite decision+action function the main flow
|
|
# calls unconditionally (same shape as swap_if_built/refuse_drain_gate, #521/#528). These four cover
|
|
# the four outcomes named in the ticket's "trap": complete, died, unknown ("cannot tell" — neither a
|
|
# pass nor a failure), and n/a (no previous daemon was actually stopped this run).
|
|
#
|
|
# Deliberately NOT run inside `$(...)`: report_shutdown_drain sets REDEPLOY_DRAIN_STATE as a global
|
|
# side effect that these tests need to read back afterward, and a command substitution forks a
|
|
# subshell that global assignment would not survive (the exact trap documented above
|
|
# detect_supervisor in redeploy-fleetd.sh, for the same reason). Plain output redirection to a file
|
|
# does not fork a subshell, so it is used to capture what was printed instead.
|
|
test_report_shutdown_drain_died_without_error_token() {
|
|
cat > "$TMP/drain-died.log" <<'LOG'
|
|
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
|
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
|
|
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
|
|
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
|
|
LOG
|
|
local error_count
|
|
error_count="$(grep -c ' ERROR ' "$TMP/drain-died.log" || true)"
|
|
[ "$error_count" = "0" ] \
|
|
|| fail "test fixture error: drain-died.log unexpectedly carries an ERROR token"
|
|
|
|
report_shutdown_drain "$TMP/drain-died.log" 1 > "$TMP/drain-died-output" 2>&1
|
|
assert_equals "died" "$REDEPLOY_DRAIN_STATE" "died fixture must set REDEPLOY_DRAIN_STATE=died"
|
|
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "died fixture uncaught-exception count"
|
|
grep -qF 'NoClassDefFoundError' "$TMP/drain-died-output" \
|
|
|| fail "report_shutdown_drain did not report the uncaught-exception shape it found"
|
|
grep -qF 'DIED' "$TMP/drain-died-output" \
|
|
|| fail "report_shutdown_drain did not report the drain as DIED"
|
|
}
|
|
|
|
test_report_shutdown_drain_complete_control() {
|
|
cat > "$TMP/drain-complete.log" <<'LOG'
|
|
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
|
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
|
|
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=3 abandoned=0 (still BUSY at the shutdown deadline)
|
|
LOG
|
|
report_shutdown_drain "$TMP/drain-complete.log" 1 > "$TMP/drain-complete-output" 2>&1
|
|
assert_equals "complete" "$REDEPLOY_DRAIN_STATE" "complete-control fixture must set REDEPLOY_DRAIN_STATE=complete"
|
|
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "complete-control fixture must find no uncaught exception"
|
|
grep -qF 'released=3 abandoned=0' "$TMP/drain-complete-output" \
|
|
|| fail "report_shutdown_drain did not report the drain-complete counts"
|
|
}
|
|
|
|
test_report_shutdown_drain_unknown_cannot_tell() {
|
|
cat > "$TMP/drain-unknown.log" <<'LOG'
|
|
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
|
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
|
|
LOG
|
|
report_shutdown_drain "$TMP/drain-unknown.log" 1 > "$TMP/drain-unknown-output" 2>&1
|
|
assert_equals "unknown" "$REDEPLOY_DRAIN_STATE" "cannot-tell fixture must set REDEPLOY_DRAIN_STATE=unknown"
|
|
grep -qF 'cannot tell' "$TMP/drain-unknown-output" \
|
|
|| fail "report_shutdown_drain did not say it could not tell"
|
|
if grep -qF ' ok' "$TMP/drain-unknown-output"; then
|
|
fail "cannot-tell outcome must not be printed via ok() — it is neither a pass nor a failure"
|
|
fi
|
|
}
|
|
|
|
# A cold start (or a restart where nothing was actually stopped) has no previous-daemon shutdown
|
|
# window to have an opinion about at all. This fixture's log content looks exactly like a died drain
|
|
# — proving the had_previous_daemon=0 gate is actually consulted, not merely documented: without it,
|
|
# this would misreport "died" or "unknown" on every clean cold start.
|
|
test_report_shutdown_drain_no_previous_daemon_is_na() {
|
|
cat > "$TMP/drain-na.log" <<'LOG'
|
|
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
|
|
LOG
|
|
report_shutdown_drain "$TMP/drain-na.log" 0 > "$TMP/drain-na-output" 2>&1
|
|
assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon fixture must set REDEPLOY_DRAIN_STATE=n/a even though the log content looks like a died drain"
|
|
grep -qF 'nothing to check' "$TMP/drain-na-output" \
|
|
|| fail "report_shutdown_drain did not report that there was nothing to check"
|
|
}
|
|
|
|
# fleetd #512 — closes the gap none of the seven tests above can: they call report_shutdown_drain
|
|
# directly, and sourcing stops before the main flow ever runs (the SOURCED guard), so none of them
|
|
# can prove the main flow still calls it at all. Same shape as test_refuse_drain_gate_call_site_present
|
|
# and test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site, plus
|
|
# an ordering check against its neighbours in the verify/result flow.
|
|
test_report_shutdown_drain_call_site_present() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
|
call_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$call_line" ] \
|
|
|| fail "could not find the main flow's report_shutdown_drain call site in redeploy-fleetd.sh"
|
|
}
|
|
|
|
test_report_shutdown_drain_ordered_after_classify_and_before_result() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" classify_line drain_line result_line
|
|
classify_line="$(grep -Fn 'classify_amqp_connection_errors "$FRESH_LOG"' "$src" | tail -1 | cut -d: -f1 || true)"
|
|
drain_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)"
|
|
[ -n "$classify_line" ] || fail "could not find the classify_amqp_connection_errors call site"
|
|
[ -n "$drain_line" ] || fail "could not find the report_shutdown_drain call site"
|
|
[ -n "$result_line" ] || fail "could not find the result section"
|
|
[ "$drain_line" -gt "$classify_line" ] \
|
|
|| fail "report_shutdown_drain (line $drain_line) is not after classify_amqp_connection_errors (line $classify_line)"
|
|
[ "$drain_line" -lt "$result_line" ] \
|
|
|| fail "report_shutdown_drain (line $drain_line) is not before the result section (line $result_line)"
|
|
}
|
|
|
|
# fleetd #512 item 4 — the summary line must not read as reassurance when the shutdown-drain check
|
|
# found something wrong (or could not tell). Sourcing stops before the main flow runs, so this is a
|
|
# source-text check like test_drain_gate_abort_message_says_no_no_build above.
|
|
test_no_error_lines_message_gated_by_drain_state() {
|
|
local src="$ROOT/scripts/redeploy-fleetd.sh" block
|
|
block="$(grep -B2 -F 'ok "no ERROR lines since restart"' "$src")"
|
|
[ -n "$block" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh"
|
|
printf '%s' "$block" | grep -qF 'REDEPLOY_DRAIN_STATE' \
|
|
|| fail "'no ERROR lines since restart' is not guarded by the shutdown-drain outcome (fleetd #512 item 4)"
|
|
}
|
|
|
|
test_detect_supervisor_launchd_only
|
|
test_detect_supervisor_systemd_only
|
|
test_detect_supervisor_none
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_systemd_probe_error_is_unclear
|
|
test_detect_supervisor_systemd_probe_setup_failure_is_unclear
|
|
test_mktemp_dash_t_templates_have_x_placeholders
|
|
test_no_unguarded_macos_only_hasher_calls
|
|
test_require_drivable_supervisor_refuses_ambiguous
|
|
test_require_drivable_supervisor_refuses_unclear
|
|
test_require_drivable_supervisor_accepts_known_kinds
|
|
test_count_daemon_pids
|
|
test_assert_single_daemon_accepts_one_pid
|
|
test_assert_single_daemon_rejects_two_pids
|
|
test_jar_id_defaults_to_live_and_reports_explicit_path
|
|
test_hash256_computes_a_real_sha256
|
|
test_jar_id_reports_absent_for_missing_file
|
|
test_jar_id_reports_unhashable_when_no_hasher_on_path
|
|
test_stage_built_jar_moves_off_live_path
|
|
test_stage_built_jar_dies_when_build_produced_nothing
|
|
test_swap_staged_jar_moves_staged_onto_live
|
|
test_swap_staged_jar_dies_without_staged_file
|
|
test_swap_staged_jar_dies_when_mv_fails
|
|
test_should_swap_true_when_build_ran
|
|
test_should_swap_false_when_build_skipped
|
|
test_swap_if_built_performs_the_swap_when_build_ran
|
|
test_swap_if_built_skips_the_swap_when_build_skipped
|
|
test_require_no_build_jar_dies_when_absent
|
|
test_require_no_build_jar_accepts_present_jar
|
|
test_wait_for_daemon_exit_returns_true_once_pid_clears
|
|
test_wait_for_daemon_exit_times_out_if_pid_never_clears
|
|
test_swap_ordered_after_wait_and_before_start
|
|
test_drain_gate_abort_message_says_no_no_build
|
|
test_drain_gate_refusal_build_ran_staged_present
|
|
test_drain_gate_refusal_build_ran_staged_absent
|
|
test_drain_gate_refusal_no_build_staged_present
|
|
test_drain_gate_refusal_no_build_staged_absent
|
|
test_refuse_drain_gate_build_ran_staged_present
|
|
test_refuse_drain_gate_build_ran_staged_absent
|
|
test_refuse_drain_gate_no_build_staged_present
|
|
test_refuse_drain_gate_no_build_staged_absent
|
|
test_refuse_drain_gate_call_site_present
|
|
test_unload_launchd_if_loaded_dies_on_real_failure
|
|
test_unload_launchd_if_loaded_tolerates_clean_negative
|
|
test_stop_systemd_if_loaded_dies_on_real_failure
|
|
test_stop_systemd_if_loaded_tolerates_clean_negative
|
|
test_stop_branches_call_tolerant_helpers_not_bare_or_true
|
|
test_no_errors
|
|
test_recovery_patterns_match_source
|
|
test_attributed_recovered_connection_error
|
|
test_source_derived_error_shapes_recover_by_connection
|
|
test_cross_connection_unattributable_errors_stay_loud
|
|
test_attributed_cross_connection_errors_stay_loud
|
|
test_attributed_unrecovered_connection_error
|
|
test_other_error_is_unexplained
|
|
test_recovery_requirement_mutation_is_caught
|
|
test_shared_counter_mutation_is_caught
|
|
test_unattributable_quiet_mutation_is_caught
|
|
test_scan_uncaught_exceptions_finds_shape_without_error_token
|
|
test_scan_uncaught_exceptions_clean_control
|
|
test_find_drain_complete_line_present
|
|
test_find_drain_complete_line_absent
|
|
test_report_shutdown_drain_died_without_error_token
|
|
test_report_shutdown_drain_complete_control
|
|
test_report_shutdown_drain_unknown_cannot_tell
|
|
test_report_shutdown_drain_no_previous_daemon_is_na
|
|
test_report_shutdown_drain_call_site_present
|
|
test_report_shutdown_drain_ordered_after_classify_and_before_result
|
|
test_no_error_lines_message_gated_by_drain_state
|
|
printf 'PASS: redeploy log classifier\n'
|