b17f37a683
Two situations were silently landing in the "none" answer, which require_drivable_supervisor accepts and the script then falls back to a raw kill + nohup — exactly the wrong move when a supervisor actually IS present: - installed-but-not-loaded, on either supervisor. `systemctl --user is-active` answers "no" for activating/deactivating/failed and while an auto-restart is pending too, and every one of those is a host that IS under systemd (or launchd) and about to act again. `*_installed` already knew this; it was only ever consulted for a warning line, never by the decision itself. - a systemd probe that could not answer at all (e.g. systemctl cannot reach the user bus over a non-lingering ssh session) looked identical to a clean negative, because both probes redirected stderr straight to /dev/null. detect_supervisor now returns a fifth answer, "unclear", for both cases. systemd_loaded/systemd_installed capture systemctl's exit status and stderr separately and set their own *_ERRORED flag only on a real tool failure (non-zero exit WITH stderr), never on a clean negative. "none" now means only: neither supervisor installed, neither loaded, neither probe errored. require_drivable_supervisor die()s on "unclear" exactly like it already does on "ambiguous", naming the specific supervisor and reason via the new SUPERVISOR_UNCLEAR_DETAIL global. Tests: 4 new cases (systemd/launchd installed-but-not-loaded, a real systemd_loaded run through a systemctl stub that errors on stderr, and the die() refusal for "unclear" naming the unit). All 3 new guards were verified by mutation: each was removed from the real script, the suite caught it (a new FAIL line naming the exact broken assertion), then the file was restored byte-identically and the suite went green again.
395 lines
20 KiB
Bash
Executable File
395 lines
20 KiB
Bash
Executable File
#!/usr/bin/env bash
|
|
# Self-contained checks for the pure log classifier in redeploy-fleetd.sh.
|
|
|
|
set -euo pipefail
|
|
|
|
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
|
TMP="$(mktemp -d "$ROOT/.redeploy-log-test.XXXXXX")"
|
|
trap 'rm -rf "$TMP"' EXIT
|
|
|
|
# Sourcing stops before redeploy-fleetd.sh can build, stop, or start the daemon.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
|
|
fail() {
|
|
printf 'FAIL: %s\n' "$*" >&2
|
|
return 1
|
|
}
|
|
|
|
assert_equals() {
|
|
local expected="$1" actual="$2" description="$3"
|
|
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
|
|
}
|
|
|
|
classify_fixture() {
|
|
local name="$1"
|
|
classify_amqp_connection_errors "$TMP/$name"
|
|
}
|
|
|
|
|
|
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
|
|
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
|
|
# being naturally absent: this machine may itself be running a real fleetd under launchd right now
|
|
# (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving
|
|
# launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state
|
|
# instead of the fixture.
|
|
test_detect_supervisor_launchd_only() {
|
|
launchd_installed() { return 0; }
|
|
launchd_loaded() { return 0; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "launchd" "$(detect_supervisor)" "launchd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_systemd_only() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 0; }
|
|
assert_equals "systemd" "$(detect_supervisor)" "systemd-only detection"
|
|
}
|
|
|
|
test_detect_supervisor_none() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "none" "$(detect_supervisor)" "unsupervised detection"
|
|
}
|
|
|
|
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
|
|
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
|
|
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
|
|
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
|
|
# auto-restart too, and every one of those is a host that IS under systemd.
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; } # the unit file IS there
|
|
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
|
|
assert_equals "unclear" "$(detect_supervisor)" "systemd installed-but-not-loaded must read as unclear, not none"
|
|
}
|
|
|
|
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
|
|
# removed, or about to be reloaded) must not read as "no supervisor" either.
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
|
|
launchd_installed() { return 0; } # the plist IS there
|
|
launchd_loaded() { return 1; } # launchctl list says not loaded
|
|
systemd_installed() { return 1; }
|
|
systemd_loaded() { return 1; }
|
|
assert_equals "unclear" "$(detect_supervisor)" "launchd installed-but-not-loaded must read as unclear, not none"
|
|
}
|
|
|
|
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
|
|
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
|
|
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
|
|
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
|
|
# the same fact as "no supervisor is loaded".
|
|
test_detect_supervisor_systemd_probe_error_is_unclear() {
|
|
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
|
|
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
|
|
# of those tests), and a bash function definition is global for the rest of the process — without
|
|
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
|
|
# a real `systemctl` call at all.
|
|
source "$ROOT/scripts/redeploy-fleetd.sh"
|
|
local bin_dir result rc=0
|
|
bin_dir="$TMP/stub-bin-systemctl-errors"
|
|
mkdir -p "$bin_dir"
|
|
cat > "$bin_dir/systemctl" <<'STUB'
|
|
#!/usr/bin/env bash
|
|
echo "Failed to connect to bus: No such file or directory" >&2
|
|
exit 1
|
|
STUB
|
|
chmod +x "$bin_dir/systemctl"
|
|
|
|
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
|
|
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
|
|
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
|
|
assert_equals "unclear" "$result" "a systemd probe error must read as unclear, not none"
|
|
}
|
|
|
|
# The other half of the ticket: an "unclear" supervisor must refuse exactly like "ambiguous" does —
|
|
# die(), never fall through to the `kill` path — and the refusal must name the specific supervisor
|
|
# and reason detect_supervisor found, not just the bare word "unclear".
|
|
test_require_drivable_supervisor_refuses_unclear() {
|
|
launchd_installed() { return 1; }
|
|
launchd_loaded() { return 1; }
|
|
systemd_installed() { return 0; }
|
|
systemd_loaded() { return 1; }
|
|
local kind output rc=0
|
|
kind="$(detect_supervisor)"
|
|
assert_equals "unclear" "$kind" "setup: expected unclear before testing the refusal"
|
|
output="$(require_drivable_supervisor "$kind" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
|
|
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "refusal message does not name the systemd unit it found installed-but-not-loaded"
|
|
}
|
|
|
|
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
|
|
# `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that
|
|
# forks a subshell, so its exit() only ends the subshell and this test script keeps running under
|
|
# `set -e`.
|
|
test_require_drivable_supervisor_refuses_ambiguous() {
|
|
local output rc=0
|
|
output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor"
|
|
printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \
|
|
|| fail "refusal message does not name the launchd label it found"
|
|
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|
|
|| fail "refusal message does not name the systemd unit it found"
|
|
}
|
|
|
|
test_require_drivable_supervisor_accepts_known_kinds() {
|
|
require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor"
|
|
require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor"
|
|
require_drivable_supervisor "none" || fail "refused the unsupervised case"
|
|
}
|
|
|
|
# fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor
|
|
# produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see
|
|
# it because either daemon alone satisfies them.
|
|
test_count_daemon_pids() {
|
|
assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list"
|
|
assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid"
|
|
assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids"
|
|
}
|
|
|
|
test_assert_single_daemon_accepts_one_pid() {
|
|
assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid"
|
|
}
|
|
|
|
test_assert_single_daemon_rejects_two_pids() {
|
|
local output rc=0
|
|
output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$?
|
|
[ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids"
|
|
printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found"
|
|
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
|
|
}
|
|
|
|
test_no_errors() {
|
|
cat > "$TMP/no-errors.log" <<'LOG'
|
|
2026-09-05 12:00:00 INFO fleetd listening
|
|
LOG
|
|
classify_fixture no-errors.log
|
|
assert_equals 0 "$REDEPLOY_ERROR_COUNT" "no-errors total"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "no-errors unexplained"
|
|
}
|
|
|
|
test_recovery_patterns_match_source() {
|
|
grep -F 'AMQP connection {}: {}' "$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "AMQP failure pattern no longer matches source"
|
|
grep -F 'AMQP connection recovered; cleared held replies for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/AmqpReplyInbox.java" > /dev/null \
|
|
|| fail "reply-inbox recovery pattern no longer matches source"
|
|
grep -F 'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery' \
|
|
"$ROOT/fleetd/src/main/java/dev/ltms/fleet/msg/LeadMailbox.java" > /dev/null \
|
|
|| fail "lead-mailbox recovery pattern no longer matches source"
|
|
}
|
|
|
|
test_attributed_recovered_connection_error() {
|
|
cat > "$TMP/attributed-recovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture attributed-recovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "attributed-recovered total"
|
|
assert_equals 1 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "attributed-recovered errors"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "attributed-recovered unexplained"
|
|
}
|
|
|
|
test_source_derived_error_shapes_recover_by_connection() {
|
|
# These ERROR shapes come from AmqpConnectionFailureLogger on main. They need a live-log check
|
|
# after redeploy because the new code has not yet written a production line.
|
|
cat > "$TMP/source-derived.log" <<'LOG'
|
|
17:37:53.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:54.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: Caught an exception during connection recovery!
|
|
17:37:55.537 ERROR [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
17:37:56.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:37:57.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: Caught an exception during connection recovery!
|
|
17:37:58.537 ERROR [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP connection fleetd-lead-mailbox: An unexpected connection driver error occurred
|
|
17:38:00.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:01.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:02.000 INFO [AMQP Connection broker:5672] d.l.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
17:38:03.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:04.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
17:38:05.000 INFO [AMQP Connection broker:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture source-derived.log
|
|
assert_equals 6 "$REDEPLOY_ERROR_COUNT" "source-derived total"
|
|
assert_equals 6 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "source-derived recovered"
|
|
assert_equals 0 "$REDEPLOY_UNEXPLAINED_ERRORS" "source-derived unexplained"
|
|
}
|
|
|
|
test_cross_connection_unattributable_errors_stay_loud() {
|
|
# This candidate has neither stable connection name, so LeadMailbox recovery must not consume it.
|
|
cat > "$TMP/cross-unattributable.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR [AMQP Connection broker:5672] unknown - AMQP connection: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO [AMQP Connection 10.10.20.13:5672] d.ltms.fleet.msg.LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-unattributable.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-unattributable total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-unattributable recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-unattributable unexplained"
|
|
}
|
|
|
|
test_attributed_cross_connection_errors_stay_loud() {
|
|
# LeadMailbox recovery cannot heal AmqpReplyInbox errors.
|
|
cat > "$TMP/cross-attributed.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:01 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
2026-09-05 12:00:02 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
2026-09-05 12:00:03 INFO LeadMailbox - AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery
|
|
LOG
|
|
classify_fixture cross-attributed.log
|
|
assert_equals 2 "$REDEPLOY_ERROR_COUNT" "cross-attributed total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "cross-attributed recovered"
|
|
assert_equals 2 "$REDEPLOY_UNEXPLAINED_ERRORS" "cross-attributed unexplained"
|
|
}
|
|
|
|
test_attributed_unrecovered_connection_error() {
|
|
cat > "$TMP/unrecovered.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR AmqpReplyInbox - AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred
|
|
LOG
|
|
classify_fixture unrecovered.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "unrecovered total"
|
|
assert_equals 0 "$REDEPLOY_RECOVERED_AMQP_ERRORS" "unrecovered AMQP errors"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "unrecovered unexplained"
|
|
}
|
|
|
|
test_other_error_is_unexplained() {
|
|
cat > "$TMP/other-error.log" <<'LOG'
|
|
2026-09-05 12:00:00 ERROR dev.ltms.fleet.Fleetd - startup failed
|
|
2026-09-05 12:00:01 INFO dev.ltms.fleet.msg.AmqpReplyInbox - AMQP connection recovered; cleared held replies for fresh redelivery
|
|
LOG
|
|
classify_fixture other-error.log
|
|
assert_equals 1 "$REDEPLOY_ERROR_COUNT" "other-error total"
|
|
assert_equals 1 "$REDEPLOY_UNEXPLAINED_ERRORS" "other-error unexplained"
|
|
}
|
|
|
|
test_recovery_requirement_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection fleetd-reply-inbox: An unexpected connection driver error occurred'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_attributed_unrecovered_connection_error > "$TMP/mutation-output" 2>&1; then
|
|
fail "mutation accepted an unrecovered connection error"
|
|
fi
|
|
grep -F 'FAIL: unrecovered AMQP errors: expected 0, got 1' "$TMP/mutation-output" > /dev/null \
|
|
|| fail "mutation failed without the expected assertion"
|
|
printf 'Recovery mutation: FAIL: unrecovered AMQP errors: expected 0, got 1\n'
|
|
}
|
|
|
|
test_shared_counter_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line pending=0
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
case "$line" in
|
|
*'fleetd-reply-inbox'*|*'fleetd-lead-mailbox'*) pending=$((pending + 1)) ;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
*'AMQP connection recovered; cleared held replies for fresh redelivery'*|*'AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery'*)
|
|
if [ "$pending" -gt 0 ]; then
|
|
pending=$((pending - 1))
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
fi
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending))
|
|
}
|
|
|
|
if test_attributed_cross_connection_errors_stay_loud > "$TMP/shared-mutation-output" 2>&1; then
|
|
fail "shared counter mutation accepted cross-connection recovery"
|
|
fi
|
|
grep -F 'FAIL: cross-attributed recovered: expected 0, got 2' "$TMP/shared-mutation-output" > /dev/null \
|
|
|| fail "shared counter mutation failed without the expected assertion"
|
|
printf 'Shared-counter mutation: FAIL: cross-attributed recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
test_unattributable_quiet_mutation_is_caught() {
|
|
classify_amqp_connection_errors() {
|
|
local log_file="$1" line
|
|
REDEPLOY_ERROR_COUNT=0
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=0
|
|
REDEPLOY_UNEXPLAINED_ERRORS=0
|
|
while IFS= read -r line || [ -n "$line" ]; do
|
|
case "$line" in
|
|
*' ERROR '*|*' SEVERE '*)
|
|
REDEPLOY_ERROR_COUNT=$((REDEPLOY_ERROR_COUNT + 1))
|
|
case "$line" in
|
|
*'AMQP connection'*'An unexpected connection driver error occurred'*|*'AMQP connection'*'Caught an exception during connection recovery!'*)
|
|
REDEPLOY_RECOVERED_AMQP_ERRORS=$((REDEPLOY_RECOVERED_AMQP_ERRORS + 1))
|
|
;;
|
|
*) REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + 1)) ;;
|
|
esac
|
|
;;
|
|
esac
|
|
done < "$log_file"
|
|
}
|
|
|
|
if test_cross_connection_unattributable_errors_stay_loud > "$TMP/unattributable-mutation-output" 2>&1; then
|
|
fail "unattributable mutation accepted an unknown connection"
|
|
fi
|
|
grep -F 'FAIL: cross-unattributable recovered: expected 0, got 2' "$TMP/unattributable-mutation-output" > /dev/null \
|
|
|| fail "unattributable mutation failed without the expected assertion"
|
|
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
|
|
}
|
|
|
|
test_detect_supervisor_launchd_only
|
|
test_detect_supervisor_systemd_only
|
|
test_detect_supervisor_none
|
|
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
|
|
test_detect_supervisor_systemd_probe_error_is_unclear
|
|
test_require_drivable_supervisor_refuses_ambiguous
|
|
test_require_drivable_supervisor_refuses_unclear
|
|
test_require_drivable_supervisor_accepts_known_kinds
|
|
test_count_daemon_pids
|
|
test_assert_single_daemon_accepts_one_pid
|
|
test_assert_single_daemon_rejects_two_pids
|
|
test_no_errors
|
|
test_recovery_patterns_match_source
|
|
test_attributed_recovered_connection_error
|
|
test_source_derived_error_shapes_recover_by_connection
|
|
test_cross_connection_unattributable_errors_stay_loud
|
|
test_attributed_cross_connection_errors_stay_loud
|
|
test_attributed_unrecovered_connection_error
|
|
test_other_error_is_unexplained
|
|
test_recovery_requirement_mutation_is_caught
|
|
test_shared_counter_mutation_is_caught
|
|
test_unattributable_quiet_mutation_is_caught
|
|
printf 'PASS: redeploy log classifier\n'
|