Compare commits
12 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e4eb3dbed4 | |||
| f1640f5dcc | |||
| c7903c1efe | |||
| 7d711942fe | |||
| bec87f987c | |||
| 7f8a8829f9 | |||
| af9589783e | |||
| 190436c9cf | |||
| 8335b12562 | |||
| d25c863118 | |||
| 7c34e8f4f9 | |||
| be07ed2033 |
@@ -96,8 +96,10 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
||||
forge MCP server it appears to have holds a blocked credential and fails every call, and a piped
|
||||
command (`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a
|
||||
fact. Its injected repo-scoped `GITEA_TOKEN` is a different credential and does work, so a worker
|
||||
reporting that it opened its own PR is reporting something it really can do.
|
||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||
@@ -200,8 +202,11 @@ simply complies has thrown away the reason there are two of you.
|
||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
||||
find there holds a deliberately blocked credential and fails every call, by design.
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge MCP server you
|
||||
may find there holds a deliberately blocked credential and fails every call, by design. That is
|
||||
not your only forge route, and the two must not be confused: the repo-scoped `GITEA_TOKEN` the
|
||||
daemon injects into your environment does work, and using it to open your own PR is part of the
|
||||
job. A blocked MCP tool is never a reason to skip that step.
|
||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
|
||||
@@ -1,15 +1,12 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.msg.LeadMailbox;
|
||||
import dev.ltms.fleet.testing.CapturedLog;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
@@ -52,29 +49,22 @@ class FleetdLeadMailboxSelectionTest {
|
||||
}
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> captureFleetdLogs() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static String joined(ListAppender<ILoggingEvent> appender, Level level) {
|
||||
return appender.list.stream().filter(e -> e.getLevel() == level)
|
||||
private static String joined(CapturedLog captured, Level level) {
|
||||
return captured.events().stream().filter(e -> e.getLevel() == level)
|
||||
.map(ILoggingEvent::getFormattedMessage).reduce("", (a, b) -> a + "\n" + b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void noCoordinatorBlockLeavesTheFeatureOffSilently() {
|
||||
var appender = captureFleetdLogs();
|
||||
var opener = new RecordingOpener();
|
||||
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||
var opener = new RecordingOpener();
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(null, Map.of(), opener));
|
||||
assertNull(Fleetd.openLeadMailbox(null, Map.of(), opener));
|
||||
|
||||
assertNull(opener.offeredUri, "nothing configured means nothing is opened");
|
||||
assertEquals("", joined(appender, Level.WARN),
|
||||
"an opt-in feature nobody asked for must not warn on every boot");
|
||||
assertNull(opener.offeredUri, "nothing configured means nothing is opened");
|
||||
assertEquals("", joined(captured, Level.WARN),
|
||||
"an opt-in feature nobody asked for must not warn on every boot");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -114,31 +104,33 @@ class FleetdLeadMailboxSelectionTest {
|
||||
|
||||
@Test
|
||||
void warnsAndStaysOffWhenSelfIdIsMissing() {
|
||||
var appender = captureFleetdLogs();
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
||||
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||
var opener = new RecordingOpener();
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, null, null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener));
|
||||
|
||||
assertNull(opener.offeredUri, "a mailbox with no owning coord-id has no queue to declare");
|
||||
String warns = joined(appender, Level.WARN);
|
||||
assertTrue(warns.contains("coordinator.selfId"), () -> "say which key is missing: " + warns);
|
||||
assertFalse(warns.contains(SECRET), () -> "the URI's password must never be logged: " + warns);
|
||||
assertNull(opener.offeredUri, "a mailbox with no owning coord-id has no queue to declare");
|
||||
String warns = joined(captured, Level.WARN);
|
||||
assertTrue(warns.contains("coordinator.selfId"), () -> "say which key is missing: " + warns);
|
||||
assertFalse(warns.contains(SECRET), () -> "the URI's password must never be logged: " + warns);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void warnsAndStaysOffWhenTheBrokerIsUnreachableAtBoot() {
|
||||
var appender = captureFleetdLogs();
|
||||
var opener = new RecordingOpener();
|
||||
opener.unreachable = true;
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||
try (var captured = CapturedLog.of(Fleetd.class)) {
|
||||
var opener = new RecordingOpener();
|
||||
opener.unreachable = true;
|
||||
var coordinator = new FleetConfig.Coordinator(RESOLVED_URI, null, "mac-opus", null, null);
|
||||
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
||||
"a down coordination broker turns the feature off; it must never take the daemon down");
|
||||
assertNull(Fleetd.openLeadMailbox(coordinator, Map.of(), opener),
|
||||
"a down coordination broker turns the feature off; it must never take the daemon down");
|
||||
|
||||
String warns = joined(appender, Level.WARN);
|
||||
assertTrue(warns.contains("coord.example"), () -> "name the host that failed: " + warns);
|
||||
assertFalse(warns.contains(SECRET), () -> "with credentials stripped: " + warns);
|
||||
assertTrue(warns.contains("Connection refused"), () -> "and the real reason: " + warns);
|
||||
String warns = joined(captured, Level.WARN);
|
||||
assertTrue(warns.contains("coord.example"), () -> "name the host that failed: " + warns);
|
||||
assertFalse(warns.contains(SECRET), () -> "with credentials stripped: " + warns);
|
||||
assertTrue(warns.contains("Connection refused"), () -> "and the real reason: " + warns);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -21,9 +21,29 @@ import java.util.List;
|
||||
* fork. try-with-resources makes "restored the appender but not the level" impossible to write,
|
||||
* because there is only one thing to close.
|
||||
*
|
||||
* <p>This is the one way to pin or capture a logger's level and output in this test tree — do not
|
||||
* hand-roll the {@code ListAppender} + {@code setLevel} + {@code finally detachAppender} pattern;
|
||||
* use {@link #at} or {@link #of} instead.
|
||||
* <p>New code must use this rather than hand-rolling the {@code ListAppender} + {@code setLevel} +
|
||||
* {@code finally detachAppender} pattern: use {@link #at} or {@link #of}. It is <em>not</em> yet
|
||||
* the only instance of the pattern in this test tree, and the earlier wording here said it was —
|
||||
* which would leave a reader who greps unable to tell a leftover from a violation.
|
||||
*
|
||||
* <p>Measured on main at af95897 (2026-09-12): nine test files still hand-roll it, with 42
|
||||
* {@code setLevel} calls on a raw logback {@code Logger} between them — {@code
|
||||
* FleetdStartupReportTest}, {@code GitHostShapeReportTest}, {@code MemberCredentialsGapReportTest},
|
||||
* {@code MemberTrustModelReportTest}, {@code FleetHealthMonitorTest}, {@code
|
||||
* ClaudeCodeLauncherTest}, {@code HerdrPeerLauncherAllowListWiringTest}, {@code
|
||||
* HerdrPeerLauncherCharterTest} and {@code OpenCodeLauncherTest}. Every one of them pairs its pin
|
||||
* with a restore, so none is the fleetd #525 leak and none was in fleetd #529's scope, which was
|
||||
* the 19 <em>unrestored</em> pins only. They are unmigrated, not broken.
|
||||
*
|
||||
* <p>Re-measure with the two commands below, from the repo root. A file that appears in the first
|
||||
* list and not the second still hand-rolls the pattern. When the first list comes back empty, this
|
||||
* paragraph is spent and the sentence above can go back to saying "the one way" — delete the
|
||||
* paragraph then rather than updating the count.
|
||||
*
|
||||
* <pre>{@code
|
||||
* grep -rlE '\.setLevel\(' fleetd/src/test/java --include='*.java' | grep -v CapturedLog.java
|
||||
* grep -rl 'CapturedLog' fleetd/src/test/java --include='*.java'
|
||||
* }</pre>
|
||||
*/
|
||||
public final class CapturedLog implements AutoCloseable {
|
||||
private final Logger logger;
|
||||
|
||||
+218
-11
@@ -42,6 +42,14 @@
|
||||
# 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if
|
||||
# more than one is alive. That is the one thing none of the checks above (healthz 200, jar id,
|
||||
# the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon.
|
||||
# 9. fleetd #512 — the ERROR-line count above is blind by construction to the exact failure #493
|
||||
# is about: an uncaught exception in a shutdown thread never passes through the logger, so it
|
||||
# never carries an ERROR (or SEVERE) token that any count could see. This script now also
|
||||
# greps the previous daemon's shutdown window for that exception's real shape, and separately
|
||||
# asserts that SessionManager's drain-complete line (fleetd #522) is present there — its
|
||||
# absence is the real signal, because a drain that dies on its first session prints nothing
|
||||
# else either. Warns loudly; never fails the redeploy, because by the time this is detectable
|
||||
# the new daemon is already up and healthy.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/redeploy-fleetd.sh # build, confirm, restart, verify
|
||||
@@ -279,6 +287,49 @@ systemd_loaded() {
|
||||
return "$rc"
|
||||
}
|
||||
|
||||
# fleetd #504: the "loaded but not currently running" branches in the main stop step (case
|
||||
# launchd/systemd, reached when $OLD_PID is empty) used to run `launchctl unload`/`systemctl --user
|
||||
# stop` with `2>/dev/null || true` and then print `ok` unconditionally — the exact conflation
|
||||
# systemd_loaded/systemd_installed above already fixed on the READ side (fleetd #492 follow-up): a
|
||||
# genuine "already stopped" answer (nonzero exit, nothing on stderr) is harmless, but a real tool
|
||||
# failure (nonzero exit WITH a stderr message — e.g. launchd or the systemd user bus is
|
||||
# unreachable) is not, and reporting `ok` on THAT means the start step below can register a fresh
|
||||
# load on top of a supervisor that never actually let go: the exact two-daemons failure fleetd #492
|
||||
# exists to prevent, reached from the one state (already odd) where a false `ok` is least
|
||||
# affordable. These two functions apply the same "capture stderr separately, flag only a nonzero
|
||||
# exit WITH stderr as a real failure" pattern to the WRITE side. No ${VAR:-default} anywhere here —
|
||||
# see the #492 follow-up constraints comment above detect_supervisor for why a default would hide a
|
||||
# lost value instead of surfacing it (fleetd #497's defect class).
|
||||
unload_launchd_if_loaded() {
|
||||
local err_file rc=0
|
||||
if ! err_file="$(mktemp -t launchd-unload-err)"; then
|
||||
die "could not create a temp file to capture 'launchctl unload' stderr — cannot tell a real
|
||||
failure from a clean already-unloaded answer, so refusing to guess. The daemon's
|
||||
supervision state was NOT touched."
|
||||
fi
|
||||
launchctl unload -w "$LAUNCHD_PLIST" >/dev/null 2>"$err_file" || rc=$?
|
||||
if [ "$rc" -ne 0 ] && [ -s "$err_file" ]; then
|
||||
die "'launchctl unload -w $LAUNCHD_PLIST' failed: $(cat "$err_file")
|
||||
The daemon may still be under supervision; investigate before retrying."
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
}
|
||||
|
||||
stop_systemd_if_loaded() {
|
||||
local err_file rc=0
|
||||
if ! err_file="$(mktemp -t systemd-stop-err)"; then
|
||||
die "could not create a temp file to capture 'systemctl --user stop' stderr — cannot tell a
|
||||
real failure from a clean already-stopped answer, so refusing to guess. The daemon's
|
||||
supervision state was NOT touched."
|
||||
fi
|
||||
systemctl --user stop "$SYSTEMD_UNIT" >/dev/null 2>"$err_file" || rc=$?
|
||||
if [ "$rc" -ne 0 ] && [ -s "$err_file" ]; then
|
||||
die "'systemctl --user stop $SYSTEMD_UNIT' failed: $(cat "$err_file")
|
||||
The daemon may still be under supervision; investigate before retrying."
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
}
|
||||
|
||||
# fleetd #492: three real answers, not two — launchd, systemd, or genuinely unsupervised — plus a
|
||||
# fourth, "ambiguous", for the one case this script cannot tell apart: both signals firing at once.
|
||||
# That is exactly "I cannot tell who supervises this process", and guessing wrong here is how two
|
||||
@@ -306,10 +357,12 @@ systemd_loaded() {
|
||||
# never assigned to a global: a global set inside a `$( )` subshell dies with that subshell.
|
||||
# This function packs BOTH values (kind and detail) onto that one stdout line, joined by
|
||||
# $SUPERVISOR_DETAIL_SEP, and the caller unpacks them on its own side of the subshell boundary.
|
||||
# 2. This script runs under `set -euo pipefail` (line 50), so an unset variable is a loud
|
||||
# failure. Do not add a `${VAR:-default}` anywhere downstream to paper over a value that
|
||||
# should always be there — that hides a lost value instead of surfacing it (fleetd #497's
|
||||
# defect class).
|
||||
# 2. This script runs under `set -u` (part of the `set -euo pipefail` at the top of the file), so
|
||||
# an unset variable is a loud failure. Do not add a `${VAR:-default}` anywhere downstream to
|
||||
# paper over a value that should always be there — that hides a lost value instead of
|
||||
# surfacing it (fleetd #497's defect class). The mechanism is named rather than cited by line
|
||||
# number on purpose: a line number in a comment goes stale on the next insert above it, and
|
||||
# this one already had — it said line 50 while the `set` line was at 54.
|
||||
# 3. Every `case` on this function's return value needs an explicit final `*)` arm, chosen by
|
||||
# whether that caller ACTS on the value (`die` — an unrecognised value must never be silently
|
||||
# driven) or only DISPLAYS it (`echo`/`warn` and continue — a diagnostic must not go silent on
|
||||
@@ -490,6 +543,114 @@ classify_amqp_connection_errors() {
|
||||
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending_inbox + pending_lead_mailbox))
|
||||
}
|
||||
|
||||
# fleetd #512 part 2 — the negative check. #493's failure (an uncaught exception in a shutdown
|
||||
# thread) never passes through the logger: the JVM's default uncaught-exception handler prints
|
||||
# straight to stderr, so the line never carries a level, so classify_amqp_connection_errors's
|
||||
# ERROR/SEVERE token scan is structurally blind to it — measured on two hosts, including one where
|
||||
# even a syslog PRIORITY filter is blind to it too (fd 1 and fd 2 collapse to one socket there, so
|
||||
# every uncaught-exception line lands at priority 6/info). The fix is to grep the shape instead of
|
||||
# the level: `Exception in thread` at the start of a line (the handler's own banner) or
|
||||
# `NoClassDefFoundError` anywhere in it (the one real instance seen so far, but not the only shape
|
||||
# this could take). Kept as its own function, never folded into classify_amqp_connection_errors —
|
||||
# this is not an AMQP concern, and the two must stay independently readable and independently
|
||||
# testable.
|
||||
#
|
||||
# Sets REDEPLOY_UNCAUGHT_EXCEPTION_COUNT (lines matched) and REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE
|
||||
# (the first matching line, "" if none) so a caller can report both a count and a concrete quote
|
||||
# without re-reading the file. Pure: reads $1, sets globals, no side effects.
|
||||
scan_uncaught_exceptions() {
|
||||
local log_file="$1" line
|
||||
REDEPLOY_UNCAUGHT_EXCEPTION_COUNT=0
|
||||
REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE=""
|
||||
while IFS= read -r line || [ -n "$line" ]; do
|
||||
case "$line" in
|
||||
'Exception in thread'*|*NoClassDefFoundError*)
|
||||
REDEPLOY_UNCAUGHT_EXCEPTION_COUNT=$((REDEPLOY_UNCAUGHT_EXCEPTION_COUNT + 1))
|
||||
[ -n "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" ] || REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE="$line"
|
||||
;;
|
||||
esac
|
||||
done < "$log_file"
|
||||
}
|
||||
|
||||
# fleetd #512 part 2 — the positive check. fleetd #522 added a `log.info` at the very end of
|
||||
# SessionManager.drainAll's normal path (never in a `finally` — see the ticket discussion for why
|
||||
# that distinction matters): "drain complete: released=N abandoned=M (still BUSY at the shutdown
|
||||
# deadline)", printed once, on every successful drain, including the all-zero case. A drain that
|
||||
# dies partway through never reaches that statement, so the line's ABSENCE is a real signal — unlike
|
||||
# the ERROR-count check above, this one does not depend on the failure happening to throw.
|
||||
#
|
||||
# Sets REDEPLOY_DRAIN_COMPLETE_LINE to the matching line (last one, though drainAll runs at most
|
||||
# once per shutdown so there should never be more than one) or "" if absent. Pure, same shape as
|
||||
# scan_uncaught_exceptions above.
|
||||
find_drain_complete_line() {
|
||||
local log_file="$1"
|
||||
REDEPLOY_DRAIN_COMPLETE_LINE="$(grep -F 'drain complete: released=' "$log_file" | tail -1 || true)"
|
||||
}
|
||||
|
||||
# fleetd #512 part 2 — THE TRAP, and the reason this is one function instead of two independent
|
||||
# checks the caller ORs together. Absence of the drain-complete line has TWO causes that need
|
||||
# OPPOSITE handling, and a naive "line absent -> the drain died" reading collapses them exactly the
|
||||
# way this whole ticket exists to stop: the line is emitted by the daemon being STOPPED, which is
|
||||
# running the OLD jar. Until a redeploy has landed fleetd #522 once, every previous daemon predates
|
||||
# the line and cannot emit it no matter how cleanly it drained — so on the very first redeploy after
|
||||
# #522 merged, "absent" means "too old to know how", not "died". Only once BOTH signals — this
|
||||
# line's absence AND scan_uncaught_exceptions' result — have been read together can the three real
|
||||
# outcomes be told apart:
|
||||
#
|
||||
# complete -> the line is present: the drain finished. Name the counts it reported.
|
||||
# died -> the line is absent AND an uncaught-exception shape was found: the drain died. Name
|
||||
# what was found.
|
||||
# unknown -> the line is absent AND no exception shape either: cannot tell. Say so, and say why
|
||||
# (predates the line, or failed without throwing) — never worded as a pass or a
|
||||
# failure, and never reassuring: "ok, no ERROR lines" one level up is the exact mistake
|
||||
# this ticket exists to fix, and this outcome must not reproduce it.
|
||||
#
|
||||
# A fourth case, n/a, covers a cold start or a "loaded but wasn't running" restart: no previous
|
||||
# daemon was actually stopped THIS run, so there is no shutdown window in $log_file to have an
|
||||
# opinion about at all — scanning it anyway would read the NEW daemon's own startup lines and could
|
||||
# misreport "cannot tell" on every clean cold start. had_previous_daemon carries that fact in from
|
||||
# the caller (it already knows $OLD_PID) rather than this function re-deriving it from log content.
|
||||
#
|
||||
# Same shape as swap_if_built/refuse_drain_gate (fleetd #521/#528): the decision (which of the four
|
||||
# outcomes applies) and the action (which ok/warn line to print, and setting REDEPLOY_DRAIN_STATE
|
||||
# for the "result" section below to consult) live together in ONE function that the main flow calls
|
||||
# unconditionally — there is no guard left in the main flow to remove, invert, or bypass
|
||||
# independently of this function. Never calls die(): #512's own decision is to warn loudly and let
|
||||
# the redeploy stand, because by the time this is detectable the new daemon is already up and
|
||||
# healthy and failing here would give the operator nothing to do differently.
|
||||
report_shutdown_drain() {
|
||||
local log_file="$1" had_previous_daemon="$2"
|
||||
REDEPLOY_UNCAUGHT_EXCEPTION_COUNT=0
|
||||
REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE=""
|
||||
REDEPLOY_DRAIN_COMPLETE_LINE=""
|
||||
|
||||
if [ "$had_previous_daemon" != 1 ]; then
|
||||
REDEPLOY_DRAIN_STATE="n/a"
|
||||
ok "no previous daemon was running before this restart — nothing to check for a died shutdown drain"
|
||||
return 0
|
||||
fi
|
||||
|
||||
find_drain_complete_line "$log_file"
|
||||
scan_uncaught_exceptions "$log_file"
|
||||
|
||||
if [ -n "$REDEPLOY_DRAIN_COMPLETE_LINE" ]; then
|
||||
REDEPLOY_DRAIN_STATE="complete"
|
||||
ok "previous daemon's shutdown drain finished: $REDEPLOY_DRAIN_COMPLETE_LINE"
|
||||
elif [ "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" -gt 0 ]; then
|
||||
REDEPLOY_DRAIN_STATE="died"
|
||||
warn "previous daemon's shutdown drain DIED — no drain-complete line, and an uncaught exception"
|
||||
warn "was found in its shutdown window ($REDEPLOY_UNCAUGHT_EXCEPTION_COUNT line(s)):"
|
||||
warn " $REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE"
|
||||
warn "Some sessions from the PREVIOUS daemon may not have been released."
|
||||
else
|
||||
REDEPLOY_DRAIN_STATE="unknown"
|
||||
warn "cannot tell whether the previous daemon's shutdown drain finished — no drain-complete line"
|
||||
warn "and no uncaught-exception shape either. This is NOT a pass and NOT a failure: it means"
|
||||
warn "either that daemon predates fleetd #522's drain-complete log line, or its drain failed"
|
||||
warn "without throwing (hung, or returned early)."
|
||||
fi
|
||||
}
|
||||
|
||||
# fleetd #517: extracted so the suite can call this decision directly, the same way #510 extracted
|
||||
# wait_for_daemon_exit so its ordering became checkable. Before this, the only test of the drain-gate
|
||||
# abort message was a grep of this script's own source for the wording — so mutating the `if` below
|
||||
@@ -519,6 +680,32 @@ drain_gate_refusal() {
|
||||
fi
|
||||
}
|
||||
|
||||
# fleetd #528 — drain_gate_refusal above is well tested (four cases, all direct), but nothing made
|
||||
# the MAIN FLOW's abort actually consult it. Before this, the main flow read
|
||||
# `die "$(drain_gate_refusal "$DO_BUILD" "$JAR_STAGED")"` directly, and mutating that one line to a
|
||||
# flat `die "aborted — nothing changed"` left the whole suite at exit 0 with zero FAIL lines and
|
||||
# byte-identical output to a clean run — every one of drain_gate_refusal's own tests still passed,
|
||||
# because they call the predicate directly and never touch this call site. That silently reinstated
|
||||
# the exact defect #517 was filed to fix. Same shape as #521/#526's should_swap/swap_if_built: a
|
||||
# predicate alone is not enough, because a test proving the predicate is right cannot also prove the
|
||||
# main flow consults it. So the decision (drain_gate_refusal) and the action (die) now live together
|
||||
# in ONE function, and the main flow calls it unconditionally instead of building the die() call
|
||||
# itself — there is no guard left in the main flow to remove, invert, or bypass independently of this
|
||||
# function. drain_gate_refusal stays separate and separately tested because the message-selection
|
||||
# logic is worth naming and testing on its own; refuse_drain_gate is the only thing that ever dies.
|
||||
#
|
||||
# What the behavioural tests above still cannot pin on their own: deleting the call to this function
|
||||
# from the main flow altogether — they call refuse_drain_gate directly, never through the main flow,
|
||||
# because sourcing stops before the main flow ever runs (see the SOURCED guard below). That gap is
|
||||
# closed the same way swap_if_built's is: test_refuse_drain_gate_call_site_present greps this script
|
||||
# for the real invocation, the same shape test_swap_ordered_after_wait_and_before_start already uses
|
||||
# for the swap call. Deliberately NOT written out here as a literal quoted string, so this comment
|
||||
# itself can never become a second match for that test's needle.
|
||||
refuse_drain_gate() {
|
||||
local do_build="$1" staged_path="$2"
|
||||
die "$(drain_gate_refusal "$do_build" "$staged_path")"
|
||||
}
|
||||
|
||||
# CB-600: sourceable for testing. When this file is SOURCED (not executed) it stops here — nothing
|
||||
# below runs — so a test harness can `source` it to call check_log_path_matches_plist (or the
|
||||
# other pure helpers above) against a throwaway plist fixture without ever reaching the mutating
|
||||
@@ -677,10 +864,11 @@ if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
||||
echo
|
||||
read -r -p " Fleet drained? type yes to restart: " reply
|
||||
if [ "$reply" != "yes" ]; then
|
||||
# fleetd #493 / #517: "nothing changed" would be a lie once a build has run and staged a jar —
|
||||
# see drain_gate_refusal above for the full decision and why each of its four cases reads the
|
||||
# way it does.
|
||||
die "$(drain_gate_refusal "$DO_BUILD" "$JAR_STAGED")"
|
||||
# fleetd #493 / #517 / #528: "nothing changed" would be a lie once a build has run and staged a
|
||||
# jar — see drain_gate_refusal above for the full decision and why each of its four cases reads
|
||||
# the way it does. refuse_drain_gate composes that message AND calls die itself, so this guard
|
||||
# has nothing left of its own to get wrong beyond whether it calls refuse_drain_gate at all.
|
||||
refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"
|
||||
fi
|
||||
fi
|
||||
|
||||
@@ -736,16 +924,20 @@ if [ -n "$OLD_PID" ]; then
|
||||
elif [ "$SUPERVISOR_KIND" = "launchd" ]; then
|
||||
# Loaded but not currently running (e.g. throttled after a crash loop). Unload it anyway so the
|
||||
# start step below does a clean load, never a load stacked on an already-loaded label.
|
||||
# fleetd #504: unload_launchd_if_loaded (above) tolerates a genuine already-unloaded answer but
|
||||
# dies on a real `launchctl` failure — never a bare `|| true` that would print `ok` either way.
|
||||
say "stop"
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
||||
launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true
|
||||
unload_launchd_if_loaded
|
||||
ok "launchd agent unloaded (was already not running)"
|
||||
elif [ "$SUPERVISOR_KIND" = "systemd" ]; then
|
||||
# Same case for systemd: the unit is known/active-capable but not currently running. `stop` on an
|
||||
# already-stopped unit is a harmless no-op — kept for symmetry with the launchd branch above.
|
||||
# fleetd #504: stop_systemd_if_loaded (above) tolerates that genuine no-op but dies on a real
|
||||
# `systemctl` failure — never a bare `|| true` that would print `ok` either way.
|
||||
say "stop"
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
||||
systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true
|
||||
stop_systemd_if_loaded
|
||||
ok "systemd --user unit stopped (was already not running)"
|
||||
else
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
||||
@@ -873,6 +1065,14 @@ trap 'rm -f "$FRESH_LOG"' EXIT
|
||||
tail -n "+$((RESTART_MARK + 1))" "$OUT" > "$FRESH_LOG" 2>/dev/null || true
|
||||
classify_amqp_connection_errors "$FRESH_LOG"
|
||||
|
||||
# fleetd #512 part 2: the previous daemon's shutdown drain, checked in the same fresh-log region —
|
||||
# see report_shutdown_drain above for the full decision (four outcomes, one of them a deliberate
|
||||
# "cannot tell"). HAD_OLD_PID crosses in whether a previous daemon was actually stopped this run;
|
||||
# see the function's own comment for why that matters.
|
||||
say "previous daemon's shutdown drain"
|
||||
HAD_OLD_PID=0; [ -n "$OLD_PID" ] && HAD_OLD_PID=1
|
||||
report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"
|
||||
|
||||
# fleetd #492: checked here, after healthz and the fresh-log check have both had time to run, so a
|
||||
# supervisor that revives the OLD jar a few seconds late is caught too. Every check above (healthz
|
||||
# 200, jar id, the fresh 'listening' line) is satisfied by EITHER daemon if two are alive — this is
|
||||
@@ -882,7 +1082,14 @@ assert_single_daemon "$(running_pid)"
|
||||
say "result"
|
||||
ok "pid $NEW_PID, jar $(jar_id)"
|
||||
if [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then
|
||||
ok "no ERROR lines since restart"
|
||||
# fleetd #512 item 4: this line must not print when the shutdown-drain check above found the
|
||||
# previous daemon's drain died, or could not tell — either would make "no ERROR lines" read as a
|
||||
# clean bill of health it is not (an uncaught exception never carries an ERROR token to begin
|
||||
# with, so this count alone cannot see that failure). "complete" and "n/a" are the only two
|
||||
# outcomes report_shutdown_drain sets that mean nothing is wrong there.
|
||||
if [ "$REDEPLOY_DRAIN_STATE" = "complete" ] || [ "$REDEPLOY_DRAIN_STATE" = "n/a" ]; then
|
||||
ok "no ERROR lines since restart"
|
||||
fi
|
||||
elif [ "$REDEPLOY_UNEXPLAINED_ERRORS" -eq 0 ]; then
|
||||
ok "$REDEPLOY_RECOVERED_AMQP_ERRORS AMQP connection reset ERROR lines recovered since restart"
|
||||
else
|
||||
|
||||
@@ -521,6 +521,210 @@ test_drain_gate_refusal_no_build_staged_absent() {
|
||||
assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording"
|
||||
}
|
||||
|
||||
# fleetd #528 — the four tests above pin drain_gate_refusal(), and that is ALL they pin: they call
|
||||
# the predicate directly and never touch the main flow's call site. That was measured to be not
|
||||
# enough, the same way test_should_swap_true_when_build_ran/test_should_swap_false_when_build_skipped
|
||||
# were not enough for #521: with the main flow reading `die "$(drain_gate_refusal "$DO_BUILD"
|
||||
# "$JAR_STAGED")"`, replacing that whole line with a flat `die "aborted — nothing changed"` left this
|
||||
# suite at exit 0 with zero FAIL lines and byte-identical output to a clean run. Nothing above could
|
||||
# tell the difference, because none of it calls anything at or above the call site itself.
|
||||
#
|
||||
# So these four call refuse_drain_gate() — the function the main flow actually calls, holding the
|
||||
# composed message and the die() together — with die() stubbed to RECORD whether it was called and
|
||||
# with what message, instead of exiting the process. That fails if refuse_drain_gate stops consulting
|
||||
# drain_gate_refusal, mangles what it passes it, or simply never calls die.
|
||||
#
|
||||
# What none of these four can catch: deleting the `refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"` line
|
||||
# from the main flow altogether — see the comment above refuse_drain_gate in redeploy-fleetd.sh for
|
||||
# why no test in this file can do better than that (sourcing stops before the main flow runs).
|
||||
DIED_CALLED=0
|
||||
DIED_MESSAGE=""
|
||||
stub_die_recorder() {
|
||||
DIED_CALLED=0
|
||||
DIED_MESSAGE=""
|
||||
die() { DIED_CALLED=1; DIED_MESSAGE="$*"; }
|
||||
}
|
||||
|
||||
test_refuse_drain_gate_build_ran_staged_present() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local dir staged
|
||||
dir="$TMP/refuse-drain-build-staged"; mkdir -p "$dir"
|
||||
staged="$dir/fleetd-new.jar"
|
||||
printf 'staged jar bytes' > "$staged"
|
||||
stub_die_recorder
|
||||
refuse_drain_gate 1 "$staged"
|
||||
[ "$DIED_CALLED" = 1 ] \
|
||||
|| fail "refuse_drain_gate build-ran+staged-present must call die, and did not"
|
||||
printf '%s' "$DIED_MESSAGE" | grep -qF "$staged" \
|
||||
|| fail "refuse_drain_gate build-ran+staged-present die message does not name the staged jar"
|
||||
printf '%s' "$DIED_MESSAGE" | grep -qF 'Rerun WITHOUT --no-build' \
|
||||
|| fail "refuse_drain_gate build-ran+staged-present die message is missing the rerun instruction"
|
||||
if printf '%s' "$DIED_MESSAGE" | grep -qF 'nothing changed'; then
|
||||
fail "refuse_drain_gate build-ran+staged-present must not claim nothing changed — the jar already moved"
|
||||
fi
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_refuse_drain_gate_build_ran_staged_absent() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local dir
|
||||
dir="$TMP/refuse-drain-build-no-staged"; mkdir -p "$dir"
|
||||
stub_die_recorder
|
||||
refuse_drain_gate 1 "$dir/fleetd-new.jar"
|
||||
[ "$DIED_CALLED" = 1 ] \
|
||||
|| fail "refuse_drain_gate build-ran+staged-absent must call die, and did not"
|
||||
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate build-ran+staged-absent die message"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_refuse_drain_gate_no_build_staged_present() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local dir staged
|
||||
dir="$TMP/refuse-drain-no-build-staged"; mkdir -p "$dir"
|
||||
staged="$dir/fleetd-new.jar"
|
||||
printf 'leftover staged jar bytes' > "$staged"
|
||||
stub_die_recorder
|
||||
refuse_drain_gate 0 "$staged"
|
||||
[ "$DIED_CALLED" = 1 ] \
|
||||
|| fail "refuse_drain_gate no-build+staged-present must call die, and did not"
|
||||
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-present die message"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_refuse_drain_gate_no_build_staged_absent() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local dir
|
||||
dir="$TMP/refuse-drain-no-build-no-staged"; mkdir -p "$dir"
|
||||
stub_die_recorder
|
||||
refuse_drain_gate 0 "$dir/fleetd-new.jar"
|
||||
[ "$DIED_CALLED" = 1 ] \
|
||||
|| fail "refuse_drain_gate no-build+staged-absent must call die, and did not"
|
||||
assert_equals "aborted — nothing changed" "$DIED_MESSAGE" "refuse_drain_gate no-build+staged-absent die message"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #528 — closes the one gap the four behavioural tests above cannot: they call
|
||||
# refuse_drain_gate directly, and sourcing stops before the main flow ever runs (the SOURCED guard),
|
||||
# so none of them can prove the main flow still CALLS refuse_drain_gate at all. Same shape as
|
||||
# test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site. This is
|
||||
# what actually kills the item-1 mutation from the ticket — replacing the main flow's call with a
|
||||
# flat `die "aborted — nothing changed"` removes this exact needle, where none of the behavioural
|
||||
# tests above would even notice.
|
||||
#
|
||||
# The grep ends `|| true`: this file runs under `set -euo pipefail`, so an ABSENT needle would fail
|
||||
# the assignment and `set -e` would kill the whole suite before the `[ -n ... ] || fail` guard below
|
||||
# ever ran — the exact dead-check shape fleetd #528 also flags as a sweep finding (see the PR body).
|
||||
test_refuse_drain_gate_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'refuse_drain_gate "$DO_BUILD" "$JAR_STAGED"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's refuse_drain_gate call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #504 — the "loaded but not currently running" branches for launchd/systemd used to run
|
||||
# `launchctl unload`/`systemctl --user stop` with `2>/dev/null || true` and print `ok`
|
||||
# unconditionally, so a real supervisor failure (e.g. it cannot reach launchd/the systemd user bus)
|
||||
# read exactly like a harmless already-stopped answer. unload_launchd_if_loaded/
|
||||
# stop_systemd_if_loaded (redeploy-fleetd.sh, right after systemd_loaded) apply systemd_loaded's own
|
||||
# "capture stderr separately — only a non-zero exit WITH stderr is a real failure" pattern to the
|
||||
# WRITE side. Both call the real `launchctl`/`systemctl` binaries directly (they are not overridable
|
||||
# wrapper functions the way launchd_loaded/systemd_loaded are), so these tests put a stub binary
|
||||
# first on PATH — the same technique test_detect_supervisor_systemd_probe_error_is_unclear above
|
||||
# already uses for `systemctl`.
|
||||
test_unload_launchd_if_loaded_dies_on_real_failure() {
|
||||
local bin_dir output rc=0 saved_plist="$LAUNCHD_PLIST"
|
||||
bin_dir="$TMP/stub-bin-launchctl-error"; mkdir -p "$bin_dir"
|
||||
cat > "$bin_dir/launchctl" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
echo "Could not find specified service" >&2
|
||||
exit 1
|
||||
STUB
|
||||
chmod +x "$bin_dir/launchctl"
|
||||
LAUNCHD_PLIST="$TMP/fake-fail.plist"
|
||||
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
|
||||
LAUNCHD_PLIST="$saved_plist"
|
||||
[ "$rc" -ne 0 ] \
|
||||
|| fail "unload_launchd_if_loaded must die when launchctl exits non-zero AND writes to stderr"
|
||||
printf '%s' "$output" | grep -qF 'launchctl unload' \
|
||||
|| fail "die message does not name the failing launchctl unload command"
|
||||
}
|
||||
|
||||
# Captured via $(...) rather than called bare: unload_launchd_if_loaded's own die() does a hard
|
||||
# `exit`, and calling it directly at this level would let a regression that makes it die on this
|
||||
# clean-negative case kill the WHOLE suite before the `|| fail` below ever ran — printing die's own
|
||||
# message instead of this test's. Inside a command substitution, that `exit` only ends the subshell
|
||||
# (a-guard-is-defeated-by-its-calling-context: the same reason the *_dies_on_real_failure tests
|
||||
# above capture this way), so this test's own message is what actually reaches the report.
|
||||
test_unload_launchd_if_loaded_tolerates_clean_negative() {
|
||||
local bin_dir saved_plist="$LAUNCHD_PLIST" output rc=0
|
||||
bin_dir="$TMP/stub-bin-launchctl-noop"; mkdir -p "$bin_dir"
|
||||
cat > "$bin_dir/launchctl" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
exit 1
|
||||
STUB
|
||||
chmod +x "$bin_dir/launchctl"
|
||||
LAUNCHD_PLIST="$TMP/fake-noop.plist"
|
||||
output="$(PATH="$bin_dir:$PATH" unload_launchd_if_loaded 2>&1)" || rc=$?
|
||||
LAUNCHD_PLIST="$saved_plist"
|
||||
[ "$rc" -eq 0 ] \
|
||||
|| fail "unload_launchd_if_loaded must tolerate a clean already-unloaded answer (non-zero exit, empty stderr): $output"
|
||||
}
|
||||
|
||||
test_stop_systemd_if_loaded_dies_on_real_failure() {
|
||||
local bin_dir output rc=0
|
||||
bin_dir="$TMP/stub-bin-systemctl-stop-error"; mkdir -p "$bin_dir"
|
||||
cat > "$bin_dir/systemctl" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
echo "Failed to connect to bus: No such file or directory" >&2
|
||||
exit 1
|
||||
STUB
|
||||
chmod +x "$bin_dir/systemctl"
|
||||
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] \
|
||||
|| fail "stop_systemd_if_loaded must die when systemctl exits non-zero AND writes to stderr"
|
||||
printf '%s' "$output" | grep -qF 'systemctl --user stop' \
|
||||
|| fail "die message does not name the failing systemctl --user stop command"
|
||||
}
|
||||
|
||||
# Same subshell-capture reasoning as test_unload_launchd_if_loaded_tolerates_clean_negative above:
|
||||
# stop_systemd_if_loaded's own die() does a hard `exit`, so this must run inside $(...) or a
|
||||
# regression here would kill the whole suite with die's message instead of this test's.
|
||||
test_stop_systemd_if_loaded_tolerates_clean_negative() {
|
||||
local bin_dir output rc=0
|
||||
bin_dir="$TMP/stub-bin-systemctl-stop-noop"; mkdir -p "$bin_dir"
|
||||
cat > "$bin_dir/systemctl" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
exit 1
|
||||
STUB
|
||||
chmod +x "$bin_dir/systemctl"
|
||||
output="$(PATH="$bin_dir:$PATH" stop_systemd_if_loaded 2>&1)" || rc=$?
|
||||
[ "$rc" -eq 0 ] \
|
||||
|| fail "stop_systemd_if_loaded must tolerate a clean already-stopped answer (non-zero exit, empty stderr): $output"
|
||||
}
|
||||
|
||||
# Closes the same gap test_refuse_drain_gate_call_site_present closes for the drain gate: the four
|
||||
# tests above call unload_launchd_if_loaded/stop_systemd_if_loaded directly, and sourcing stops
|
||||
# before the main flow ever runs (the SOURCED guard), so none of them can prove the main flow still
|
||||
# CALLS these two functions instead of the original bare `2>/dev/null || true`. A source-text check,
|
||||
# like test_swap_ordered_after_wait_and_before_start. The call-site needle is anchored (`^ name$`)
|
||||
# so it cannot be satisfied by the comment lines above each call site that merely mention the
|
||||
# function by name.
|
||||
test_stop_branches_call_tolerant_helpers_not_bare_or_true() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" unload_call_line stop_call_line
|
||||
unload_call_line="$(grep -n '^ unload_launchd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
stop_call_line="$(grep -n '^ stop_systemd_if_loaded$' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$unload_call_line" ] \
|
||||
|| fail "could not find the main flow's call to unload_launchd_if_loaded in redeploy-fleetd.sh"
|
||||
[ -n "$stop_call_line" ] \
|
||||
|| fail "could not find the main flow's call to stop_systemd_if_loaded in redeploy-fleetd.sh"
|
||||
if grep -qF 'launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true' "$src"; then
|
||||
fail "the bare 'launchctl unload ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
|
||||
fi
|
||||
if grep -qF 'systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true' "$src"; then
|
||||
fail "the bare 'systemctl --user stop ... 2>/dev/null || true' defect (fleetd #504) is back in redeploy-fleetd.sh"
|
||||
fi
|
||||
}
|
||||
|
||||
test_no_errors() {
|
||||
cat > "$TMP/no-errors.log" <<'LOG'
|
||||
2026-09-05 12:00:00 INFO fleetd listening
|
||||
@@ -720,6 +924,164 @@ test_unattributable_quiet_mutation_is_caught() {
|
||||
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
|
||||
}
|
||||
|
||||
# fleetd #512 part 2 — the negative check (scan_uncaught_exceptions). The heart of this half of the
|
||||
# ticket: a fixture with the uncaught-exception shape and NO line carrying an ERROR token at all,
|
||||
# proving the scan finds it without one. A fixture that also carried an ERROR line would pass for
|
||||
# the wrong reason.
|
||||
test_scan_uncaught_exceptions_finds_shape_without_error_token() {
|
||||
cat > "$TMP/scan-died.log" <<'LOG'
|
||||
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
||||
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
|
||||
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
|
||||
LOG
|
||||
local error_count
|
||||
error_count="$(grep -c ' ERROR ' "$TMP/scan-died.log" || true)"
|
||||
[ "$error_count" = "0" ] \
|
||||
|| fail "test fixture error: scan-died.log unexpectedly carries an ERROR token"
|
||||
scan_uncaught_exceptions "$TMP/scan-died.log"
|
||||
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "scan must find the exception without an ERROR token"
|
||||
printf '%s' "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" | grep -qF 'NoClassDefFoundError' \
|
||||
|| fail "scan did not capture the matching line as the sample"
|
||||
}
|
||||
|
||||
test_scan_uncaught_exceptions_clean_control() {
|
||||
cat > "$TMP/scan-clean.log" <<'LOG'
|
||||
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
||||
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=0 abandoned=0 (still BUSY at the shutdown deadline)
|
||||
LOG
|
||||
scan_uncaught_exceptions "$TMP/scan-clean.log"
|
||||
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "clean control must find no uncaught exception"
|
||||
assert_equals "" "$REDEPLOY_UNCAUGHT_EXCEPTION_SAMPLE" "clean control sample must be empty"
|
||||
}
|
||||
|
||||
# fleetd #512 part 2 — the positive check (find_drain_complete_line). Both halves of #522's line:
|
||||
# present, and absent.
|
||||
test_find_drain_complete_line_present() {
|
||||
cat > "$TMP/drain-line-present.log" <<'LOG'
|
||||
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=2 abandoned=1 (still BUSY at the shutdown deadline)
|
||||
LOG
|
||||
find_drain_complete_line "$TMP/drain-line-present.log"
|
||||
printf '%s' "$REDEPLOY_DRAIN_COMPLETE_LINE" | grep -qF 'released=2 abandoned=1' \
|
||||
|| fail "find_drain_complete_line did not capture the present line"
|
||||
}
|
||||
|
||||
test_find_drain_complete_line_absent() {
|
||||
cat > "$TMP/drain-line-absent.log" <<'LOG'
|
||||
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
||||
LOG
|
||||
find_drain_complete_line "$TMP/drain-line-absent.log"
|
||||
assert_equals "" "$REDEPLOY_DRAIN_COMPLETE_LINE" "find_drain_complete_line must report empty when absent"
|
||||
}
|
||||
|
||||
# fleetd #512 part 2 — report_shutdown_drain, the composite decision+action function the main flow
|
||||
# calls unconditionally (same shape as swap_if_built/refuse_drain_gate, #521/#528). These four cover
|
||||
# the four outcomes named in the ticket's "trap": complete, died, unknown ("cannot tell" — neither a
|
||||
# pass nor a failure), and n/a (no previous daemon was actually stopped this run).
|
||||
#
|
||||
# Deliberately NOT run inside `$(...)`: report_shutdown_drain sets REDEPLOY_DRAIN_STATE as a global
|
||||
# side effect that these tests need to read back afterward, and a command substitution forks a
|
||||
# subshell that global assignment would not survive (the exact trap documented above
|
||||
# detect_supervisor in redeploy-fleetd.sh, for the same reason). Plain output redirection to a file
|
||||
# does not fork a subshell, so it is used to capture what was printed instead.
|
||||
test_report_shutdown_drain_died_without_error_token() {
|
||||
cat > "$TMP/drain-died.log" <<'LOG'
|
||||
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
||||
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
|
||||
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
|
||||
at dev.ltms.fleet.session.SessionManager.drainAll(SessionManager.java:1081)
|
||||
LOG
|
||||
local error_count
|
||||
error_count="$(grep -c ' ERROR ' "$TMP/drain-died.log" || true)"
|
||||
[ "$error_count" = "0" ] \
|
||||
|| fail "test fixture error: drain-died.log unexpectedly carries an ERROR token"
|
||||
|
||||
report_shutdown_drain "$TMP/drain-died.log" 1 > "$TMP/drain-died-output" 2>&1
|
||||
assert_equals "died" "$REDEPLOY_DRAIN_STATE" "died fixture must set REDEPLOY_DRAIN_STATE=died"
|
||||
assert_equals 1 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "died fixture uncaught-exception count"
|
||||
grep -qF 'NoClassDefFoundError' "$TMP/drain-died-output" \
|
||||
|| fail "report_shutdown_drain did not report the uncaught-exception shape it found"
|
||||
grep -qF 'DIED' "$TMP/drain-died-output" \
|
||||
|| fail "report_shutdown_drain did not report the drain as DIED"
|
||||
}
|
||||
|
||||
test_report_shutdown_drain_complete_control() {
|
||||
cat > "$TMP/drain-complete.log" <<'LOG'
|
||||
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
||||
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
|
||||
2026-09-12 10:15:05 INFO dev.ltms.fleet.session.SessionManager - drain complete: released=3 abandoned=0 (still BUSY at the shutdown deadline)
|
||||
LOG
|
||||
report_shutdown_drain "$TMP/drain-complete.log" 1 > "$TMP/drain-complete-output" 2>&1
|
||||
assert_equals "complete" "$REDEPLOY_DRAIN_STATE" "complete-control fixture must set REDEPLOY_DRAIN_STATE=complete"
|
||||
assert_equals 0 "$REDEPLOY_UNCAUGHT_EXCEPTION_COUNT" "complete-control fixture must find no uncaught exception"
|
||||
grep -qF 'released=3 abandoned=0' "$TMP/drain-complete-output" \
|
||||
|| fail "report_shutdown_drain did not report the drain-complete counts"
|
||||
}
|
||||
|
||||
test_report_shutdown_drain_unknown_cannot_tell() {
|
||||
cat > "$TMP/drain-unknown.log" <<'LOG'
|
||||
2026-09-12 10:15:00 INFO fleetd listening on 127.0.0.1:8765
|
||||
2026-09-12 10:15:05 INFO dev.ltms.fleet.Fleetd - shutting down
|
||||
LOG
|
||||
report_shutdown_drain "$TMP/drain-unknown.log" 1 > "$TMP/drain-unknown-output" 2>&1
|
||||
assert_equals "unknown" "$REDEPLOY_DRAIN_STATE" "cannot-tell fixture must set REDEPLOY_DRAIN_STATE=unknown"
|
||||
grep -qF 'cannot tell' "$TMP/drain-unknown-output" \
|
||||
|| fail "report_shutdown_drain did not say it could not tell"
|
||||
if grep -qF ' ok' "$TMP/drain-unknown-output"; then
|
||||
fail "cannot-tell outcome must not be printed via ok() — it is neither a pass nor a failure"
|
||||
fi
|
||||
}
|
||||
|
||||
# A cold start (or a restart where nothing was actually stopped) has no previous-daemon shutdown
|
||||
# window to have an opinion about at all. This fixture's log content looks exactly like a died drain
|
||||
# — proving the had_previous_daemon=0 gate is actually consulted, not merely documented: without it,
|
||||
# this would misreport "died" or "unknown" on every clean cold start.
|
||||
test_report_shutdown_drain_no_previous_daemon_is_na() {
|
||||
cat > "$TMP/drain-na.log" <<'LOG'
|
||||
Exception in thread "Thread-0" java.lang.NoClassDefFoundError: reactor/core/Exceptions
|
||||
LOG
|
||||
report_shutdown_drain "$TMP/drain-na.log" 0 > "$TMP/drain-na-output" 2>&1
|
||||
assert_equals "n/a" "$REDEPLOY_DRAIN_STATE" "no-previous-daemon fixture must set REDEPLOY_DRAIN_STATE=n/a even though the log content looks like a died drain"
|
||||
grep -qF 'nothing to check' "$TMP/drain-na-output" \
|
||||
|| fail "report_shutdown_drain did not report that there was nothing to check"
|
||||
}
|
||||
|
||||
# fleetd #512 — closes the gap none of the seven tests above can: they call report_shutdown_drain
|
||||
# directly, and sourcing stops before the main flow ever runs (the SOURCED guard), so none of them
|
||||
# can prove the main flow still calls it at all. Same shape as test_refuse_drain_gate_call_site_present
|
||||
# and test_swap_ordered_after_wait_and_before_start: a source-text grep for the real call site, plus
|
||||
# an ordering check against its neighbours in the verify/result flow.
|
||||
test_report_shutdown_drain_call_site_present() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" call_line
|
||||
call_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$call_line" ] \
|
||||
|| fail "could not find the main flow's report_shutdown_drain call site in redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_report_shutdown_drain_ordered_after_classify_and_before_result() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" classify_line drain_line result_line
|
||||
classify_line="$(grep -Fn 'classify_amqp_connection_errors "$FRESH_LOG"' "$src" | tail -1 | cut -d: -f1 || true)"
|
||||
drain_line="$(grep -Fn 'report_shutdown_drain "$FRESH_LOG" "$HAD_OLD_PID"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
result_line="$(grep -Fn 'say "result"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$classify_line" ] || fail "could not find the classify_amqp_connection_errors call site"
|
||||
[ -n "$drain_line" ] || fail "could not find the report_shutdown_drain call site"
|
||||
[ -n "$result_line" ] || fail "could not find the result section"
|
||||
[ "$drain_line" -gt "$classify_line" ] \
|
||||
|| fail "report_shutdown_drain (line $drain_line) is not after classify_amqp_connection_errors (line $classify_line)"
|
||||
[ "$drain_line" -lt "$result_line" ] \
|
||||
|| fail "report_shutdown_drain (line $drain_line) is not before the result section (line $result_line)"
|
||||
}
|
||||
|
||||
# fleetd #512 item 4 — the summary line must not read as reassurance when the shutdown-drain check
|
||||
# found something wrong (or could not tell). Sourcing stops before the main flow runs, so this is a
|
||||
# source-text check like test_drain_gate_abort_message_says_no_no_build above.
|
||||
test_no_error_lines_message_gated_by_drain_state() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" block
|
||||
block="$(grep -B2 -F 'ok "no ERROR lines since restart"' "$src")"
|
||||
[ -n "$block" ] || fail "could not find the 'no ERROR lines since restart' line in redeploy-fleetd.sh"
|
||||
printf '%s' "$block" | grep -qF 'REDEPLOY_DRAIN_STATE' \
|
||||
|| fail "'no ERROR lines since restart' is not guarded by the shutdown-drain outcome (fleetd #512 item 4)"
|
||||
}
|
||||
|
||||
test_detect_supervisor_launchd_only
|
||||
test_detect_supervisor_systemd_only
|
||||
test_detect_supervisor_none
|
||||
@@ -753,6 +1115,16 @@ test_drain_gate_refusal_build_ran_staged_present
|
||||
test_drain_gate_refusal_build_ran_staged_absent
|
||||
test_drain_gate_refusal_no_build_staged_present
|
||||
test_drain_gate_refusal_no_build_staged_absent
|
||||
test_refuse_drain_gate_build_ran_staged_present
|
||||
test_refuse_drain_gate_build_ran_staged_absent
|
||||
test_refuse_drain_gate_no_build_staged_present
|
||||
test_refuse_drain_gate_no_build_staged_absent
|
||||
test_refuse_drain_gate_call_site_present
|
||||
test_unload_launchd_if_loaded_dies_on_real_failure
|
||||
test_unload_launchd_if_loaded_tolerates_clean_negative
|
||||
test_stop_systemd_if_loaded_dies_on_real_failure
|
||||
test_stop_systemd_if_loaded_tolerates_clean_negative
|
||||
test_stop_branches_call_tolerant_helpers_not_bare_or_true
|
||||
test_no_errors
|
||||
test_recovery_patterns_match_source
|
||||
test_attributed_recovered_connection_error
|
||||
@@ -764,4 +1136,15 @@ test_other_error_is_unexplained
|
||||
test_recovery_requirement_mutation_is_caught
|
||||
test_shared_counter_mutation_is_caught
|
||||
test_unattributable_quiet_mutation_is_caught
|
||||
test_scan_uncaught_exceptions_finds_shape_without_error_token
|
||||
test_scan_uncaught_exceptions_clean_control
|
||||
test_find_drain_complete_line_present
|
||||
test_find_drain_complete_line_absent
|
||||
test_report_shutdown_drain_died_without_error_token
|
||||
test_report_shutdown_drain_complete_control
|
||||
test_report_shutdown_drain_unknown_cannot_tell
|
||||
test_report_shutdown_drain_no_previous_daemon_is_na
|
||||
test_report_shutdown_drain_call_site_present
|
||||
test_report_shutdown_drain_ordered_after_classify_and_before_result
|
||||
test_no_error_lines_message_gated_by_drain_state
|
||||
printf 'PASS: redeploy log classifier\n'
|
||||
|
||||
Reference in New Issue
Block a user