Compare commits
35 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| be07ed2033 | |||
| a6415f3e52 | |||
| bb6fc9e0d7 | |||
| 3366590dbe | |||
| 01adc841fa | |||
| 08771e270b | |||
| b5843ab43f | |||
| d8985719eb | |||
| 5b1e13ca3d | |||
| c89a375e5d | |||
| 37dcefa834 | |||
| 6a7342b1f0 | |||
| a5ad7c6561 | |||
| c71ac231e5 | |||
| 33720c42b3 | |||
| 3833d8e52b | |||
| b37def9238 | |||
| 8f02576df6 | |||
| 525bc1c5f4 | |||
| d59ece6dec | |||
| 32408d1e64 | |||
| 6e23bf8309 | |||
| aa4c0b84c3 | |||
| 40c593cd09 | |||
| 979adf82eb | |||
| 36870836aa | |||
| 136312fb11 | |||
| 4f28da62a3 | |||
| 708f1795ad | |||
| 599419f9e6 | |||
| ac351ee1de | |||
| 274afafde6 | |||
| 8f59019305 | |||
| b17f37a683 | |||
| dcd505286f |
@@ -96,8 +96,10 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
||||
forge MCP server it appears to have holds a blocked credential and fails every call, and a piped
|
||||
command (`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a
|
||||
fact. Its injected repo-scoped `GITEA_TOKEN` is a different credential and does work, so a worker
|
||||
reporting that it opened its own PR is reporting something it really can do.
|
||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||
@@ -200,8 +202,11 @@ simply complies has thrown away the reason there are two of you.
|
||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
||||
find there holds a deliberately blocked credential and fails every call, by design.
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge MCP server you
|
||||
may find there holds a deliberately blocked credential and fails every call, by design. That is
|
||||
not your only forge route, and the two must not be confused: the repo-scoped `GITEA_TOKEN` the
|
||||
daemon injects into your environment does work, and using it to open your own PR is part of the
|
||||
job. A blocked MCP tool is never a reason to skip that step.
|
||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
|
||||
@@ -80,6 +80,7 @@ import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
@@ -270,15 +271,12 @@ public final class Fleetd {
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
HerdrAwaitOutcome herdrOutcome = awaitHerdr(herdr, System::nanoTime, Fleetd::sleepHerdrPoll);
|
||||
boolean herdrUp = logHerdrWaitOutcomeAndShouldReap(herdrOutcome);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
@@ -696,7 +694,7 @@ public final class Fleetd {
|
||||
}, outagePolicy);
|
||||
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics,
|
||||
primaryRegistry, callers, FleetMcp.AuthorizationMode.ENFORCED, metrics,
|
||||
capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
||||
new FleetMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
@@ -1696,12 +1694,70 @@ public final class Fleetd {
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
* How {@link #awaitHerdr} ended (fleetd #498). The old code returned a bare {@code boolean},
|
||||
* which collapsed two different facts onto the same {@code false}: the configured wait budget
|
||||
* genuinely running out, and the waiting thread being interrupted possibly milliseconds in.
|
||||
* Those need different operator messages — see {@link #logHerdrWaitOutcomeAndShouldReap} — so
|
||||
* this is a third state, not a better number (the same shape fleetd #497 named). Never treat
|
||||
* {@link #INTERRUPTED} as if it were {@link #DEADLINE_PASSED}: only the latter means herdr was
|
||||
* actually given the full {@link #HERDR_WAIT_SECONDS} and still failed to answer.
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
enum HerdrWaitResult {
|
||||
/** herdr answered {@code ping} before the deadline. */
|
||||
ANSWERED,
|
||||
/** the configured {@link #HERDR_WAIT_SECONDS} budget elapsed with no answer. */
|
||||
DEADLINE_PASSED,
|
||||
/**
|
||||
* the waiting thread was interrupted before the budget ran out — a different event from
|
||||
* {@link #DEADLINE_PASSED} and must never be reported as "did not answer within Ns".
|
||||
*/
|
||||
INTERRUPTED
|
||||
}
|
||||
|
||||
/**
|
||||
* The outcome of one {@link #awaitHerdr} call, carrying the MEASURED elapsed wait time
|
||||
* alongside {@link #result}. {@code elapsedNanos} is always measured against the {@code nanos}
|
||||
* supplier passed to {@link #awaitHerdr} — never assume it equals the configured budget, the
|
||||
* same defect fleetd #494 already fixed once in {@code LeadRollover}.
|
||||
*/
|
||||
record HerdrAwaitOutcome(HerdrWaitResult result, long elapsedNanos) {}
|
||||
|
||||
/**
|
||||
* The real per-poll wait {@link #main} passes to {@link #awaitHerdr}: sleep
|
||||
* {@link #HERDR_WAIT_POLL_MILLIS}, and on interruption re-set the thread's interrupt flag
|
||||
* rather than throwing — {@link #awaitHerdr} detects an interruption by checking {@link
|
||||
* Thread#isInterrupted()} right after this returns, so a poller that swallowed the flag
|
||||
* instead of restoring it would make that check silently miss the interruption.
|
||||
*/
|
||||
private static void sleepHerdrPoll() {
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers, the configured {@link #HERDR_WAIT_SECONDS}
|
||||
* budget elapses, or the waiting thread is interrupted (CB-504, fleetd #498).
|
||||
*
|
||||
* <p>{@code nanos} and {@code poller} are required parameters with no defaulted overload
|
||||
* (fleetd #415's shape: a defaulted overload is a silent survivor a green suite would vouch
|
||||
* for) — the previous version read {@link System#nanoTime()} and called {@link Thread#sleep}
|
||||
* directly, so nothing could drive it from a test. The one production call site in {@link
|
||||
* #main} passes {@code System::nanoTime} and {@link #sleepHerdrPoll}.
|
||||
*
|
||||
* @param nanos a monotonic elapsed-time clock, e.g. {@code System::nanoTime} — never a
|
||||
* wall-clock source, since only elapsed time (not a timestamp) is measured here
|
||||
* @param poller called once per failed ping while the budget remains; must, on an
|
||||
* {@link InterruptedException}, re-set the thread's interrupt flag rather than
|
||||
* throw or swallow it — this method's interruption check reads that flag right
|
||||
* after {@code poller.run()} returns
|
||||
* @return the outcome and the measured elapsed wait time — see {@link HerdrAwaitOutcome}
|
||||
*/
|
||||
static HerdrAwaitOutcome awaitHerdr(HerdrClient herdr, LongSupplier nanos, Runnable poller) {
|
||||
long start = nanos.getAsLong();
|
||||
long deadline = start + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
@@ -1709,25 +1765,57 @@ public final class Fleetd {
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
return new HerdrAwaitOutcome(HerdrWaitResult.ANSWERED, nanos.getAsLong() - start);
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
if (nanos.getAsLong() >= deadline) {
|
||||
return new HerdrAwaitOutcome(HerdrWaitResult.DEADLINE_PASSED, nanos.getAsLong() - start);
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
poller.run();
|
||||
if (Thread.currentThread().isInterrupted()) {
|
||||
return new HerdrAwaitOutcome(HerdrWaitResult.INTERRUPTED, nanos.getAsLong() - start);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Log the right message for {@code outcome} — never the configured {@link #HERDR_WAIT_SECONDS}
|
||||
* budget alone, always the measured elapsed time next to it — and say whether {@link #main}
|
||||
* should now reap orphan worker panes (fleetd #498).
|
||||
*
|
||||
* <p>Extracted out of {@link #main} so this decision is drivable from a test: {@link #main}
|
||||
* boots the whole daemon and cannot itself be run in a unit test, but this is the exact,
|
||||
* unmodified code {@link #main} calls for the decision, not a re-derivation of it.
|
||||
*
|
||||
* @return true only for {@link HerdrWaitResult#ANSWERED} — orphan workers are reaped only
|
||||
* then, exactly as before this ticket
|
||||
*/
|
||||
static boolean logHerdrWaitOutcomeAndShouldReap(HerdrAwaitOutcome outcome) {
|
||||
long elapsedMillis = TimeUnit.NANOSECONDS.toMillis(outcome.elapsedNanos());
|
||||
if (outcome.result() == HerdrWaitResult.ANSWERED) {
|
||||
return true;
|
||||
}
|
||||
if (outcome.result() == HerdrWaitResult.DEADLINE_PASSED) {
|
||||
log.warn("herdr did not answer within the configured wait (configured={}s elapsed={}ms) "
|
||||
+ "— starting anyway; /healthz will report degraded until it comes up. Orphaned "
|
||||
+ "worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS, elapsedMillis);
|
||||
return false;
|
||||
}
|
||||
// HerdrWaitResult.INTERRUPTED — a different fact from DEADLINE_PASSED (fleetd #498): the
|
||||
// wait was cut short, not exhausted, and must never be reported as "did not answer within
|
||||
// Ns" — that claim would be false and would send an operator to debug herdr for nothing.
|
||||
log.warn("herdr wait was interrupted before the configured wait ran out (configured={}s "
|
||||
+ "elapsed={}ms) — starting anyway; /healthz will report degraded until it comes "
|
||||
+ "up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS, elapsedMillis);
|
||||
return false;
|
||||
}
|
||||
|
||||
private Fleetd() {
|
||||
}
|
||||
}
|
||||
|
||||
@@ -244,7 +244,15 @@ public final class CallerResolver {
|
||||
// already names what happens if that case is handed the primary role: a worker→primary
|
||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||
return isLoopback(remoteAddr) && c.resolved() ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
//
|
||||
// fleetd #505: the OTHER way a real pid can wrongly reach here with a null terminal — not a
|
||||
// failed lsof lookup, but a herdr error partway through PaneLocator's pane scan. c.resolved()
|
||||
// says nothing about that; it only tests the lsof sentinel (by design — see
|
||||
// ConnectionIdentity.Caller#resolved). c.scanComplete() is the separate signal: a scan that
|
||||
// could not check every pane must not be read as "checked everywhere, no match" — the pane it
|
||||
// could not check might have been the caller's own. So both must hold before this promotes.
|
||||
return isLoopback(remoteAddr) && c.resolved() && c.scanComplete()
|
||||
? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
@@ -36,6 +38,8 @@ import java.util.Set;
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(PaneLocator.class);
|
||||
|
||||
/**
|
||||
* Bound on how many ancestor generations {@link #ancestorsOf} walks. This runs on every MCP
|
||||
* call, so a cycle or a pathologically deep process tree must not hang identity resolution;
|
||||
@@ -73,22 +77,46 @@ public final class PaneLocator {
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane on any searched daemon owns it (e.g. the caller is the
|
||||
* primary, or off-host).
|
||||
* The outcome of a {@link #terminalForPid} scan: the {@code terminal_id} of the agent pane
|
||||
* whose process tree contains the pid ({@link #terminal} is {@code null} if none matched),
|
||||
* and whether the scan that produced that answer ran to completion on every daemon searched.
|
||||
*
|
||||
* <p>{@link #complete} is {@code false} exactly when some {@code pane.process_info} call
|
||||
* failed and, despite that, no pane was ever found to own the pid. In that case a {@code null}
|
||||
* {@link #terminal} means "could not tell", not "definitely not a worker" — fleetd #505: a
|
||||
* transient herdr error on the very pane that <em>does</em> own the caller's pid must not read
|
||||
* as a clean negative and fall through to {@code Principal.primary}, the same way #317's
|
||||
* {@code Caller.resolved()} already guards a failed lsof lookup. Callers ({@code
|
||||
* ConnectionIdentity}, {@code CallerResolver}) must refuse rather than promote on an incomplete
|
||||
* scan.
|
||||
*
|
||||
* <p>When a pane genuinely owns the pid, {@link #complete} is {@code true} regardless of
|
||||
* whether some other, unrelated pane failed to answer earlier in the same scan — a positive
|
||||
* match is definitive and does not need every pane to have been checked (a pane that "vanished
|
||||
* mid-scan" but was never the match is still a clean, complete result).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
public record Lookup(String terminal, boolean complete) {
|
||||
private static final Lookup NOT_FOUND = new Lookup(null, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve {@code pid} to the agent pane whose process tree contains it, across every searched
|
||||
* herdr daemon. See {@link Lookup} for how to read a {@code null} terminal.
|
||||
*/
|
||||
public Lookup terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
return Lookup.NOT_FOUND;
|
||||
}
|
||||
Set<Long> ancestry = ancestorsOf(pid);
|
||||
for (HerdrClient herdr : herdrs) {
|
||||
String terminal = terminalForPid(herdr, ancestry);
|
||||
if (terminal != null) {
|
||||
return terminal;
|
||||
boolean complete = true;
|
||||
for (int i = 0; i < herdrs.size(); i++) {
|
||||
Lookup outcome = scan(herdrs.get(i), i, herdrs.size(), ancestry);
|
||||
if (outcome.terminal() != null) {
|
||||
return outcome; // a definite match — no need to finish checking other clients
|
||||
}
|
||||
complete = complete && outcome.complete();
|
||||
}
|
||||
return null;
|
||||
return new Lookup(null, complete);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -117,31 +145,51 @@ public final class PaneLocator {
|
||||
return ancestry;
|
||||
}
|
||||
|
||||
private static String terminalForPid(HerdrClient herdr, Set<Long> ancestry) {
|
||||
/** Whether a pane owns one of the scanned pid's ancestors, or the check of it failed outright. */
|
||||
private enum Ownership { OWNS, DOES_NOT_OWN, UNKNOWN }
|
||||
|
||||
private static Lookup scan(HerdrClient herdr, int clientIndex, int clientCount, Set<Long> ancestry) {
|
||||
boolean complete = true;
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsAnyOf(herdr, paneId, ancestry)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
if (paneId == null) {
|
||||
continue;
|
||||
}
|
||||
Ownership owns = paneOwnsAnyOf(herdr, clientIndex, clientCount, paneId, ancestry);
|
||||
if (owns == Ownership.OWNS) {
|
||||
return new Lookup(pane.path("terminal_id").asText(null), true);
|
||||
}
|
||||
if (owns == Ownership.UNKNOWN) {
|
||||
complete = false;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
return new Lookup(null, complete);
|
||||
}
|
||||
|
||||
private static boolean paneOwnsAnyOf(HerdrClient herdr, String paneId, Set<Long> ancestry) {
|
||||
private static Ownership paneOwnsAnyOf(HerdrClient herdr, int clientIndex, int clientCount,
|
||||
String paneId, Set<Long> ancestry) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
// fleetd #505: this used to be read as a clean "does not own it" (the pane vanished
|
||||
// mid-scan, just skip it) — one boolean carrying two different facts. It is UNKNOWN
|
||||
// now: if THIS pane is the one that owns the pid, the caller must not be told "no pane
|
||||
// owns it", because that reads as a real primary and is promoted under loopback-trust.
|
||||
log.warn("pane.process_info failed for pane {} on herdr client {} of {} during a "
|
||||
+ "pid-owner scan — treating it as \"could not tell\", not a clean "
|
||||
+ "negative (fleetd #505): {}",
|
||||
paneId, clientIndex + 1, clientCount, e.getMessage());
|
||||
return Ownership.UNKNOWN;
|
||||
}
|
||||
if (ancestry.contains(info.path("shell_pid").asLong(-1))) {
|
||||
return true;
|
||||
return Ownership.OWNS;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (ancestry.contains(p.path("pid").asLong(-1))) {
|
||||
return true;
|
||||
return Ownership.OWNS;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
return Ownership.DOES_NOT_OWN;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -15,6 +15,7 @@ import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
@@ -94,6 +95,14 @@ public final class Injector {
|
||||
private final TurnListener turnListener;
|
||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
||||
/**
|
||||
* Wall-clock source for the readiness-grace elapsed time logged in {@link #onStatus} (fleetd
|
||||
* #501). Production constructors default this to {@code System::currentTimeMillis}; the
|
||||
* package-private constructors below take it explicitly so a test can supply a stub whose
|
||||
* advance does not track {@link #POLL_INTERVAL_MILLIS} — copying the shape {@code LeadRollover}
|
||||
* already uses for the same purpose.
|
||||
*/
|
||||
private final LongSupplier nowMillis;
|
||||
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
||||
|
||||
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
||||
@@ -125,20 +134,34 @@ public final class Injector {
|
||||
*/
|
||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this(agents, turnListener, ready, forget, System::currentTimeMillis);
|
||||
}
|
||||
|
||||
/** Full constructor — for tests: an injectable wall-clock supplier (fleetd #501). */
|
||||
Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget, LongSupplier nowMillis) {
|
||||
this.agents = agents;
|
||||
this.router = null;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
this.nowMillis = nowMillis;
|
||||
}
|
||||
|
||||
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this(router, turnListener, ready, forget, System::currentTimeMillis);
|
||||
}
|
||||
|
||||
/** Full constructor — for tests: an injectable wall-clock supplier (fleetd #501). */
|
||||
Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget, LongSupplier nowMillis) {
|
||||
this.agents = null;
|
||||
this.router = router;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
this.nowMillis = nowMillis;
|
||||
}
|
||||
|
||||
private AgentControl agentsFor(String target) {
|
||||
@@ -209,6 +232,7 @@ public final class Injector {
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
long notReadySinceMillis; // wall-clock time of the FIRST non-ready sample in the current notReadySincePoll streak (fleetd #501); reset alongside it
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
boolean postTurnObserved;
|
||||
@@ -302,6 +326,7 @@ public final class Injector {
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
t.notReadySincePoll = 0;
|
||||
t.notReadySinceMillis = 0;
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
t.unknownSinceTurn = 0;
|
||||
@@ -353,6 +378,7 @@ public final class Injector {
|
||||
Pending p = t.queue.peek();
|
||||
if (p != null && ready.test(target)) {
|
||||
t.notReadySincePoll = 0;
|
||||
t.notReadySinceMillis = 0;
|
||||
try {
|
||||
agentsFor(target).send(target, p.text());
|
||||
t.queue.poll();
|
||||
@@ -370,23 +396,52 @@ public final class Injector {
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
} else if (p != null && ++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
||||
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
||||
// never connected the bridge MCP (crashed during boot, or wedged on a
|
||||
// startup prompt). The readiness gate would hold this message forever, so
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
for (Pending pending : notReady) {
|
||||
pending.state = Pending.State.NOT_DELIVERED;
|
||||
} else if (p != null) {
|
||||
// fleetd #501: stamp the wall-clock time of the FIRST non-ready sample in
|
||||
// this streak, so the expiry log below can print how long the target
|
||||
// actually sat non-ready — not just how many polls that took.
|
||||
if (t.notReadySincePoll == 0) {
|
||||
t.notReadySinceMillis = nowMillis.getAsLong();
|
||||
}
|
||||
if (++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
||||
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
||||
// never connected the bridge MCP (crashed during boot, or wedged on a
|
||||
// startup prompt). The readiness gate would hold this message forever, so
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
for (Pending pending : notReady) {
|
||||
pending.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
// fleetd #501: t.notReadySincePoll — the loop's own counter, already in
|
||||
// scope — is printed here instead of the READINESS_GRACE_POLLS constant.
|
||||
// On this branch the counter has JUST reached the threshold, so the two
|
||||
// agree by construction and no test can tell them apart. Printed anyway:
|
||||
// it gives this line one source of truth instead of two, so a later
|
||||
// change to the loop above cannot leave this message reporting a number
|
||||
// the loop no longer produces.
|
||||
//
|
||||
// elapsedMillis is a different case: it is NOT equal-by-construction to
|
||||
// the truth. notReadySincePoll only increments on a sample that reaches
|
||||
// this branch (p != null, not ready) — a poll that misses that condition
|
||||
// advances real time without advancing the counter — and this loop's real
|
||||
// period is not guaranteed to equal POLL_INTERVAL_MILLIS (load, or a host
|
||||
// sleep, can widen the real gap far past it). READINESS_GRACE_POLLS *
|
||||
// POLL_INTERVAL_MILLIS / 1000 is arithmetic on two constants, not a
|
||||
// measurement, so it stays here only as the labelled CONFIGURED budget,
|
||||
// never presented as elapsed time.
|
||||
long elapsedMillis = nowMillis.getAsLong() - t.notReadySinceMillis;
|
||||
log.warn("readiness grace for {} expired after {} polls (configured={} "
|
||||
+ "polls/{}s elapsed={}ms): target never became "
|
||||
+ "deliverable, so failing {} queued message(s) that "
|
||||
+ "never reached its pane",
|
||||
target, t.notReadySincePoll, READINESS_GRACE_POLLS,
|
||||
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, elapsedMillis,
|
||||
notReady.size());
|
||||
t.queue.clear();
|
||||
t.notReadySincePoll = 0;
|
||||
t.notReadySinceMillis = 0;
|
||||
}
|
||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
||||
+ "became deliverable, so failing {} queued message(s) that never "
|
||||
+ "reached its pane",
|
||||
target, READINESS_GRACE_POLLS,
|
||||
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, notReady.size());
|
||||
t.queue.clear();
|
||||
t.notReadySincePoll = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -33,9 +33,10 @@ public final class ConnectionIdentity {
|
||||
|
||||
/**
|
||||
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
||||
* primary / an off-host client), its {@code pid} (or {@code -1} if not resolvable), and whether
|
||||
* the pane scan behind {@code terminal} ran to completion ({@link #scanComplete}).
|
||||
*/
|
||||
public record Caller(String terminal, long pid) {
|
||||
public record Caller(String terminal, long pid, boolean scanComplete) {
|
||||
|
||||
/**
|
||||
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
||||
@@ -51,6 +52,10 @@ public final class ConnectionIdentity {
|
||||
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
||||
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
||||
* "one rule, two copies" shape that let #305 drift.
|
||||
*
|
||||
* <p>This method is deliberately NOT widened for fleetd #505's failure (a herdr error
|
||||
* during the pane scan, not a failed lsof lookup) — it still tests only the sentinel it is
|
||||
* named for. #505 is a different axis, carried separately in {@link #scanComplete}.
|
||||
*/
|
||||
public boolean resolved() {
|
||||
return pid > 0;
|
||||
@@ -60,10 +65,11 @@ public final class ConnectionIdentity {
|
||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||
public Caller resolve(String remoteAddr, int remotePort) {
|
||||
if (!isLoopback(remoteAddr)) {
|
||||
return new Caller(null, -1); // only same-host callers can be workers
|
||||
return new Caller(null, -1, true); // only same-host callers can be workers
|
||||
}
|
||||
long pid = pids.pidForLocalPort(remotePort);
|
||||
return new Caller(panes.terminalForPid(pid), pid);
|
||||
PaneLocator.Lookup lookup = panes.terminalForPid(pid);
|
||||
return new Caller(lookup.terminal(), pid, lookup.complete());
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -93,7 +93,18 @@ public final class FleetMcp {
|
||||
|
||||
private final HttpServletStreamableServerTransportProvider transport;
|
||||
private final McpSyncServer server;
|
||||
private final CallerResolver authz; // CB-501: null → authorization not enforced (legacy)
|
||||
/**
|
||||
* fleetd #518: whether {@link #denyFor} enforces the CB-505 policy table at all. Replaces the
|
||||
* old {@code CallerResolver authz} field, whose null-ness used to decide BOTH this AND which
|
||||
* principal-resolution code path {@link #contextExtractor} ran — reaching "authorization off"
|
||||
* by simply not passing a {@link CallerResolver} also meant the resolved {@link Principal}
|
||||
* came from a second, separately-maintained heuristic ({@code legacyPrincipal}, now deleted)
|
||||
* that nothing ever exercised. There is now exactly one resolution path ({@code callers},
|
||||
* required and non-null below) and a separate, explicitly-chosen {@link AuthorizationMode}
|
||||
* for this flag — so a caller can turn enforcement off without silently swapping in a second,
|
||||
* untested identity heuristic.
|
||||
*/
|
||||
private final boolean authorizationEnforced;
|
||||
private final Metrics metrics; // CB-502: null → auth failures not counted
|
||||
private final CapacitySource capacity;
|
||||
private final HealthCoverageSource healthCoverage;
|
||||
@@ -250,6 +261,17 @@ public final class FleetMcp {
|
||||
public static CoordinationSource none() { return new CoordinationSource(null, List.of()); }
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #518: whether {@link #denyFor} enforces the CB-505 policy table. A required
|
||||
* constructor parameter with no default, so "authorization is off" can only be reached by a
|
||||
* caller explicitly saying so — never by omitting a {@link CallerResolver} the way the old
|
||||
* {@code callers == null} idiom allowed. {@code callers} itself is required either way: even
|
||||
* under {@link #UNENFORCED}, the one real {@link CallerResolver} still resolves every caller's
|
||||
* {@link Principal} (so {@code markSpawnedMemberPresent}/{@code recordPrimarySingleton} see a
|
||||
* real identity), and {@link #denyFor} is the only thing that changes.
|
||||
*/
|
||||
public enum AuthorizationMode { ENFORCED, UNENFORCED }
|
||||
|
||||
/**
|
||||
* The only constructor (fleetd #480 Unit C correction round). Every field below used to have
|
||||
* its own defaulting overload — {@code leadChannel}/{@code outage}/{@code leadSeats}/
|
||||
@@ -268,10 +290,16 @@ public final class FleetMcp {
|
||||
* {@link OutageSource#none()}, {@link LeadSeatSource#none()}, {@code List.of()} are all still
|
||||
* perfectly fine values, just never an implicit default reached by omission.
|
||||
*
|
||||
* @param callers resolves each call's {@link Principal}; {@code null} disables
|
||||
* authorization. This surface needs its own enforcement: {@code /mcp} is a
|
||||
* raw servlet on Jetty's context handler and never passes through
|
||||
* Javalin's {@code before} filter, so the REST guard does not cover it.
|
||||
* @param callers resolves each call's {@link Principal}. Required, never {@code null} —
|
||||
* fleetd #518: use {@link AuthorizationMode#UNENFORCED} to disable
|
||||
* enforcement, not a missing resolver. This surface needs its own
|
||||
* enforcement: {@code /mcp} is a raw servlet on Jetty's context handler and
|
||||
* never passes through Javalin's {@code before} filter, so the REST guard
|
||||
* does not cover it.
|
||||
* @param authorizationMode fleetd #518: whether {@link #denyFor} enforces the CB-505 policy
|
||||
* table ({@link AuthorizationMode#ENFORCED}) or leaves the gate open
|
||||
* ({@link AuthorizationMode#UNENFORCED}, for the pre-CB-513 test suite that
|
||||
* does not exercise authorization). Required, with no default.
|
||||
* @param metrics registry for auth-failure counting; may be {@code null}
|
||||
* @param quarantine CB-578 stage B facts for {@code fleet_profiles}; pass
|
||||
* {@link QuarantineSource#none()} for a caller that does not want the
|
||||
@@ -301,9 +329,13 @@ public final class FleetMcp {
|
||||
*/
|
||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
CallerResolver callers, AuthorizationMode authorizationMode, Metrics metrics,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||
LeadSeatSource leadSeats, List<String> peers, LeadRollover leadRollover) {
|
||||
Objects.requireNonNull(callers, "callers");
|
||||
this.authorizationEnforced = Objects.requireNonNull(authorizationMode, "authorizationMode")
|
||||
== AuthorizationMode.ENFORCED;
|
||||
this.leadChannel = leadChannel;
|
||||
this.peers = peers == null ? List.of() : List.copyOf(peers);
|
||||
this.capacity = capacity;
|
||||
@@ -322,11 +354,11 @@ public final class FleetMcp {
|
||||
// (CB-113) — its MCP initialize is the reliable "the agent is up" signal.
|
||||
.contextExtractor(req -> {
|
||||
// One resolution per call, shared with the REST surface via CallerResolver so
|
||||
// the two paths cannot drift on who a caller is.
|
||||
Principal p = callers != null
|
||||
? callers.resolve(req.getRemoteAddr(), req.getRemotePort(),
|
||||
req.getHeader("Authorization"))
|
||||
: legacyPrincipal(identity, req.getRemoteAddr(), req.getRemotePort());
|
||||
// the two paths cannot drift on who a caller is. fleetd #518: callers is
|
||||
// required (never null) so there is no second, untested resolution path to
|
||||
// fall back to here — AuthorizationMode governs enforcement, not identity.
|
||||
Principal p = callers.resolve(req.getRemoteAddr(), req.getRemotePort(),
|
||||
req.getHeader("Authorization"));
|
||||
// CB-532: guard on the ROLE, not on the terminal being null. This excludes a
|
||||
// lead, which carries its pane too, while including every spawned member role.
|
||||
// Enrolling a lead would count it as an available member in the roster.
|
||||
@@ -443,7 +475,7 @@ public final class FleetMcp {
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_list", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
leadSeats, callers == null ? Map.of() : callers.leads(),
|
||||
leadSeats, callers.leads(),
|
||||
callerTerminal(exchange),
|
||||
new CoordinationSource(leadChannel, peers),
|
||||
coordinatorVisibleTo(principal(exchange)));
|
||||
@@ -522,21 +554,9 @@ public final class FleetMcp {
|
||||
.toolCall(fleetWhoami, whoamiHandler)
|
||||
.toolCall(fleetHandover, handoverHandler)
|
||||
.build();
|
||||
this.authz = callers;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-CB-501 identity: worker if the connection maps to a pane, otherwise the primary. Used
|
||||
* only by the legacy constructor, where authorization is not enforced anyway.
|
||||
*/
|
||||
private static Principal legacyPrincipal(ConnectionIdentity identity, String addr, int port) {
|
||||
ConnectionIdentity.Caller c = identity.resolve(addr, port);
|
||||
return c.terminal() != null
|
||||
? Principal.worker(c.terminal(), c.pid())
|
||||
: Principal.primary(c.pid());
|
||||
}
|
||||
|
||||
/** The caller reconstructed from the transport context. */
|
||||
private static Principal principal(McpSyncServerExchange exchange) {
|
||||
return principalFrom(exchange.transportContext().get(CALLER_ROLE),
|
||||
@@ -591,8 +611,8 @@ public final class FleetMcp {
|
||||
McpSchema.CallToolResult denyFor(Principal caller, Authz.Action action, String target) {
|
||||
// The enforcement switch lives HERE rather than in the exchange-facing wrapper: any future
|
||||
// tool that calls this directly must not be able to skip the gate by accident.
|
||||
if (authz == null) {
|
||||
return null; // legacy constructor: authorization not enforced
|
||||
if (!authorizationEnforced) {
|
||||
return null; // AuthorizationMode.UNENFORCED: authorization not enforced (fleetd #518)
|
||||
}
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ) {
|
||||
|
||||
@@ -302,9 +302,10 @@ public final class SessionManager implements TurnListener {
|
||||
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
||||
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
private MemberSession release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||
return removed;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1064,16 +1065,39 @@ public final class SessionManager implements TurnListener {
|
||||
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
||||
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
||||
* {@link #roster()} call.
|
||||
*
|
||||
* <p>fleetd #512: a drain that releases every session cleanly used to log nothing at all — the
|
||||
* only log calls in this method and {@link #drainSnapshot} sit on abnormal paths, so "nothing
|
||||
* logged" was indistinguishable from "died on the first session". The {@code log.info} at the
|
||||
* end below is a positive assertion that the drain actually finished, on the normal path,
|
||||
* every time — including the all-zero case, which is a common and legitimate outcome (no
|
||||
* members were live) and must still produce the line. Both {@link #drainSnapshot} passes (the
|
||||
* main snapshot and the straggler sweep) are folded into the one line: a caller reading two
|
||||
* lines could not tell a two-pass drain from two separate drains.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
draining.set(true);
|
||||
drainSnapshot(roster(), deadline);
|
||||
DrainTally tally = drainSnapshot(roster(), deadline);
|
||||
List<MemberSession> stragglers = roster();
|
||||
if (!stragglers.isEmpty()) {
|
||||
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
||||
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
||||
drainSnapshot(stragglers, deadline);
|
||||
tally = tally.plus(drainSnapshot(stragglers, deadline));
|
||||
}
|
||||
log.info("drain complete: released={} abandoned={} (still BUSY at the shutdown deadline)",
|
||||
tally.released(), tally.abandoned());
|
||||
}
|
||||
|
||||
/**
|
||||
* Running count for one {@link #drainAll} invocation, folded across both {@link #drainSnapshot}
|
||||
* passes (fleetd #512). {@code abandoned} counts sessions that were still {@code BUSY} at the
|
||||
* moment they were released — i.e. the whole-drain deadline passed before they left {@code BUSY}
|
||||
* on their own (see {@link #drainSnapshot}) — a subset of {@code released}, not additional to it.
|
||||
*/
|
||||
private record DrainTally(int released, int abandoned) {
|
||||
private DrainTally plus(DrainTally other) {
|
||||
return new DrainTally(released + other.released, abandoned + other.abandoned);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1081,8 +1105,12 @@ public final class SessionManager implements TurnListener {
|
||||
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
||||
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
||||
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
||||
* Returns how many sessions this pass released, and how many of those were still {@code BUSY}
|
||||
* (abandoned mid-turn) at the moment of release.
|
||||
*/
|
||||
private void drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||
private DrainTally drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||
int released = 0;
|
||||
int abandoned = 0;
|
||||
for (MemberSession s : snapshot) {
|
||||
try {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
@@ -1100,11 +1128,16 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
}
|
||||
}
|
||||
release(s.paneId(), ReleaseCause.SHUTDOWN);
|
||||
MemberSession removed = release(s.paneId(), ReleaseCause.SHUTDOWN);
|
||||
released++;
|
||||
if (removed != null && removed.state() == MemberSession.State.BUSY) {
|
||||
abandoned++;
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
||||
}
|
||||
}
|
||||
return new DrainTally(released, abandoned);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,212 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assertions.fail;
|
||||
|
||||
/**
|
||||
* fleetd #498: {@code Fleetd.awaitHerdr} used to return a bare {@code boolean}, collapsing "the
|
||||
* configured wait budget genuinely ran out" and "the waiting thread was interrupted, possibly
|
||||
* milliseconds in" onto the same {@code false} — and the caller's log line printed only the
|
||||
* configured budget, never how long the wait actually ran. This class covers both halves of the
|
||||
* fix:
|
||||
* <ul>
|
||||
* <li>the seam — {@link Fleetd#awaitHerdr} itself, driven with an injected clock and a stub
|
||||
* {@link HerdrClient}, one test per {@link Fleetd.HerdrWaitResult};</li>
|
||||
* <li>the call site — {@link Fleetd#logHerdrWaitOutcomeAndShouldReap}, the exact decision {@code
|
||||
* main} calls (extracted here because {@code main} itself boots the whole daemon and cannot
|
||||
* be driven from a unit test), pinning the three distinct log messages it emits.</li>
|
||||
* </ul>
|
||||
* Every expected message below is a plain literal, not built from {@code HERDR_WAIT_SECONDS} or
|
||||
* any other production constant — a test that derives its expectation the way the code does
|
||||
* cannot see a change to either (fleetd #496's identical trap).
|
||||
*/
|
||||
class FleetdAwaitHerdrTest {
|
||||
|
||||
// ---- the seam: Fleetd.awaitHerdr ----------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void answeredReturnsImmediatelyWithZeroElapsedAndNeverPolls() {
|
||||
HerdrStub herdr = new HerdrStub(0); // succeeds on the very first call
|
||||
LongSupplier clock = fixedClock(1_000L);
|
||||
AtomicBoolean polled = new AtomicBoolean(false);
|
||||
Runnable poller = () -> polled.set(true);
|
||||
|
||||
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||
|
||||
assertEquals(Fleetd.HerdrWaitResult.ANSWERED, outcome.result());
|
||||
assertEquals(0L, outcome.elapsedNanos(), "a fixed clock must measure zero elapsed time");
|
||||
assertFalse(polled.get(), "herdr answering on the first try must never poll");
|
||||
}
|
||||
|
||||
@Test
|
||||
void deadlinePassedIsMeasuredNotAssumed() {
|
||||
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||
// call order inside awaitHerdr: start, then per failed attempt: deadline-check, elapsed-calc
|
||||
ScriptedClock clock = new ScriptedClock(0L, 30_500_000_000L, 30_500_000_000L);
|
||||
Runnable poller = () -> fail("the deadline was already exceeded on the first attempt — must not poll");
|
||||
|
||||
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||
|
||||
assertEquals(Fleetd.HerdrWaitResult.DEADLINE_PASSED, outcome.result());
|
||||
assertEquals(30_500_000_000L, outcome.elapsedNanos(),
|
||||
"elapsed must be the MEASURED clock delta, not the configured budget");
|
||||
}
|
||||
|
||||
@Test
|
||||
void interruptedIsDistinctFromDeadlinePassedAndPreservesTheInterruptFlag() {
|
||||
HerdrStub herdr = new HerdrStub(-1); // never succeeds
|
||||
// start=0, deadline-check returns 500ms (well under the 30s budget) -> not deadline-passed,
|
||||
// then the poller interrupts, and the elapsed-calc call returns 750ms.
|
||||
ScriptedClock clock = new ScriptedClock(0L, 500_000_000L, 750_000_000L);
|
||||
Runnable poller = () -> Thread.currentThread().interrupt();
|
||||
|
||||
try {
|
||||
Fleetd.HerdrAwaitOutcome outcome = Fleetd.awaitHerdr(herdr, clock, poller);
|
||||
|
||||
assertEquals(Fleetd.HerdrWaitResult.INTERRUPTED, outcome.result());
|
||||
assertEquals(750_000_000L, outcome.elapsedNanos(),
|
||||
"elapsed must be measured even when the wait ends via interruption, not the deadline");
|
||||
assertTrue(Thread.currentThread().isInterrupted(),
|
||||
"the interrupt flag the old code re-set must still be set on return");
|
||||
} finally {
|
||||
Thread.interrupted(); // clear it so it cannot leak into another test on this thread
|
||||
}
|
||||
}
|
||||
|
||||
// ---- the call site: Fleetd.logHerdrWaitOutcomeAndShouldReap -------------------------------
|
||||
|
||||
@Test
|
||||
void answeredLogsNothingAndSaysReap() {
|
||||
ListAppender<ILoggingEvent> events = attach();
|
||||
try {
|
||||
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.ANSWERED, 0L));
|
||||
|
||||
assertTrue(shouldReap, "only ANSWERED should tell main to reap orphan workers");
|
||||
assertEquals(0, events.list.size(), "the answered path logs nothing itself");
|
||||
} finally {
|
||||
detach(events);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void deadlinePassedLogsConfiguredAndMeasuredElapsedTogether() {
|
||||
ListAppender<ILoggingEvent> events = attach();
|
||||
try {
|
||||
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.DEADLINE_PASSED, 30_500_000_000L));
|
||||
|
||||
assertFalse(shouldReap, "a deadline-passed wait must not tell main to reap");
|
||||
assertEquals(1, events.list.size());
|
||||
ILoggingEvent event = events.list.getFirst();
|
||||
assertEquals(Level.WARN, event.getLevel());
|
||||
assertEquals("herdr did not answer within the configured wait (configured=30s "
|
||||
+ "elapsed=30500ms) — starting anyway; /healthz will report degraded until it "
|
||||
+ "comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
event.getFormattedMessage());
|
||||
} finally {
|
||||
detach(events);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void interruptedLogsItsOwnMessageAndNeverClaimsTheBudgetElapsed() {
|
||||
ListAppender<ILoggingEvent> events = attach();
|
||||
try {
|
||||
// 3ms: the ticket's own example of "a few milliseconds in", not the 30s budget.
|
||||
boolean shouldReap = Fleetd.logHerdrWaitOutcomeAndShouldReap(
|
||||
new Fleetd.HerdrAwaitOutcome(Fleetd.HerdrWaitResult.INTERRUPTED, 3_000_000L));
|
||||
|
||||
assertFalse(shouldReap, "an interrupted wait must not tell main to reap");
|
||||
assertEquals(1, events.list.size());
|
||||
ILoggingEvent event = events.list.getFirst();
|
||||
assertEquals(Level.WARN, event.getLevel());
|
||||
String message = event.getFormattedMessage();
|
||||
assertEquals("herdr wait was interrupted before the configured wait ran out "
|
||||
+ "(configured=30s elapsed=3ms) — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
message);
|
||||
assertFalse(message.contains("did not answer"),
|
||||
"an interrupted wait must not be reported as if herdr failed to answer within the budget");
|
||||
} finally {
|
||||
detach(events);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- fixtures --------------------------------------------------------------------------
|
||||
|
||||
/** Always returns the same value, i.e. a clock that measures zero elapsed time. */
|
||||
private static LongSupplier fixedClock(long value) {
|
||||
return () -> value;
|
||||
}
|
||||
|
||||
/** Returns each value in order, then repeats the last one for any call beyond the list. */
|
||||
private static final class ScriptedClock implements LongSupplier {
|
||||
private final long[] values;
|
||||
private int index;
|
||||
|
||||
ScriptedClock(long... values) {
|
||||
this.values = values;
|
||||
}
|
||||
|
||||
@Override
|
||||
public long getAsLong() {
|
||||
long v = values[Math.min(index, values.length - 1)];
|
||||
if (index < values.length - 1) {
|
||||
index++;
|
||||
}
|
||||
return v;
|
||||
}
|
||||
}
|
||||
|
||||
/** Fails {@code failuresBeforeSuccess} times, then succeeds forever; {@code -1} never succeeds. */
|
||||
private static final class HerdrStub implements HerdrClient {
|
||||
private final int failuresBeforeSuccess;
|
||||
private int calls;
|
||||
|
||||
HerdrStub(int failuresBeforeSuccess) {
|
||||
this.failuresBeforeSuccess = failuresBeforeSuccess;
|
||||
}
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) throws HerdrException {
|
||||
calls++;
|
||||
if (failuresBeforeSuccess < 0 || calls <= failuresBeforeSuccess) {
|
||||
throw new HerdrException("herdr not up yet");
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
logger.setLevel(Level.DEBUG);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
}
|
||||
@@ -186,6 +186,43 @@ class CallerResolverTest {
|
||||
assertEquals(Role.PRIMARY, r.resolve("127.0.0.1", 99, "BEARER s3cret").role());
|
||||
}
|
||||
|
||||
// ── fleetd #505: a herdr error DURING THE SCAN must not be conflated with "not a worker" ──────
|
||||
// #317 (above) covers a failed lsof lookup. This is the other input to the same decision: the
|
||||
// lsof lookup succeeds (a real pid), but PaneLocator's own pane scan hits a herdr error on the
|
||||
// pane that owns that pid — so c.resolved() is true and c.terminal() is null, exactly like a
|
||||
// real primary. c.scanComplete() is what tells them apart.
|
||||
|
||||
/**
|
||||
* The discriminating case named in the ticket: the error must land on the pane that DOES own
|
||||
* the caller's pid, or the test proves nothing (any other pane's failure is invisible to the
|
||||
* scan's outcome, since a match found elsewhere is definitive regardless).
|
||||
*/
|
||||
@Test
|
||||
void aHerdrErrorOnTheOwningPaneDuringTheScanIsRefusedNotPromotedToPrimary() {
|
||||
FakeHerdr failing = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||
ConnectionIdentity incomplete = new ConnectionIdentity(new PaneLocator(failing), _ -> FakeHerdr.WORKER_PID);
|
||||
|
||||
Principal p = new CallerResolver(incomplete).resolve("127.0.0.1", 55555, null);
|
||||
|
||||
assertEquals(Role.ANONYMOUS, p.role(),
|
||||
"an incomplete pane scan must never be read as a clean negative and promoted to primary");
|
||||
}
|
||||
|
||||
/**
|
||||
* The companion invariant: a herdr error on a DIFFERENT, non-owning pane must not turn every
|
||||
* mid-scan teardown into a refusal — the real match is still found and resolves as a worker.
|
||||
*/
|
||||
@Test
|
||||
void aHerdrErrorOnANonOwningPaneStillResolvesTheRealWorker() {
|
||||
FakeHerdr vanishedElsewhere = new FakeHerdr().processInfoFailsForPane("w2:p9", "pane_not_found");
|
||||
ConnectionIdentity id = new ConnectionIdentity(new PaneLocator(vanishedElsewhere), _ -> FakeHerdr.WORKER_PID);
|
||||
|
||||
Principal p = new CallerResolver(id).resolve("127.0.0.1", 55555, null);
|
||||
|
||||
assertEquals(Role.WORKER, p.role());
|
||||
assertEquals("term_a", p.terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonLoopbackCallerIsNeverThePrimaryUnderLoopbackTrust() {
|
||||
// Defence in depth: startup already refuses this pairing (validateAuthExposure), but if a
|
||||
|
||||
@@ -44,6 +44,7 @@ public final class FakeHerdr implements HerdrClient {
|
||||
private int workerTabPaneCount = 1;
|
||||
private String paneCloseErrorCode = null;
|
||||
private final Map<String, String> paneCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||
private final Map<String, String> processInfoErrorCodeFor = new ConcurrentHashMap<>();
|
||||
private String tabCloseErrorCode = null;
|
||||
private final Map<String, String> tabCloseErrorCodeFor = new ConcurrentHashMap<>();
|
||||
private String agentSendErrorCode = null;
|
||||
@@ -144,6 +145,18 @@ public final class FakeHerdr implements HerdrClient {
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code pane.process_info} fail with this herdr error code, but only for the given
|
||||
* {@code pane_id} — every other pane's {@code pane.process_info} still succeeds. Models a
|
||||
* transient herdr failure partway through a {@link PaneLocator} pid→pane scan (fleetd #505):
|
||||
* the scan must be able to tell "this pane does not own the pid" apart from "the scan could
|
||||
* not check this pane at all", instead of collapsing both into one {@code false}.
|
||||
*/
|
||||
public FakeHerdr processInfoFailsForPane(String paneId, String code) {
|
||||
this.processInfoErrorCodeFor.put(paneId, code);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Set the {@code agent_status} that {@code agent.get} reports (drives the injector). */
|
||||
public FakeHerdr agentStatus(String status) {
|
||||
this.agentStatus = status;
|
||||
@@ -407,6 +420,13 @@ public final class FakeHerdr implements HerdrClient {
|
||||
{"pane_id":"w2:p9","terminal_id":"term_shell","workspace_id":"w2","tab_id":"w2:t8"}]}""");
|
||||
case "pane.process_info" -> {
|
||||
Object paneId = params instanceof java.util.Map<?, ?> m ? m.get("pane_id") : null;
|
||||
String failCode = paneId == null ? null
|
||||
: processInfoErrorCodeFor.get(String.valueOf(paneId));
|
||||
if (failCode != null) {
|
||||
throw new HerdrException(
|
||||
"herdr error [" + failCode + "]: pane.process_info failed",
|
||||
failCode, null);
|
||||
}
|
||||
yield "w2:p7".equals(paneId)
|
||||
? mapper.readTree(("""
|
||||
{"type":"pane_process_info","process_info":{"pane_id":"w2:p7","shell_pid":%d,
|
||||
|
||||
@@ -37,7 +37,7 @@ class PaneLocatorContractTest {
|
||||
.path("pane").path("terminal_id").asText(null);
|
||||
assertNotNull(terminalId, "seed pane should carry a terminal_id");
|
||||
|
||||
assertEquals(terminalId, new PaneLocator(herdr).terminalForPid(shellPid),
|
||||
assertEquals(terminalId, new PaneLocator(herdr).terminalForPid(shellPid).terminal(),
|
||||
"a real PID must resolve back to its own pane's terminal_id");
|
||||
} finally {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
|
||||
@@ -15,18 +15,20 @@ class PaneLocatorTest {
|
||||
|
||||
@Test
|
||||
void resolvesTerminalForAForegroundPid() {
|
||||
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForAPidInNoPane() {
|
||||
assertNull(loc.terminalForPid(999_999));
|
||||
PaneLocator.Lookup outcome = loc.terminalForPid(999_999);
|
||||
assertNull(outcome.terminal());
|
||||
assertTrue(outcome.complete(), "a full, error-free scan that finds no match is complete");
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForNonPositivePid() {
|
||||
assertNull(loc.terminalForPid(0));
|
||||
assertNull(loc.terminalForPid(-1));
|
||||
assertNull(loc.terminalForPid(0).terminal());
|
||||
assertNull(loc.terminalForPid(-1).terminal());
|
||||
}
|
||||
|
||||
// --- two-daemon fallback (CB-185) -----------------------------------------
|
||||
@@ -38,7 +40,7 @@ class PaneLocatorTest {
|
||||
HerdrClient lead = new FakeHerdr().withNoPanes();
|
||||
HerdrClient member = new FakeHerdr();
|
||||
PaneLocator two = new PaneLocator(lead, member);
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -48,13 +50,13 @@ class PaneLocatorTest {
|
||||
HerdrClient lead = new FakeHerdr();
|
||||
HerdrClient member = new FakeHerdr().withNoPanes();
|
||||
PaneLocator two = new PaneLocator(lead, member);
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullWhenNeitherClientHasTheMatch() {
|
||||
PaneLocator two = new PaneLocator(new FakeHerdr().withNoPanes(), new FakeHerdr().withNoPanes());
|
||||
assertNull(two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
assertNull(two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -63,7 +65,7 @@ class PaneLocatorTest {
|
||||
// must behave exactly like the one-arg constructor, including making only one herdr call.
|
||||
FakeHerdr shared = new FakeHerdr();
|
||||
PaneLocator two = new PaneLocator(shared, shared);
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
assertEquals("term_a", two.terminalForPid(FakeHerdr.WORKER_PID).terminal());
|
||||
long paneListCalls = shared.calls.stream().filter(c -> c.method().equals("pane.list")).count();
|
||||
assertEquals(1, paneListCalls, "same-object lead/member must scan exactly once, not twice");
|
||||
}
|
||||
@@ -75,7 +77,7 @@ class PaneLocatorTest {
|
||||
// Regression: a pid with no parent chain at all — no ancestry walk is needed to match it.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
||||
assertEquals("term_x", loc.terminalForPid(5000));
|
||||
assertEquals("term_x", loc.terminalForPid(5000).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -83,7 +85,7 @@ class PaneLocatorTest {
|
||||
// Regression: same as above, but matching via the foreground-processes list.
|
||||
OnePaneHerdr pane = new OnePaneHerdr("term_x", "pX", 5000, 6000);
|
||||
PaneLocator loc = new PaneLocator(pane, new FakeParentResolver());
|
||||
assertEquals("term_x", loc.terminalForPid(6000));
|
||||
assertEquals("term_x", loc.terminalForPid(6000).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -97,7 +99,7 @@ class PaneLocatorTest {
|
||||
.parent(7002, 7001) // grandchild -> child
|
||||
.parent(7001, 5000); // child -> shell (the pane's shell_pid)
|
||||
PaneLocator loc = new PaneLocator(pane, parents);
|
||||
assertEquals("term_x", loc.terminalForPid(7002));
|
||||
assertEquals("term_x", loc.terminalForPid(7002).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -110,7 +112,7 @@ class PaneLocatorTest {
|
||||
.parent(9002, 9001)
|
||||
.parent(9001, 9000); // chain never reaches 5000 or 6000
|
||||
PaneLocator loc = new PaneLocator(pane, parents);
|
||||
assertNull(loc.terminalForPid(9002));
|
||||
assertNull(loc.terminalForPid(9002).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -122,7 +124,7 @@ class PaneLocatorTest {
|
||||
.parent(100, 101)
|
||||
.parent(101, 100); // cycle, never reaches the pane's pids
|
||||
PaneLocator loc = new PaneLocator(pane, parents);
|
||||
assertNull(loc.terminalForPid(100));
|
||||
assertNull(loc.terminalForPid(100).terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -141,11 +143,69 @@ class PaneLocatorTest {
|
||||
};
|
||||
HerdrClient noPanes = new FakeHerdr().withNoPanes();
|
||||
PaneLocator two = new PaneLocator(noPanes, pane, counting);
|
||||
assertEquals("term_x", two.terminalForPid(7002));
|
||||
assertEquals("term_x", two.terminalForPid(7002).terminal());
|
||||
assertEquals(3, calls.get(), "ancestry must be walked once (3 lookups: 7002, 7001, 5000), "
|
||||
+ "not re-walked per herdr client");
|
||||
}
|
||||
|
||||
// --- fleetd #505: a herdr error during the scan must not read as a clean negative ---------
|
||||
|
||||
@Test
|
||||
void anErrorOnThePaneThatOwnsThePidMakesTheScanIncompleteNotAClearNegative() {
|
||||
// The discriminating case: pane.process_info fails for exactly the pane that DOES own the
|
||||
// caller's pid ("w2:p7", term_a). Before the fix, that failure was swallowed into a plain
|
||||
// "does not own it" and the scan finished with a clean-looking null — indistinguishable
|
||||
// from a real primary. It must now report incomplete, not a definite null.
|
||||
FakeHerdr herdr = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||
PaneLocator loc = new PaneLocator(herdr);
|
||||
|
||||
PaneLocator.Lookup outcome = loc.terminalForPid(FakeHerdr.WORKER_PID);
|
||||
|
||||
assertNull(outcome.terminal(), "the failing pane's ownership could not be confirmed");
|
||||
assertFalse(outcome.complete(),
|
||||
"a scan that could not check the owning pane must not report as complete");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aVanishedPaneThatIsNotTheMatchLeavesAnOtherwiseSuccessfulScanComplete() {
|
||||
// The companion invariant: a DIFFERENT pane (not the caller's own) failing mid-scan must
|
||||
// not turn every mid-scan teardown into a refusal — the real match is still found, and the
|
||||
// scan is still reported complete.
|
||||
FakeHerdr herdr = new FakeHerdr().processInfoFailsForPane("w2:p9", "pane_not_found");
|
||||
PaneLocator loc = new PaneLocator(herdr);
|
||||
|
||||
PaneLocator.Lookup outcome = loc.terminalForPid(FakeHerdr.WORKER_PID);
|
||||
|
||||
assertEquals("term_a", outcome.terminal());
|
||||
assertTrue(outcome.complete(), "a positive match elsewhere in the scan is definitive");
|
||||
}
|
||||
|
||||
// --- fleetd #509: the completeness fold across clients must not collapse to "last wins" ----
|
||||
|
||||
@Test
|
||||
void anEarlierClientsErrorSurvivesALaterClientsCleanNegative() {
|
||||
// terminalForPid folds each client's Lookup.complete() with
|
||||
// complete = complete && outcome.complete();
|
||||
// (PaneLocator.java:117). With a SINGLE client, a fold that keeps only the last outcome
|
||||
// (dropping the "complete &&" prefix) agrees with the real fold — which is why 14 of the
|
||||
// 15 pre-existing tests never catch that mutation: none of them vary the number of clients.
|
||||
// Here the LEAD client errors on exactly the pane that would have owned the pid (so its
|
||||
// scan is incomplete AND finds no match), and the MEMBER client cleanly reports no panes
|
||||
// at all (a complete, negative scan). The real fold ANDs the two into false. A fold that
|
||||
// just keeps the last client's outcome would read this as a clean true — the earlier
|
||||
// error is erased, and CallerResolver.java:254 would read scanComplete() as true and
|
||||
// promote an unverified caller to the primary.
|
||||
HerdrClient lead = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||
HerdrClient member = new FakeHerdr().withNoPanes();
|
||||
PaneLocator two = new PaneLocator(lead, member);
|
||||
|
||||
PaneLocator.Lookup outcome = two.terminalForPid(FakeHerdr.WORKER_PID);
|
||||
|
||||
assertNull(outcome.terminal(), "the pane that could have owned the pid was never checked");
|
||||
assertFalse(outcome.complete(),
|
||||
"an earlier client's error must survive a later client's clean negative");
|
||||
}
|
||||
|
||||
/** Minimal single-pane {@link HerdrClient} fake, purpose-built for the ancestry tests above. */
|
||||
private static final class OnePaneHerdr implements HerdrClient {
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
@@ -19,6 +19,8 @@ import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -521,6 +523,97 @@ class InjectorTest {
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void readinessGraceExpiryLogsTheMeasuredPollCountNextToTheConfiguredBudget() {
|
||||
// fleetd #501, defect 1: READINESS_GRACE_POLLS (240) used to be printed twice — once as
|
||||
// "after {} polls" and once inside the parenthesised budget — even though the loop's own
|
||||
// counter (Target.notReadySincePoll) was in scope at the same call site. On THIS branch the
|
||||
// counter has just reached the threshold, so it equals the constant by construction and this
|
||||
// test cannot tell the two apart — it only pins that the message still carries both a poll
|
||||
// count and a labelled configured budget, using literal numbers (240, 60), never
|
||||
// READINESS_GRACE_POLLS or POLL_INTERVAL_MILLIS, so the assertion can't silently track a
|
||||
// constant change instead of catching a real regression.
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger injectorLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(Injector.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
injectorLog.addAppender(appender);
|
||||
injectorLog.setLevel(Level.WARN);
|
||||
try {
|
||||
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> false, _ -> {
|
||||
});
|
||||
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||
|
||||
for (int i = 0; i < READINESS_SAMPLES; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no grace-expiry WARN logged");
|
||||
assertTrue(warn.contains("after 240 polls"),
|
||||
"must print the measured poll count as a plain number: " + warn);
|
||||
assertTrue(warn.contains("configured=240 polls/60s"),
|
||||
"must print the configured budget, clearly labelled: " + warn);
|
||||
} finally {
|
||||
injectorLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void readinessGraceExpiryLogsTheMeasuredElapsedTimeNotArithmeticOnConstants() {
|
||||
// fleetd #501, defect 2: the old line computed "({}s)" as READINESS_GRACE_POLLS *
|
||||
// POLL_INTERVAL_MILLIS / 1000 — arithmetic on two constants, never a measurement, and wrong
|
||||
// in the direction that says everything ran on schedule. This stub clock returns two FIXED
|
||||
// values (1_000ms at the first non-ready sample, 318_412ms at the poll that trips the grace)
|
||||
// whose difference — 317_412ms — does NOT equal 240 * POLL_INTERVAL_MILLIS (=60_000ms).
|
||||
// Asserting on that literal, non-derived number is what makes this test able to fail if the
|
||||
// production code goes back to printing the constant-arithmetic value instead of the
|
||||
// injected clock's measurement.
|
||||
long[] readings = {1_000L, 318_412L};
|
||||
AtomicInteger call = new AtomicInteger(0);
|
||||
LongSupplier stubClock = () -> {
|
||||
int i = call.getAndIncrement();
|
||||
if (i >= readings.length) {
|
||||
throw new AssertionError("nowMillis read more times than this fixture expects (" + i
|
||||
+ "); the readiness-not-ready branch should read the clock exactly twice — "
|
||||
+ "once to stamp the first non-ready sample, once at grace expiry");
|
||||
}
|
||||
return readings[i];
|
||||
};
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger injectorLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(Injector.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
injectorLog.addAppender(appender);
|
||||
injectorLog.setLevel(Level.WARN);
|
||||
try {
|
||||
Injector inj = new Injector(new AgentControl(herdr), TurnListener.NOOP, _ -> false, _ -> {
|
||||
}, stubClock);
|
||||
inj.enqueue(T, "task", TestTurnTokens.inert(T));
|
||||
|
||||
for (int i = 0; i < READINESS_SAMPLES; i++) inj.onStatus(T, AgentStatus.IDLE);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no grace-expiry WARN logged");
|
||||
assertTrue(warn.contains("elapsed=317412ms"), "must print the MEASURED elapsed time from "
|
||||
+ "the injected clock (318412 - 1000 = 317412), not an arithmetic value: " + warn);
|
||||
assertFalse(warn.contains("elapsed=60000ms"), "must not print "
|
||||
+ "READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS (240 * 250 = 60000ms) as if it "
|
||||
+ "were the measured elapsed time: " + warn);
|
||||
} finally {
|
||||
injectorLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerThatBecomesReadyWithinTheGraceIsDeliveredNormally() {
|
||||
// The readiness grace must not fail a worker that is merely slow to boot: once it becomes
|
||||
|
||||
@@ -57,6 +57,23 @@ class ConnectionIdentityTest {
|
||||
// must read as "resolved" — the distinction #317 turns on.
|
||||
ConnectionIdentity.Caller c = with(_ -> 999_999).resolve("127.0.0.1", 55555);
|
||||
assertTrue(c.resolved());
|
||||
assertTrue(c.scanComplete(), "no herdr error happened, so the scan is complete");
|
||||
}
|
||||
|
||||
@Test
|
||||
void scanIsIncompleteWhenHerdrErrorsOnThePaneThatOwnsThePid() {
|
||||
// fleetd #505: a transient herdr error on exactly the pane that DOES own the caller's pid
|
||||
// must be visible as an incomplete scan, distinct from a real primary (resolved(), null
|
||||
// terminal, complete scan). Both have pid > 0 and a null terminal — scanComplete is the
|
||||
// only thing that tells them apart.
|
||||
FakeHerdr failing = new FakeHerdr().processInfoFailsForPane("w2:p7", "transient");
|
||||
ConnectionIdentity id = new ConnectionIdentity(new PaneLocator(failing), _ -> FakeHerdr.WORKER_PID);
|
||||
|
||||
ConnectionIdentity.Caller c = id.resolve("127.0.0.1", 55555);
|
||||
|
||||
assertTrue(c.resolved(), "the pid itself resolved fine — this is not #317's failure");
|
||||
assertNull(c.terminal(), "the owning pane could not be confirmed");
|
||||
assertFalse(c.scanComplete(), "the scan could not check the pane that owns this pid");
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
@@ -78,10 +78,15 @@ class FleetMcpAuthzTest {
|
||||
// fleetd #480 correction round: FleetMcp has one constructor now (no defaulting
|
||||
// overloads — see its javadoc), so every feature this test does not exercise is passed
|
||||
// its explicit "off" value here rather than being omitted.
|
||||
//
|
||||
// fleetd #518: callers is now required (never null) either way — the resolver that used
|
||||
// to be omitted to reach "legacy" is now always real, and AuthorizationMode is the
|
||||
// separate, explicit choice that governs enforcement.
|
||||
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||
new PrimaryRegistry(null),
|
||||
enforce ? CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
Map::of, new MemberRegistry(null)) : null,
|
||||
CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
Map::of, new MemberRegistry(null)),
|
||||
enforce ? FleetMcp.AuthorizationMode.ENFORCED : FleetMcp.AuthorizationMode.UNENFORCED,
|
||||
metrics, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
||||
FleetMcp.LeadSeatSource.none(), List.of(), null);
|
||||
@@ -187,7 +192,31 @@ class FleetMcpAuthzTest {
|
||||
// The 22 pre-existing FleetMcpTest cases rely on no authorization being enforced.
|
||||
FleetMcp m = mcp(false);
|
||||
assertNull(m.denyFor(ANON, Authz.Action.SPAWN, null),
|
||||
"no CallerResolver supplied ⇒ authorization not enforced (legacy behaviour)");
|
||||
"AuthorizationMode.UNENFORCED chosen explicitly ⇒ authorization not enforced "
|
||||
+ "(legacy behaviour) — fleetd #518 replaced the old callers == null idiom");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #509 was originally proven against {@code FleetMcp.legacyPrincipal} — a second,
|
||||
* separately-maintained principal-resolution heuristic that only ran when {@code callers} was
|
||||
* omitted (null). fleetd #518 deleted that whole heuristic: {@code callers} is now required
|
||||
* and non-null under every {@link FleetMcp.AuthorizationMode}, so the ONE real
|
||||
* {@link CallerResolver} resolves every caller, enforced or not, and #509's property (a
|
||||
* non-loopback / unresolved caller must never earn the primary's authority) is exactly what
|
||||
* {@code CallerResolverTest.aNonLoopbackCallerIsNeverThePrimaryUnderLoopbackTrust} already
|
||||
* proves on that one real path. There is no longer a second heuristic here to test.
|
||||
*/
|
||||
@Test
|
||||
void anUnresolvedNonLoopbackCallerIsAnonymousUnderTheOneRealResolver() {
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> 999_999);
|
||||
CallerResolver resolver = CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
Map::of, new MemberRegistry(null));
|
||||
// A non-loopback address never even reaches the pane scan — resolve() short-circuits it
|
||||
// to Caller(null, -1, true), the same "no terminal" shape a genuine primary's connection
|
||||
// produces on loopback. The real resolver must not conflate the two.
|
||||
Principal p = resolver.resolve("8.8.8.8", 1234, null);
|
||||
assertEquals(Principal.anonymous(), p,
|
||||
"an unresolved, non-loopback caller must earn no authority, not the primary's");
|
||||
}
|
||||
|
||||
// --- fleetd #439: who may see fleet_list's coordinator row ----------------------------------
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.session.FakeWorktrees;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import io.modelcontextprotocol.client.McpClient;
|
||||
import io.modelcontextprotocol.client.McpSyncClient;
|
||||
import io.modelcontextprotocol.client.transport.HttpClientStreamableHttpTransport;
|
||||
import io.modelcontextprotocol.spec.McpClientTransport;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.eclipse.jetty.server.Server;
|
||||
import org.eclipse.jetty.server.ServerConnector;
|
||||
import org.eclipse.jetty.servlet.ServletContextHandler;
|
||||
import org.eclipse.jetty.servlet.ServletHolder;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.net.http.HttpRequest;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #518 — Part 2: drive the {@code contextExtractor} closure for real.
|
||||
*
|
||||
* <p>{@code FleetMcp.deny()}/{@code denyFor()} has a full policy table of tests
|
||||
* ({@code FleetMcpAuthzTest}), and {@code CallerResolver.resolve()} has its own full suite
|
||||
* ({@code CallerResolverTest}). Neither one ever exercises the closure that WIRES them together
|
||||
* inside {@code FleetMcp}'s constructor: it is built once, handed to the MCP SDK's transport, and
|
||||
* only ever runs when a real MCP client makes a real HTTP request. Every existing test either
|
||||
* calls {@code denyFor(Principal, ...)} with a hand-built {@link dev.ltms.fleet.auth.Principal}
|
||||
* (never asking who the transport would actually have resolved) or drives a static handler method
|
||||
* directly. A mutation that swapped the whole resolution decision for an unconditional fallback —
|
||||
* bypassing {@link CallerResolver} entirely — passed the full suite, including every
|
||||
* {@code FleetMcpAuthzTest} case, because none of them go through the transport at all.
|
||||
*
|
||||
* <p>This test boots the real {@code HttpServletStreamableServerTransportProvider} on a real
|
||||
* Jetty server, drives it with a real MCP client over HTTP, and checks a result that only the
|
||||
* real {@link CallerResolver} can produce: token-mode inspects the {@code Authorization} header
|
||||
* and grants {@code PRIMARY} only for the right bearer token. The connection never resolves to a
|
||||
* worker pane (the fake peer-pid lookup always misses), so the ONLY way {@code fleet_whoami} can
|
||||
* come back as {@code primary} is if the closure actually called {@code callers.resolve(...)} and
|
||||
* read that header — a behaviour the deleted {@code legacyPrincipal} heuristic never had at all.
|
||||
*/
|
||||
class FleetMcpContextExtractorTest {
|
||||
|
||||
private static final String TOKEN = "s3cret-mcp-token";
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private FleetMcp mcp;
|
||||
private Server server;
|
||||
|
||||
@AfterEach
|
||||
void tearDown() throws Exception {
|
||||
if (server != null) {
|
||||
server.stop();
|
||||
}
|
||||
if (mcp != null) {
|
||||
mcp.close();
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRealMcpRequestIsResolvedByTheRealCallerResolverNotAFallback() throws Exception {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(agents, new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "tok");
|
||||
SessionManager sessions = new SessionManager(workers, new FakeWorktrees());
|
||||
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(),
|
||||
new InMemoryReplyInbox());
|
||||
// The peer-pid lookup always misses (-1), so no connection here is ever resolved to a
|
||||
// worker pane — every call falls through to CallerResolver's token check, the one branch
|
||||
// that is unreachable through the deleted legacy heuristic.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> -1);
|
||||
CallerResolver callers = CallerResolver.withLeadsAndMembers(identity, true, TOKEN,
|
||||
Map::of, new MemberRegistry(null));
|
||||
|
||||
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||
new PrimaryRegistry(null), callers, FleetMcp.AuthorizationMode.ENFORCED,
|
||||
null, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
||||
FleetMcp.LeadSeatSource.none(), List.of(), null);
|
||||
|
||||
ServletContextHandler handler = new ServletContextHandler();
|
||||
handler.setContextPath("/");
|
||||
handler.addServlet(new ServletHolder(mcp.servlet()), "/mcp");
|
||||
server = new Server(0);
|
||||
server.setHandler(handler);
|
||||
server.start();
|
||||
String baseUrl = "http://127.0.0.1:"
|
||||
+ ((ServerConnector) server.getConnectors()[0]).getLocalPort();
|
||||
|
||||
// The right bearer token: the real CallerResolver grants PRIMARY, which fleet_whoami's
|
||||
// READ gate lets through.
|
||||
McpSchema.CallToolResult authorized = callWhoami(baseUrl, "Bearer " + TOKEN);
|
||||
assertFalse(authorized.isError(), "a valid bearer token must resolve as PRIMARY and pass "
|
||||
+ "fleet_whoami's READ gate: " + textOf(authorized));
|
||||
assertTrue(textOf(authorized).contains("\"role\":\"primary\""),
|
||||
"fleet_whoami must report the role the real CallerResolver resolved over this "
|
||||
+ "connection, not a fallback: " + textOf(authorized));
|
||||
|
||||
// No credential at all, over the SAME wiring: the real resolver refuses it as ANONYMOUS.
|
||||
// legacyPrincipal never looked at the Authorization header, so it could not have told
|
||||
// these two calls apart at all -- this is the assertion the deleted mutation would fail.
|
||||
McpSchema.CallToolResult unauthorized = callWhoami(baseUrl, null);
|
||||
assertTrue(unauthorized.isError(), "no credential must be refused, not silently let "
|
||||
+ "through: " + textOf(unauthorized));
|
||||
}
|
||||
|
||||
private static McpSchema.CallToolResult callWhoami(String baseUrl, String authorizationHeader) {
|
||||
HttpRequest.Builder requestTemplate = HttpRequest.newBuilder();
|
||||
if (authorizationHeader != null) {
|
||||
requestTemplate.header("Authorization", authorizationHeader);
|
||||
}
|
||||
McpClientTransport transport = HttpClientStreamableHttpTransport.builder(baseUrl)
|
||||
.endpoint("/mcp")
|
||||
.requestBuilder(requestTemplate)
|
||||
.build();
|
||||
try (McpSyncClient client = McpClient.sync(transport).build()) {
|
||||
client.initialize();
|
||||
return client.callTool(McpSchema.CallToolRequest.builder("fleet_whoami").arguments(Map.of()).build());
|
||||
}
|
||||
}
|
||||
|
||||
private static String textOf(McpSchema.CallToolResult r) {
|
||||
return ((McpSchema.TextContent) r.content().getFirst()).text();
|
||||
}
|
||||
}
|
||||
@@ -89,6 +89,7 @@ class FleetMcpHandoverTest {
|
||||
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||
new PrimaryRegistry(null),
|
||||
CallerResolver.withLeadsAndMembers(identity, false, null, Map::of, new MemberRegistry(null)),
|
||||
FleetMcp.AuthorizationMode.ENFORCED,
|
||||
null, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), null, FleetMcp.OutageSource.none(),
|
||||
FleetMcp.LeadSeatSource.none(), List.of(), leadRollover);
|
||||
|
||||
@@ -25,7 +25,12 @@ import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.AfterAll;
|
||||
import org.junit.jupiter.api.BeforeAll;
|
||||
import org.junit.jupiter.api.MethodOrderer;
|
||||
import org.junit.jupiter.api.Order;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.TestMethodOrder;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -50,9 +55,46 @@ import static org.junit.jupiter.api.Assertions.*;
|
||||
* CB-301 / CB-303 acceptance tests for the authoritative session registry, one-shot lifecycle FSM,
|
||||
* and configurable lifecycle limits (idle TTL, context cap, drain).
|
||||
* No live herdr — everything runs against the same {@link FakeHerdr} the rest of the project uses.
|
||||
*
|
||||
* <p>fleetd #525: only {@link #onTurnFailedIsLoggedAtWarnWithThePriorState} (explicitly
|
||||
* {@link Order#value() @Order(1)}) and the proving test right after it
|
||||
* ({@link #sharedSessionManagerLoggerLevelIsRestoredAfterOnTurnFailedPinsWarn}, {@code @Order(2)})
|
||||
* care about method order — every other test here has no {@code @Order} and so runs after both of
|
||||
* these (JUnit 5's {@link MethodOrderer.OrderAnnotation} gives an unannotated method the lowest
|
||||
* priority), in whatever relative order it already ran in.
|
||||
*/
|
||||
@TestMethodOrder(MethodOrderer.OrderAnnotation.class)
|
||||
class SessionManagerTest {
|
||||
|
||||
/**
|
||||
* fleetd #525: the level {@link SessionManager}'s logger had when this class started, captured
|
||||
* before any test here — including the leak this ticket fixes — can touch it. {@code
|
||||
* pinSessionManagerLoggerToAKnownBaseline} then forces a distinctive, known value (DEBUG) so
|
||||
* {@link #sharedSessionManagerLoggerLevelIsRestoredAfterOnTurnFailedPinsWarn} can tell "the
|
||||
* level came back to what it was" apart from "the level happens to already be WARN because
|
||||
* some earlier test class in this JVM fork (surefire reuses forks by default) left it there" —
|
||||
* a real risk, since {@code ch.qos.logback.classic.Logger} instances are cached per class and
|
||||
* shared across the whole JVM, and this exact logger is also touched by
|
||||
* {@code WorktreeSessionManagerTest#releasePreservesDirtyWorktreeAndLogsWarn}, which has the
|
||||
* same unfixed leak (reported, not fixed — out of this ticket's scope).
|
||||
*/
|
||||
private static Level sessionManagerLevelBeforeThisClass;
|
||||
|
||||
@BeforeAll
|
||||
static void pinSessionManagerLoggerToAKnownBaseline() {
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
sessionManagerLevelBeforeThisClass = sessionLog.getLevel();
|
||||
sessionLog.setLevel(Level.DEBUG);
|
||||
}
|
||||
|
||||
@AfterAll
|
||||
static void restoreSessionManagerLoggerLevel() {
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
sessionLog.setLevel(sessionManagerLevelBeforeThisClass);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr) {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
@@ -81,6 +123,55 @@ class SessionManagerTest {
|
||||
return new SessionManager(workers, worktrees, clock);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #525: captures a logger's output and, on {@link #close}, restores <em>both</em> the
|
||||
* appender and the level to what they were before. A bare {@code addAppender}/{@code
|
||||
* setLevel} pair whose {@code finally} only detaches the appender leaves the level pinned —
|
||||
* {@code ch.qos.logback.classic.Logger} instances are cached per class and shared across the
|
||||
* whole JVM, so a level set by one test in this class is still in effect for every test that
|
||||
* runs after it, in this class or any other. try-with-resources makes "restored the appender
|
||||
* but not the level" impossible to write, because there is only one thing to close.
|
||||
*/
|
||||
private static final class CapturedLog implements AutoCloseable {
|
||||
private final ch.qos.logback.classic.Logger logger;
|
||||
private final Level originalLevel;
|
||||
private final ListAppender<ILoggingEvent> appender;
|
||||
|
||||
private CapturedLog(Class<?> loggerClass, Level pinnedLevel) {
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
this.logger = (ch.qos.logback.classic.Logger) LoggerFactory.getLogger(loggerClass);
|
||||
this.originalLevel = logger.getLevel();
|
||||
this.appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
if (pinnedLevel != null) {
|
||||
logger.setLevel(pinnedLevel);
|
||||
}
|
||||
}
|
||||
|
||||
/** Capture {@code loggerClass}'s output, pinning its level to {@code pinnedLevel} for the
|
||||
* duration of the try-with-resources block. */
|
||||
static CapturedLog at(Class<?> loggerClass, Level pinnedLevel) {
|
||||
return new CapturedLog(loggerClass, pinnedLevel);
|
||||
}
|
||||
|
||||
/** Capture {@code loggerClass}'s output without changing its level. */
|
||||
static CapturedLog of(Class<?> loggerClass) {
|
||||
return new CapturedLog(loggerClass, null);
|
||||
}
|
||||
|
||||
List<ILoggingEvent> events() {
|
||||
return appender.list;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(originalLevel);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-581: a {@link Worktrees} test double whose {@code hasUncommitted} and {@code remove} can
|
||||
* be told to throw, so {@link SessionManager#release} can be exercised against exactly the
|
||||
@@ -466,24 +557,15 @@ class SessionManagerTest {
|
||||
|
||||
@Test
|
||||
void backendErrorForUnknownTargetIsWarnedAndDoesNotCreateASession() {
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
try {
|
||||
try (CapturedLog log = CapturedLog.of(SessionManager.class)) {
|
||||
SessionManager sessions = sessionManager(new FakeHerdr());
|
||||
|
||||
assertFalse(sessions.onBackendError("term_missing", "backend exited"));
|
||||
|
||||
assertTrue(sessions.roster().isEmpty(), "unknown target must not create a session");
|
||||
assertTrue(appender.list.stream().anyMatch(e -> e.getLevel().equals(Level.WARN)
|
||||
assertTrue(log.events().stream().anyMatch(e -> e.getLevel().equals(Level.WARN)
|
||||
&& e.getFormattedMessage().contains("term_missing")),
|
||||
"unknown target is logged at WARN");
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -516,19 +598,12 @@ class SessionManagerTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
@Order(1)
|
||||
void onTurnFailedIsLoggedAtWarnWithThePriorState() {
|
||||
// CB-564: this transition used to be a bare DEBUG "session marked failed" — a symptom with no
|
||||
// cause. A member that can no longer be delegated to must be at least WARN, and should name
|
||||
// what stage it failed at (here: BUSY, i.e. a turn was in flight and never resolved).
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
try {
|
||||
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
@@ -538,18 +613,37 @@ class SessionManagerTest {
|
||||
|
||||
sessions.onTurnFailed(terminal);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
String warn = log.events().stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no turn-failed WARN logged");
|
||||
assertTrue(warn.contains(terminal), "the log names the member: " + warn);
|
||||
assertTrue(warn.contains("BUSY"), "the log names the stage it failed at: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #525: proves the leak in {@link #onTurnFailedIsLoggedAtWarnWithThePriorState} above
|
||||
* (which runs immediately before this, via {@code @Order}) is closed. That test pins the
|
||||
* shared {@link SessionManager} logger to WARN through a {@link CapturedLog}; if {@link
|
||||
* CapturedLog#close} only detached the appender — the original bug, before this ticket's fix —
|
||||
* the level would still read WARN here instead of the {@code DEBUG} baseline this class's
|
||||
* {@code @BeforeAll} set. Runs at {@code @Order(2)}, guaranteed after {@code @Order(1)} and
|
||||
* before every other (unannotated) test in this class.
|
||||
*/
|
||||
@Test
|
||||
@Order(2)
|
||||
void sharedSessionManagerLoggerLevelIsRestoredAfterOnTurnFailedPinsWarn() {
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
assertEquals(Level.DEBUG, sessionLog.getLevel(),
|
||||
"onTurnFailedIsLoggedAtWarnWithThePriorState pins the shared SessionManager logger "
|
||||
+ "to WARN; its cleanup must restore the level it captured (DEBUG, set by "
|
||||
+ "this class's @BeforeAll) rather than leaving WARN pinned for every test "
|
||||
+ "that runs after it");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #226: a contended slot is refused through the real {@link SessionManager#acquire}
|
||||
* path before the real launcher can hand an architect charter to a process.
|
||||
@@ -576,28 +670,18 @@ class SessionManagerTest {
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberRegistry members = architectRegistry();
|
||||
sessions.setMemberLifecycle(bindFailureAfterReservation(members));
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger registryLog = (ch.qos.logback.classic.Logger)
|
||||
LoggerFactory.getLogger(MemberRegistry.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
registryLog.addAppender(appender);
|
||||
registryLog.setLevel(Level.WARN);
|
||||
try {
|
||||
try (CapturedLog log = CapturedLog.at(MemberRegistry.class, Level.WARN)) {
|
||||
MemberSession session = sessions.acquire("ltms-local", MemberRole.ARCHITECT, null,
|
||||
"/caller", "term_primary", null);
|
||||
|
||||
assertEquals(MemberRole.DEV, session.role(), "a failed reservation bind must use the fallback");
|
||||
String warn = appender.list.stream()
|
||||
String warn = log.events().stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no slot-exhaustion WARN logged");
|
||||
assertTrue(warn.contains("ltms-local"), "the WARN names the profile: " + warn);
|
||||
assertTrue(warn.contains(session.terminalId()), "the WARN names the terminal: " + warn);
|
||||
} finally {
|
||||
registryLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -902,6 +986,82 @@ class SessionManagerTest {
|
||||
.count();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #512: a drain that releases every session cleanly used to log nothing at all — the
|
||||
* two log calls in {@code drainAll}/{@code drainSnapshot} both sit on abnormal paths, so
|
||||
* "clean drain" and "died on the first session" were indistinguishable. This asserts the new
|
||||
* {@code log.info} line fires on the ordinary, nothing-went-wrong path, and that its numbers
|
||||
* are the real counts (two released, zero abandoned) rather than just a non-empty string.
|
||||
*/
|
||||
@Test
|
||||
void drainAllLogsACompletionLineWithTheRealCountsOnACleanDrain() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession first = sessions.acquire("ltms-local", "/one", "/caller", "ownerOne");
|
||||
MemberSession second = sessions.acquire("ltms-local", "/two", "/caller", "ownerTwo");
|
||||
sessions.asPresence().markPresent(first.terminalId());
|
||||
sessions.asPresence().markPresent(second.terminalId());
|
||||
// Both stay READY — neither is delivered a turn, so neither is BUSY and the drain below
|
||||
// has nothing abnormal to hit.
|
||||
|
||||
// Pin INFO explicitly: fleetd #525 made CapturedLog itself restore the level it pins, but
|
||||
// this pin stays anyway as belt-and-braces — a later change to the sweep must not be able
|
||||
// to make this INFO assertion vacuous again by leaving some other test's WARN pin in place.
|
||||
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.INFO)) {
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(sessions.roster().isEmpty(), "precondition: the drain actually ran");
|
||||
String info = log.events().stream()
|
||||
.filter(e -> e.getLevel().equals(Level.INFO))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains("drain complete"))
|
||||
.findFirst()
|
||||
.orElse("no drain-complete INFO logged");
|
||||
assertTrue(info.contains("released=2"),
|
||||
"both released sessions must be counted: " + info);
|
||||
assertTrue(info.contains("abandoned=0"),
|
||||
"neither session was BUSY, so nothing was abandoned mid-turn: " + info);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #512: the same completion line must also report a non-zero abandoned count when a
|
||||
* session is still {@code BUSY} once the whole-drain deadline passes — the case the ticket
|
||||
* calls out as the one a script needs to be able to see. Reuses the same BUSY/READY mix as
|
||||
* {@link #drainAllReleasesBusyAndReadySessionsAndWaitsForBusy}, which already forces the busy
|
||||
* session to spin until the real-time deadline expires (its state never leaves BUSY on its
|
||||
* own), and adds the log assertion that test does not make.
|
||||
*/
|
||||
@Test
|
||||
void drainAllLogsANonZeroAbandonedCountForASessionStillBusyAtTheDeadline() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "ownerR");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "ownerB");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId(), TestTurnTokens.inert(busy.terminalId()));
|
||||
// busy never leaves BUSY — no completion is delivered — so the drain below must spin the
|
||||
// full timeout and then release it anyway, counting it abandoned.
|
||||
|
||||
// Pin INFO explicitly — see the comment in drainAllLogsACompletionLineWithTheRealCountsOnACleanDrain.
|
||||
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.INFO)) {
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(sessions.roster().isEmpty(), "precondition: the drain actually ran");
|
||||
String info = log.events().stream()
|
||||
.filter(e -> e.getLevel().equals(Level.INFO))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains("drain complete"))
|
||||
.findFirst()
|
||||
.orElse("no drain-complete INFO logged");
|
||||
assertTrue(info.contains("released=2"),
|
||||
"both the ready and the busy session are released: " + info);
|
||||
assertTrue(info.contains("abandoned=1"),
|
||||
"the busy session hit the deadline still BUSY and must be counted: " + info);
|
||||
}
|
||||
}
|
||||
|
||||
// --- fleetd #308: a spawn accepted while the shutdown drain is running must not orphan ---
|
||||
|
||||
@Test
|
||||
@@ -1202,21 +1362,13 @@ class SessionManagerTest {
|
||||
new WorktreeRequest("cb-581a", null));
|
||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
try {
|
||||
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||
"a throwing dirty check must not abort the release");
|
||||
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"the worktree is preserved when its dirty state cannot be determined");
|
||||
String warn = appender.list.stream()
|
||||
String warn = log.events().stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains(s.worktree()))
|
||||
@@ -1224,8 +1376,6 @@ class SessionManagerTest {
|
||||
.orElse("no warn logged naming the worktree");
|
||||
assertTrue(warn.contains(s.paneId()), "the WARN names the pane: " + warn);
|
||||
assertTrue(warn.contains(s.terminalId()), "the WARN names the terminal: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1392,20 +1542,12 @@ class SessionManagerTest {
|
||||
// this itself, so it no longer propagates out of release() at all.
|
||||
worktrees.failRemoveFor(b.worktree());
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
int reaped;
|
||||
try {
|
||||
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||
clock[0] = 100;
|
||||
reaped = sessions.reapIdle(10);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
String warn = log.events().stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains(b.paneId()))
|
||||
@@ -1413,8 +1555,6 @@ class SessionManagerTest {
|
||||
.orElse("no worktree-removal-failure WARN logged");
|
||||
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
||||
assertTrue(warn.contains(b.worktree()), "the WARN names the failed session's worktree: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertEquals(3, reaped,
|
||||
@@ -1460,28 +1600,18 @@ class SessionManagerTest {
|
||||
// trigger reapIdle's own guard is for, now that #283 closed the worktree-removal trigger.
|
||||
herdr.paneCloseFailsForPane("w9:pRoot_2", "internal_error");
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
int reaped;
|
||||
try {
|
||||
try (CapturedLog log = CapturedLog.at(SessionManager.class, Level.WARN)) {
|
||||
clock[0] = 100;
|
||||
reaped = sessions.reapIdle(10);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
String warn = log.events().stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains("reap failed") && m.contains(b.paneId()))
|
||||
.findFirst()
|
||||
.orElse("no reap-failed WARN logged for the failing session");
|
||||
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertEquals(2, reaped,
|
||||
|
||||
@@ -64,7 +64,106 @@
|
||||
# The two outputs side by side are the finding: any name whose hash matches between them is a
|
||||
# credential the member holds in full.
|
||||
#
|
||||
# One parse pass: line 1 = present (true/false/null), line 2 = policy mode (possibly blank),
|
||||
# lines 3-5 = knownCount/allowedCount/blockedCount, remaining lines = the known[] names. A single
|
||||
# pass avoids re-parsing (and re-risking a truthiness bug) five separate times.
|
||||
#
|
||||
# This used to feed the parser straight into `mapfile -t _FIELDS < <(producer)`. That form cannot
|
||||
# see the producer fail: `<` `<(...)` is a process substitution, not a pipeline, so `set -o
|
||||
# pipefail` does not reach inside it, and mapfile's own exit status reports whether the BUILTIN
|
||||
# ran, not whether the command substituted into it succeeded — a failing jq or python3 there still
|
||||
# leaves mapfile at rc=0 with an empty array, read as a parse that genuinely found nothing (fleetd
|
||||
# #500). Capturing the parser's output with command substitution first, and checking ITS exit
|
||||
# status, reports the producer's real failure while the fact still exists — before it is handed to
|
||||
# mapfile at all.
|
||||
#
|
||||
# mapfile then reads from that captured string with `<<<` (a herestring), not `< <(...)`: `<<<`
|
||||
# materialises the whole string in memory first, where `< <(...)` would stream it. That only
|
||||
# matters for a large producer; this one is a short credential-name policy response, so the
|
||||
# tradeoff is irrelevant here — noted because it would not be for every producer.
|
||||
parse_policy_fields() {
|
||||
if command -v jq >/dev/null 2>&1; then
|
||||
_FIELDS_RAW="$(printf '%s' "$POLICY_JSON" | jq -r '
|
||||
(.present | tostring),
|
||||
(.policy // ""),
|
||||
(.knownCount // 0 | tostring),
|
||||
(.allowedCount // 0 | tostring),
|
||||
(.blockedCount // 0 | tostring),
|
||||
(.known[]? // empty)')"
|
||||
_PARSE_STATUS=$?
|
||||
_PARSER_NAME="jq"
|
||||
else
|
||||
_FIELDS_RAW="$(printf '%s' "$POLICY_JSON" | python3 - <<'PY'
|
||||
import json, sys
|
||||
data = json.load(sys.stdin)
|
||||
print(str(data.get("present")))
|
||||
print(data.get("policy") or "")
|
||||
print(data.get("knownCount") if data.get("knownCount") is not None else 0)
|
||||
print(data.get("allowedCount") if data.get("allowedCount") is not None else 0)
|
||||
print(data.get("blockedCount") if data.get("blockedCount") is not None else 0)
|
||||
for n in (data.get("known") or []):
|
||||
print(n)
|
||||
PY
|
||||
)"
|
||||
_PARSE_STATUS=$?
|
||||
_PARSER_NAME="python3"
|
||||
fi
|
||||
|
||||
if [ "$_PARSE_STATUS" -ne 0 ]; then
|
||||
echo "refusing to run: could not parse the policy fetched from $POLICY_URL — $_PARSER_NAME exited" \
|
||||
"non-zero (status $_PARSE_STATUS). That is a parser failure, not a claim about the policy" \
|
||||
"itself; the policy response has not been read." >&2
|
||||
return 4
|
||||
fi
|
||||
|
||||
# A herestring adds a newline, so mapfile would turn an empty parser result into one empty field.
|
||||
# Keep that case separate so the refusal reports what the parser actually returned: zero fields.
|
||||
if [ -z "$_FIELDS_RAW" ]; then
|
||||
_FIELDS=()
|
||||
else
|
||||
mapfile -t _FIELDS <<< "$_FIELDS_RAW"
|
||||
fi
|
||||
|
||||
# Arity check — the CORRECTNESS fix (fleetd #500). A parser that exits 0 can still return fewer
|
||||
# than the 5 fixed fields (present, policy mode, 3 counts) that every fixed-field read in main() expects,
|
||||
# whatever the reason: a producer that printed nothing, malformed JSON that jq/python3 still
|
||||
# accepted, or a schema change upstream. main()'s slice (`_FIELDS[@]:5`) does not fire
|
||||
# `set -u` on an unset OR a short array, and every fixed-field read there used a `:-` default, so
|
||||
# without this check a short `_FIELDS` reaches the "0 known names" guard further down with the
|
||||
# same look as a policy that genuinely has 0 names. Check the count here, at the one point the
|
||||
# fact is still present, before the slice consumes it.
|
||||
if (( ${#_FIELDS[@]} < 5 )); then
|
||||
echo "refusing to run: the policy parser ($_PARSER_NAME) returned ${#_FIELDS[@]} field(s); at" \
|
||||
"least 5 are required (present, policy mode, knownCount, allowedCount, blockedCount). The" \
|
||||
"parse ran but its shape is wrong — this is not a claim about how many names the policy" \
|
||||
"knows." >&2
|
||||
return 5
|
||||
fi
|
||||
}
|
||||
|
||||
main() {
|
||||
set -uo pipefail
|
||||
# `pipefail` is not what catches the parser failure handled in parse_policy_fields() above (fleetd #500): in
|
||||
# `printf '%s' "$POLICY_JSON" | jq -r '...'`, jq is the LAST element of the pipe, so the pipeline's
|
||||
# own exit status is already jq's status, with or without pipefail. It is kept as insurance for if
|
||||
# a post-processing stage is ever appended after the parser (e.g. `| tail -n +2`) — at that point
|
||||
# the parser would sit upstream and pipefail becomes the only thing that still reports its status.
|
||||
|
||||
# --- refuse on an interpreter that cannot run this script (fleetd #500) -------------------------
|
||||
#
|
||||
# mapfile, used below to parse the policy response, was added in bash 4.0. macOS ships bash 3.2.57
|
||||
# at /bin/bash, which predates it. This script's own `set -uo pipefail` does not catch a missing
|
||||
# mapfile: the builtin just fails with "command not found" on stderr, and every line below that
|
||||
# reads the array it would have filled uses a `:-` default or a slice, neither of which `set -u`
|
||||
# catches on an unset array. Left unguarded, that chain ends in the "0 known names" refusal further
|
||||
# down — a claim about the POLICY, for a failure that is actually about the INTERPRETER. So the
|
||||
# interpreter is checked once, explicitly, before it is asked to do anything mapfile depends on.
|
||||
if (( ${BASH_VERSINFO[0]} < 4 )); then
|
||||
echo "refusing to run: this script uses mapfile, which needs bash 4 or newer. This shell is bash" \
|
||||
"${BASH_VERSION:-<unknown, no \$BASH_VERSION>}. Re-run it under a newer bash, for example:" \
|
||||
"\"\$(command -v bash)\" \"$0\"" "$@" >&2
|
||||
exit 3
|
||||
fi
|
||||
|
||||
FLEETD_HOST="${FLEETD_HOST:-http://127.0.0.1:8765}"
|
||||
POLICY_URL="${FLEETD_HOST%/}/member-credentials"
|
||||
@@ -121,31 +220,10 @@ EOF
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# One parse pass: line 1 = present (true/false/null), line 2 = policy mode (possibly blank),
|
||||
# lines 3-5 = knownCount/allowedCount/blockedCount, remaining lines = the known[] names. A single
|
||||
# pass avoids re-parsing (and re-risking a truthiness bug) five separate times.
|
||||
if command -v jq >/dev/null 2>&1; then
|
||||
mapfile -t _FIELDS < <(printf '%s' "$POLICY_JSON" | jq -r '
|
||||
(.present | tostring),
|
||||
(.policy // ""),
|
||||
(.knownCount // 0 | tostring),
|
||||
(.allowedCount // 0 | tostring),
|
||||
(.blockedCount // 0 | tostring),
|
||||
(.known[]? // empty)')
|
||||
else
|
||||
mapfile -t _FIELDS < <(printf '%s' "$POLICY_JSON" | python3 - <<'PY'
|
||||
import json, sys
|
||||
data = json.load(sys.stdin)
|
||||
print(str(data.get("present")))
|
||||
print(data.get("policy") or "")
|
||||
print(data.get("knownCount") if data.get("knownCount") is not None else 0)
|
||||
print(data.get("allowedCount") if data.get("allowedCount") is not None else 0)
|
||||
print(data.get("blockedCount") if data.get("blockedCount") is not None else 0)
|
||||
for n in (data.get("known") or []):
|
||||
print(n)
|
||||
PY
|
||||
)
|
||||
fi
|
||||
# Parse the policy in one pass and refuse on any of the three failure causes. The decision, the
|
||||
# three refusals and the reasoning behind each live in parse_policy_fields() above — kept there
|
||||
# with the code rather than here, so the explanation cannot drift away from what it explains.
|
||||
parse_policy_fields || exit $?
|
||||
|
||||
PRESENT="${_FIELDS[0]:-null}"
|
||||
POLICY_MODE="${_FIELDS[1]:-}"
|
||||
@@ -164,19 +242,21 @@ case "$KNOWN_COUNT_REPORTED" in
|
||||
;;
|
||||
esac
|
||||
|
||||
# --- guard the denominator explicitly — never proceed on a zero/short count ---------------------
|
||||
# --- guard the denominator explicitly — never proceed on a zero count ---------------------------
|
||||
#
|
||||
# This is the exact trap named in the ticket: an empty (or truncated) NAMES array passes every
|
||||
# subsequent "is it set" check vacuously and prints a table that LOOKS complete. So this is checked
|
||||
# before anything else runs, with a message that says why, not just that it failed.
|
||||
# This is the exact trap named in the ticket: an empty NAMES array passes every subsequent "is it
|
||||
# set" check vacuously and prints a table that LOOKS complete. By this point the interpreter gate,
|
||||
# the parser-exit-status check, and the arity check above have already ruled out "the interpreter
|
||||
# couldn't run mapfile", "the parser failed", and "the parser returned the wrong shape" — so a zero
|
||||
# count reaching here really does mean the policy itself reports 0 known names, not a swallowed
|
||||
# failure upstream. That is still checked before anything else runs, with a message that says so.
|
||||
if [ "${#NAMES[@]}" -eq 0 ] || [ "$KNOWN_COUNT_REPORTED" -eq 0 ]; then
|
||||
cat >&2 <<EOF
|
||||
refusing to run: the policy fetched from $POLICY_URL contains 0 known names (present=${PRESENT:-unknown}).
|
||||
|
||||
Either memberCredentials: is absent/empty on the running daemon (nothing is protected — see fleetd's
|
||||
own startup warning), or the response could not be parsed. Either way, checking zero names would
|
||||
print a clean-looking table for a policy that protects nothing, or for a probe that read nothing.
|
||||
This is refused rather than reported as a pass.
|
||||
memberCredentials: is absent or empty on the running daemon — nothing is protected (see fleetd's own
|
||||
startup warning). Checking zero names would print a clean-looking table for a policy that protects
|
||||
nothing. This is refused rather than reported as a pass.
|
||||
EOF
|
||||
exit 1
|
||||
fi
|
||||
@@ -251,3 +331,8 @@ How to read this:
|
||||
hardcoded list did. If the daemon's policy changes, the next run of this script reflects it
|
||||
with no edit to this file.
|
||||
EOF
|
||||
}
|
||||
|
||||
if [[ "${BASH_SOURCE[0]}" == "$0" ]]; then
|
||||
main "$@"
|
||||
fi
|
||||
|
||||
+532
-64
@@ -5,7 +5,7 @@
|
||||
# A merge is not a deployment: the running daemon holds the jar it was started with, so code merged
|
||||
# to main does nothing until this runs. See CLAUDE.md -> "Redeploying the daemon".
|
||||
#
|
||||
# This script exists to turn six remembered traps into one auditable command:
|
||||
# This script exists to turn eight remembered traps into one auditable command:
|
||||
#
|
||||
# 1. A piped `mvn` hides BUILD FAILURE behind a zero exit, so the build here is never piped.
|
||||
# 2. The daemon must start from a LOGIN shell, or the tokens it hands to members are empty:
|
||||
@@ -27,6 +27,21 @@
|
||||
# restart of the OLD jar. So this script detects whether the agent is loaded and, only then,
|
||||
# swaps `kill` + manual `nohup` for `launchctl unload`/`load` — the one supervisor in control
|
||||
# at any moment is whichever one you asked to act, never both.
|
||||
# 7. fleetd #492 — a systemd --user unit is a THIRD possible supervisor (seen on a second host):
|
||||
# Restart=on-failure treats this JVM's SIGTERM exit code (143, per CB-594 above) as a failure
|
||||
# too, so a bare `kill` there would race systemd's own restart of the OLD jar exactly like
|
||||
# launchd would. This script now tells launchd, systemd, and "genuinely unsupervised" apart as
|
||||
# three different answers, drives whichever one it finds through its own control plane
|
||||
# (`launchctl` / `systemctl --user`), and REFUSES outright — never falls back to `kill` — when
|
||||
# it finds a supervision signal it cannot map to exactly one of the two it knows how to drive.
|
||||
# A wrong guess here is how two daemons end up running against one herdr session. Follow-up:
|
||||
# "not currently loaded" is not the same fact as "unsupervised" — a unit that is installed but
|
||||
# activating/failed/pending-restart, or a `systemctl` call that could not answer at all (e.g.
|
||||
# no user-bus access), both now read as a fifth answer, "unclear", and REFUSE the same way
|
||||
# "ambiguous" does, rather than silently falling through to "none".
|
||||
# 8. fleetd #492 — a post-restart check counts running fleetd processes and fails the whole run if
|
||||
# more than one is alive. That is the one thing none of the checks above (healthz 200, jar id,
|
||||
# the fresh "listening" line) can see: every one of them is satisfied by EITHER daemon.
|
||||
#
|
||||
# Usage:
|
||||
# scripts/redeploy-fleetd.sh # build, confirm, restart, verify
|
||||
@@ -41,6 +56,13 @@ set -euo pipefail
|
||||
REPO="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
MODULE="$REPO/fleetd"
|
||||
JAR="$MODULE/target/fleetd.jar"
|
||||
# fleetd #493: never build into the path a running process holds. The build writes here first
|
||||
# (Maven's shade plugin has finalName=fleetd, so `clean install` still lands its output at
|
||||
# target/fleetd.jar — that part is unchanged and out of this script's control), but this script
|
||||
# now moves it out to JAR_STAGED immediately, and only swaps it back to JAR (a plain `mv`, so a
|
||||
# rename, never a byte-by-byte overwrite) after the OLD daemon has been confirmed exited. See
|
||||
# stage_built_jar/swap_staged_jar below.
|
||||
JAR_STAGED="$MODULE/target/fleetd-new.jar"
|
||||
OUT="$MODULE/fleetd.out"
|
||||
# Matches BOTH the absolute form and the relative `java -jar target/fleetd.jar` a hand-start
|
||||
# produces from inside fleetd/. Anchoring on the absolute path alone was a real bug: the daemon
|
||||
@@ -55,13 +77,42 @@ HEALTH_WAIT=60 # seconds to wait for /healthz to answer after start
|
||||
LAUNCHD_LABEL='dev.ltms.fleetd'
|
||||
LAUNCHD_PLIST="$HOME/Library/LaunchAgents/$LAUNCHD_LABEL.plist"
|
||||
|
||||
# fleetd #492: the systemd --user unit this script must not fight with either (see trap 7 above).
|
||||
# Measured on the second host: `systemctl --user cat fleetd` names the unit "fleetd" (not
|
||||
# "dev.ltms.fleetd" — systemd user units here are not namespaced the way the launchd label is).
|
||||
SYSTEMD_UNIT='fleetd'
|
||||
|
||||
# fleetd #492 follow-up: detect_supervisor packs TWO values (kind, detail) onto the one stdout
|
||||
# line that survives its $(...) call — see the constraints comment above that function. This is
|
||||
# the separator between them: the ASCII "unit separator" byte, chosen because it never occurs in
|
||||
# any of the prose detail strings and needs no escaping in a `case`/glob pattern.
|
||||
SUPERVISOR_DETAIL_SEP=$'\x1f'
|
||||
|
||||
# fleetd #492 follow-up: set by systemd_loaded/systemd_installed when the underlying `systemctl`
|
||||
# call could not answer cleanly — it exited non-zero AND wrote something to stderr, which is a real
|
||||
# tool failure (e.g. it cannot reach the user bus over a non-lingering ssh session), never the same
|
||||
# fact as a clean negative answer ("not active", no stderr). Initialized here, not just inside the
|
||||
# probes, so detect_supervisor can read them under `set -u` even before either probe has ever run,
|
||||
# and so a test that stubs a probe with a plain `return 0`/`return 1` body (leaving these untouched)
|
||||
# reads a deterministic 0 rather than whatever a previous probe call left behind.
|
||||
SYSTEMD_LOADED_ERRORED=0
|
||||
SYSTEMD_INSTALLED_ERRORED=0
|
||||
# fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL is the specific supervisor/reason that
|
||||
# require_drivable_supervisor's die() names on an "unclear" answer. Deliberately NOT pre-declared
|
||||
# here (unlike the two flags above): it is set only by the real call site, right after it unpacks
|
||||
# detect_supervisor's stdout (see the constraints comment above detect_supervisor). If that call
|
||||
# site is ever skipped or broken, a bare `set -u` reference to this variable in
|
||||
# require_drivable_supervisor must fail loudly with "unbound variable" — a pre-declared empty
|
||||
# default would instead silently print an empty reason, hiding exactly the value this ticket
|
||||
# exists to surface.
|
||||
|
||||
DO_BUILD=1; ASSUME_YES=0; CHECK_ONLY=0
|
||||
for arg in "$@"; do
|
||||
case "$arg" in
|
||||
--yes|-y) ASSUME_YES=1 ;;
|
||||
--no-build) DO_BUILD=0 ;;
|
||||
--check) CHECK_ONLY=1 ;;
|
||||
-h|--help) sed -n '3,37p' "${BASH_SOURCE[0]}"; exit 0 ;;
|
||||
-h|--help) sed -n '3,48p' "${BASH_SOURCE[0]}"; exit 0 ;;
|
||||
*) echo "unknown option: $arg (try --help)" >&2; exit 2 ;;
|
||||
esac
|
||||
done
|
||||
@@ -71,14 +122,293 @@ ok() { printf ' ok %s\n' "$*"; }
|
||||
warn() { printf ' WARN %s\n' "$*"; }
|
||||
die() { printf '\n FAIL %s\n\n' "$*" >&2; exit 1; }
|
||||
|
||||
jar_id() { [ -f "$JAR" ] && shasum -a 256 "$JAR" | cut -c1-12 || echo "absent"; }
|
||||
# Reports the hash of $JAR by default, or of whatever path is passed — used to report the STAGED
|
||||
# jar right after a build (before it has been swapped in) without ever changing what a bare
|
||||
# `jar_id` (no args) means: the live path, $JAR. --check and the final "pid ..., jar ..." line
|
||||
# both call it with no args on purpose, so neither can ever be fooled by a leftover staged file.
|
||||
jar_id() { local f="${1:-$JAR}"; [ -f "$f" ] && shasum -a 256 "$f" | cut -c1-12 || echo "absent"; }
|
||||
running_pid() { pgrep -f "$PATTERN" || true; }
|
||||
|
||||
# fleetd #493 — three small, independently testable pieces of "never build into the path a
|
||||
# running process holds":
|
||||
#
|
||||
# stage_built_jar moves the jar Maven just produced OUT of the live path and onto the staging
|
||||
# path, immediately after a successful build. Dies (leaving the OLD daemon
|
||||
# untouched — this runs before the stop step) if Maven reported success but
|
||||
# left no jar behind, or if the move itself fails.
|
||||
# require_no_build_jar the --no-build path never builds or stages anything: it must find a
|
||||
# jar already sitting at the live path from an earlier successful run, and
|
||||
# die with the same truthful message this script has always used if not.
|
||||
# wait_for_daemon_exit polls running_pid() for up to $1 seconds and reports whether the OLD
|
||||
# daemon actually exited — extracted to its own function so the main flow
|
||||
# can be relied on to call swap_staged_jar only AFTER this returns success,
|
||||
# and so a test can prove that ordering by reading the script's own source.
|
||||
# swap_staged_jar the actual swap: a plain `mv` of the staged jar onto the live path. Called
|
||||
# only once the OLD daemon is confirmed gone (see wait_for_daemon_exit above),
|
||||
# so this is never a write into a path a running process holds — by the time
|
||||
# it runs, nothing holds that path anymore. If it fails, the caller must not
|
||||
# start a new daemon: die() below already refuses that by exiting the script.
|
||||
stage_built_jar() {
|
||||
[ -f "$JAR" ] || die "build succeeded but produced no jar at $JAR — cannot stage it for restart.
|
||||
The running daemon was NOT touched."
|
||||
mv -f "$JAR" "$JAR_STAGED" \
|
||||
|| die "could not move the freshly built jar from $JAR to the staging path $JAR_STAGED.
|
||||
The running daemon was NOT touched."
|
||||
}
|
||||
|
||||
require_no_build_jar() {
|
||||
[ -f "$JAR" ] || die "no jar at $JAR — run without --no-build"
|
||||
}
|
||||
|
||||
wait_for_daemon_exit() {
|
||||
local timeout="$1" _i
|
||||
for _i in $(seq "$timeout"); do
|
||||
[ -z "$(running_pid)" ] && return 0
|
||||
sleep 1
|
||||
done
|
||||
[ -z "$(running_pid)" ]
|
||||
}
|
||||
|
||||
swap_staged_jar() {
|
||||
local staged="$1" live="$2"
|
||||
[ -f "$staged" ] || die "no staged jar at $staged to swap in — the daemon was NOT started."
|
||||
mv -f "$staged" "$live" \
|
||||
|| die "could not move the staged jar from $staged into place at $live — the daemon was NOT
|
||||
started. The built jar is still sitting at $staged; a manual 'mv \"$staged\" \"$live\"'
|
||||
may recover this once you find out why the move failed."
|
||||
}
|
||||
|
||||
# fleetd #521 — the swap decision, and the step that acts on it.
|
||||
#
|
||||
# The defect: the swap step used to be guarded inline by `if [ "$DO_BUILD" = 1 ]` in the main flow.
|
||||
# Changing that to `if false` left the suite green and the swap never ran, so a redeploy reported
|
||||
# every step succeeding while the daemon started on no jar at all (stage_built_jar has already moved
|
||||
# the freshly built one to $JAR_STAGED by then) or on a stale one.
|
||||
# test_swap_ordered_after_wait_and_before_start could not catch it: it reads this script's own text
|
||||
# and compares line positions, and a same-line edit moves no line.
|
||||
#
|
||||
# Why these are TWO functions, and why the second one exists at all. Extracting only the predicate
|
||||
# — `should_swap`, which is what #521 asked for — is not enough, and this was measured, not guessed:
|
||||
# with the main flow calling `if should_swap "$DO_BUILD"; then`, changing THAT to `if false; then`
|
||||
# still left the whole suite at exit 0 with no failures. Tests that call a predicate directly prove
|
||||
# the predicate is right; nothing makes the code that does the work consult it. Extraction had moved
|
||||
# the untested decision one level up rather than removing it.
|
||||
#
|
||||
# So the decision and the action live together in swap_if_built, and the main flow has no guard of
|
||||
# its own to get wrong — it calls one function unconditionally. A test then calls swap_if_built with
|
||||
# both values of do_build and checks whether the swap actually happened, which fails if the guard is
|
||||
# removed, inverted, or stops being consulted. should_swap stays a separate predicate because it is
|
||||
# the decision itself and is worth naming and testing on its own.
|
||||
#
|
||||
# What this still does not pin: deleting the swap_if_built call from the main flow altogether. That
|
||||
# is the ordering test's job — its needle is that call site — and no test in this file can do better,
|
||||
# because sourcing stops before the main flow ever runs (see the SOURCED guard below).
|
||||
should_swap() {
|
||||
local do_build="$1"
|
||||
[ "$do_build" = 1 ]
|
||||
}
|
||||
|
||||
swap_if_built() {
|
||||
local do_build="$1"
|
||||
should_swap "$do_build" || return 0
|
||||
say "swap"
|
||||
swap_staged_jar "$JAR_STAGED" "$JAR"
|
||||
ok "jar in place: $(jar_id)"
|
||||
}
|
||||
|
||||
# `launchctl list <label>` exits 0 iff the label is loaded (registered with launchd) — true whether
|
||||
# or not it is currently running, which is exactly "supervision is active" for our purposes. Read-
|
||||
# only: neither helper below changes anything, so both are also safe under --check.
|
||||
launchd_installed() { [ -f "$LAUNCHD_PLIST" ]; }
|
||||
launchd_loaded() { launchctl list "$LAUNCHD_LABEL" >/dev/null 2>&1; }
|
||||
|
||||
# fleetd #492: same two questions for systemd --user. Kept as separate, overridable functions
|
||||
# (never an inline `systemctl` call at each use site) so a test on a box with no systemd at all
|
||||
# (this repo is developed on macOS) can substitute each one independently — the same seam
|
||||
# launchd_installed/launchd_loaded above already use.
|
||||
#
|
||||
# fleetd #492 follow-up: both functions used to throw `systemctl`'s stderr straight into
|
||||
# /dev/null, which meant "systemctl answered no" and "systemctl could not answer at all" (e.g. it
|
||||
# cannot reach the user bus over a non-lingering ssh session) looked identical — both a plain
|
||||
# nonzero exit. They now capture stderr separately and set their own *_ERRORED flag ONLY when the
|
||||
# call exited non-zero AND wrote something to stderr — a real tool failure, never a clean "not
|
||||
# installed"/"not active" answer (which exits non-zero with empty stderr). detect_supervisor reads
|
||||
# the flag right after calling the probe, so a probe that could not answer routes to "unclear",
|
||||
# never silently becomes "none".
|
||||
#
|
||||
# "installed": a unit FILE by this name exists, regardless of its current state — the systemd
|
||||
# analogue of the plist file existing on disk. `list-unit-files` reads unit definitions without
|
||||
# depending on runtime state, so this stays read-only and safe under --check.
|
||||
systemd_installed() {
|
||||
SYSTEMD_INSTALLED_ERRORED=0
|
||||
command -v systemctl >/dev/null 2>&1 || return 1
|
||||
local err_file out rc=0
|
||||
if ! err_file="$(mktemp -t systemd-installed-err)"; then
|
||||
SYSTEMD_INSTALLED_ERRORED=1
|
||||
return 1
|
||||
fi
|
||||
out="$(systemctl --user list-unit-files "$SYSTEMD_UNIT.service" --no-legend 2>"$err_file")" || rc=$?
|
||||
if [ "$rc" -ne 0 ]; then
|
||||
if [ -s "$err_file" ]; then
|
||||
SYSTEMD_INSTALLED_ERRORED=1
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
return "$rc"
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
printf '%s' "$out" | grep -q .
|
||||
}
|
||||
# "loaded": systemd currently supervises this unit as an active job — the systemd analogue of
|
||||
# `launchctl list <label>` succeeding. Measured on the second host: `systemctl --user is-active
|
||||
# fleetd` -> "active". A clean "no" (inactive/failed/activating/deactivating) exits non-zero with
|
||||
# nothing on stderr; a probe that could not reach systemd at all exits non-zero WITH a stderr
|
||||
# message — see the fleetd #492 follow-up note above.
|
||||
systemd_loaded() {
|
||||
SYSTEMD_LOADED_ERRORED=0
|
||||
command -v systemctl >/dev/null 2>&1 || return 1
|
||||
local err_file rc=0
|
||||
if ! err_file="$(mktemp -t systemd-loaded-err)"; then
|
||||
SYSTEMD_LOADED_ERRORED=1
|
||||
return 1
|
||||
fi
|
||||
systemctl --user is-active "$SYSTEMD_UNIT" >/dev/null 2>"$err_file" || rc=$?
|
||||
if [ "$rc" -ne 0 ] && [ -s "$err_file" ]; then
|
||||
SYSTEMD_LOADED_ERRORED=1
|
||||
fi
|
||||
rm -f "$err_file"
|
||||
return "$rc"
|
||||
}
|
||||
|
||||
# fleetd #492: three real answers, not two — launchd, systemd, or genuinely unsupervised — plus a
|
||||
# fourth, "ambiguous", for the one case this script cannot tell apart: both signals firing at once.
|
||||
# That is exactly "I cannot tell who supervises this process", and guessing wrong here is how two
|
||||
# daemons end up running against one herdr session (see trap 7 in the header).
|
||||
#
|
||||
# fleetd #492 follow-up: a fifth answer, "unclear", for two more situations that must NEVER be read
|
||||
# as "none" (measured — see the report this ticket is a follow-up to):
|
||||
# - installed-but-not-loaded, on EITHER supervisor. `systemctl --user is-active` answers "no" for
|
||||
# `activating`, `deactivating`, `failed`, and while an auto-restart is pending — every one of
|
||||
# those is a host that IS under systemd (or launchd) and whose supervisor is about to act again.
|
||||
# `*_installed` already knows the unit/agent exists; this is the first place that fact is
|
||||
# actually consulted in the decision, not just printed as a warning.
|
||||
# - a probe that could not answer at all. systemd_loaded/systemd_installed set their own
|
||||
# *_ERRORED flag (see the comment above them) when `systemctl` exits non-zero WITH a stderr
|
||||
# message — a real tool failure, e.g. it cannot reach the user bus over a non-lingering ssh
|
||||
# session — never conflated with a clean negative answer.
|
||||
# "none" now means only: neither supervisor is installed, neither is loaded, and neither probe
|
||||
# errored.
|
||||
#
|
||||
# fleetd #492 follow-up — constraints every caller of this function depends on (learned the hard
|
||||
# way: an earlier version of this fix set a SUPERVISOR_UNCLEAR_DETAIL global from inside here and
|
||||
# it was silently lost, because every real call site invokes this as `$(detect_supervisor)`):
|
||||
# 1. It is called as `$(detect_supervisor)`, so ONLY STDOUT crosses back to the caller. Anything
|
||||
# this function needs to tell its caller — the "unclear" detail included — must be printed,
|
||||
# never assigned to a global: a global set inside a `$( )` subshell dies with that subshell.
|
||||
# This function packs BOTH values (kind and detail) onto that one stdout line, joined by
|
||||
# $SUPERVISOR_DETAIL_SEP, and the caller unpacks them on its own side of the subshell boundary.
|
||||
# 2. This script runs under `set -euo pipefail` (line 50), so an unset variable is a loud
|
||||
# failure. Do not add a `${VAR:-default}` anywhere downstream to paper over a value that
|
||||
# should always be there — that hides a lost value instead of surfacing it (fleetd #497's
|
||||
# defect class).
|
||||
# 3. Every `case` on this function's return value needs an explicit final `*)` arm, chosen by
|
||||
# whether that caller ACTS on the value (`die` — an unrecognised value must never be silently
|
||||
# driven) or only DISPLAYS it (`echo`/`warn` and continue — a diagnostic must not go silent on
|
||||
# exactly the value it most needs to report).
|
||||
#
|
||||
# Pure and side-effect-free besides the two *_ERRORED flags (read back within this same call, never
|
||||
# by the caller — see the constraints above): reads the four probes and decides — never mutates
|
||||
# anything, so it is safe under --check and testable by overriding
|
||||
# launchd_installed/launchd_loaded/systemd_installed/systemd_loaded after sourcing.
|
||||
detect_supervisor() {
|
||||
local ld=0 sd=0 li=0 si=0 kind detail=""
|
||||
SYSTEMD_LOADED_ERRORED=0
|
||||
SYSTEMD_INSTALLED_ERRORED=0
|
||||
|
||||
launchd_loaded && ld=1
|
||||
systemd_loaded && sd=1
|
||||
launchd_installed && li=1
|
||||
systemd_installed && si=1
|
||||
|
||||
if [ "$SYSTEMD_LOADED_ERRORED" = 1 ] || [ "$SYSTEMD_INSTALLED_ERRORED" = 1 ]; then
|
||||
detail="the systemd --user probe for '$SYSTEMD_UNIT' could not answer cleanly (systemctl exited non-zero and reported an error on stderr, not a clean negative — e.g. it cannot reach the user bus)"
|
||||
kind="unclear"
|
||||
elif [ "$ld" = 1 ] && [ "$sd" = 1 ]; then
|
||||
kind="ambiguous"
|
||||
elif [ "$li" = 1 ] && [ "$ld" = 0 ]; then
|
||||
detail="the launchd agent ($LAUNCHD_LABEL) is installed ($LAUNCHD_PLIST exists) but is not currently loaded"
|
||||
kind="unclear"
|
||||
elif [ "$si" = 1 ] && [ "$sd" = 0 ]; then
|
||||
detail="the systemd --user unit ($SYSTEMD_UNIT) is installed but not currently active — it may be activating, deactivating, failed, or waiting on an auto-restart"
|
||||
kind="unclear"
|
||||
elif [ "$ld" = 1 ]; then
|
||||
kind="launchd"
|
||||
elif [ "$sd" = 1 ]; then
|
||||
kind="systemd"
|
||||
else
|
||||
kind="none"
|
||||
fi
|
||||
|
||||
printf '%s%s%s' "$kind" "$SUPERVISOR_DETAIL_SEP" "$detail"
|
||||
}
|
||||
|
||||
# fleetd #492: turns anything detect_supervisor returns that is NOT exactly one of the two
|
||||
# supervisors this script knows how to drive into a die() — never a fall-through to the `kill`
|
||||
# path. Kept as its own function so a test can call it directly (in a subshell, since it die()s)
|
||||
# without running the whole report-state flow or needing a real launchd/systemd.
|
||||
require_drivable_supervisor() {
|
||||
local kind="$1"
|
||||
case "$kind" in
|
||||
launchd|systemd|none) ;;
|
||||
ambiguous)
|
||||
die "both launchd ($LAUNCHD_LABEL) and systemd --user ($SYSTEMD_UNIT) report themselves as
|
||||
loaded for this daemon at the same time. This script cannot tell which one actually
|
||||
supervises the running process, and driving either alone risks the OTHER reviving the
|
||||
OLD jar out from under it — the exact failure this ticket (fleetd #492) exists to
|
||||
prevent. Stop one of the two supervisors by hand, confirm only one remains loaded, then
|
||||
rerun." ;;
|
||||
unclear)
|
||||
# fleetd #492 follow-up: SUPERVISOR_UNCLEAR_DETAIL crosses back from detect_supervisor's
|
||||
# subshell via its stdout, unpacked by the caller BEFORE it calls this function (see the
|
||||
# constraints comment above detect_supervisor). No ${VAR:-default} here on purpose: if the
|
||||
# detail is somehow missing, `set -u` makes this reference fail loudly instead of silently
|
||||
# naming nothing — a default that hides a lost value is the same defect class as fleetd
|
||||
# #497.
|
||||
die "a supervisor looks present but this script cannot tell whether it actually drives this
|
||||
daemon: $SUPERVISOR_UNCLEAR_DETAIL. Guessing wrong here is the same failure 'ambiguous'
|
||||
above exists to prevent: driving the daemon while an unseen supervisor revives the OLD
|
||||
jar out from under it (fleetd #492). Check 'launchctl list $LAUNCHD_LABEL' and
|
||||
'systemctl --user status $SYSTEMD_UNIT' by hand, resolve whichever looks unclear, then
|
||||
rerun." ;;
|
||||
*)
|
||||
die "detect_supervisor returned an unrecognized value '$kind' — refusing to guess which
|
||||
supervisor, if any, controls this daemon." ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# fleetd #492: the exact symptom a racing supervisor produces — count how many fleetd processes are
|
||||
# alive right now. Takes the pid list as a parameter (rather than calling running_pid() itself) so a
|
||||
# test can pass a canned two-line string without a real second process running. Pure except for the
|
||||
# die() in assert_single_daemon below.
|
||||
count_daemon_pids() {
|
||||
local pids="$1"
|
||||
if [ -z "$pids" ]; then
|
||||
echo 0
|
||||
else
|
||||
printf '%s\n' "$pids" | grep -c .
|
||||
fi
|
||||
}
|
||||
assert_single_daemon() {
|
||||
local pids="$1" count
|
||||
count="$(count_daemon_pids "$pids")"
|
||||
if [ "$count" -gt 1 ]; then
|
||||
die "more than one fleetd process is running after this restart (pids: $(printf '%s' "$pids" | tr '\n' ' ')).
|
||||
This is the exact failure a racing supervisor produces: the OLD jar was revived by its
|
||||
supervisor while this script started a NEW copy. Two daemons on one herdr session kill
|
||||
each other's members. Investigate with 'pgrep -f \"$PATTERN\"' and stop the wrong one by
|
||||
hand — do not assume either pid is the one you want."
|
||||
fi
|
||||
}
|
||||
|
||||
# CB-600: the script computes its own log path from where it sits on disk (REPO, above); the
|
||||
# plist hard-codes an absolute StandardOutPath. Nothing forced the two to agree — if this script
|
||||
# were ever run from a checkout other than the one the loaded plist names, launchd would start and
|
||||
@@ -160,6 +490,35 @@ classify_amqp_connection_errors() {
|
||||
REDEPLOY_UNEXPLAINED_ERRORS=$((REDEPLOY_UNEXPLAINED_ERRORS + pending_inbox + pending_lead_mailbox))
|
||||
}
|
||||
|
||||
# fleetd #517: extracted so the suite can call this decision directly, the same way #510 extracted
|
||||
# wait_for_daemon_exit so its ordering became checkable. Before this, the only test of the drain-gate
|
||||
# abort message was a grep of this script's own source for the wording — so mutating the `if` below
|
||||
# to `if false` (making the branch unreachable) left every test green, because the wording was still
|
||||
# sitting in the file. Pure: only decides which message applies and prints it, no side effects, so a
|
||||
# test can call it directly with an in-memory staged path instead of driving the real drain-gate flow
|
||||
# (which needs a live $OLD_PID and an interactive prompt neither test can supply).
|
||||
#
|
||||
# The four cases:
|
||||
# build ran, staged jar present -> names the staged jar and how to finish or discard it
|
||||
# build ran, staged jar absent -> "nothing changed" (nothing was staged this run either)
|
||||
# --no-build, staged jar present -> ALSO "nothing changed", deliberately: --no-build itself builds
|
||||
# and stages nothing (see require_no_build_jar above), so a staged jar found here is a leftover
|
||||
# from an earlier, unrelated run. THIS run truly changed nothing, and the next DO_BUILD=1 run
|
||||
# wipes that leftover before it builds (`rm -f "$JAR_STAGED"` in the build section above) — so
|
||||
# there is nothing here for the operator to lose track of.
|
||||
# --no-build, staged jar absent -> "nothing changed"
|
||||
drain_gate_refusal() {
|
||||
local do_build="$1" staged_path="$2"
|
||||
if [ "$do_build" = 1 ] && [ -f "$staged_path" ]; then
|
||||
printf 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at
|
||||
%s, not yet swapped into %s. Rerun WITHOUT --no-build to finish the restart —
|
||||
the freshly built jar is no longer at the live path that --no-build requires — or
|
||||
remove %s by hand if you want to discard this build.' "$staged_path" "$JAR" "$staged_path"
|
||||
else
|
||||
printf 'aborted — nothing changed'
|
||||
fi
|
||||
}
|
||||
|
||||
# CB-600: sourceable for testing. When this file is SOURCED (not executed) it stops here — nothing
|
||||
# below runs — so a test harness can `source` it to call check_log_path_matches_plist (or the
|
||||
# other pure helpers above) against a throwaway plist fixture without ever reaching the mutating
|
||||
@@ -182,25 +541,57 @@ fi
|
||||
ok "jar on disk: $(jar_id) ($([ -f "$JAR" ] && date -r "$JAR" '+%Y-%m-%d %H:%M:%S' || echo 'none'))"
|
||||
ok "HEAD: $(git -C "$REPO" log --oneline -1)"
|
||||
|
||||
# CB-594: supervision state. Installed and loaded are different facts — a copied-but-never-loaded
|
||||
# plist supervises nothing, and a loaded label with no file backing it (rare, but possible after an
|
||||
# edited/moved plist) is still what launchd will act on.
|
||||
# CB-594 / fleetd #492: supervision state. Installed and loaded are different facts — a
|
||||
# copied-but-never-loaded plist (or an unloaded systemd unit) supervises nothing, and a loaded
|
||||
# label/unit with no file backing it is still what its supervisor will act on.
|
||||
if launchd_installed; then
|
||||
ok "launchd agent installed: $LAUNCHD_PLIST"
|
||||
else
|
||||
warn "launchd agent NOT installed (no supervision — a crash will not restart the daemon)."
|
||||
warn "launchd agent NOT installed."
|
||||
fi
|
||||
SUPERVISED=0
|
||||
if launchd_loaded; then
|
||||
SUPERVISED=1
|
||||
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
||||
# CB-600: fail loudly here, before ANY other check runs, if this script and the loaded plist
|
||||
# would read different log files — every check after this point is worthless otherwise.
|
||||
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
||||
if systemd_installed; then
|
||||
ok "systemd --user unit installed: $SYSTEMD_UNIT"
|
||||
else
|
||||
warn "launchd agent not loaded — this script is the only thing that will restart the daemon."
|
||||
warn "systemd --user unit NOT installed ($SYSTEMD_UNIT)."
|
||||
fi
|
||||
|
||||
# fleetd #492: decide which of the two (if either) actually supervises this daemon, and refuse
|
||||
# outright — before touching anything — if that cannot be told apart (see require_drivable_
|
||||
# supervisor above). --check reaches this same line, so a host with an undrivable supervisor is
|
||||
# reported as a failure even in --check, without ever reaching the build/stop/start steps.
|
||||
# fleetd #492 follow-up: detect_supervisor runs as $(...), so only the printed line survives —
|
||||
# unpack kind and detail from it HERE, in this shell, before calling anything downstream. See the
|
||||
# constraints comment above detect_supervisor for why this cannot be done any other way.
|
||||
SUPERVISOR_RAW="$(detect_supervisor)"
|
||||
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
||||
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
||||
require_drivable_supervisor "$SUPERVISOR_KIND"
|
||||
ok "supervisor detected: $SUPERVISOR_KIND"
|
||||
SUPERVISED=0
|
||||
case "$SUPERVISOR_KIND" in
|
||||
launchd)
|
||||
SUPERVISED=1
|
||||
ok "launchd agent loaded ($LAUNCHD_LABEL) — launchd supervises this daemon"
|
||||
# CB-600: fail loudly here, before ANY other check runs, if this script and the loaded plist
|
||||
# would read different log files — every check after this point is worthless otherwise.
|
||||
check_log_path_matches_plist "$OUT" "$LAUNCHD_PLIST"
|
||||
;;
|
||||
systemd)
|
||||
SUPERVISED=1
|
||||
ok "systemd --user unit active ($SYSTEMD_UNIT) — systemd supervises this daemon"
|
||||
;;
|
||||
none)
|
||||
warn "no supervisor loaded — this script is the only thing that will restart the daemon."
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block only DISPLAYS state, it changes nothing yet — so a value
|
||||
# it doesn't recognise gets reported, not an abort that goes silent on exactly the state most
|
||||
# worth seeing. (Unreachable today: require_drivable_supervisor above already died on
|
||||
# "ambiguous"/"unclear" before this case runs. Guards the value nobody has invented yet.)
|
||||
warn "unrecognised supervisor kind: '$SUPERVISOR_KIND' — detect_supervisor returned a value this block does not know; continuing to report the rest of the state."
|
||||
;;
|
||||
esac
|
||||
|
||||
# The trap with no log line. Checked in a LOGIN shell, because that is how the daemon is started
|
||||
# below. Never prints the value — only whether it resolved.
|
||||
if zsh -lc '[ -n "${WORKER_GITEA_TOKEN:-}" ]' 2>/dev/null; then
|
||||
@@ -249,6 +640,9 @@ fi
|
||||
|
||||
if [ "$DO_BUILD" = 1 ]; then
|
||||
say "build"
|
||||
# fleetd #493: wipe a leftover staged jar from a previous failed/interrupted run BEFORE doing
|
||||
# anything else, so that run's leftovers can never be mistaken for this run's output.
|
||||
rm -f "$JAR_STAGED"
|
||||
BUILD_LOG="$(mktemp -t fleetd-build)"
|
||||
echo " log: $BUILD_LOG"
|
||||
if ! mvn -f "$MODULE/pom.xml" clean install > "$BUILD_LOG" 2>&1; then
|
||||
@@ -258,13 +652,19 @@ if [ "$DO_BUILD" = 1 ]; then
|
||||
fi
|
||||
grep -E '^\[INFO\] Tests run:.*Failures' "$BUILD_LOG" | tail -1 | sed 's/^\[INFO\] / /' || true
|
||||
ok "BUILD SUCCESS"
|
||||
ok "jar now: $(jar_id)"
|
||||
# fleetd #493: move the freshly built jar off the live path immediately — the running (OLD)
|
||||
# daemon, if any, is still up at this point (build always runs before stop). From here until the
|
||||
# swap step below (after the OLD daemon is confirmed gone), $JAR_STAGED is the only artefact this
|
||||
# script treats as "the new jar" — $JAR itself is not touched again until the swap.
|
||||
stage_built_jar
|
||||
ok "jar now: $(jar_id "$JAR_STAGED")"
|
||||
else
|
||||
say "build skipped (--no-build)"
|
||||
# fleetd #493: --no-build never builds or stages anything — it restarts whatever jar is already
|
||||
# sitting at the live path from an earlier successful run. Same check, same message as before.
|
||||
require_no_build_jar
|
||||
fi
|
||||
|
||||
[ -f "$JAR" ] || die "no jar at $JAR — run without --no-build"
|
||||
|
||||
# ----------------------------------------------------------------- drain gate
|
||||
|
||||
if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
||||
@@ -276,82 +676,143 @@ if [ -n "$OLD_PID" ] && [ "$ASSUME_YES" = 0 ]; then
|
||||
echo " you still want, BEFORE continuing."
|
||||
echo
|
||||
read -r -p " Fleet drained? type yes to restart: " reply
|
||||
[ "$reply" = "yes" ] || die "aborted — nothing changed"
|
||||
if [ "$reply" != "yes" ]; then
|
||||
# fleetd #493 / #517: "nothing changed" would be a lie once a build has run and staged a jar —
|
||||
# see drain_gate_refusal above for the full decision and why each of its four cases reads the
|
||||
# way it does.
|
||||
die "$(drain_gate_refusal "$DO_BUILD" "$JAR_STAGED")"
|
||||
fi
|
||||
fi
|
||||
|
||||
# ------------------------------------------------------------------ stop
|
||||
#
|
||||
# CB-594: when SUPERVISED, launchd owns the stop — never a raw `kill` here. A bare SIGTERM makes
|
||||
# this JVM exit 143 even with its shutdown hook running to completion (verified separately: a
|
||||
# throwaway Java process with an equivalent shutdown hook, sent SIGTERM from a login shell that
|
||||
# could `wait` on it directly, reported exit code 143 every time — never 0). launchd's
|
||||
# KeepAlive.SuccessfulExit=false treats any nonzero exit as a crash and restarts the OLD jar,
|
||||
# which would race this script's own restart of the NEW one. `launchctl unload` avoids that race
|
||||
# by deregistering the job first, so no KeepAlive is left armed when the process actually stops.
|
||||
# CB-594 / fleetd #492: when SUPERVISED, the supervisor owns the stop — never a raw `kill` here. A
|
||||
# bare SIGTERM makes this JVM exit 143 even with its shutdown hook running to completion (verified
|
||||
# separately: a throwaway Java process with an equivalent shutdown hook, sent SIGTERM from a login
|
||||
# shell that could `wait` on it directly, reported exit code 143 every time — never 0). launchd's
|
||||
# KeepAlive.SuccessfulExit=false and systemd's Restart=on-failure both treat any nonzero exit as a
|
||||
# crash and restart the OLD jar, which would race this script's own restart of the NEW one.
|
||||
# `launchctl unload` avoids that race by deregistering the job first, so no KeepAlive is left
|
||||
# armed when the process actually stops. `systemctl --user stop` needs no such dance: unlike
|
||||
# KeepAlive, systemd's Restart= does not fire on a deliberate stop, only on an unexpected exit of
|
||||
# an active unit.
|
||||
|
||||
if [ -n "$OLD_PID" ]; then
|
||||
say "stop"
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)" # verify a FRESH line appears later
|
||||
if [ "$SUPERVISED" = 1 ]; then
|
||||
echo " supervision is ON: using 'launchctl unload' (not kill) so launchd's own KeepAlive"
|
||||
echo " cannot restart the OLD jar out from under this script — see the CB-594 comment above."
|
||||
launchctl unload -w "$LAUNCHD_PLIST" \
|
||||
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
|
||||
else
|
||||
kill "$OLD_PID"
|
||||
fi
|
||||
for _ in $(seq "$STOP_WAIT"); do
|
||||
[ -z "$(running_pid)" ] && break
|
||||
sleep 1
|
||||
done
|
||||
if [ -n "$(running_pid)" ]; then
|
||||
case "$SUPERVISOR_KIND" in
|
||||
launchd)
|
||||
echo " supervision is ON (launchd): using 'launchctl unload' (not kill) so launchd's own"
|
||||
echo " KeepAlive cannot restart the OLD jar out from under this script — see the CB-594"
|
||||
echo " comment above."
|
||||
launchctl unload -w "$LAUNCHD_PLIST" \
|
||||
|| die "launchctl unload failed — the daemon may still be under supervision; investigate before retrying"
|
||||
;;
|
||||
systemd)
|
||||
echo " supervision is ON (systemd --user): using 'systemctl --user stop' (not kill) so"
|
||||
echo " systemd's own Restart=on-failure cannot restart the OLD jar out from under this"
|
||||
echo " script — see the fleetd #492 comment above."
|
||||
systemctl --user stop "$SYSTEMD_UNIT" \
|
||||
|| die "'systemctl --user stop $SYSTEMD_UNIT' failed — the daemon may still be under supervision; investigate before retrying"
|
||||
;;
|
||||
none)
|
||||
kill "$OLD_PID"
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block ACTS (stops the daemon one specific way per kind) — an
|
||||
# unrecognised value must never fall through to a default action, silently picking the wrong
|
||||
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
||||
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
||||
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the stop step —
|
||||
refusing to guess how to stop a daemon under an unknown supervisor. The daemon was NOT
|
||||
stopped." ;;
|
||||
esac
|
||||
if ! wait_for_daemon_exit "$STOP_WAIT"; then
|
||||
die "pid $OLD_PID still alive after ${STOP_WAIT}s. Not escalating to kill -9 automatically:
|
||||
the shutdown hook releases sessions and worktrees in order, and killing it hard can
|
||||
leave worktrees and panes behind. Investigate, then kill -9 by hand if you accept that."
|
||||
fi
|
||||
ok "pid $OLD_PID exited"
|
||||
elif [ "$SUPERVISED" = 1 ]; then
|
||||
elif [ "$SUPERVISOR_KIND" = "launchd" ]; then
|
||||
# Loaded but not currently running (e.g. throttled after a crash loop). Unload it anyway so the
|
||||
# start step below does a clean load, never a load stacked on an already-loaded label.
|
||||
say "stop"
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
||||
launchctl unload -w "$LAUNCHD_PLIST" 2>/dev/null || true
|
||||
ok "launchd agent unloaded (was already not running)"
|
||||
elif [ "$SUPERVISOR_KIND" = "systemd" ]; then
|
||||
# Same case for systemd: the unit is known/active-capable but not currently running. `stop` on an
|
||||
# already-stopped unit is a harmless no-op — kept for symmetry with the launchd branch above.
|
||||
say "stop"
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
||||
systemctl --user stop "$SYSTEMD_UNIT" 2>/dev/null || true
|
||||
ok "systemd --user unit stopped (was already not running)"
|
||||
else
|
||||
RESTART_MARK="$(wc -l < "$OUT" 2>/dev/null || echo 0)"
|
||||
fi
|
||||
|
||||
# ------------------------------------------------------------------ swap
|
||||
#
|
||||
# fleetd #493: every branch above has now either confirmed the OLD daemon actually exited
|
||||
# (wait_for_daemon_exit, above) or established there was never one running to begin with. Only
|
||||
# NOW is it safe to put the freshly built jar at the path the NEXT `java -jar` (direct, or via
|
||||
# launchd/systemd's ExecStart) will read from — this mv is the one and only write to $JAR anywhere
|
||||
# in this script's mutating flow. If it fails, do not start: die() below exits before "start" runs.
|
||||
swap_if_built "$DO_BUILD"
|
||||
|
||||
# ------------------------------------------------------------------ start
|
||||
# Unsupervised: login shell (zsh -l) is what puts the secrets on the daemon's environment, and cwd
|
||||
# must be fleetd/ because the daemon resolves fleetd.yaml, logs/ and target/ relative to it.
|
||||
# Supervised: launchd does both — deploy/dev.ltms.fleetd.plist points ProgramArguments at
|
||||
# Supervised (launchd): launchd does both — deploy/dev.ltms.fleetd.plist points ProgramArguments at
|
||||
# scripts/fleetd-launchd-wrapper.sh (CB-594), which is what execs the login shell in launchd's
|
||||
# place, and WorkingDirectory in the plist already pins fleetd/.
|
||||
# Supervised (systemd --user): the unit does both too — measured on the second host, ExecStart is
|
||||
# `/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"` (a login shell, same reason as
|
||||
# above) and WorkingDirectory is already pinned to fleetd/.
|
||||
|
||||
say "start"
|
||||
if [ "$SUPERVISED" = 1 ]; then
|
||||
echo " supervision is ON: using 'launchctl load' so launchd starts and keeps supervising this"
|
||||
echo " process, instead of a manual nohup that launchd would know nothing about."
|
||||
# CB-600: 'launchctl unload -w' above already persisted Disabled=true for this label. A load -w
|
||||
# that succeeds clears it; a load -w that FAILS leaves the agent both stopped and disabled — worse
|
||||
# than before this script ran, because a later reboot or login will not bring it back either. One
|
||||
# retry covers a transient race (e.g. launchd not yet fully done deregistering); if it still fails,
|
||||
# die with the exact recovery command rather than a bare "failed".
|
||||
if ! launchctl load -w "$LAUNCHD_PLIST" 2>/dev/null; then
|
||||
warn "launchctl load failed on the first attempt — retrying once after a short pause"
|
||||
sleep 2
|
||||
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed twice.
|
||||
The agent is now STOPPED and DISABLED — it will NOT come back on its own, not even after a
|
||||
reboot or login, because 'launchctl unload -w' above persisted Disabled=true and load -w
|
||||
never got the chance to clear it. Recover with:
|
||||
launchctl load -w \"$LAUNCHD_PLIST\"
|
||||
If that still fails, check 'launchctl list $LAUNCHD_LABEL', validate the plist with
|
||||
'plutil -lint \"$LAUNCHD_PLIST\"', and check $OUT before assuming a retry will succeed."
|
||||
fi
|
||||
else
|
||||
# Absolute jar path so `ps` names which checkout is running.
|
||||
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
||||
fi
|
||||
case "$SUPERVISOR_KIND" in
|
||||
launchd)
|
||||
echo " supervision is ON (launchd): using 'launchctl load' so launchd starts and keeps"
|
||||
echo " supervising this process, instead of a manual nohup that launchd would know nothing"
|
||||
echo " about."
|
||||
# CB-600: 'launchctl unload -w' above already persisted Disabled=true for this label. A load -w
|
||||
# that succeeds clears it; a load -w that FAILS leaves the agent both stopped and disabled — worse
|
||||
# than before this script ran, because a later reboot or login will not bring it back either. One
|
||||
# retry covers a transient race (e.g. launchd not yet fully done deregistering); if it still fails,
|
||||
# die with the exact recovery command rather than a bare "failed".
|
||||
if ! launchctl load -w "$LAUNCHD_PLIST" 2>/dev/null; then
|
||||
warn "launchctl load failed on the first attempt — retrying once after a short pause"
|
||||
sleep 2
|
||||
launchctl load -w "$LAUNCHD_PLIST" || die "launchctl load failed twice.
|
||||
The agent is now STOPPED and DISABLED — it will NOT come back on its own, not even after a
|
||||
reboot or login, because 'launchctl unload -w' above persisted Disabled=true and load -w
|
||||
never got the chance to clear it. Recover with:
|
||||
launchctl load -w \"$LAUNCHD_PLIST\"
|
||||
If that still fails, check 'launchctl list $LAUNCHD_LABEL', validate the plist with
|
||||
'plutil -lint \"$LAUNCHD_PLIST\"', and check $OUT before assuming a retry will succeed."
|
||||
fi
|
||||
;;
|
||||
systemd)
|
||||
echo " supervision is ON (systemd --user): using 'systemctl --user start' so systemd starts"
|
||||
echo " and keeps supervising this process, instead of a manual nohup it would know nothing"
|
||||
echo " about."
|
||||
systemctl --user start "$SYSTEMD_UNIT" || die "'systemctl --user start $SYSTEMD_UNIT' failed.
|
||||
Check 'systemctl --user status $SYSTEMD_UNIT' and $OUT before assuming a retry will succeed."
|
||||
;;
|
||||
none)
|
||||
# Absolute jar path so `ps` names which checkout is running.
|
||||
( cd "$MODULE" && zsh -lc "nohup java -jar '$JAR' >> fleetd.out 2>&1 &" )
|
||||
;;
|
||||
*)
|
||||
# fleetd #492 follow-up: this block ACTS (starts the daemon one specific way per kind) — an
|
||||
# unrecognised value must never fall through to a default action, silently picking the wrong
|
||||
# one (or none at all) while reporting success. (Unreachable today: require_drivable_
|
||||
# supervisor already died before this runs. Guards the value nobody has invented yet.)
|
||||
die "detect_supervisor returned an unrecognized value '$SUPERVISOR_KIND' at the start step —
|
||||
refusing to guess how to start a daemon under an unknown supervisor. The daemon was NOT
|
||||
started." ;;
|
||||
esac
|
||||
|
||||
for _ in $(seq 10); do
|
||||
NEW_PID="$(running_pid)"
|
||||
@@ -411,6 +872,13 @@ FRESH_LOG="$(mktemp -t fleetd-fresh-log)"
|
||||
trap 'rm -f "$FRESH_LOG"' EXIT
|
||||
tail -n "+$((RESTART_MARK + 1))" "$OUT" > "$FRESH_LOG" 2>/dev/null || true
|
||||
classify_amqp_connection_errors "$FRESH_LOG"
|
||||
|
||||
# fleetd #492: checked here, after healthz and the fresh-log check have both had time to run, so a
|
||||
# supervisor that revives the OLD jar a few seconds late is caught too. Every check above (healthz
|
||||
# 200, jar id, the fresh 'listening' line) is satisfied by EITHER daemon if two are alive — this is
|
||||
# the only one that can tell.
|
||||
assert_single_daemon "$(running_pid)"
|
||||
|
||||
say "result"
|
||||
ok "pid $NEW_PID, jar $(jar_id)"
|
||||
if [ "$REDEPLOY_ERROR_COUNT" -eq 0 ]; then
|
||||
|
||||
Executable
+109
@@ -0,0 +1,109 @@
|
||||
#!/usr/bin/env bash
|
||||
# Self-contained checks for the policy parsing guards in probe-member-credentials.sh.
|
||||
|
||||
set -euo pipefail
|
||||
|
||||
ROOT="$(cd "$(dirname "${BASH_SOURCE[0]}")/.." && pwd)"
|
||||
PROBE="$ROOT/scripts/probe-member-credentials.sh"
|
||||
TMP="$(mktemp -d "$ROOT/.probe-member-credentials-test.XXXXXX")"
|
||||
trap 'rm -rf "$TMP"' EXIT
|
||||
|
||||
# The SOURCED guard exposes this pure parser without contacting POLICY_URL.
|
||||
source "$PROBE"
|
||||
|
||||
fail() {
|
||||
printf 'FAIL: %s\n' "$*" >&2
|
||||
return 1
|
||||
}
|
||||
|
||||
assert_equals() {
|
||||
local expected="$1" actual="$2" description="$3"
|
||||
[ "$expected" = "$actual" ] || fail "$description: expected $expected, got $actual"
|
||||
}
|
||||
|
||||
assert_contains() {
|
||||
local needle="$1" text="$2" description="$3"
|
||||
printf '%s' "$text" | grep -qF "$needle" || fail "$description: missing $needle"
|
||||
}
|
||||
|
||||
make_jq() {
|
||||
local body="$1"
|
||||
mkdir -p "$TMP/bin"
|
||||
printf '%s\n' '#!/usr/bin/env bash' "$body" > "$TMP/bin/jq"
|
||||
chmod +x "$TMP/bin/jq"
|
||||
}
|
||||
|
||||
run_parser() {
|
||||
local output rc=0
|
||||
POLICY_JSON="$(< "$TMP/policy.json")"
|
||||
POLICY_URL="fixture://member-credentials"
|
||||
output="$(PATH="$TMP/bin:$PATH" parse_policy_fields 2>&1)" || rc=$?
|
||||
PARSER_OUTPUT="$output"
|
||||
PARSER_RC="$rc"
|
||||
}
|
||||
|
||||
test_bash_older_than_four_refuses() {
|
||||
local output rc=0 version
|
||||
version="$(/bin/bash -c 'printf %s "$BASH_VERSION"')"
|
||||
output="$(/bin/bash "$PROBE" 2>&1)" || rc=$?
|
||||
assert_equals 3 "$rc" "bash 3 refusal status"
|
||||
assert_contains 'This shell is bash' "$output" "bash 3 refusal"
|
||||
assert_contains "$version" "$output" "bash 3 refusal version"
|
||||
}
|
||||
|
||||
test_parser_non_zero_refuses() {
|
||||
make_jq 'exit 17'
|
||||
run_parser
|
||||
assert_equals 4 "$PARSER_RC" "parser failure status"
|
||||
assert_contains 'jq exited non-zero (status 17)' "$PARSER_OUTPUT" "parser failure message"
|
||||
}
|
||||
|
||||
test_short_parser_output_refuses() {
|
||||
make_jq "printf '%s\\n' true enforce 3 2"
|
||||
run_parser
|
||||
assert_equals 5 "$PARSER_RC" "short parser output status"
|
||||
assert_contains 'policy parser (jq) returned 4 field(s)' "$PARSER_OUTPUT" "short parser output count"
|
||||
}
|
||||
|
||||
test_empty_parser_output_reports_zero_fields() {
|
||||
make_jq ':'
|
||||
run_parser
|
||||
assert_equals 5 "$PARSER_RC" "empty parser output status"
|
||||
assert_contains 'policy parser (jq) returned 0 field(s)' "$PARSER_OUTPUT" "empty parser output count"
|
||||
}
|
||||
|
||||
test_well_formed_policy_prints_name_table() {
|
||||
local output rc=0
|
||||
make_jq "cat '$TMP/policy.fields'"
|
||||
# Shell functions cannot be passed in an environment assignment. Run the executable through bash.
|
||||
output="$(BRIDGED_MEMBER=1 FIXTURE="$TMP/policy.json" PROBE="$PROBE" PATH="$TMP/bin:$PATH" bash -c '
|
||||
curl() { cat "$FIXTURE"; }
|
||||
export -f curl
|
||||
exec "$PROBE"
|
||||
' 2>&1)" || rc=$?
|
||||
assert_equals 0 "$rc" "well-formed policy status"
|
||||
assert_contains 'ALPHA_TOKEN' "$output" "name table"
|
||||
assert_contains 'BETA_TOKEN' "$output" "name table"
|
||||
assert_contains 'GAMMA_TOKEN' "$output" "name table"
|
||||
}
|
||||
|
||||
cat > "$TMP/policy.json" <<'JSON'
|
||||
{"present":true,"policy":"enforce","knownCount":3,"allowedCount":2,"blockedCount":1,"known":["ALPHA_TOKEN","BETA_TOKEN","GAMMA_TOKEN"]}
|
||||
JSON
|
||||
cat > "$TMP/policy.fields" <<'FIELDS'
|
||||
true
|
||||
enforce
|
||||
3
|
||||
2
|
||||
1
|
||||
ALPHA_TOKEN
|
||||
BETA_TOKEN
|
||||
GAMMA_TOKEN
|
||||
FIELDS
|
||||
|
||||
test_bash_older_than_four_refuses
|
||||
test_parser_non_zero_refuses
|
||||
test_short_parser_output_refuses
|
||||
test_empty_parser_output_reports_zero_fields
|
||||
test_well_formed_policy_prints_name_table
|
||||
printf 'PASS: probe member credentials guards\n'
|
||||
@@ -25,6 +25,502 @@ classify_fixture() {
|
||||
classify_amqp_connection_errors "$TMP/$name"
|
||||
}
|
||||
|
||||
# fleetd #492 follow-up: detect_supervisor's stdout is now "kind<SEP>detail" (see the constraints
|
||||
# comment above detect_supervisor in redeploy-fleetd.sh) — every test below that only cares about
|
||||
# the kind must split it out with the SAME in-shell parameter expansion the real call site (:438)
|
||||
# uses, never a bare string comparison against the raw output.
|
||||
supervisor_kind_of() {
|
||||
printf '%s' "${1%%"$SUPERVISOR_DETAIL_SEP"*}"
|
||||
}
|
||||
supervisor_detail_of() {
|
||||
printf '%s' "${1#*"$SUPERVISOR_DETAIL_SEP"}"
|
||||
}
|
||||
|
||||
|
||||
# fleetd #492 — supervisor detection. Detect_supervisor() reads launchd_loaded/systemd_loaded, so
|
||||
# each test overrides BOTH pairs (installed + loaded) explicitly, rather than relying on either
|
||||
# being naturally absent: this machine may itself be running a real fleetd under launchd right now
|
||||
# (see CLAUDE.md/MEMORY.md — launchd supervision has been live here since 2026-08-26), so leaving
|
||||
# launchd_loaded unmocked in a "systemd only" test would silently read this host's own live state
|
||||
# instead of the fixture.
|
||||
test_detect_supervisor_launchd_only() {
|
||||
launchd_installed() { return 0; }
|
||||
launchd_loaded() { return 0; }
|
||||
systemd_installed() { return 1; }
|
||||
systemd_loaded() { return 1; }
|
||||
assert_equals "launchd" "$(supervisor_kind_of "$(detect_supervisor)")" "launchd-only detection"
|
||||
}
|
||||
|
||||
test_detect_supervisor_systemd_only() {
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 0; }
|
||||
systemd_loaded() { return 0; }
|
||||
assert_equals "systemd" "$(supervisor_kind_of "$(detect_supervisor)")" "systemd-only detection"
|
||||
}
|
||||
|
||||
test_detect_supervisor_none() {
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 1; }
|
||||
systemd_loaded() { return 1; }
|
||||
assert_equals "none" "$(supervisor_kind_of "$(detect_supervisor)")" "unsupervised detection"
|
||||
}
|
||||
|
||||
# fleetd #492 follow-up — detect_supervisor must never answer "none" when the truth is "could not
|
||||
# tell". `systemd_installed`/`systemd_loaded` already know a unit file exists; this proves that
|
||||
# fact is now actually consulted, not just printed as a warning: an installed-but-not-loaded unit
|
||||
# reads as unclear, because is-active answers "no" for activating/deactivating/failed/pending
|
||||
# auto-restart too, and every one of those is a host that IS under systemd.
|
||||
test_detect_supervisor_systemd_installed_not_loaded_is_unclear() {
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 0; } # the unit file IS there
|
||||
systemd_loaded() { return 1; } # is-active says no — could be activating/failed/pending restart
|
||||
local raw
|
||||
raw="$(detect_supervisor)"
|
||||
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "systemd installed-but-not-loaded must read as unclear, not none"
|
||||
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$SYSTEMD_UNIT" \
|
||||
|| fail "detail does not name the systemd unit it found installed-but-not-loaded"
|
||||
}
|
||||
|
||||
# Same fact, the launchd side: a plist on disk that is not currently loaded (unloaded without being
|
||||
# removed, or about to be reloaded) must not read as "no supervisor" either.
|
||||
test_detect_supervisor_launchd_installed_not_loaded_is_unclear() {
|
||||
launchd_installed() { return 0; } # the plist IS there
|
||||
launchd_loaded() { return 1; } # launchctl list says not loaded
|
||||
systemd_installed() { return 1; }
|
||||
systemd_loaded() { return 1; }
|
||||
local raw
|
||||
raw="$(detect_supervisor)"
|
||||
assert_equals "unclear" "$(supervisor_kind_of "$raw")" "launchd installed-but-not-loaded must read as unclear, not none"
|
||||
printf '%s' "$(supervisor_detail_of "$raw")" | grep -qF "$LAUNCHD_LABEL" \
|
||||
|| fail "detail does not name the launchd label it found installed-but-not-loaded"
|
||||
}
|
||||
|
||||
# Drives the REAL systemd_loaded/systemd_installed bodies (never stubbed) through a `systemctl`
|
||||
# stub placed first on PATH that exits non-zero AND writes to stderr — the shape of a systemctl
|
||||
# that runs but cannot reach the user bus (measured elsewhere as a headless ssh session with no
|
||||
# lingering). This must read as unclear, never none: a probe that could not answer at all is not
|
||||
# the same fact as "no supervisor is loaded".
|
||||
test_detect_supervisor_systemd_probe_error_is_unclear() {
|
||||
# Re-source first to restore the REAL launchd_*/systemd_* probe bodies. Earlier tests in this
|
||||
# file permanently override them with stub `return 0`/`return 1` bodies (that is the whole point
|
||||
# of those tests), and a bash function definition is global for the rest of the process — without
|
||||
# this, systemd_loaded here would still be whatever the previous test left it as, never touching
|
||||
# a real `systemctl` call at all.
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local bin_dir result rc=0
|
||||
bin_dir="$TMP/stub-bin-systemctl-errors"
|
||||
mkdir -p "$bin_dir"
|
||||
cat > "$bin_dir/systemctl" <<'STUB'
|
||||
#!/usr/bin/env bash
|
||||
echo "Failed to connect to bus: No such file or directory" >&2
|
||||
exit 1
|
||||
STUB
|
||||
chmod +x "$bin_dir/systemctl"
|
||||
|
||||
PATH="$bin_dir:$PATH" systemd_loaded && rc=0 || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "systemd_loaded must not report loaded=true when systemctl only errored"
|
||||
assert_equals "1" "$SYSTEMD_LOADED_ERRORED" "systemd_loaded must flag a probe error, not a clean negative"
|
||||
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
result="$(PATH="$bin_dir:$PATH" detect_supervisor)"
|
||||
assert_equals "unclear" "$(supervisor_kind_of "$result")" "a systemd probe error must read as unclear, not none"
|
||||
printf '%s' "$(supervisor_detail_of "$result")" | grep -qF "$SYSTEMD_UNIT" \
|
||||
|| fail "detail does not name the systemd unit whose probe errored"
|
||||
}
|
||||
|
||||
# fleetd #492 follow-up (Item 1): this must go through the REAL call-site shape at :437-440, not a
|
||||
# hand-constructed "unclear" value — a test that builds "unclear" directly proves the switch, not
|
||||
# the handoff, and that is exactly the gap that let SUPERVISOR_UNCLEAR_DETAIL never reach the real
|
||||
# caller in b17f37a. detect_supervisor runs as $(detect_supervisor): a subshell. Only stdout
|
||||
# survives that boundary, so kind AND detail must both cross on it — this test proves they do.
|
||||
test_require_drivable_supervisor_refuses_unclear() {
|
||||
launchd_installed() { return 1; }
|
||||
launchd_loaded() { return 1; }
|
||||
systemd_installed() { return 0; }
|
||||
systemd_loaded() { return 1; }
|
||||
|
||||
local SUPERVISOR_RAW SUPERVISOR_KIND SUPERVISOR_UNCLEAR_DETAIL output rc=0
|
||||
# Exactly what :437-439 does — do not shortcut this by constructing "unclear" by hand.
|
||||
SUPERVISOR_RAW="$(detect_supervisor)"
|
||||
SUPERVISOR_KIND="${SUPERVISOR_RAW%%"$SUPERVISOR_DETAIL_SEP"*}"
|
||||
SUPERVISOR_UNCLEAR_DETAIL="${SUPERVISOR_RAW#*"$SUPERVISOR_DETAIL_SEP"}"
|
||||
|
||||
assert_equals "unclear" "$SUPERVISOR_KIND" "setup: expected unclear before testing the refusal"
|
||||
[ -n "$SUPERVISOR_UNCLEAR_DETAIL" ] \
|
||||
|| fail "detail did not survive the \$(...) call-site boundary — SUPERVISOR_UNCLEAR_DETAIL is empty in the parent shell"
|
||||
printf '%s' "$SUPERVISOR_UNCLEAR_DETAIL" | grep -qF "$SYSTEMD_UNIT" \
|
||||
|| fail "detail that crossed the subshell boundary does not name the systemd unit it found installed-but-not-loaded"
|
||||
|
||||
output="$(require_drivable_supervisor "$SUPERVISOR_KIND" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an unclear (undrivable) supervisor"
|
||||
# Check for the ACTUAL DETAIL TEXT, not just "$SYSTEMD_UNIT" — the die() message's boilerplate
|
||||
# recovery instructions name the unit unconditionally either way ("systemctl --user status
|
||||
# $SYSTEMD_UNIT"), so a bare unit-name grep here would pass even on a lost/fallback detail. Only
|
||||
# the specific detail string proves the crossed value, not the boilerplate, reached the message.
|
||||
printf '%s' "$output" | grep -qF "$SUPERVISOR_UNCLEAR_DETAIL" \
|
||||
|| fail "refusal message does not contain the specific detail that crossed the subshell boundary"
|
||||
}
|
||||
|
||||
# The heart of the ticket: a supervisor this script cannot drive must refuse, never fall through to
|
||||
# `kill`. require_drivable_supervisor die()s, so it is invoked inside a command substitution — that
|
||||
# forks a subshell, so its exit() only ends the subshell and this test script keeps running under
|
||||
# `set -e`.
|
||||
test_require_drivable_supervisor_refuses_ambiguous() {
|
||||
local output rc=0
|
||||
output="$(require_drivable_supervisor "ambiguous" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "require_drivable_supervisor accepted an ambiguous (undrivable) supervisor"
|
||||
printf '%s' "$output" | grep -qF "$LAUNCHD_LABEL" \
|
||||
|| fail "refusal message does not name the launchd label it found"
|
||||
printf '%s' "$output" | grep -qF "$SYSTEMD_UNIT" \
|
||||
|| fail "refusal message does not name the systemd unit it found"
|
||||
}
|
||||
|
||||
test_require_drivable_supervisor_accepts_known_kinds() {
|
||||
require_drivable_supervisor "launchd" || fail "refused a drivable launchd supervisor"
|
||||
require_drivable_supervisor "systemd" || fail "refused a drivable systemd supervisor"
|
||||
require_drivable_supervisor "none" || fail "refused the unsupervised case"
|
||||
}
|
||||
|
||||
# fleetd #492 — the one-daemon check. Two live pids is the exact symptom a racing supervisor
|
||||
# produces, and none of the other post-restart checks (healthz, jar id, the fresh log line) can see
|
||||
# it because either daemon alone satisfies them.
|
||||
test_count_daemon_pids() {
|
||||
assert_equals 0 "$(count_daemon_pids "")" "count of an empty pid list"
|
||||
assert_equals 1 "$(count_daemon_pids "4242")" "count of a single pid"
|
||||
assert_equals 2 "$(count_daemon_pids "$(printf '4242\n4343\n')")" "count of two pids"
|
||||
}
|
||||
|
||||
test_assert_single_daemon_accepts_one_pid() {
|
||||
assert_single_daemon "4242" || fail "assert_single_daemon rejected a single running pid"
|
||||
}
|
||||
|
||||
test_assert_single_daemon_rejects_two_pids() {
|
||||
local output rc=0
|
||||
output="$(assert_single_daemon "$(printf '4242\n4343\n')" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "assert_single_daemon accepted two simultaneously running pids"
|
||||
printf '%s' "$output" | grep -qF '4242' || fail "refusal message does not list the pids it found"
|
||||
printf '%s' "$output" | grep -qF '4343' || fail "refusal message does not list the pids it found"
|
||||
}
|
||||
|
||||
# fleetd #511 — jar_id()'s no-argument default was unpinned by any test: nothing proved it reports
|
||||
# $JAR (the live path) rather than $JAR_STAGED. Both halves matter, so this pins both: the bare call
|
||||
# must hash the live jar, and an explicit path argument must hash THAT file, not fall back to $JAR.
|
||||
# Two files with different content, so a default pointed at the wrong one reports the wrong hash
|
||||
# rather than accidentally matching.
|
||||
test_jar_id_defaults_to_live_and_reports_explicit_path() {
|
||||
local dir saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
||||
local live_hash staged_hash default_result explicit_result
|
||||
dir="$TMP/jar-id"; mkdir -p "$dir"
|
||||
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
||||
printf 'live jar bytes' > "$JAR"
|
||||
printf 'staged jar bytes, not the same content' > "$JAR_STAGED"
|
||||
live_hash="$(shasum -a 256 "$JAR" | cut -c1-12)"
|
||||
staged_hash="$(shasum -a 256 "$JAR_STAGED" | cut -c1-12)"
|
||||
default_result="$(jar_id)"
|
||||
explicit_result="$(jar_id "$JAR_STAGED")"
|
||||
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
||||
[ "$live_hash" != "$staged_hash" ] || fail "test fixture error: live and staged jars hashed the same"
|
||||
assert_equals "$live_hash" "$default_result" "jar_id with no arguments must report the hash of \$JAR"
|
||||
assert_equals "$staged_hash" "$explicit_result" "jar_id \"\$JAR_STAGED\" must report the hash of the staged jar, not fall back to \$JAR"
|
||||
}
|
||||
|
||||
# fleetd #517 — jar_id()'s "absent" branch was unpinned by any test: the existing test above (#511)
|
||||
# proves both halves of the present-file contract but never exercises the missing-file path. This
|
||||
# word matters more than a string usually would: "absent" is the #413 signal that a `mvn clean`
|
||||
# deleted the running daemon's jar out from under it, and the `redeploy-fleetd` skill points
|
||||
# operators at `--check` for exactly this. Covers both the no-argument default and an explicit path,
|
||||
# since the mutation (`absent` -> `present`) sits on the single shared `|| echo` and would flip both.
|
||||
test_jar_id_reports_absent_for_missing_file() {
|
||||
local saved_jar="$JAR" dir default_result explicit_result
|
||||
dir="$TMP/jar-id-absent"; mkdir -p "$dir"
|
||||
JAR="$dir/does-not-exist.jar"
|
||||
[ ! -f "$JAR" ] || fail "test fixture error: \$JAR unexpectedly exists at $JAR"
|
||||
default_result="$(jar_id)"
|
||||
explicit_result="$(jar_id "$dir/also-does-not-exist.jar")"
|
||||
JAR="$saved_jar"
|
||||
assert_equals "absent" "$default_result" "jar_id with no arguments must report absent when \$JAR does not exist"
|
||||
assert_equals "absent" "$explicit_result" "jar_id with an explicit missing path must report absent"
|
||||
}
|
||||
|
||||
# fleetd #493 — never build into the path a running process holds. stage_built_jar/swap_staged_jar
|
||||
# are exercised directly against real files on disk (not stubs), because the whole point is file
|
||||
# behavior (does the content move, does the source disappear, does a failure leave both sides
|
||||
# intact) that a stubbed function cannot prove.
|
||||
test_stage_built_jar_moves_off_live_path() {
|
||||
local dir jar staged saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
||||
dir="$TMP/stage-ok"; mkdir -p "$dir"
|
||||
jar="$dir/fleetd.jar"; staged="$dir/fleetd-new.jar"
|
||||
printf 'built jar bytes' > "$jar"
|
||||
JAR="$jar"; JAR_STAGED="$staged"
|
||||
stage_built_jar || fail "stage_built_jar rejected a real build output"
|
||||
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
||||
[ ! -f "$jar" ] || fail "stage_built_jar left the jar behind at the live path $jar"
|
||||
[ -f "$staged" ] || fail "stage_built_jar did not create the staged jar at $staged"
|
||||
grep -qF 'built jar bytes' "$staged" || fail "staged jar does not carry the built content"
|
||||
}
|
||||
|
||||
test_stage_built_jar_dies_when_build_produced_nothing() {
|
||||
local dir output rc=0 saved_jar="$JAR" saved_staged="$JAR_STAGED"
|
||||
dir="$TMP/stage-missing"; mkdir -p "$dir"
|
||||
JAR="$dir/fleetd.jar"; JAR_STAGED="$dir/fleetd-new.jar"
|
||||
output="$(stage_built_jar 2>&1)" || rc=$?
|
||||
JAR="$saved_jar"; JAR_STAGED="$saved_staged"
|
||||
[ "$rc" -ne 0 ] || fail "stage_built_jar accepted a missing build output"
|
||||
printf '%s' "$output" | grep -qF "$dir/fleetd.jar" \
|
||||
|| fail "refusal message does not name the missing jar path"
|
||||
}
|
||||
|
||||
test_swap_staged_jar_moves_staged_onto_live() {
|
||||
local dir staged live
|
||||
dir="$TMP/swap-ok"; mkdir -p "$dir"
|
||||
staged="$dir/fleetd-new.jar"; live="$dir/fleetd.jar"
|
||||
printf 'swapped jar bytes' > "$staged"
|
||||
swap_staged_jar "$staged" "$live" || fail "swap_staged_jar rejected a real staged jar"
|
||||
[ ! -f "$staged" ] || fail "swap_staged_jar left the staged file behind at $staged"
|
||||
[ -f "$live" ] || fail "swap_staged_jar did not create the live jar at $live"
|
||||
grep -qF 'swapped jar bytes' "$live" || fail "live jar does not carry the staged content"
|
||||
}
|
||||
|
||||
# The heart of the ticket's item 3: a failed swap must refuse to start. This function dies on
|
||||
# failure, and die() exits — so like the require_drivable_supervisor tests above, the call goes
|
||||
# inside a command substitution to contain that exit to a subshell.
|
||||
test_swap_staged_jar_dies_without_staged_file() {
|
||||
local dir output rc=0
|
||||
dir="$TMP/swap-missing"; mkdir -p "$dir"
|
||||
output="$(swap_staged_jar "$dir/fleetd-new.jar" "$dir/fleetd.jar" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a missing staged jar"
|
||||
[ ! -f "$dir/fleetd.jar" ] || fail "swap_staged_jar must not create the live jar when nothing was staged"
|
||||
printf '%s' "$output" | grep -qF "$dir/fleetd-new.jar" \
|
||||
|| fail "refusal message does not name the missing staged path"
|
||||
}
|
||||
|
||||
test_swap_staged_jar_dies_when_mv_fails() {
|
||||
local dir staged live output rc=0
|
||||
dir="$TMP/swap-fail"; mkdir -p "$dir/src"
|
||||
staged="$dir/src/fleetd-new.jar"
|
||||
printf 'fake jar bytes' > "$staged"
|
||||
live="$dir/no-such-dir/fleetd.jar" # parent directory does not exist -> mv fails
|
||||
output="$(swap_staged_jar "$staged" "$live" 2>&1)" || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "swap_staged_jar accepted a failing mv"
|
||||
[ -f "$staged" ] || fail "swap_staged_jar must leave the staged jar in place when the move fails"
|
||||
[ ! -f "$live" ] || fail "swap_staged_jar must not report success when the move failed"
|
||||
printf '%s' "$output" | grep -qF "$staged" \
|
||||
|| fail "refusal message does not name the staged path that could not be moved"
|
||||
}
|
||||
|
||||
# --no-build must still resolve $JAR (never the staged path — there is nothing to stage on this
|
||||
# path) and must still die with the exact wording documented in the script's own header comment.
|
||||
test_require_no_build_jar_dies_when_absent() {
|
||||
local saved_jar="$JAR" output rc=0 missing="$TMP/no-build-absent/fleetd.jar"
|
||||
JAR="$missing"
|
||||
output="$(require_no_build_jar 2>&1)" || rc=$?
|
||||
JAR="$saved_jar"
|
||||
[ "$rc" -ne 0 ] || fail "require_no_build_jar accepted a missing jar"
|
||||
printf '%s' "$output" | grep -qF "no jar at $missing — run without --no-build" \
|
||||
|| fail "refusal message does not match the documented --no-build wording"
|
||||
}
|
||||
|
||||
test_require_no_build_jar_accepts_present_jar() {
|
||||
local saved_jar="$JAR" dir
|
||||
dir="$TMP/no-build-present"; mkdir -p "$dir"
|
||||
JAR="$dir/fleetd.jar"
|
||||
printf 'existing jar' > "$JAR"
|
||||
require_no_build_jar || fail "require_no_build_jar rejected an existing jar"
|
||||
JAR="$saved_jar"
|
||||
}
|
||||
|
||||
# wait_for_daemon_exit is the seam the swap ordering depends on: it must not report success while
|
||||
# running_pid() still answers, and must report success the moment it clears. `sleep` is shadowed so
|
||||
# the timeout-loop test does not actually wait out its budget.
|
||||
test_wait_for_daemon_exit_returns_true_once_pid_clears() {
|
||||
# running_pid() runs inside a $(...) — a subshell — every time wait_for_daemon_exit calls it, so
|
||||
# a plain shell variable it increments would reset on each call instead of accumulating. Count in
|
||||
# a file instead, which is the one thing that actually survives across those subshells.
|
||||
local counter_file="$TMP/wait-exit-calls" final_calls
|
||||
printf '0' > "$counter_file"
|
||||
running_pid() {
|
||||
local n
|
||||
n="$(cat "$counter_file")"
|
||||
n=$((n + 1))
|
||||
printf '%s' "$n" > "$counter_file"
|
||||
if [ "$n" -lt 3 ]; then printf '4242'; else printf ''; fi
|
||||
}
|
||||
sleep() { :; }
|
||||
wait_for_daemon_exit 10 || fail "wait_for_daemon_exit did not report success once the pid cleared"
|
||||
final_calls="$(cat "$counter_file")"
|
||||
[ "$final_calls" -ge 3 ] || fail "wait_for_daemon_exit returned before actually re-checking running_pid"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
||||
}
|
||||
|
||||
test_wait_for_daemon_exit_times_out_if_pid_never_clears() {
|
||||
local rc=0
|
||||
running_pid() { printf '4242'; }
|
||||
sleep() { :; }
|
||||
wait_for_daemon_exit 3 || rc=$?
|
||||
[ "$rc" -ne 0 ] || fail "wait_for_daemon_exit reported success while the pid never cleared"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh" # restore the real running_pid/sleep for later tests
|
||||
}
|
||||
|
||||
# fleetd #521 — the swap step's guard, at two levels.
|
||||
#
|
||||
# The first two tests call the predicate should_swap() directly. They pin its logic, and that is all
|
||||
# they pin. On their own they did NOT close #521, and this was measured rather than argued: with the
|
||||
# main flow reading `if should_swap "$DO_BUILD"; then`, changing that line to `if false; then` left
|
||||
# this whole suite at exit 0 with zero FAIL lines, because nothing here made the code that performs
|
||||
# the swap consult the predicate at all. Extracting the decision had moved the untested decision up
|
||||
# a level, not removed it.
|
||||
#
|
||||
# So the last two tests call swap_if_built() — the function the main flow actually calls, holding the
|
||||
# guard and the swap together — with a recording stub in place of the real `mv`. Those fail if the
|
||||
# guard is removed, inverted, or stops being consulted.
|
||||
#
|
||||
# What none of these four can catch: deleting the `swap_if_built "$DO_BUILD"` line from the main flow
|
||||
# altogether. That is test_swap_ordered_after_wait_and_before_start's job below, because sourcing
|
||||
# stops before the main flow runs, so no test in this file can invoke it.
|
||||
test_should_swap_true_when_build_ran() {
|
||||
should_swap 1 || fail "should_swap 1 (a build ran and staged a jar) must return true"
|
||||
}
|
||||
|
||||
test_should_swap_false_when_build_skipped() {
|
||||
if should_swap 0; then
|
||||
fail "should_swap 0 (--no-build; nothing was staged this run) must return false"
|
||||
fi
|
||||
}
|
||||
|
||||
# Both of these re-source redeploy-fleetd.sh at the START, because a bash function definition is
|
||||
# global for the rest of the process and an earlier test may have left swap_staged_jar or jar_id
|
||||
# overridden (see the longer note on this at test_detect_supervisor_systemd_probe_error_is_unclear),
|
||||
# and again at the END, so their own stubs do not leak into every test that runs after them.
|
||||
test_swap_if_built_performs_the_swap_when_build_ran() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local marker="$TMP/swap-if-built-ran"
|
||||
rm -f "$marker"
|
||||
swap_staged_jar() { printf '%s -> %s\n' "$1" "$2" > "$marker"; }
|
||||
jar_id() { printf 'stubbed\n'; }
|
||||
swap_if_built 1 > /dev/null
|
||||
[ -f "$marker" ] \
|
||||
|| fail "swap_if_built 1 (a build ran and staged a jar) must perform the swap, and did not"
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
test_swap_if_built_skips_the_swap_when_build_skipped() {
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
local marker="$TMP/swap-if-built-skipped"
|
||||
rm -f "$marker"
|
||||
swap_staged_jar() { printf 'swapped\n' > "$marker"; }
|
||||
jar_id() { printf 'stubbed\n'; }
|
||||
swap_if_built 0 > /dev/null
|
||||
if [ -f "$marker" ]; then
|
||||
fail "swap_if_built 0 (--no-build; nothing was staged this run) must not swap, but it did"
|
||||
fi
|
||||
source "$ROOT/scripts/redeploy-fleetd.sh"
|
||||
}
|
||||
|
||||
# fleetd #493 item 2: "put the swap after that wait, before the start." Sourcing stops before the
|
||||
# main flow ever runs (see the SOURCED guard in redeploy-fleetd.sh), so the ordering guarantee
|
||||
# itself — as opposed to the pure functions it's built from — can only be checked by reading the
|
||||
# script's own call sites, the same way test_recovery_patterns_match_source below checks Java
|
||||
# source shape instead of behavior it cannot invoke directly.
|
||||
#
|
||||
# Two details about the three greps below, both of which have already gone wrong here.
|
||||
#
|
||||
# The needle for the swap is the MAIN FLOW's call site, `swap_if_built "$DO_BUILD"` — not
|
||||
# `swap_staged_jar "$JAR_STAGED" "$JAR"`. Since fleetd #521 that second string lives inside
|
||||
# swap_if_built's body, which is defined near the top of the script, far ABOVE the stop step. Using
|
||||
# it made this test report "swap_staged_jar (line 215) is not after wait_for_daemon_exit (line 730)"
|
||||
# — a true statement about a function definition, and nothing at all about the order of the steps.
|
||||
#
|
||||
# Each grep ends in `|| true`. This file runs under `set -euo pipefail`, and `pipefail` makes the
|
||||
# pipeline's status grep's status, so a needle that is simply ABSENT failed the assignment and `set
|
||||
# -e` killed the whole suite on the spot — before reaching the `[ -n ... ] || fail` line written to
|
||||
# report exactly that. Measured: the suite exited 1 having printed zero bytes, no FAIL line and no
|
||||
# name of the missing call site. `|| true` lets the assignment succeed empty so the guard can speak.
|
||||
test_swap_ordered_after_wait_and_before_start() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" wait_line swap_line start_line
|
||||
wait_line="$(grep -Fn 'wait_for_daemon_exit "$STOP_WAIT"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
swap_line="$(grep -Fn 'swap_if_built "$DO_BUILD"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
start_line="$(grep -Fn 'say "start"' "$src" | head -1 | cut -d: -f1 || true)"
|
||||
[ -n "$wait_line" ] || fail "could not find the wait-for-exit call site in redeploy-fleetd.sh"
|
||||
[ -n "$swap_line" ] || fail "could not find the swap call site in redeploy-fleetd.sh"
|
||||
[ -n "$start_line" ] || fail "could not find the start section in redeploy-fleetd.sh"
|
||||
[ "$swap_line" -gt "$wait_line" ] \
|
||||
|| fail "swap_if_built (line $swap_line) is not after wait_for_daemon_exit (line $wait_line)"
|
||||
[ "$swap_line" -lt "$start_line" ] \
|
||||
|| fail "swap_if_built (line $swap_line) is not before the start section (line $start_line)"
|
||||
}
|
||||
|
||||
# fleetd #511: the drain-gate abort message (fired when a build has staged a jar but the operator
|
||||
# declines the drain confirmation) used to tell the operator to "Rerun (with or without --no-build)"
|
||||
# to finish the restart. That is wrong — by the time this message can fire, stage_built_jar has
|
||||
# already moved the jar off $JAR, so a rerun WITH --no-build hits require_no_build_jar's own refusal
|
||||
# ("no jar at $JAR — run without --no-build"). Like test_swap_ordered_after_wait_and_before_start
|
||||
# above, this code path is never reached by sourcing (the SOURCED guard stops before the main flow),
|
||||
# so the only way to pin its exact wording is to read the source.
|
||||
test_drain_gate_abort_message_says_no_no_build() {
|
||||
local src="$ROOT/scripts/redeploy-fleetd.sh" msg
|
||||
msg="$(grep -A3 -F 'aborted — the running daemon was NOT touched, but the freshly built jar is sitting at' "$src")"
|
||||
[ -n "$msg" ] || fail "could not find the drain-gate staged-jar abort message in redeploy-fleetd.sh"
|
||||
if printf '%s' "$msg" | grep -qF 'with or without --no-build'; then
|
||||
fail "abort message still claims a rerun WITH --no-build can finish the restart"
|
||||
fi
|
||||
printf '%s' "$msg" | grep -qF 'WITHOUT --no-build' \
|
||||
|| fail "abort message does not tell the operator to rerun without --no-build"
|
||||
printf '%s' "$msg" | grep -qF 'no longer at the live path' \
|
||||
|| fail "abort message does not say why --no-build cannot finish the restart"
|
||||
}
|
||||
|
||||
# fleetd #517 — the drain-gate abort branch itself. Before this, the only test of this message was
|
||||
# a source-text grep (test_drain_gate_abort_message_says_no_no_build, below): it greps this script's
|
||||
# own file for the wording, which stays in the file even if the `if` guarding it is mutated to
|
||||
# `if false` and the branch can never run. These four tests call drain_gate_refusal directly instead,
|
||||
# so they fail if the branch is unreachable OR if its wording regresses — the grep test is KEPT
|
||||
# alongside these, not replaced, because it catches a different regression (a re-wording that still
|
||||
# reaches the right branch would not change which case fires here, but would still be worth pinning).
|
||||
test_drain_gate_refusal_build_ran_staged_present() {
|
||||
local dir staged result
|
||||
dir="$TMP/drain-refusal-build-staged"; mkdir -p "$dir"
|
||||
staged="$dir/fleetd-new.jar"
|
||||
printf 'staged jar bytes' > "$staged"
|
||||
result="$(drain_gate_refusal 1 "$staged")"
|
||||
printf '%s' "$result" | grep -qF "$staged" \
|
||||
|| fail "build-ran+staged-present refusal does not name the staged jar path"
|
||||
printf '%s' "$result" | grep -qF 'Rerun WITHOUT --no-build' \
|
||||
|| fail "build-ran+staged-present refusal does not tell the operator how to finish the restart"
|
||||
if printf '%s' "$result" | grep -qF 'nothing changed'; then
|
||||
fail "build-ran+staged-present refusal must not claim nothing changed — the jar already moved"
|
||||
fi
|
||||
}
|
||||
|
||||
test_drain_gate_refusal_build_ran_staged_absent() {
|
||||
local dir result
|
||||
dir="$TMP/drain-refusal-build-no-staged"; mkdir -p "$dir"
|
||||
result="$(drain_gate_refusal 1 "$dir/fleetd-new.jar")"
|
||||
assert_equals "aborted — nothing changed" "$result" "build-ran+staged-absent refusal wording"
|
||||
}
|
||||
|
||||
# --no-build itself never builds or stages anything (require_no_build_jar, above), so a staged jar
|
||||
# found here is a leftover from an earlier, unrelated run — THIS run truly changed nothing. See the
|
||||
# comment above drain_gate_refusal in redeploy-fleetd.sh for the full reasoning.
|
||||
test_drain_gate_refusal_no_build_staged_present() {
|
||||
local dir staged result
|
||||
dir="$TMP/drain-refusal-no-build-staged"; mkdir -p "$dir"
|
||||
staged="$dir/fleetd-new.jar"
|
||||
printf 'leftover staged jar bytes' > "$staged"
|
||||
result="$(drain_gate_refusal 0 "$staged")"
|
||||
assert_equals "aborted — nothing changed" "$result" "no-build+staged-present refusal must deliberately say nothing changed"
|
||||
}
|
||||
|
||||
test_drain_gate_refusal_no_build_staged_absent() {
|
||||
local dir result
|
||||
dir="$TMP/drain-refusal-no-build-no-staged"; mkdir -p "$dir"
|
||||
result="$(drain_gate_refusal 0 "$dir/fleetd-new.jar")"
|
||||
assert_equals "aborted — nothing changed" "$result" "no-build+staged-absent refusal wording"
|
||||
}
|
||||
|
||||
test_no_errors() {
|
||||
cat > "$TMP/no-errors.log" <<'LOG'
|
||||
2026-09-05 12:00:00 INFO fleetd listening
|
||||
@@ -224,6 +720,39 @@ test_unattributable_quiet_mutation_is_caught() {
|
||||
printf 'Unattributable mutation: FAIL: cross-unattributable recovered: expected 0, got 2\n'
|
||||
}
|
||||
|
||||
test_detect_supervisor_launchd_only
|
||||
test_detect_supervisor_systemd_only
|
||||
test_detect_supervisor_none
|
||||
test_detect_supervisor_systemd_installed_not_loaded_is_unclear
|
||||
test_detect_supervisor_launchd_installed_not_loaded_is_unclear
|
||||
test_detect_supervisor_systemd_probe_error_is_unclear
|
||||
test_require_drivable_supervisor_refuses_ambiguous
|
||||
test_require_drivable_supervisor_refuses_unclear
|
||||
test_require_drivable_supervisor_accepts_known_kinds
|
||||
test_count_daemon_pids
|
||||
test_assert_single_daemon_accepts_one_pid
|
||||
test_assert_single_daemon_rejects_two_pids
|
||||
test_jar_id_defaults_to_live_and_reports_explicit_path
|
||||
test_jar_id_reports_absent_for_missing_file
|
||||
test_stage_built_jar_moves_off_live_path
|
||||
test_stage_built_jar_dies_when_build_produced_nothing
|
||||
test_swap_staged_jar_moves_staged_onto_live
|
||||
test_swap_staged_jar_dies_without_staged_file
|
||||
test_swap_staged_jar_dies_when_mv_fails
|
||||
test_should_swap_true_when_build_ran
|
||||
test_should_swap_false_when_build_skipped
|
||||
test_swap_if_built_performs_the_swap_when_build_ran
|
||||
test_swap_if_built_skips_the_swap_when_build_skipped
|
||||
test_require_no_build_jar_dies_when_absent
|
||||
test_require_no_build_jar_accepts_present_jar
|
||||
test_wait_for_daemon_exit_returns_true_once_pid_clears
|
||||
test_wait_for_daemon_exit_times_out_if_pid_never_clears
|
||||
test_swap_ordered_after_wait_and_before_start
|
||||
test_drain_gate_abort_message_says_no_no_build
|
||||
test_drain_gate_refusal_build_ran_staged_present
|
||||
test_drain_gate_refusal_build_ran_staged_absent
|
||||
test_drain_gate_refusal_no_build_staged_present
|
||||
test_drain_gate_refusal_no_build_staged_absent
|
||||
test_no_errors
|
||||
test_recovery_patterns_match_source
|
||||
test_attributed_recovered_connection_error
|
||||
|
||||
Reference in New Issue
Block a user