Compare commits
11 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 275ac0d251 | |||
| 204da67d66 | |||
| b091c51eee | |||
| 84d631b030 | |||
| 3f8c38fc54 | |||
| a4dbc8f8b7 | |||
| ed2fd6646a | |||
| 5441a2b321 | |||
| 384867dfa3 | |||
| 034e17bb32 | |||
| e20ccab1eb |
@@ -144,7 +144,7 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
| Confirm your own role | `fleet_whoami` |
|
||||
| See backends available | `fleet_profiles` |
|
||||
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `fleet_status{sessionId}` |
|
||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) + `loopHealth` (`RUNNING`, `STALLED`, or `STOPPED` for `statusPoller` and `sessionReaper`) · one peer's state: `fleet_status{sessionId}` |
|
||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
|
||||
@@ -21,6 +21,7 @@ import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.inject.TurnListener;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
@@ -79,6 +80,7 @@ import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.BooleanSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Predicate;
|
||||
@@ -491,42 +493,10 @@ public final class Fleetd {
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
// fleetd #561: extracted to a static factory (see turnListener below) — the anonymous class
|
||||
// this replaced had two bare, unguarded statements per callback, and nothing enforced that
|
||||
// the completion half went first beyond call order in the source.
|
||||
TurnListener turnListener = turnListener(completion, sessions);
|
||||
Predicate<String> deliverable = deliverableTo(presence, leads);
|
||||
// fleetd #556: registration is wired directly to `completion`, not folded into the
|
||||
// `turnListener` fan-out above — so it survives `sessions.onDelivered` (or any future
|
||||
@@ -696,10 +666,13 @@ public final class Fleetd {
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, outagePolicy);
|
||||
|
||||
FleetMcp.LoopHealthSource loopHealth = new FleetMcp.LoopHealthSource(poller::health,
|
||||
() -> reaper == null ? LoopWatchdog.State.STOPPED : reaper.health());
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, FleetMcp.AuthorizationMode.ENFORCED, metrics,
|
||||
capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
||||
healthCoverageSource(config),
|
||||
loopHealth,
|
||||
quarantineSource,
|
||||
leadMailbox,
|
||||
outageSource,
|
||||
@@ -791,7 +764,7 @@ public final class Fleetd {
|
||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable,
|
||||
() -> MemberCredentialPolicyView.of(config.get().memberCredentials()),
|
||||
quarantineSource, outageSource).build();
|
||||
quarantineSource, outageSource, loopHealth).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
@@ -1096,6 +1069,157 @@ public final class Fleetd {
|
||||
.orElse(null);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #561: compose the production {@link TurnListener} from its two halves — the
|
||||
* completion resolver (which resolves a blocked {@code fleet_send}'s waiter) and the session
|
||||
* manager (which drives the member's lifecycle state) — hardened so a throw from either
|
||||
* half's callback can never suppress the other half's callback for the same event.
|
||||
*
|
||||
* <p>Before this, each method below was two bare, unguarded statements: whichever ran first
|
||||
* throwing meant the second one never ran at all, and nothing beyond call order in the
|
||||
* source enforced "the completion half goes first". #556 fixed the identical shape for {@code
|
||||
* onDelivered}'s registration by moving it off the fan-out entirely (see {@code
|
||||
* TurnRegistrar}); this fixes the four remaining callbacks — {@code onTurnComplete}, {@code
|
||||
* onTurnCompleteWithPostAction}, and both {@code onTurnFailed} overloads — by hardening the
|
||||
* fan-out itself instead, since none of them can be pulled out of the listener the way
|
||||
* registration was.
|
||||
*
|
||||
* <p>The invariant this composition guarantees: <b>a throwing session-listener must not
|
||||
* prevent the completion resolver from being told the turn ended.</b> The completion half is
|
||||
* always attempted first, and — as a bonus the resolver does not depend on — its own throw
|
||||
* does not stop the session half from running either. Whatever escapes (from one half or
|
||||
* both) is rethrown once both have been attempted, with a second failure recorded via {@link
|
||||
* Throwable#addSuppressed} on the first, so it still reaches {@link StatusPoller}'s {@code
|
||||
* catch (Throwable)} and logs at ERROR. Nothing here swallows a failure to make the two halves
|
||||
* look safe.
|
||||
*
|
||||
* <p>{@code onTurnCompleteWithPostAction} is the one place order is not just fault-tolerance
|
||||
* but a functional requirement: {@code completion.resolveBeforePostAction} must resolve the
|
||||
* scrape before {@code sessions.onTurnCompleteWithPostAction}'s adapter housekeeping can erase
|
||||
* the pane's rendered output (see {@link CompletionResolver#resolveBeforePostAction}). That is
|
||||
* why this composition does not treat the pair symmetrically the way {@link #bothMustRun}
|
||||
* does for the other three callbacks: the mirror case (completion half throws, session half's
|
||||
* return value still observed) is not preserved here — once the completion half's failure
|
||||
* escapes, the session half's return value is discarded, matching how the {@link Injector}
|
||||
* already treats any throw from this callback as "the action did not start" (see the {@code
|
||||
* started} default at its call site, {@code Injector.java} ~line 616).
|
||||
*
|
||||
* <p>Package-private so {@code FleetdTurnListenerCompositionTest} can build this listener
|
||||
* directly from a real {@link CompletionResolver} and a fake {@link TurnListener} standing in
|
||||
* for {@code sessions}, without booting the rest of {@code main}.
|
||||
*/
|
||||
static TurnListener turnListener(CompletionResolver completion, TurnListener sessions) {
|
||||
return new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
bothMustRun(() -> completion.onTurnComplete(target), () -> sessions.onTurnComplete(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
return bothMustRunKeepingSecondResult(() -> completion.resolveBeforePostAction(target),
|
||||
() -> sessions.onTurnCompleteWithPostAction(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
||||
// fleetd #561: left unguarded on purpose, not because registration survives a
|
||||
// throw elsewhere. completion.onDelivered runs CompletionResolver.captureBaseline,
|
||||
// which already wraps its scrape read in its own catch (RuntimeException) and
|
||||
// fails open (baseline = null) — so this call does not realistically throw, and
|
||||
// there is nothing here for bothMustRun to protect.
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
bothMustRun(() -> completion.onTurnFailed(target), () -> sessions.onTurnFailed(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
bothMustRun(() -> completion.onTurnFailed(target, reason), () -> sessions.onTurnFailed(target));
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #561: run two listener-callback halves for one lifecycle event, guaranteeing the
|
||||
* SECOND always runs even when the FIRST throws. Whatever is thrown is rethrown once both
|
||||
* halves have been attempted — a second failure is attached to the first via {@link
|
||||
* Throwable#addSuppressed} rather than dropped. Never swallows.
|
||||
*/
|
||||
private static void bothMustRun(Runnable completionHalf, Runnable sessionsHalf) {
|
||||
Throwable failure = null;
|
||||
try {
|
||||
completionHalf.run();
|
||||
} catch (Throwable t) {
|
||||
failure = t;
|
||||
}
|
||||
try {
|
||||
sessionsHalf.run();
|
||||
} catch (Throwable t) {
|
||||
if (failure == null) {
|
||||
failure = t;
|
||||
} else {
|
||||
failure.addSuppressed(t);
|
||||
}
|
||||
}
|
||||
if (failure != null) {
|
||||
throwUnchecked(failure);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #561: like {@link #bothMustRun}, but for {@code onTurnCompleteWithPostAction}, whose
|
||||
* session half returns the value the {@link Injector} needs. The second (session) half's
|
||||
* result is what this method returns; if the first (completion) half throws, the second half
|
||||
* still runs and its result is still computed here, but the throw is rethrown afterward
|
||||
* regardless — so that result is discarded at the {@link Injector} call site exactly as it
|
||||
* already is today when this callback throws (see {@link #turnListener}'s javadoc).
|
||||
*/
|
||||
private static boolean bothMustRunKeepingSecondResult(Runnable completionHalf,
|
||||
BooleanSupplier sessionsHalf) {
|
||||
Throwable failure = null;
|
||||
try {
|
||||
completionHalf.run();
|
||||
} catch (Throwable t) {
|
||||
failure = t;
|
||||
}
|
||||
boolean result = false;
|
||||
try {
|
||||
result = sessionsHalf.getAsBoolean();
|
||||
} catch (Throwable t) {
|
||||
if (failure == null) {
|
||||
failure = t;
|
||||
} else {
|
||||
failure.addSuppressed(t);
|
||||
}
|
||||
}
|
||||
if (failure != null) {
|
||||
throwUnchecked(failure);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #561: rethrow a captured {@link Throwable} without a checked-exception wrapper. The
|
||||
* two callback halves above never declare a checked exception (both existing production
|
||||
* halves — {@code CompletionResolver} and {@code SessionManager} — only ever throw unchecked),
|
||||
* so this only ever actually rethrows a {@link RuntimeException} or {@link Error}; the generic
|
||||
* cast is the standard "sneaky throw" idiom, not a claim that a checked exception is expected.
|
||||
*/
|
||||
@SuppressWarnings("unchecked")
|
||||
private static <T extends Throwable> void throwUnchecked(Throwable t) throws T {
|
||||
throw (T) t;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #480: construct the {@link LeadRollover} executor only when {@code leadRollover:} is
|
||||
* present at startup — the same presence gate {@code leadHeartbeat:} uses just above this
|
||||
|
||||
@@ -20,6 +20,7 @@ import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
@@ -108,6 +109,7 @@ public final class FleetMcp {
|
||||
private final Metrics metrics; // CB-502: null → auth failures not counted
|
||||
private final CapacitySource capacity;
|
||||
private final HealthCoverageSource healthCoverage;
|
||||
private final LoopHealthSource loopHealth;
|
||||
private final QuarantineSource quarantine;
|
||||
/** fleetd #201 Unit 5: SEPARATE from {@link #quarantine} — see {@link OutageSource}'s doc. */
|
||||
private final OutageSource outage;
|
||||
@@ -136,6 +138,15 @@ public final class FleetMcp {
|
||||
/** Coverage is supplied by the health wiring, not inferred from a missing dependency. */
|
||||
public record HealthCoverageSource(Supplier<String> value) { }
|
||||
|
||||
/** Progress states for fleetd's singleton background loops, read by {@code fleet_list} and {@code /healthz}. */
|
||||
public record LoopHealthSource(Supplier<LoopWatchdog.State> statusPoller,
|
||||
Supplier<LoopWatchdog.State> sessionReaper) {
|
||||
/** Inert source for callers that do not wire the background loops. */
|
||||
public static LoopHealthSource none() {
|
||||
return new LoopHealthSource(() -> LoopWatchdog.State.STOPPED, () -> LoopWatchdog.State.STOPPED);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage B quarantine facts used by {@code fleet_profiles}: a profile → credential id
|
||||
* lookup, plus the shared {@link BackendQuarantine} to read remaining cooldowns off.
|
||||
@@ -328,11 +339,22 @@ public final class FleetMcp {
|
||||
* instead of throwing. See {@link #handover}.
|
||||
*/
|
||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, AuthorizationMode authorizationMode, Metrics metrics,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||
LeadSeatSource leadSeats, List<String> peers, LeadRollover leadRollover) {
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, AuthorizationMode authorizationMode, Metrics metrics,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||
LeadSeatSource leadSeats, List<String> peers, LeadRollover leadRollover) {
|
||||
this(messages, workers, sessions, identity, presence, primaryRegistry, callers, authorizationMode, metrics,
|
||||
capacity, healthCoverage, LoopHealthSource.none(), quarantine, leadChannel, outage, leadSeats,
|
||||
peers, leadRollover);
|
||||
}
|
||||
|
||||
public FleetMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, AuthorizationMode authorizationMode, Metrics metrics,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage, LoopHealthSource loopHealth,
|
||||
QuarantineSource quarantine, LeadChannel leadChannel, OutageSource outage,
|
||||
LeadSeatSource leadSeats, List<String> peers, LeadRollover leadRollover) {
|
||||
Objects.requireNonNull(callers, "callers");
|
||||
this.authorizationEnforced = Objects.requireNonNull(authorizationMode, "authorizationMode")
|
||||
== AuthorizationMode.ENFORCED;
|
||||
@@ -343,6 +365,7 @@ public final class FleetMcp {
|
||||
this.outage = Objects.requireNonNull(outage, "outage");
|
||||
this.leadSeats = Objects.requireNonNull(leadSeats, "leadSeats");
|
||||
this.healthCoverage = healthCoverage;
|
||||
this.loopHealth = Objects.requireNonNull(loopHealth, "loopHealth");
|
||||
this.leadRollover = leadRollover;
|
||||
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
||||
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
||||
@@ -474,7 +497,7 @@ public final class FleetMcp {
|
||||
(exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_list", Map.of()), null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine, outage,
|
||||
leadSeats, callers.leads(),
|
||||
callerTerminal(exchange),
|
||||
new CoordinationSource(leadChannel, peers),
|
||||
@@ -1546,25 +1569,42 @@ public final class FleetMcp {
|
||||
* @param selfTerm the calling pane's terminal id, or blank for a caller with no pane
|
||||
*/
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, null, CapacitySource.none(), new HealthCoverageSource(() -> "off"),
|
||||
QuarantineSource.none(), leads, selfTerm);
|
||||
LoopHealthSource.none(), QuarantineSource.none(), leads, selfTerm);
|
||||
}
|
||||
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, leads, selfTerm,
|
||||
CoordinationSource.none());
|
||||
}
|
||||
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, leads, selfTerm,
|
||||
LoopHealthSource loopHealth, QuarantineSource quarantine,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine, leads, selfTerm,
|
||||
CoordinationSource.none());
|
||||
}
|
||||
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
LoopHealthSource loopHealth, QuarantineSource quarantine,
|
||||
Map<String, String> leads, String selfTerm,
|
||||
CoordinationSource coordination) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, loopHealth, quarantine,
|
||||
OutageSource.none(), LeadSeatSource.none(), leads, selfTerm, coordination, false);
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, CoordinationSource.none());
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, CoordinationSource.none(), false);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1580,8 +1620,8 @@ public final class FleetMcp {
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, Map<String, String> leads, String selfTerm,
|
||||
CoordinationSource coordination) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, OutageSource.none(),
|
||||
LeadSeatSource.none(), leads, selfTerm, coordination);
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, OutageSource.none(),
|
||||
LeadSeatSource.none(), leads, selfTerm, coordination, false);
|
||||
}
|
||||
|
||||
/** As above, plus fleetd #201 Unit 5 cool-off facts (see {@link OutageSource}). */
|
||||
@@ -1589,8 +1629,8 @@ public final class FleetMcp {
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
Map<String, String> leads, String selfTerm, CoordinationSource coordination) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, coordination);
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
|
||||
LeadSeatSource.none(), leads, selfTerm, coordination, false);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1611,7 +1651,7 @@ public final class FleetMcp {
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||
CoordinationSource coordination) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, quarantine, outage,
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine, outage,
|
||||
leadSeats, leads, selfTerm, coordination, false);
|
||||
}
|
||||
|
||||
@@ -1631,8 +1671,18 @@ public final class FleetMcp {
|
||||
* an explicit {@code true}
|
||||
*/
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||
CoordinationSource coordination, boolean callerIsPrimary) {
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage, LoopHealthSource.none(), quarantine,
|
||||
outage, leadSeats, leads, selfTerm, coordination, callerIsPrimary);
|
||||
}
|
||||
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
LoopHealthSource loopHealth,
|
||||
QuarantineSource quarantine, OutageSource outage,
|
||||
LeadSeatSource leadSeats, Map<String, String> leads, String selfTerm,
|
||||
CoordinationSource coordination, boolean callerIsPrimary) {
|
||||
try {
|
||||
@@ -1656,6 +1706,9 @@ public final class FleetMcp {
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("leads", leadRows); result.put("members", out);
|
||||
result.put("healthCoverage", healthCoverage.value().get());
|
||||
result.put("loopHealth", Map.of(
|
||||
"statusPoller", loopHealth.statusPoller().get().name(),
|
||||
"sessionReaper", loopHealth.sessionReaper().get().name()));
|
||||
// fleetd #439: coordinator/coordinatorView is lead-to-lead coordination state and must
|
||||
// never reach a worker or an architect -- gate BEFORE assembling it, not after, so the
|
||||
// key is absent rather than present-and-empty.
|
||||
@@ -2135,9 +2188,10 @@ public final class FleetMcp {
|
||||
+ "cannot reliably re-identify: some backends (e.g. opencode) resolve it from the "
|
||||
+ "member's working directory, which only uniquely identifies a member when it "
|
||||
+ "was spawned into its own fleetd-provisioned worktree (worktree:true/<slug>); a "
|
||||
+ "member spawned without one shares its directory with others and never reports "
|
||||
+ "an id, however long it runs (fleetd #249). An empty 'members' "
|
||||
+ "means no members are spawned; it says nothing about peers. When capacity "
|
||||
+ "member spawned without one shares its directory with others and never reports "
|
||||
+ "an id, however long it runs (fleetd #249). An empty 'members' "
|
||||
+ "means no members are spawned; it says nothing about peers. 'loopHealth' reports "
|
||||
+ "the RUNNING, STALLED, or STOPPED state of statusPoller and sessionReaper. When capacity "
|
||||
+ "facts are configured, a 'capacity' row per profile reports 'free' — the "
|
||||
+ "slots a fresh fleet_spawn on that profile will actually be granted right "
|
||||
+ "now (max(0, maxLoad - live)), the same check the spawn gate itself runs. A "
|
||||
|
||||
@@ -1135,21 +1135,33 @@ public final class MessageService {
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
||||
// resumed turn can re-associate the async ticket with its new turnId via
|
||||
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
||||
// markAsyncQuestion silently returns null.
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
clearAsyncQuestion(turnId, false);
|
||||
// fleetd #575: this try used to open below, AFTER the Task lookup/registration and the
|
||||
// STALE_TURN early return that follows it — so that return was covered only by a
|
||||
// hand-rolled copy of the finally's own cleanup pair, not the finally itself. Widening the
|
||||
// try up to wrap the registration closes the gap structurally: every exit from here on,
|
||||
// STALE_TURN included, now runs through the one finally below exactly once, and the
|
||||
// duplicated pair is gone. This did not fix a live leak — see the ticket: neither
|
||||
// rendezvous.answerAsk nor clearAsyncQuestion(turnId, false) can throw, so nothing ever
|
||||
// actually left through the old gap uncovered — but #572 found this exact drift (one
|
||||
// finally asserted, an identical sibling not) on this same file, and the hand-rolled copy
|
||||
// was the wrong shape to keep regardless of whether it was ever exercised.
|
||||
try {
|
||||
// #282: mirror send()'s registration (:802) so a SECOND fleet_ask inside this same
|
||||
// resumed turn can re-associate the async ticket with its new turnId via
|
||||
// markAsyncQuestion — without this, that second ask has no Task to attach to, and
|
||||
// markAsyncQuestion silently returns null.
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
if (answerAskLapseRaceHookForTest != null) {
|
||||
// Test-only (fleetd #575): see the field's own javadoc.
|
||||
answerAskLapseRaceHookForTest.run();
|
||||
}
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
clearAsyncQuestion(turnId, false);
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
// #282: this waiter can resolve with a FRESH question rather than a terminal reply —
|
||||
@@ -1649,6 +1661,30 @@ public final class MessageService {
|
||||
this.askTimeoutRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/**
|
||||
* Null in production; test seam for fleetd #575 — invoked from {@link #answer}, right after this
|
||||
* call's own {@code Task} registration and right before its {@code rendezvous.answerAsk(turnId,
|
||||
* content)} call. A test installs this to complete the SAME turnId's ask directly via {@link
|
||||
* Rendezvous#answerAsk} from inside that exact window, deterministically reproducing what a
|
||||
* second, concurrent {@code answer()} call racing to unblock the same ask can otherwise only
|
||||
* win by timing luck: this call's own {@code askSession(turnId)} lookup at the top already saw
|
||||
* the ask as open, but by the time it reaches {@code rendezvous.answerAsk} here, the other call
|
||||
* already completed it (or the worker's own {@code ask()} teardown already closed it) — so this
|
||||
* call must see {@code false} and return {@link Outcome#STALE_TURN}, exactly the "lapsed between
|
||||
* the lookup and the unblock" case named at that call site. Proves the #575 fix (widening this
|
||||
* method's try so a single finally covers this exit) does not change that outcome and still
|
||||
* cleans this call's own {@code reply} up exactly once.
|
||||
*/
|
||||
private volatile Runnable answerAskLapseRaceHookForTest;
|
||||
|
||||
/**
|
||||
* Test-only (fleetd #575): install {@link #answerAskLapseRaceHookForTest}. Package-private so the
|
||||
* test, in the same package, can reach it without widening any production API.
|
||||
*/
|
||||
void setAnswerAskLapseRaceHookForTest(Runnable hook) {
|
||||
this.answerAskLapseRaceHookForTest = hook;
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
private boolean hasAsyncQuestion(String target) {
|
||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||
|
||||
@@ -94,6 +94,7 @@ public final class FleetApp {
|
||||
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
||||
private final FleetMcp.QuarantineSource quarantine;
|
||||
private final FleetMcp.OutageSource outage;
|
||||
private final FleetMcp.LoopHealthSource loopHealth;
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
@@ -155,7 +156,8 @@ public final class FleetApp {
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none(),
|
||||
FleetMcp.LoopHealthSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -169,8 +171,18 @@ public final class FleetApp {
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics, deliverable,
|
||||
memberCredentials, quarantine, outage, FleetMcp.LoopHealthSource.none());
|
||||
}
|
||||
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage,
|
||||
FleetMcp.LoopHealthSource loopHealth) {
|
||||
this.herdr = herdr;
|
||||
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
||||
this.workers = workers;
|
||||
@@ -183,6 +195,7 @@ public final class FleetApp {
|
||||
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
||||
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
||||
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
||||
this.loopHealth = loopHealth != null ? loopHealth : FleetMcp.LoopHealthSource.none();
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
@@ -291,15 +304,19 @@ public final class FleetApp {
|
||||
* spotted by comparing two numbers by eye.
|
||||
*/
|
||||
private void healthz(Context ctx) {
|
||||
HealthzResponse response = healthzResponse(herdr, memberHerdr, loopHealth);
|
||||
ctx.status(response.status()).json(response.body());
|
||||
}
|
||||
|
||||
record HealthzResponse(int status, Map<String, Object> body) { }
|
||||
|
||||
static HealthzResponse healthzResponse(HerdrClient herdr, HerdrClient memberHerdr,
|
||||
FleetMcp.LoopHealthSource loopHealth) {
|
||||
JsonNode pong;
|
||||
try {
|
||||
pong = herdr.call("ping");
|
||||
} catch (HerdrException e) {
|
||||
ctx.status(503).json(Map.of(
|
||||
"status", "degraded",
|
||||
"herdr", "unreachable",
|
||||
"detail", e.getMessage()));
|
||||
return;
|
||||
return degradedResponse("unreachable", e.getMessage(), loopHealth);
|
||||
}
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("status", "ok");
|
||||
@@ -311,11 +328,11 @@ public final class FleetApp {
|
||||
try {
|
||||
memberPong = memberHerdr.call("ping");
|
||||
} catch (HerdrException e) {
|
||||
ctx.status(503).json(Map.of(
|
||||
return new HealthzResponse(503, Map.of(
|
||||
"status", "degraded",
|
||||
"herdr", "member unreachable",
|
||||
"detail", e.getMessage()));
|
||||
return;
|
||||
"detail", e.getMessage(),
|
||||
"loopHealth", loopHealthView(loopHealth)));
|
||||
}
|
||||
int leadProtocol = pong.path("protocol").asInt();
|
||||
int memberProtocol = memberPong.path("protocol").asInt();
|
||||
@@ -326,7 +343,20 @@ public final class FleetApp {
|
||||
body.put("protocolMismatch", true);
|
||||
}
|
||||
}
|
||||
ctx.status(200).json(body);
|
||||
body.put("loopHealth", loopHealthView(loopHealth));
|
||||
return new HealthzResponse(200, body);
|
||||
}
|
||||
|
||||
private static Map<String, String> loopHealthView(FleetMcp.LoopHealthSource loopHealth) {
|
||||
return Map.of("statusPoller", loopHealth.statusPoller().get().name(),
|
||||
"sessionReaper", loopHealth.sessionReaper().get().name());
|
||||
}
|
||||
|
||||
private static HealthzResponse degradedResponse(String herdr, String detail,
|
||||
FleetMcp.LoopHealthSource loopHealth) {
|
||||
return new HealthzResponse(503, Map.of(
|
||||
"status", "degraded", "herdr", herdr, "detail", detail,
|
||||
"loopHealth", loopHealthView(loopHealth)));
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,286 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.inject.TurnListener;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #561: {@code Fleetd.turnListener} composes the completion resolver and the session
|
||||
* manager into one {@link TurnListener}. Each of the four callbacks below used to be two bare,
|
||||
* unguarded statements (completion first, then sessions) — a throw from the session half used to
|
||||
* skip nothing <em>after</em> it (there was nothing after it), but nothing enforced that the
|
||||
* completion half had to come first either, beyond call order in the source. {@code onDelivered}
|
||||
* had the identical shape and was fixed by #556 (moving its registration off the fan-out
|
||||
* entirely); these four callbacks cannot be fixed that way, so the fan-out itself is hardened
|
||||
* instead (see {@link Fleetd#turnListener} and {@code Fleetd.bothMustRun}/{@code
|
||||
* bothMustRunKeepingSecondResult}).
|
||||
*
|
||||
* <p>The invariant under test: <b>a throwing session-listener must not prevent the completion
|
||||
* resolver from being told the turn ended.</b> Each test below builds the exact production
|
||||
* composition ({@link Fleetd#turnListener}) from a real {@link CompletionResolver} and a fake
|
||||
* {@code sessions} half that throws, then asserts the completion half's effect (the captured
|
||||
* {@link Rendezvous} waiter resolving) happened anyway — never by inspecting call order directly.
|
||||
*/
|
||||
class FleetdTurnListenerCompositionTest {
|
||||
|
||||
/** Records which callbacks ran and can be told to throw from a chosen one. */
|
||||
private static final class RecordingSessions implements TurnListener {
|
||||
final Set<String> called = new LinkedHashSet<>();
|
||||
private final Set<String> throwing;
|
||||
|
||||
RecordingSessions(String... throwingMethods) {
|
||||
this.throwing = Set.of(throwingMethods);
|
||||
}
|
||||
|
||||
private void maybeThrow(String method) {
|
||||
called.add(method);
|
||||
if (throwing.contains(method)) {
|
||||
throw new IllegalStateException("boom: sessions." + method);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
maybeThrow("onTurnComplete");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
maybeThrow("onTurnCompleteWithPostAction");
|
||||
return true;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
maybeThrow("onTurnFailed");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
maybeThrow("onTurnFailedWithReason");
|
||||
}
|
||||
}
|
||||
|
||||
private static CompletionResolver newResolver(FakeHerdr herdr, Rendezvous rendezvous) {
|
||||
return new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* A resolver with a controllable clock, and the clock itself, so a test can register a turn
|
||||
* "delivered" at time 0 and then jump the clock past {@link CompletionResolver#MIN_TURN_NANOS}
|
||||
* before resolving it — otherwise {@code onTurnComplete}/{@code onTurnCompleteWithPostAction}
|
||||
* resolve within microseconds of registering in-test, well inside the fleetd#164 floor, and
|
||||
* get classified as a too-fast crash rather than a real completion. Mirrors the {@code
|
||||
* LongSupplier} clock-injection pattern fleetd#164's own tests use.
|
||||
*/
|
||||
private static CompletionResolver newResolverPastTheFloor(FakeHerdr herdr, Rendezvous rendezvous,
|
||||
AtomicLong clock) {
|
||||
LongSupplier nowNanos = clock::get;
|
||||
return new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), nowNanos);
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnCompleteResolvesTheWaiterEvenWhenTheSessionHalfThrows() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ real answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
AtomicLong clock = new AtomicLong(0);
|
||||
CompletionResolver completion = newResolverPastTheFloor(herdr, rendezvous, clock);
|
||||
var waiter = rendezvous.open("term_a");
|
||||
completion.register("term_a", new TurnToken("term_a", waiter)); // delivered at clock=0
|
||||
clock.set(CompletionResolver.MIN_TURN_NANOS + 1_000_000); // past the too-fast floor
|
||||
|
||||
RecordingSessions sessions = new RecordingSessions("onTurnComplete");
|
||||
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||
|
||||
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||
() -> composed.onTurnComplete("term_a"),
|
||||
"the session half's throw must still escape the composed listener");
|
||||
assertEquals("boom: sessions.onTurnComplete", thrown.getMessage());
|
||||
assertTrue(sessions.called.contains("onTurnComplete"), "the session half must have run");
|
||||
|
||||
// onTurnComplete resolves off-thread (a virtual thread) — wait on the waiter itself,
|
||||
// exactly like MessageServiceTest.completionFallbackIsNeverQueued does.
|
||||
Rendezvous.Resolution resolution = waiter.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, resolution.kind(),
|
||||
"the completion resolver must still have resolved the send, despite the session "
|
||||
+ "half throwing");
|
||||
assertEquals("real answer", resolution.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedResolvesTheWaiterAsFailedEvenWhenTheSessionHalfThrows() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ crash context\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||
var waiter = rendezvous.open("term_b");
|
||||
completion.register("term_b", new TurnToken("term_b", waiter));
|
||||
|
||||
RecordingSessions sessions = new RecordingSessions("onTurnFailed");
|
||||
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||
|
||||
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||
() -> composed.onTurnFailed("term_b"),
|
||||
"the session half's throw must still escape the composed listener");
|
||||
assertEquals("boom: sessions.onTurnFailed", thrown.getMessage());
|
||||
assertTrue(sessions.called.contains("onTurnFailed"), "the session half must have run");
|
||||
|
||||
Rendezvous.Resolution resolution = waiter.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(Rendezvous.Kind.FAILED, resolution.kind(),
|
||||
"the completion resolver must still have failed the send, despite the session "
|
||||
+ "half throwing");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedWithReasonResolvesTheWaiterEvenWhenTheSessionHalfThrows() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||
var waiter = rendezvous.open("term_c");
|
||||
completion.register("term_c", new TurnToken("term_c", waiter));
|
||||
|
||||
// fleetd #561: the composed listener calls sessions.onTurnFailed(target) — the ONE-arg
|
||||
// overload — for this reason-carrying event too (matching the pre-existing production
|
||||
// behaviour: SessionManager never overrides the two-arg overload either), so the
|
||||
// throwing key here is "onTurnFailed", not a distinct "...WithReason" one.
|
||||
RecordingSessions sessions = new RecordingSessions("onTurnFailed");
|
||||
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||
|
||||
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||
() -> composed.onTurnFailed("term_c", "worker unreachable"),
|
||||
"the session half's throw must still escape the composed listener");
|
||||
assertEquals("boom: sessions.onTurnFailed", thrown.getMessage());
|
||||
assertTrue(sessions.called.contains("onTurnFailed"), "the session half must have run");
|
||||
|
||||
Rendezvous.Resolution resolution = waiter.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(Rendezvous.Kind.FAILED, resolution.kind(),
|
||||
"the completion resolver must still have failed the send, despite the session "
|
||||
+ "half throwing");
|
||||
assertEquals("worker unreachable", resolution.text(),
|
||||
"the explicit reason must still reach the resolved send");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnCompleteWithPostActionResolvesTheWaiterEvenWhenTheSessionHalfThrows() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ answer before reset\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
AtomicLong clock = new AtomicLong(0);
|
||||
CompletionResolver completion = newResolverPastTheFloor(herdr, rendezvous, clock);
|
||||
var waiter = rendezvous.open("term_d");
|
||||
completion.register("term_d", new TurnToken("term_d", waiter)); // delivered at clock=0
|
||||
clock.set(CompletionResolver.MIN_TURN_NANOS + 1_000_000); // past the too-fast floor
|
||||
|
||||
RecordingSessions sessions = new RecordingSessions("onTurnCompleteWithPostAction");
|
||||
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||
|
||||
IllegalStateException thrown = assertThrows(IllegalStateException.class,
|
||||
() -> composed.onTurnCompleteWithPostAction("term_d"),
|
||||
"the session half's throw must still escape the composed listener");
|
||||
assertEquals("boom: sessions.onTurnCompleteWithPostAction", thrown.getMessage());
|
||||
assertTrue(sessions.called.contains("onTurnCompleteWithPostAction"), "the session half must have run");
|
||||
|
||||
// resolveBeforePostAction is synchronous by design (it must run before the context reset
|
||||
// can erase the pane) — the waiter is already resolved by the time the throw propagates.
|
||||
assertTrue(waiter.isDone(), "resolveBeforePostAction is synchronous — the send must "
|
||||
+ "already be resolved once the composed call returns (by throwing)");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("answer before reset", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
/**
|
||||
* The mirror case (comment 17041's acceptance item 2): the completion half throws, and the
|
||||
* session half still ran. The production {@link CompletionResolver} is deliberately defensive
|
||||
* (its scrape reads are wrapped in {@code catch (RuntimeException)}, by the same fail-open
|
||||
* design {@link CompletionResolver#captureBaseline} documents) so it essentially never throws
|
||||
* synchronously in normal operation — forcing it to do so needs a genuinely-real trigger, not
|
||||
* a fabricated one. {@code ConcurrentHashMap.get(null)} is that trigger: passing a {@code null}
|
||||
* target makes {@code onTurnComplete}'s {@code inFlight.get(target)} throw a
|
||||
* {@link NullPointerException} before it ever starts its resolving thread — a real code path,
|
||||
* not a contrived one.
|
||||
*/
|
||||
@Test
|
||||
void sessionHalfStillRunsWhenTheCompletionHalfThrowsSynchronously() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||
|
||||
RecordingSessions sessions = new RecordingSessions(); // throws from nothing
|
||||
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||
|
||||
assertThrows(NullPointerException.class, () -> composed.onTurnComplete(null),
|
||||
"ConcurrentHashMap.get(null) inside CompletionResolver.onTurnComplete must still "
|
||||
+ "escape the composed listener");
|
||||
assertTrue(sessions.called.contains("onTurnComplete"),
|
||||
"the session half must still have run even though the completion half threw first");
|
||||
}
|
||||
|
||||
/**
|
||||
* The mirror case above only exercises {@code onTurnComplete}/{@code bothMustRun}. {@code
|
||||
* onTurnCompleteWithPostAction} is composed through the OTHER helper,
|
||||
* {@code bothMustRunKeepingSecondResult}, and nothing previously asserted that its session half
|
||||
* still runs when its completion half throws — that helper could be reverted to the pre-#561
|
||||
* broken shape (run the session half only if the completion half did not throw) and the suite
|
||||
* would still stay green. Uses the same real, non-fabricated trigger as the test above: a
|
||||
* {@code null} target makes {@code resolveBeforePostAction}'s {@code inFlight.get(target)}
|
||||
* throw a {@link NullPointerException} before {@code resolve} is ever entered.
|
||||
*/
|
||||
@Test
|
||||
void sessionHalfStillRunsWhenTheCompletionHalfThrowsSynchronouslyForPostAction() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||
|
||||
RecordingSessions sessions = new RecordingSessions(); // throws from nothing
|
||||
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||
|
||||
assertThrows(NullPointerException.class, () -> composed.onTurnCompleteWithPostAction(null),
|
||||
"ConcurrentHashMap.get(null) inside CompletionResolver.resolveBeforePostAction must "
|
||||
+ "still escape the composed listener");
|
||||
assertTrue(sessions.called.contains("onTurnCompleteWithPostAction"),
|
||||
"the session half must still have run even though the completion half threw first");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bothMustRun} must not just run both halves — it must not DROP a second failure when
|
||||
* both halves throw. Forces the completion half to throw (the same real {@code
|
||||
* inFlight.get(null)} NullPointerException trigger used above) while the session half throws a
|
||||
* distinct {@link IllegalStateException}, and asserts the completion half's throwable is what
|
||||
* escapes while the session half's throwable survives as a suppressed exception rather than
|
||||
* being silently discarded.
|
||||
*/
|
||||
@Test
|
||||
void bothFailuresEscapeWhenBothHalvesThrowDistinctExceptions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = newResolver(herdr, rendezvous);
|
||||
|
||||
RecordingSessions sessions = new RecordingSessions("onTurnComplete");
|
||||
TurnListener composed = Fleetd.turnListener(completion, sessions);
|
||||
|
||||
NullPointerException thrown = assertThrows(NullPointerException.class,
|
||||
() -> composed.onTurnComplete(null),
|
||||
"the completion half's throw (NPE from inFlight.get(null)) must be what escapes");
|
||||
assertTrue(sessions.called.contains("onTurnComplete"), "the session half must still have run");
|
||||
assertEquals(1, thrown.getSuppressed().length,
|
||||
"the session half's distinct failure must be recorded as suppressed, not dropped");
|
||||
assertEquals(IllegalStateException.class, thrown.getSuppressed()[0].getClass());
|
||||
assertEquals("boom: sessions.onTurnComplete", thrown.getSuppressed()[0].getMessage());
|
||||
}
|
||||
}
|
||||
@@ -8,6 +8,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
@@ -1053,6 +1054,36 @@ class FleetMcpTest {
|
||||
assertFalse(out.contains("quarantinedForSeconds"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void loopHealthReportsStalledStatusPoller() {
|
||||
String out = loopHealth(LoopWatchdog.State.STALLED, LoopWatchdog.State.RUNNING);
|
||||
assertTrue(out.contains("\"statusPoller\":\"STALLED\""),
|
||||
"fleet_list must report a stalled StatusPoller: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void loopHealthReportsStoppedSessionReaperAsStopped() {
|
||||
String out = loopHealth(LoopWatchdog.State.RUNNING, LoopWatchdog.State.STOPPED);
|
||||
assertTrue(out.contains("\"sessionReaper\":\"STOPPED\""),
|
||||
"fleet_list must report a deliberately stopped SessionReaper as STOPPED, not an alarm: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void loopHealthReportsRunningStatusPoller() {
|
||||
String out = loopHealth(LoopWatchdog.State.RUNNING, LoopWatchdog.State.STOPPED);
|
||||
assertTrue(out.contains("\"statusPoller\":\"RUNNING\""),
|
||||
"fleet_list must report a running StatusPoller: " + out);
|
||||
}
|
||||
|
||||
private static String loopHealth(LoopWatchdog.State statusPoller, LoopWatchdog.State sessionReaper) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
return textOf(FleetMcp.listFleet(workerService(herdr, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
new SessionManager(workerService(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"))), null,
|
||||
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
new FleetMcp.LoopHealthSource(() -> statusPoller, () -> sessionReaper),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), ""));
|
||||
}
|
||||
|
||||
@Test
|
||||
void capacityIncludesConfiguredProfileWithoutMembers() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
|
||||
@@ -19,6 +19,7 @@ import java.util.concurrent.atomic.AtomicLong;
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertInstanceOf;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
@@ -207,6 +208,31 @@ class LeadMailboxTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link LeadMailbox#inspect} opens a third channel after the mailbox's consume and publish
|
||||
* channels. Limit this connection to three channels, then require a replacement channel after
|
||||
* the successful inspect. If inspect leaves its probe open, the broker refuses that replacement.
|
||||
*/
|
||||
@Test
|
||||
void inspectClosesItsSuccessfulProbeChannel() throws Exception {
|
||||
var factory = LeadMailbox.connectionFactory(uri());
|
||||
factory.setRequestedChannelMax(3);
|
||||
Connection connection = factory.newConnection();
|
||||
try (LeadMailbox mailbox = new LeadMailbox(connection, coordId("lead-inspect-probe-close"))) {
|
||||
LeadChannel.MailboxState state = mailbox.inspect(mailbox.selfCoordId());
|
||||
assertTrue(state.exists(), "the owned mailbox must be found before checking the probe channel");
|
||||
|
||||
Channel replacement = connection.createChannel();
|
||||
assertNotNull(replacement,
|
||||
"inspect must close its successful probe channel; the replacement channel was null");
|
||||
try {
|
||||
assertTrue(replacement.isOpen(), "the replacement channel must be open after inspect returns");
|
||||
} finally {
|
||||
replacement.close();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #440: {@code heldDurable()} must be derived from what {@link LeadMailbox#own} actually
|
||||
* did against the real broker — a durable queue declare plus a manual-ack consumer — not a
|
||||
|
||||
@@ -22,6 +22,7 @@ import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
@@ -505,6 +506,52 @@ class MessageServiceTest {
|
||||
"an answer to a turn that never existed (or already lapsed) is stale, not a hang");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #575. {@code answer()}'s STALE_TURN return for "the ask lapsed between the lookup and
|
||||
* the unblock" ({@code rendezvous.answerAsk(turnId, content)} returning {@code false} even though
|
||||
* this call's own {@code rendezvous.askSession(turnId)} check at the top saw the ask as open) used
|
||||
* to be covered only by a hand-rolled copy of the cleanup pair its own finally already runs, not
|
||||
* the finally itself — a structural gap, closed by widening the try up to cover the registration
|
||||
* above it. This test pins the exact race with {@link
|
||||
* MessageService#setAnswerAskLapseRaceHookForTest}, which fires right before this call's own
|
||||
* {@code rendezvous.answerAsk} call and completes the same turnId's ask directly — reproducing
|
||||
* what a second, concurrent {@code answer()} winning that race could otherwise only do by timing
|
||||
* luck. Proves the fix changed no behaviour on this path: it still returns {@code STALE_TURN},
|
||||
* and this call's own forward waiter is still closed exactly once (not left open, and not closed
|
||||
* twice — there is now only one cleanup site left to run).
|
||||
*/
|
||||
@Test
|
||||
void answerLosingTheRaceToAnAlreadyAnsweredAskStillReturnsStaleTurnAndCleansUpOnce() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
injectDelivery();
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.TaskView asking = awaitTicketPhase(ticket, MessageService.Phase.ASKING);
|
||||
String turnId = asking.turnId();
|
||||
assertNotNull(turnId, "an ASKING view carries the turnId to answer on");
|
||||
|
||||
messages.setAnswerAskLapseRaceHookForTest(() -> rendezvous.answerAsk(turnId, "raced in first"));
|
||||
try {
|
||||
MessageService.Reply r = messages.answer(turnId, "too late", 500);
|
||||
assertEquals(MessageService.Outcome.STALE_TURN, r.outcome(),
|
||||
"an ask already answered by the race must be seen as lapsed, not double-delivered");
|
||||
} finally {
|
||||
messages.setAnswerAskLapseRaceHookForTest(null);
|
||||
}
|
||||
|
||||
// Cleanup ran exactly once: the forward waiter THIS call opened is closed, not leaked.
|
||||
assertNull(rendezvous.currentWaiter(T),
|
||||
"the forward waiter this answer() call opened must be closed after a STALE_TURN return");
|
||||
|
||||
// The worker's own ask() call, unblocked by the hook's direct answerAsk, still completes
|
||||
// normally — the race this test simulates does not strand it.
|
||||
MessageService.AskResult a = ask.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a.outcome());
|
||||
assertEquals("raced in first", a.answer());
|
||||
}
|
||||
|
||||
// --- timeout, answer, poll, and lock-contention edges ----------------------------------
|
||||
|
||||
@Test
|
||||
@@ -579,6 +626,183 @@ class MessageServiceTest {
|
||||
assertEquals("config.yaml", a.answer());
|
||||
}
|
||||
|
||||
// --- fleetd #572: answer() must release the session lock on EVERY exit, not just return it -----
|
||||
//
|
||||
// answer()'s session lock is released in an outer `finally` (MessageService.java:1217-1219) that
|
||||
// wraps its whole body. Removing that one line survives the entire suite: coverage is not the
|
||||
// gap, assertion is — every existing answer() test checks the RETURN VALUE, never that the lock
|
||||
// it took is reacquirable afterward. If it is not, the session is wedged forever with no
|
||||
// exception and no log line. Each test below drives answer() through one of its four exits
|
||||
// (normal reply, TIMED_OUT_WORKING, ExecutionException, InterruptedException) and then proves
|
||||
// reacquisition the only way that actually proves it: a bounded follow-up `send` on the SAME
|
||||
// session must not come back BUSY. A `send` only reports BUSY when `tryLock` itself timed out
|
||||
// (MessageService.java:926-928) — every other early return in `send` still passes through its own
|
||||
// lock-acquired `try`, so a non-BUSY probe result is specifically evidence the lock was free.
|
||||
|
||||
/** The REPLIED exit (the happy path) — the resumed worker's real {@code fleet_reply} arrives. */
|
||||
@Test
|
||||
void answerReleasesTheSessionLockAfterANormalReply() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||
ask.get(5, TimeUnit.SECONDS); // worker resumed with the answer
|
||||
|
||||
awaitWaiting(); // the answering call has (re)opened its own forward waiter
|
||||
assertTrue(rendezvous.resolve(T, "done"), "the worker's final reply resolves the answering send");
|
||||
MessageService.Reply done = answer.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, done.outcome());
|
||||
|
||||
MessageService.Reply probe = messages.send(T, "probe after normal reply", 300);
|
||||
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||
"the session lock must be released after a normal REPLIED answer(), or this bounded "
|
||||
+ "follow-up send would come back BUSY instead of timing out on its own work");
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code TIMED_OUT_WORKING} exit. Same setup as {@link #answerTimesOutWhenTheResumedWorkerNeverReplies}
|
||||
* (which only checks {@code answer()}'s return value — exactly the assertion the fleetd #572
|
||||
* mutation survives), plus the reacquisition proof that test does not make.
|
||||
*
|
||||
* <p>{@code answer()} MUST run on its own thread here, not the test's main thread: {@code lock}
|
||||
* is a {@link java.util.concurrent.locks.ReentrantLock}, so a probe {@code send} issued by the
|
||||
* SAME thread that (under the mutation) still "holds" it would reenter for free and report a
|
||||
* false pass — reentrancy, not release. Measured while writing this test: with {@code answer()}
|
||||
* called inline, this test stayed green under the mutation while its three siblings correctly
|
||||
* went red.
|
||||
*/
|
||||
@Test
|
||||
void answerReleasesTheSessionLockAfterATimedOutWorkingReturn() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
|
||||
// The primary answers, unblocking the worker; the worker never sends its follow-up
|
||||
// fleet_reply, so the answering call rides out its short window as still-working.
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 200));
|
||||
MessageService.Reply answered = answer.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answered.outcome(),
|
||||
"an answered worker that never replies times out as still working");
|
||||
ask.get(5, TimeUnit.SECONDS); // drain: the worker resumed with the answer
|
||||
|
||||
MessageService.Reply probe = messages.send(T, "probe after timed-out-working", 300);
|
||||
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||
"the session lock must be released after a TIMED_OUT_WORKING answer(), or this bounded "
|
||||
+ "follow-up send would come back BUSY instead of timing out on its own work");
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code ExecutionException} exit. Forces it directly on answer()'s own reopened forward
|
||||
* waiter — {@code rendezvous.currentWaiter(T)} is the exact {@code CompletableFuture} its
|
||||
* {@code reply.get(...)} is blocked on — rather than trying to make a real worker fail, since the
|
||||
* failure mode under test is in answer()'s own wait, not in how it got triggered.
|
||||
*/
|
||||
@Test
|
||||
void answerReleasesTheSessionLockWhenTheReplyFutureFailsExceptionally() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
String turnId = q.turnId();
|
||||
|
||||
java.util.concurrent.atomic.AtomicReference<Throwable> caught =
|
||||
new java.util.concurrent.atomic.AtomicReference<>();
|
||||
Thread answerer = new Thread(() -> {
|
||||
try {
|
||||
messages.answer(turnId, "config.yaml", 5000);
|
||||
caught.set(new AssertionError("expected answer() to throw"));
|
||||
} catch (Throwable t) {
|
||||
caught.set(t);
|
||||
}
|
||||
});
|
||||
answerer.start();
|
||||
awaitWaiting(); // answer() re-opened its forward waiter and is about to block on it
|
||||
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(T);
|
||||
assertNotNull(waiter, "answer() should have registered a forward waiter for the worker session");
|
||||
waiter.completeExceptionally(new RuntimeException("boom"));
|
||||
|
||||
answerer.join(5000);
|
||||
assertFalse(answerer.isAlive(), "answer() should have left by throwing once its reply future failed");
|
||||
Throwable thrown = caught.get();
|
||||
assertNotNull(thrown, "answer() should have thrown");
|
||||
assertTrue(thrown instanceof RuntimeException, "a RuntimeException cause is rethrown as-is: " + thrown);
|
||||
assertEquals("boom", thrown.getMessage());
|
||||
|
||||
ask.get(5, TimeUnit.SECONDS); // drain: the worker resumed with the answer
|
||||
|
||||
MessageService.Reply probe = messages.send(T, "probe after exceptional failure", 300);
|
||||
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||
"the session lock must be released when answer()'s reply future fails exceptionally, "
|
||||
+ "or this bounded follow-up send would come back BUSY instead of timing out on its own work");
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code InterruptedException} exit. Same shape as the {@code ExecutionException} case above,
|
||||
* but the thread blocked in {@code reply.get(...)} is interrupted instead of the future failing.
|
||||
*/
|
||||
@Test
|
||||
void answerReleasesTheSessionLockWhenInterrupted() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
String turnId = q.turnId();
|
||||
|
||||
java.util.concurrent.atomic.AtomicReference<Throwable> caught =
|
||||
new java.util.concurrent.atomic.AtomicReference<>();
|
||||
Thread answerer = new Thread(() -> {
|
||||
try {
|
||||
messages.answer(turnId, "config.yaml", 5000);
|
||||
caught.set(new AssertionError("expected answer() to throw"));
|
||||
} catch (Throwable t) {
|
||||
caught.set(t);
|
||||
}
|
||||
});
|
||||
answerer.start();
|
||||
awaitWaiting(); // answer() re-opened its forward waiter and is about to block on it
|
||||
|
||||
answerer.interrupt();
|
||||
answerer.join(5000);
|
||||
assertFalse(answerer.isAlive(), "answer() should have left by throwing once its thread was interrupted");
|
||||
Throwable thrown = caught.get();
|
||||
assertNotNull(thrown, "answer() should have thrown");
|
||||
assertTrue(thrown instanceof IllegalStateException,
|
||||
"an interrupted wait is wrapped in IllegalStateException: " + thrown);
|
||||
|
||||
ask.get(5, TimeUnit.SECONDS); // drain: the worker resumed with the answer
|
||||
|
||||
MessageService.Reply probe = messages.send(T, "probe after interruption", 300);
|
||||
assertNotEquals(MessageService.Outcome.BUSY, probe.outcome(),
|
||||
"the session lock must be released when answer()'s wait is interrupted, or this bounded "
|
||||
+ "follow-up send would come back BUSY instead of timing out on its own work");
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollReturnsNullForAnUnknownTicket() {
|
||||
assertNull(messages.poll("task-999999"), "a ticket that was never minted is unknown");
|
||||
|
||||
@@ -8,6 +8,7 @@ import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.LoopWatchdog;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
@@ -165,6 +166,29 @@ class FleetAppTest {
|
||||
assertEquals("degraded", mapper.readTree(res.body()).get("status").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzKeepsOkStatusAndReportsLoopHealthInItsBody() {
|
||||
FleetMcp.LoopHealthSource loops = new FleetMcp.LoopHealthSource(
|
||||
() -> LoopWatchdog.State.RUNNING, () -> LoopWatchdog.State.STOPPED);
|
||||
|
||||
FleetApp.HealthzResponse ok = FleetApp.healthzResponse(new FakeHerdr(), new FakeHerdr(), loops);
|
||||
assertEquals(200, ok.status(), "a healthy herdr must keep /healthz at 200 regardless of loop states");
|
||||
assertEquals(Map.of("statusPoller", "RUNNING", "sessionReaper", "STOPPED"), ok.body().get("loopHealth"),
|
||||
"the /healthz body must report each loop state without making STOPPED an alarm");
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzKeepsDegradedStatusAndReportsLoopHealthInItsBody() {
|
||||
FleetMcp.LoopHealthSource loops = new FleetMcp.LoopHealthSource(
|
||||
() -> LoopWatchdog.State.RUNNING, () -> LoopWatchdog.State.STOPPED);
|
||||
FleetApp.HealthzResponse degraded = FleetApp.healthzResponse(new FakeHerdr().healthy(false),
|
||||
new FakeHerdr(), loops);
|
||||
assertEquals(503, degraded.status(),
|
||||
"an unreachable herdr must keep /healthz at 503 regardless of loop states");
|
||||
assertEquals(Map.of("statusPoller", "RUNNING", "sessionReaper", "STOPPED"),
|
||||
degraded.body().get("loopHealth"), "the degraded /healthz body must retain loop states");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sessionsMapsWorkspaceList() throws Exception {
|
||||
int port = startHealthy();
|
||||
|
||||
Reference in New Issue
Block a user