Compare commits
26 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 776743cbe2 | |||
| 0f08b93659 | |||
| 432c1d92d1 | |||
| 60b7e67b42 | |||
| 3fbd43fe3f | |||
| 32ebf065ac | |||
| 1178b3f684 | |||
| 2a434ced2f | |||
| cc919aa2b6 | |||
| 6417b0edd9 | |||
| 39c7ce76f3 | |||
| b40f477210 | |||
| 5c56cb347f | |||
| e694deace3 | |||
| 748367b7d6 | |||
| dcf5fb3be3 | |||
| f9d2ee2a2b | |||
| 866c7f2e9a | |||
| d9168de43e | |||
| c3fa1136d4 | |||
| cabcd87b66 | |||
| bf0e09b1a2 | |||
| e1eb50ce65 | |||
| 049e7d9d54 | |||
| ff3b49cd1e | |||
| 445a45f6e1 |
@@ -169,6 +169,12 @@ public final class Fleetd {
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// fleetd #175: the daemon's real ExhaustionSink can only be built once `sessions` exists
|
||||
// (below), but `sessions` needs `workers`, which needs the adapters built right here — a
|
||||
// genuine cycle. Break it exactly like liveCountRef below: a forwarding sink built now,
|
||||
// pointed at the real one once it exists.
|
||||
AtomicReference<ExhaustionSink> exhaustionSinkRef = new AtomicReference<>(ExhaustionSink.none());
|
||||
ExhaustionSink forwardingExhaustionSink = (target, reason) -> exhaustionSinkRef.get().onExhausted(target, reason);
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
@@ -184,7 +190,7 @@ public final class Fleetd {
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials(), config::get));
|
||||
() -> config.get().memberCredentials(), config::get, forwardingExhaustionSink));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
@@ -334,6 +340,11 @@ public final class Fleetd {
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
//
|
||||
// fleetd #175: this sink is now also the quarantine target for OpenCodeLauncher's
|
||||
// model-mismatch check — it needs nothing profile-specific from the caller beyond `target`
|
||||
// (a herdr terminal id) and `reason`, so reusing it here is exactly "the existing
|
||||
// ExhaustionSink path", not a new mechanism.
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
@@ -342,10 +353,12 @@ public final class Fleetd {
|
||||
.ifPresent(profile -> {
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}' classified "
|
||||
+ "BACKEND_EXHAUSTED): {}", credentialId,
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}'): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
});
|
||||
// fleetd #175: point the forwarding sink handed to OpenCodeLauncher above at the real one,
|
||||
// now that `sessions` exists to resolve target -> session -> profile.
|
||||
exhaustionSinkRef.set(exhaustionSink);
|
||||
AgentControl agents = router.memberAgents();
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns, exhaustionSink);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
|
||||
@@ -5,17 +5,70 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
/** A slot held before a member process starts. */
|
||||
record SlotReservation(String slot, String profile) { }
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return role; // no registry configured — nothing to bind against, so the request stands
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
// no registry configured — nothing to validate against, so nothing is refused
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
/**
|
||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||
*
|
||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||
* normal operation this fallback should not happen once a reservation has been bound.
|
||||
* It remains the honest answer if a caller has no reservation, or if binding a
|
||||
* reservation unexpectedly fails.
|
||||
*/
|
||||
MemberRole acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
|
||||
/**
|
||||
* Refuse an acquire before anything spawns when {@code role} requires a live slot binding and
|
||||
* no configured slot carries {@code profile} (CB-619 / fleetd #123). A no-op for a role with
|
||||
* no slot-binding semantics.
|
||||
*
|
||||
* @throws IllegalArgumentException naming the role, the profile, and the pools that do carry it
|
||||
*/
|
||||
void requireSlotFor(MemberRole role, String profile);
|
||||
|
||||
/** Reserve a matching slot before launch, or refuse before a charter can be delivered. */
|
||||
SlotReservation reserve(MemberRole role, String profile);
|
||||
|
||||
/** Convert a reservation into a live terminal binding. */
|
||||
boolean bind(SlotReservation reservation, String terminal);
|
||||
|
||||
/** Return an unbound reservation after a failed launch. */
|
||||
void release(SlotReservation reservation);
|
||||
}
|
||||
|
||||
@@ -8,6 +8,7 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
@@ -55,8 +56,10 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
@@ -166,7 +169,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
if (terminalToSlot.containsValue(slot) || reservedSlots.contains(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
@@ -206,19 +209,116 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*
|
||||
* <p>CB-619 / fleetd #123: the return value is the role this session actually holds, and the
|
||||
* caller is required to record THAT — never the requested {@code role} — on the session. Before
|
||||
* this fix the caller kept the requested role regardless of whether the bind below succeeded, so
|
||||
* a demoted session's {@code GET /members} row still said {@code "architect"} while
|
||||
* {@code fleet_whoami} (which reads the live binding, not the request) correctly said
|
||||
* {@code "worker"} — three sources of truth that disagreed about one live member, silently.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
return role;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
return MemberRole.ARCHITECT;
|
||||
}
|
||||
}
|
||||
// fleetd #123: at least WARN — a role downgrade that the roster must now also reflect is
|
||||
// not routine bookkeeping. requireSlotFor already refuses the config-gap case (no slot at
|
||||
// all carries this profile) before a process ever spawns; reaching here means the config DID
|
||||
// carry a matching slot but every one of them was already bound to a different terminal — a
|
||||
// race this pre-spawn check cannot close on its own (see requireSlotFor's javadoc).
|
||||
log.warn("member slot: no free architect slot for profile={} terminal={}; holding the session "
|
||||
+ "as {} instead of the architect it asked for — every configured slot for this "
|
||||
+ "profile is already bound to a different terminal", profile, terminal,
|
||||
MemberRole.DEV.wireName());
|
||||
return MemberRole.DEV;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||
* documented operator override, not a defect.
|
||||
*
|
||||
* <p>This closes the config-gap case, not the live-capacity case: a profile that DOES carry a
|
||||
* slot can still lose the race to a concurrent spawn between this check and the actual
|
||||
* {@link #bind}, which is why {@link #acquired} must still answer honestly even after this
|
||||
* check has passed.
|
||||
*/
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return;
|
||||
}
|
||||
boolean hasSlot = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.anyMatch(e -> Objects.equals(profile, e.profile()));
|
||||
if (hasSlot) {
|
||||
return;
|
||||
}
|
||||
List<String> pools = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.map(Entry::profile)
|
||||
.distinct()
|
||||
.toList();
|
||||
throw new IllegalArgumentException(
|
||||
"no " + role.wireName() + " slot for profile '" + profile + "' — an architect's "
|
||||
+ "identity IS the slot it is bound to, so there is nothing to bind this "
|
||||
+ "session's identity to. fleet." + role.configKey() + " carries profiles: "
|
||||
+ (pools.isEmpty() ? "(none configured)" : String.join(", ", pools))
|
||||
+ "; add profile '" + profile + "' there, or spawn " + role.wireName()
|
||||
+ " on one of those profiles instead");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if ((profile == null || profile.isBlank() || Objects.equals(profile, entry.profile()))
|
||||
&& !terminalToSlot.containsValue(entry.key()) && reservedSlots.add(entry.key())) {
|
||||
return new SlotReservation(entry.key(), entry.profile());
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("no free architect slot for profile '" + profile
|
||||
+ "' — every matching slot is already bound or reserved");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
if (reservation == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!reservedSlots.remove(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
if (!isSlot(reservation.slot()) || terminalToSlot.containsKey(terminal)
|
||||
|| terminalToSlot.containsValue(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
terminalToSlot.put(terminal, reservation.slot());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
if (reservation != null) {
|
||||
synchronized (terminalToSlot) {
|
||||
reservedSlots.remove(reservation.slot());
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
|
||||
@@ -0,0 +1,35 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured backend-error pattern (fleetd#201 / #227): how
|
||||
* {@link CompletionResolver} recognizes a backend that failed a turn outright (a credential
|
||||
* outage, a provider 5xx) from whatever ends up on the member's pane, apart from a genuine
|
||||
* completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its failure differently, so a hardcoded sentence would only ever match one of them. A
|
||||
* target with no configured pattern is not "off" the way {@link ExhaustedPatternLookup#none()}
|
||||
* is — {@link CompletionResolver} falls back to its narrow {@code (?i)\bAPI Error\s*:}
|
||||
* compatibility pattern instead, so classification still happens, just without a profile-specific
|
||||
* match. {@link #legacy()} is the explicit stand-in existing (pre-fleetd#201) callers pass to get
|
||||
* exactly that fallback-only behavior until a caller wires a real, profile-driven lookup
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Legacy lookup — no profile has a configured pattern, so {@link CompletionResolver} classifies
|
||||
* every target using only its built-in {@code (?i)\bAPI Error\s*:} compatibility pattern. The
|
||||
* explicit stand-in existing constructors pass so their behavior is unchanged until a caller
|
||||
* wires a real, profile-driven lookup.
|
||||
*/
|
||||
static BackendErrorPatternLookup legacy() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a typed backend-error classification
|
||||
* to a waiting send (fleetd#201 / #227) — never on a race that lost. {@link CompletionResolver}
|
||||
* calls this only after {@code Rendezvous.resolveFailure} returns {@code true} for that exact
|
||||
* waiter, mirroring the win-only race rule {@link ExhaustionSink} already uses.
|
||||
*
|
||||
* <p>The public send result is unchanged by this classification — it is still a failed send
|
||||
* ({@code Rendezvous.Kind#FAILED}); this sink is the internal seam a later stage (fleetd#201 Unit
|
||||
* 5, the policy that cools a credential off after two such failures in 60 seconds and tells the
|
||||
* lead) consumes. {@link CompletionResolver} knows only {@code target} (a herdr terminal id); it
|
||||
* has no notion of profiles or credentials, so mapping {@code target} to whatever should be
|
||||
* quarantined is entirely the sink's job — exactly like {@link ExhaustionSink}.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified as a backend error
|
||||
* @param matchedLine the single pane line that matched the backend-error pattern
|
||||
* @param reason the full failure reason carried by the classification (the matched line
|
||||
* plus whatever pane context the classifying path attaches)
|
||||
*/
|
||||
void onBackendError(String target, String matchedLine, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on a backend-error classification. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload,
|
||||
* exactly like {@link ExhaustionSink#none()}.
|
||||
*/
|
||||
static BackendErrorSink none() {
|
||||
return (target, matchedLine, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -89,6 +89,8 @@ public final class CompletionResolver implements TurnListener {
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
private final BackendErrorPatternLookup backendErrorPatterns;
|
||||
private final BackendErrorSink backendErrorSink;
|
||||
private final LongSupplier nowNanos;
|
||||
|
||||
/**
|
||||
@@ -129,10 +131,48 @@ public final class CompletionResolver implements TurnListener {
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*
|
||||
* <p>Transition constructor (fleetd#201 Unit 1): delegates to the full constructor below with
|
||||
* {@link BackendErrorPatternLookup#legacy()} and {@link BackendErrorSink#none()}, so every
|
||||
* existing caller keeps today's behavior — the narrow built-in {@code API Error:} match, no
|
||||
* sink notified — until a caller wires a real, profile-driven backend-error lookup and sink
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, System::nanoTime);
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime);
|
||||
}
|
||||
|
||||
/**
|
||||
* Transition constructor (fleetd#201 Unit 1): same legacy backend-error defaults as the 4-arg
|
||||
* constructor above, but with the injectable clock. Kept so existing fleetd#164 timing tests
|
||||
* (and any other caller that wants a controllable clock but not the backend-error feature)
|
||||
* keep compiling unchanged.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), nowNanos);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor (fleetd#201 Unit 1): adds the target-keyed backend-error pattern
|
||||
* lookup and sink alongside the existing exhaustion pair. Required, like {@code exhaustedPatterns}
|
||||
* and {@code exhaustionSink} — pass {@link BackendErrorPatternLookup#legacy()} and
|
||||
* {@link BackendErrorSink#none()} to opt out.
|
||||
*
|
||||
* @param backendErrorPatterns fleetd#201: per-target lookup for a profile's configured
|
||||
* backend-error pattern; a target with none configured is matched
|
||||
* against the built-in compatibility pattern instead (never "off").
|
||||
* @param backendErrorSink fleetd#201: notified when a backend-error classification actually
|
||||
* resolves a waiter — never on a race that lost.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
System::nanoTime);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -145,11 +185,14 @@ public final class CompletionResolver implements TurnListener {
|
||||
* package.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
this.backendErrorPatterns = Objects.requireNonNull(backendErrorPatterns, "backendErrorPatterns");
|
||||
this.backendErrorSink = Objects.requireNonNull(backendErrorSink, "backendErrorSink");
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
}
|
||||
|
||||
@@ -232,16 +275,18 @@ public final class CompletionResolver implements TurnListener {
|
||||
// screen, since that is usually the backend's own error.
|
||||
long elapsedNanos = nowNanos.getAsLong() - turn.deliveredAtNanos();
|
||||
if (elapsedNanos < MIN_TURN_NANOS) {
|
||||
fail(target, turn, tooFastReason(target, elapsedNanos));
|
||||
failTooFast(target, turn, waiter, elapsedNanos);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
String rawScrape = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
rawScrape = agents.read(target, SCRAPE_SOURCE);
|
||||
assistantBlock = lastAssistantBlock(rawScrape);
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
@@ -256,6 +301,20 @@ public final class CompletionResolver implements TurnListener {
|
||||
// member, so a caller (including a lead deciding whether to delegate again) can tell a lost
|
||||
// turn from a real empty answer.
|
||||
if (scrapeFailed || tail.isEmpty()) {
|
||||
// fleetd#211: lastAssistantBlock() found nothing usable — most often a pane with no ⏺
|
||||
// marker at all, whose boundary scan then starts at the top of the raw screen and breaks
|
||||
// immediately on the first line of TUI chrome (╭, │, ❯, …). Before giving up as a lost
|
||||
// turn, run the same exhaustion/backend-error classification against the RAW scrape as a
|
||||
// fallback, ONLY here. A pane that already yielded a usable block never reaches this
|
||||
// branch, so the narrow (trimmed) match on the normal path below is completely unchanged
|
||||
// — zero new false positives there. Every pane this fallback examines was already headed
|
||||
// for the empty-scrape failure, so a wrong label here is strictly less bad than silently
|
||||
// losing an exhaustion signal: the alternative outcome is already a failure, just one that
|
||||
// never quarantines the credential. lastAssistantBlock stays the source of the reply
|
||||
// TEXT everywhere else; only classification ever consults the raw scrape, and only here.
|
||||
if (rawScrape != null && classifyRawScrapeFallback(target, turn, waiter, rawScrape)) {
|
||||
return;
|
||||
}
|
||||
fail(target, turn, emptyScrapeReason(target, scrapeFailed));
|
||||
return;
|
||||
}
|
||||
@@ -286,18 +345,27 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#164 (part 2): a scrape that read cleanly and produced content still isn't a real
|
||||
// reply when that content is the backend's own rejection (e.g. an HTTP 400 before the worker
|
||||
// did any work). Classify it as a failure naming the member, rather than handing the caller a
|
||||
// scrape that reads like a completed answer.
|
||||
String backendError = firstMatchingLine(assistantBlock, BACKEND_ERROR);
|
||||
// fleetd#164 (part 2) / fleetd#201: a scrape that read cleanly and produced content still
|
||||
// isn't a real reply when that content is the backend's own rejection (e.g. an HTTP 400
|
||||
// before the worker did any work). Classify it as a failure naming the member, rather than
|
||||
// handing the caller a scrape that reads like a completed answer, and — only on the
|
||||
// resolution that actually wins the race, mirroring the exhaustion sink above — notify the
|
||||
// typed backend-error sink so a later stage can act on repeated failures.
|
||||
String backendError = firstMatchingLine(assistantBlock, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the whole scrape, not just the matched line. The pattern is a heuristic: a member
|
||||
// that forgot fleet_reply while reporting *about* a backend error matches it too. Failing
|
||||
// is still right — the caller must not read a scrape as an answer — but dropping the rest
|
||||
// of the pane would destroy the report, which is the same defect fleetd#164 is about.
|
||||
fail(target, turn, "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail);
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race — a late
|
||||
// duplicate must never double-count one backend failure.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
@@ -313,6 +381,49 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#211: the raw-scrape fallback classification, run only when {@link #lastAssistantBlock}
|
||||
* found nothing usable (see the call site in {@link #resolve}). Mirrors the two classifications
|
||||
* the normal path already applies to the trimmed assistant block — exhaustion first, then the
|
||||
* narrow {@link #BACKEND_ERROR} pattern — against {@code raw} instead, and reports whether one of
|
||||
* them handled the turn (resolved the waiter or failed it) so the caller skips the empty-scrape
|
||||
* failure. Never runs on the normal (non-empty-block) path, and never touches the reply text.
|
||||
*/
|
||||
private boolean classifyRawScrapeFallback(String target, InFlight turn,
|
||||
CompletableFuture<Rendezvous.Resolution> waiter, String raw) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(raw, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED from the raw scrape (no "
|
||||
+ "usable assistant block; no fleet_reply): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
String backendError = firstMatchingLine(raw, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the pane, not just the matched line — the same fleetd#164 rule the normal path
|
||||
// above applies. Here it matters more, not less: the trimmed block was empty, so the raw
|
||||
// scrape is the ONLY copy of whatever the member managed to say. Clipped to the same cap
|
||||
// the normal path uses, since a raw screen has no boundary trimming to bound it.
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + clip(raw);
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback from the raw scrape: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
@@ -349,22 +460,53 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164: the failure reason for a turn that settled inside {@link #MIN_TURN_NANOS} — names
|
||||
* the member and both timings, and appends whatever the pane shows (usually the backend's own
|
||||
* error) so the caller sees the cause, not just "it failed".
|
||||
* fleetd#164 (floor) / fleetd#201 (classification): fail a turn that settled inside
|
||||
* {@link #MIN_TURN_NANOS} — a crash signature (e.g. a backend HTTP 400 before the worker did
|
||||
* anything) that a bare {@code BUSY -> DONE} transition cannot be told apart from a genuinely
|
||||
* fast completion. Runs the same backend-error classification the normal and raw-scrape paths
|
||||
* apply, against whatever is on screen right now: a match is a typed failure that notifies
|
||||
* {@link #backendErrorSink} (only on the resolution that wins the race); a non-match stays the
|
||||
* original generic too-fast failure, naming the member and both timings, with whatever the pane
|
||||
* shows appended so the caller sees the cause, not just "it failed".
|
||||
*/
|
||||
private String tooFastReason(String target, long elapsedNanos) {
|
||||
private void failTooFast(String target, InFlight turn, CompletableFuture<Rendezvous.Resolution> waiter,
|
||||
long elapsedNanos) {
|
||||
String scrape;
|
||||
try {
|
||||
scrape = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
scrape = agents.read(target, SCRAPE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
scrape = "";
|
||||
}
|
||||
String reason = String.format(
|
||||
String clippedScrape = clip(scrape);
|
||||
String baseReason = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms) — too fast to be real work, most "
|
||||
+ "likely a backend error before any work started",
|
||||
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
|
||||
return scrape.isBlank() ? reason : reason + ": " + scrape;
|
||||
String backendError = firstMatchingLine(scrape, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
String reason = baseReason + ": " + clippedScrape;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
fail(target, turn, clippedScrape.isBlank() ? baseReason : baseReason + ": " + clippedScrape);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#201: the pattern to classify a backend error against for {@code target} — its
|
||||
* profile's configured {@link BackendErrorPatternLookup} entry when there is one, else the
|
||||
* built-in {@link #BACKEND_ERROR} compatibility pattern. A target with no configured pattern is
|
||||
* never "off": it always falls back to this narrow default, exactly like the classification
|
||||
* behaved before fleetd#201 (Unit 5 reports a target relying on this fallback separately, as
|
||||
* legacy-default coverage rather than full per-profile coverage).
|
||||
*/
|
||||
private Pattern backendErrorPatternOrFallback(String target) {
|
||||
Pattern configured = backendErrorPatterns.patternFor(target);
|
||||
return configured != null ? configured : BACKEND_ERROR;
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -278,18 +278,22 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
* session id — the resume handle. A resume passes the prior id via {@code -r} and returns
|
||||
* that id; every other spawn mints a new UUID, passes it via {@code --session-id}, and
|
||||
* returns the mint. The bridge's logical name rides along as {@code -n} when present.
|
||||
*
|
||||
* <p>fleetd #214: the mint is unconditional. A plain {@code fleet_spawn} passes neither
|
||||
* sessionName nor resumeSessionId, yet the member must still be resumable, and this id is
|
||||
* the only resume handle a claude-code member has — unlike opencode, nothing resolves it
|
||||
* after the launch. Checked against the real binary (claude 2.1.252): the flag is safe on
|
||||
* every spawn. The binary takes only a valid UUID — it refuses any other value at argument
|
||||
* parsing ("Invalid session ID. Must be a valid UUID.") — so the {@code UUID.randomUUID()}
|
||||
* mint is required, not incidental. The only flag interaction the binary documents is with
|
||||
* {@code -r} (both claim the session id), and the resume branch above never combines the two.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
@@ -320,11 +324,19 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
*
|
||||
* <p>CB-618: Claude Code refuses to start when BOTH {@code --append-system-prompt} and
|
||||
* {@code --append-system-prompt-file} are on the command line ("Cannot use both ... Please use
|
||||
* only one"), so the two charters can never travel on separate flags. When both are present they
|
||||
* are concatenated into the one file, role charter first and reply charter last — last is where
|
||||
* the reply rule must sit, because it is the rule that must survive. When only the reply charter
|
||||
* is present it keeps its proven inline {@code --append-system-prompt} delivery, which is also
|
||||
* the only form that reaches a member with no repo checkout.
|
||||
* only one"), so the two charters can never travel on separate flags. They are concatenated
|
||||
* into the one file, role charter first and reply charter last — last is where the reply rule
|
||||
* must sit, because it is the rule that must survive.
|
||||
*
|
||||
* <p>fleetd #220: a lone reply charter used to ride inline on {@code --append-system-prompt},
|
||||
* which put ~800 bytes of prose on the command line herdr types into the pane. That line is
|
||||
* capped at {@value HerdrPeerLauncher#PANE_COMMAND_BYTE_LIMIT} bytes by the pty itself, and
|
||||
* everything past the cap is dropped with no error from any layer. The charter alone left about
|
||||
* 50 bytes of headroom, so adding one flag ({@code --session-id}, fleetd #214) truncated the
|
||||
* LAST argument instead — {@code --autocompact 250000} arrived as {@code --autocompact 25},
|
||||
* claude rejected it, and every claude-code spawn died as an unexplained readiness timeout.
|
||||
* The charter now always travels as a file, which takes the prose off the command line for
|
||||
* good; {@link HerdrPeerLauncher#checkPaneCommandFits} is the backstop for whatever grows next.
|
||||
*/
|
||||
private List<String> argvWithFleet(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
String roleCharter = nonBlank(spec.roleCharter());
|
||||
@@ -341,16 +353,13 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpConfigJson(cfg));
|
||||
}
|
||||
// Combine the charters in order role -> reply, dropping any that are absent. When two
|
||||
// or more survive they must ride one --append-system-prompt-file (CB-618 forbids the inline
|
||||
// flag and the file flag together). A lone reply charter keeps its proven inline delivery.
|
||||
// Combine the charters in order role -> reply, dropping any that are absent. They ride one
|
||||
// --append-system-prompt-file (CB-618 forbids the inline flag and the file flag together),
|
||||
// always — fleetd #220: charter prose on the command line overruns the pane's byte cap.
|
||||
List<String> charters = new java.util.ArrayList<>(2);
|
||||
if (roleCharter != null) charters.add(roleCharter);
|
||||
if (replyCharter != null) charters.add(replyCharter);
|
||||
if (charters.size() == 1 && replyCharter != null && roleCharter == null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(replyCharter);
|
||||
} else if (!charters.isEmpty()) {
|
||||
if (!charters.isEmpty()) {
|
||||
argv.add("--append-system-prompt-file");
|
||||
argv.add(writeCharterFile(String.join("\n\n", charters)).toString());
|
||||
}
|
||||
@@ -450,12 +459,83 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* {@link OpenCodeLauncher#writeConfig} already uses for its charter file, since the process that
|
||||
* reads this file (the spawned peer) outlives this JVM call and there is no spawn-scoped teardown
|
||||
* hook to delete it synchronously.
|
||||
*
|
||||
* <p><b>fleetd #222.</b> {@code Files.createTempFile(prefix, suffix)} with no directory argument
|
||||
* resolves against {@code java.io.tmpdir} — on macOS the per-user {@code $TMPDIR} under
|
||||
* {@code /var/folders/...}, mode {@code 0700}, both resolved against FLEETD's own OS user. Under
|
||||
* {@code memberHerdrSocket:} the member pane runs as a DIFFERENT OS user, so that user cannot even
|
||||
* traverse the directory, let alone read the file — and since fleetd #220 the charter file is the
|
||||
* ONLY delivery path for {@code --append-system-prompt-file}, always, not merely the fallback it
|
||||
* used to be. A member handed a path it cannot read is not degraded, it is broken: see this
|
||||
* method's refusal branch below.
|
||||
*
|
||||
* <p><b>Measured severity (fleetd #222 real-binary check, claude 2.1.258):</b> an unreadable
|
||||
* {@code --append-system-prompt-file} is the LOUD failure, not the silent one. {@code claude}
|
||||
* checks the file before touching auth or the network — invoked with a bogus API key against a
|
||||
* {@code chmod 000} file, it printed {@code Error reading append system prompt file: EACCES:
|
||||
* permission denied, open '<path>'} and exited 1 immediately (a nonexistent path gets {@code
|
||||
* Error: Append system prompt file not found: <path>}, same exit code). So the pre-fix bug did
|
||||
* NOT leave a charter-less member silently occupying a pane and never calling {@code
|
||||
* fleet_reply} — it made the herdr pane exit immediately, which the CB-306 spawn-readiness gate
|
||||
* (this launcher's {@code spawnReadyTimeoutMs} poll) would have surfaced as "did not reach
|
||||
* injectable state", the same unexplained-timeout shape fleetd #220 already describes. Still a
|
||||
* real defect (every claude-code member under {@code memberHerdrSocket} would have failed to
|
||||
* spawn), but not the worse, undetectable failure mode.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only live mode): byte-identical to before this
|
||||
* fix — {@code Files.createTempFile("fleetd-role-charter-", ".md")} with no directory
|
||||
* argument, i.e. still resolved against {@code java.io.tmpdir}.</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: a fresh per-spawn directory is created under {@code
|
||||
* worktreeRoot} (never {@code java.io.tmpdir}) holding just the charter file, then shared
|
||||
* read-only with {@code worktreeGroup} via {@link EnvAllowListScrub#shareWithGroup} — the
|
||||
* SAME mechanism fleetd #213 built for the ZDOTDIR scrub and fleetd #219 reused for {@link
|
||||
* OpenCodeLauncher#writeConfig}'s {@code opencode.json} directory, reused here rather than
|
||||
* duplicated a third time. A per-spawn subdirectory (not {@code worktreeRoot} itself) is the
|
||||
* unit {@code shareWithGroup} chmods, so this never touches permissions on anything else
|
||||
* under {@code worktreeRoot}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><b>Follows fleetd #219's REFUSAL decision, not #213's degrade decision.</b> The ZDOTDIR
|
||||
* scrub is a credential CONTROL — a degraded control (the CB-596 sentinel overlay) still has
|
||||
* value, so #213 falls back rather than refusing. A charter is NOT a control, it is the member's
|
||||
* TURN CONTRACT (the rule that ends every turn with {@code fleet_reply}). A member spawned with no
|
||||
* charter is not degraded, it is broken: either claude-code exits on the unreadable
|
||||
* {@code --append-system-prompt-file} path and the spawn dies at the readiness gate (loud), or it
|
||||
* starts anyway with no charter and never calls {@code fleet_reply} — the sender silently gets
|
||||
* nothing (silent). Neither outcome is worth trading for "spawn something." So a missing {@code
|
||||
* worktreeRoot}/{@code worktreeGroup} under {@code memberHerdrSocket} refuses the spawn here,
|
||||
* naming the missing key, exactly like {@link OpenCodeLauncher#configParentDir()}.
|
||||
*
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* worktreeRoot} and/or {@code worktreeGroup} is not
|
||||
*/
|
||||
private static Path writeCharterFile(String charterText) {
|
||||
private Path writeCharterFile(String charterText) {
|
||||
try {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
if (!memberHerdrSocketConfigured()) {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
return file;
|
||||
}
|
||||
Path parentDir = memberScrubParentDir();
|
||||
String group = memberGroup();
|
||||
if (parentDir == null || group == null) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the role/reply "
|
||||
+ "charter file (mounted via --append-system-prompt-file) must be placed where "
|
||||
+ "the member's OS user can read it — worktreeRoot, shared via worktreeGroup — "
|
||||
+ "but " + (parentDir == null ? "worktreeRoot" : "worktreeGroup") + " is not "
|
||||
+ "configured. Refusing to spawn rather than hand the member a charter path it "
|
||||
+ "cannot read: that member's turn contract (the fleet_reply rule) would never "
|
||||
+ "reach it. Configure both worktreeRoot and worktreeGroup to enable claude-code "
|
||||
+ "member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path dir = Files.createTempDirectory(parentDir, "fleetd-role-charter-");
|
||||
dir.toFile().deleteOnExit();
|
||||
Path file = dir.resolve("charter.md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
EnvAllowListScrub.shareWithGroup(dir, group);
|
||||
return file;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot write role charter temp file", e);
|
||||
|
||||
@@ -143,17 +143,24 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
|
||||
/**
|
||||
* chgrp/chmod-equivalent over the freshly generated directory and the startup files already
|
||||
* written into it: owner keeps full access, {@code group} gets traverse+read on the directory
|
||||
* ({@code rwxr-x---}, so a login shell under that group can find and source the files) and
|
||||
* read-only on each file ({@code rw-r-----}) — deliberately no group WRITE anywhere, since a
|
||||
* member never needs to add or change fleetd's own generated scrub. (The scrub script's own
|
||||
* report write inside the pane consequently fails closed rather than open — see {@code
|
||||
* scrub.zsh}'s trailing {@code 2>/dev/null} — which {@link
|
||||
* chgrp/chmod-equivalent over a freshly generated directory and the flat files already written
|
||||
* into it: owner keeps full access, {@code group} gets traverse+read on the directory ({@code
|
||||
* rwxr-x---}, so a member process — a login shell reading it via {@code ZDOTDIR}, or another
|
||||
* process simply opening a file under it — running under that group can find and read the
|
||||
* files) and read-only on each file ({@code rw-r-----}) — deliberately no group WRITE anywhere,
|
||||
* since a member never needs to add or change what fleetd generated. (For the ZDOTDIR scrub
|
||||
* specifically, this also means the scrub script's own report write inside the pane fails
|
||||
* closed rather than open — see {@code scrub.zsh}'s trailing {@code 2>/dev/null} — which {@link
|
||||
* dev.ltms.fleet.member.HerdrPeerLauncher#releaseZdotdir} already treats as "cannot be
|
||||
* confirmed to have run" rather than success.)
|
||||
*
|
||||
* <p>Package-private and named generically on purpose: fleetd #213 built this for the ZDOTDIR
|
||||
* scrub directory, and fleetd #219 reuses it verbatim for {@link
|
||||
* dev.ltms.fleet.member.OpenCodeLauncher}'s ephemeral {@code opencode.json} directory — both are
|
||||
* "a fleetd-generated directory of flat files that a different-uid member process must read but
|
||||
* never write," so the sharing mechanism is shared rather than copied a second time.
|
||||
*/
|
||||
private static void shareWithGroup(Path dir, String group) {
|
||||
static void shareWithGroup(Path dir, String group) {
|
||||
try {
|
||||
GroupPrincipal principal = dir.getFileSystem().getUserPrincipalLookupService()
|
||||
.lookupPrincipalByGroupName(group);
|
||||
@@ -164,11 +171,11 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot share generated ZDOTDIR " + dir + " with group '"
|
||||
throw new UncheckedIOException("cannot share generated directory " + dir + " with group '"
|
||||
+ group + "' — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
} catch (UnsupportedOperationException e) {
|
||||
throw new UncheckedIOException("cannot share generated ZDOTDIR " + dir + " with group '"
|
||||
throw new UncheckedIOException("cannot share generated directory " + dir + " with group '"
|
||||
+ group + "' — this filesystem does not support POSIX group ownership",
|
||||
new IOException(e));
|
||||
}
|
||||
|
||||
@@ -3,11 +3,13 @@ package dev.ltms.fleet.member;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.StatusRefiner;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
@@ -151,6 +153,14 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
private final LongSupplier nowMillis; // monotonic clock (injectable for tests)
|
||||
private final Runnable sleeper; // sleep/wait hook (injectable for tests; never real-sleep in unit tests)
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 2: the same UNKNOWN-refinement {@code dev.ltms.fleet.inject.StatusPoller}
|
||||
* uses, reused here for the spawn-readiness gate. Constructed once from {@link #agents} — see
|
||||
* {@link #refinedInjectable(String, Agent)} for the corroboration that keeps it from firing on
|
||||
* a dead pane's bare shell prompt.
|
||||
*/
|
||||
private final StatusRefiner statusRefiner;
|
||||
|
||||
// Per-process token mixed into each peer name so a fresh process (nameSeq back at 0) cannot
|
||||
// collide with same-profile peers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
@@ -272,6 +282,7 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
this.memberCredentials = memberCredentials;
|
||||
this.hostEnvNames = hostEnvNames != null ? hostEnvNames : () -> System.getenv().keySet();
|
||||
this.config = config;
|
||||
this.statusRefiner = new StatusRefiner(agents);
|
||||
}
|
||||
|
||||
// --- adapter seams -------------------------------------------------------------------------
|
||||
@@ -682,6 +693,7 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
checkPaneCommandFits(cfg, argv);
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
@@ -697,6 +709,59 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
throw last;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220: herdr does not exec the launch command — it TYPES it into the pane as one line,
|
||||
* and a pty line buffer holds only {@value #PANE_COMMAND_BYTE_LIMIT} bytes (BSD/macOS {@code
|
||||
* MAX_CANON}). Everything past that byte is dropped. Nothing reports it: herdr answers "agent
|
||||
* started", the backend exits on the mangled argument it was handed, the pane closes, and the
|
||||
* only symptom is {@link #waitUntilInjectableOrThrow} timing out 20 seconds later with no
|
||||
* reason. That is exactly how #214 broke every claude-code spawn — one 50-byte flag pushed a
|
||||
* 978-byte command to 1028, and the tail that got cut was {@code --autocompact 250000}.
|
||||
*
|
||||
* <p>So measure it here and refuse, loudly and immediately, rather than spawn something that
|
||||
* cannot work. The estimate is deliberately conservative: fleetd cannot see herdr's quoting, so
|
||||
* every argument is charged its own bytes plus a separator and a quote pair. An over-estimate
|
||||
* costs a clear error at a length that was already unsafe; an under-estimate would let the
|
||||
* silent truncation back in.
|
||||
*
|
||||
* @throws PeerUnreachableException when the command cannot fit — the same failure the spawn
|
||||
* would have hit anyway, named at the point it is still
|
||||
* explainable
|
||||
*/
|
||||
private void checkPaneCommandFits(FleetConfig.Profile cfg, List<String> argv) {
|
||||
int bytes = 0;
|
||||
String longest = null;
|
||||
int longestBytes = 0;
|
||||
for (String arg : argv) {
|
||||
int argBytes = arg == null ? 0 : arg.getBytes(java.nio.charset.StandardCharsets.UTF_8).length;
|
||||
bytes += argBytes + QUOTING_OVERHEAD_PER_ARG;
|
||||
if (argBytes > longestBytes) {
|
||||
longestBytes = argBytes;
|
||||
longest = arg;
|
||||
}
|
||||
}
|
||||
if (bytes <= PANE_COMMAND_BYTE_LIMIT) {
|
||||
return;
|
||||
}
|
||||
String culprit = longest == null ? "<none>"
|
||||
: longest.substring(0, Math.min(longest.length(), 60)) + (longest.length() > 60 ? "…" : "");
|
||||
throw new PeerUnreachableException(
|
||||
"launch command for profile " + cfg.profile() + " is about " + bytes + " bytes, over the "
|
||||
+ PANE_COMMAND_BYTE_LIMIT + "-byte limit of the pane line herdr types it into. "
|
||||
+ "The pty would drop the tail silently and the backend would exit on a mangled "
|
||||
+ "argument. Longest argument is " + longestBytes + " bytes: " + culprit
|
||||
+ " — move it off the command line (a file flag) or shorten it.");
|
||||
}
|
||||
|
||||
/**
|
||||
* The pty line buffer herdr types a launch command into: BSD/macOS {@code MAX_CANON}. Not a
|
||||
* fleetd choice and not configurable — see {@link #checkPaneCommandFits}.
|
||||
*/
|
||||
static final int PANE_COMMAND_BYTE_LIMIT = 1024;
|
||||
|
||||
/** Per-argument allowance for the separating space and a shell quote pair fleetd cannot see. */
|
||||
private static final int QUOTING_OVERHEAD_PER_ARG = 3;
|
||||
|
||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
||||
HerdrException busy = null;
|
||||
@@ -855,25 +920,135 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
// --- spawn-readiness gate (CB-306) ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until the pane reports an injectable state or the configured
|
||||
* Poll {@link AgentControl#get} until the pane reports an injectable state or the configured
|
||||
* timeout elapses. On timeout, close the pane (self-reap) and throw.
|
||||
*
|
||||
* <p>fleetd #176 fix 1: {@code agents.get} was previously unguarded here, so a herdr
|
||||
* {@code *_not_found} answer — which is what happens when the backend process EXITED rather
|
||||
* than being slow — propagated as a raw {@link HerdrException} instead of the
|
||||
* {@link PeerUnreachableException} every other failure path of this gate produces, and skipped
|
||||
* teardown ({@link #stop}) entirely, leaking the pane/tab. {@link #failFastOnGoneBackend} closes
|
||||
* that gap: it stops waiting immediately (never burns the rest of the timeout), runs the same
|
||||
* teardown the timeout path below runs, and throws with a message that says the backend exited
|
||||
* rather than that the pane was slow. Any other {@link HerdrException} still propagates
|
||||
* unchanged — this gate does not know how to recover from it.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long deadline = nowMillis.getAsLong() + spawnReadyTimeoutMs;
|
||||
long start = nowMillis.getAsLong();
|
||||
long deadline = start + spawnReadyTimeoutMs;
|
||||
AgentStatus lastStatus = null;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (agents.status(paneId).injectable()) {
|
||||
Agent sample;
|
||||
try {
|
||||
sample = agents.get(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (isAlreadyGone(e)) {
|
||||
failFastOnGoneBackend(paneId, e, nowMillis.getAsLong() - start);
|
||||
}
|
||||
throw e; // any other herdr failure is not ours to interpret — let it propagate
|
||||
}
|
||||
lastStatus = sample.status();
|
||||
if (lastStatus.injectable() || refinedInjectable(paneId, sample)) {
|
||||
log.debug("peer pane={} reached injectable state", paneId);
|
||||
return;
|
||||
}
|
||||
sleeper.run();
|
||||
}
|
||||
log.warn("peer pane={} did not become injectable within {}ms — closing", paneId, spawnReadyTimeoutMs);
|
||||
// fleetd #220: read the pane BEFORE stop() closes it. Without this the gate says only that
|
||||
// it timed out, which is true of every cause — a backend that never launched, a binary that
|
||||
// rejected an argument and exited, a trust prompt, a login shell that hung. The pane holds
|
||||
// the one copy of that answer and it is destroyed a line later.
|
||||
log.warn("peer pane={} did not become injectable within {}ms (last status {}) — closing. "
|
||||
+ "Pane tail:\n{}",
|
||||
paneId, spawnReadyTimeoutMs, lastStatus, readPaneQuietly(paneId));
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " did not reach injectable state within "
|
||||
+ spawnReadyTimeoutMs + "ms");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 1: the backend process exited while the gate was still waiting — herdr
|
||||
* answered {@code *_not_found} instead of ever reporting an injectable status. Fails
|
||||
* immediately (never burns the rest of {@link #spawnReadyTimeoutMs}), runs the exact same
|
||||
* teardown {@link #waitUntilInjectableOrThrow}'s timeout path runs, and throws a
|
||||
* {@link PeerUnreachableException} whose message says the process exited rather than that the
|
||||
* pane was slow.
|
||||
*
|
||||
* @throws PeerUnreachableException always — this method never returns normally
|
||||
*/
|
||||
private void failFastOnGoneBackend(String paneId, HerdrException cause, long elapsedMs) {
|
||||
String tail = readPaneQuietly(paneId); // read before stop() closes the pane
|
||||
log.warn("peer pane={} backend process exited after {}ms while waiting for injectable "
|
||||
+ "state (herdr: {}) — closing. Pane tail:\n{}",
|
||||
paneId, elapsedMs, cause.getMessage(), tail);
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " backend process exited after " + elapsedMs
|
||||
+ "ms while waiting to become injectable (herdr reported: " + cause.getMessage()
|
||||
+ "). Pane tail:\n" + tail);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 2: resolve a raw {@link AgentStatus#UNKNOWN} sample into a trustworthy
|
||||
* injectable state via pane content, the same refinement {@code StatusPoller} applies during a
|
||||
* peer's working life — but corroborated, because the trap this gate is exposed to that the
|
||||
* poller is not: a pane whose backend has already exited settles at a plain shell prompt, and
|
||||
* that prompt commonly contains the same {@code ❯} glyph {@link StatusRefiner#classify} treats
|
||||
* as "idle at the Claude Code TUI prompt". Naively wiring the refiner in would turn "the backend
|
||||
* died" into "ready to inject" — strictly worse than today's timeout.
|
||||
*
|
||||
* <p>Two guards, both required:
|
||||
* <ul>
|
||||
* <li>{@link StatusRefiner#classify} is written for the Claude Code TUI only (its own javadoc
|
||||
* says so), so refinement only ever runs for the {@code claude} adapter — never for
|
||||
* {@code opencode} or any future non-Claude backend, whatever its pane looks like.</li>
|
||||
* <li>The refined status is accepted only when the <em>same</em> {@code agents.get} sample
|
||||
* ({@code sample}, taken once by the caller) still reports a non-null
|
||||
* {@link Agent#agentType()} — herdr's own "a supported backend is still detected here"
|
||||
* signal, read from the very sample the raw status came from so the two can never
|
||||
* disagree. A bare shell prompt left by an exited backend reports no {@code agentType}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Called only when {@code sample.status() == UNKNOWN}, so a healthy spawn — which never sees
|
||||
* {@code UNKNOWN} — triggers zero extra herdr calls; only a persistently-{@code unknown} pane
|
||||
* pays the one extra {@code agent.read} {@link StatusRefiner#refine} performs.
|
||||
*/
|
||||
private boolean refinedInjectable(String paneId, Agent sample) {
|
||||
if (sample.status() != AgentStatus.UNKNOWN) {
|
||||
return false;
|
||||
}
|
||||
if (!"claude".equals(namePrefix)) {
|
||||
return false; // StatusRefiner.classify reads a Claude Code TUI prompt specifically
|
||||
}
|
||||
if (sample.agentType() == null) {
|
||||
return false; // no corroborating liveness signal — could be a dead pane's bare shell
|
||||
}
|
||||
return statusRefiner.refine(paneId, AgentStatus.UNKNOWN, agents).injectable();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220: the pane's recent output, clipped, for the readiness-gate timeout log — or a
|
||||
* short note when it cannot be read. Best-effort by construction: this runs on a path that is
|
||||
* already failing, so it must never replace the real error with one of its own.
|
||||
*/
|
||||
private String readPaneQuietly(String paneId) {
|
||||
try {
|
||||
String pane = agents.read(paneId, "recent");
|
||||
if (pane == null || pane.isBlank()) {
|
||||
return "<pane read returned nothing>";
|
||||
}
|
||||
return pane.length() <= SPAWN_FAILURE_PANE_CHARS
|
||||
? pane
|
||||
: pane.substring(pane.length() - SPAWN_FAILURE_PANE_CHARS);
|
||||
} catch (RuntimeException e) {
|
||||
return "<pane could not be read: " + e.getMessage() + ">";
|
||||
}
|
||||
}
|
||||
|
||||
/** How much of a failed spawn's pane the timeout log carries. */
|
||||
private static final int SPAWN_FAILURE_PANE_CHARS = 4000;
|
||||
|
||||
/**
|
||||
* A concrete {@link PeerHandle} wrapping herdr agent coordinates, the profile that spawned it,
|
||||
* the session identity the launch resolved (CB-547a): the bridge's logical name and the peer's
|
||||
@@ -1190,8 +1365,13 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* provisioning a worktree that will exist regardless, whereas an unconfigured value here means
|
||||
* fleetd has no operator-endorsed location to put a credential-bearing directory a different OS
|
||||
* user must reach, so falling back to the overlay is the honest answer, not a guess.
|
||||
*
|
||||
* <p>Package-private (fleetd #219) so {@link OpenCodeLauncher} can reuse the exact same
|
||||
* "different OS user, put it under worktreeRoot instead of java.io.tmpdir" resolution for its
|
||||
* own ephemeral {@code opencode.json} directory, rather than re-reading {@code config} a second
|
||||
* time with a second copy of this null/blank handling.
|
||||
*/
|
||||
private Path memberScrubParentDir() {
|
||||
Path memberScrubParentDir() {
|
||||
FleetConfig cfg = config == null ? null : config.get();
|
||||
if (cfg == null || cfg.worktreeRoot() == null || cfg.worktreeRoot().isBlank()) {
|
||||
return null;
|
||||
@@ -1204,8 +1384,11 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* the group {@link dev.ltms.fleet.session.Worktrees#shareWithGroup} already establishes for
|
||||
* provisioned worktrees, rather than a second group key — see {@link
|
||||
* #applyEnvironmentAllowListPolicy}.
|
||||
*
|
||||
* <p>Package-private (fleetd #219) — reused by {@link OpenCodeLauncher} alongside {@link
|
||||
* #memberScrubParentDir()}; see that method's javadoc.
|
||||
*/
|
||||
private String memberGroup() {
|
||||
String memberGroup() {
|
||||
FleetConfig cfg = config == null ? null : config.get();
|
||||
if (cfg == null || cfg.worktreeGroup() == null || cfg.worktreeGroup().isBlank()) {
|
||||
return null;
|
||||
@@ -1378,8 +1561,12 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* through (every production {@code HerdrPeerLauncher} does; a handful of older tests do not) —
|
||||
* treated the same as "not configured", which is the correct, permissive default: it is exactly
|
||||
* today's single-daemon behaviour.
|
||||
*
|
||||
* <p>Package-private (fleetd #219) — {@link OpenCodeLauncher} reuses this same gate to decide
|
||||
* where its own ephemeral {@code opencode.json} directory (site 1) and its opencode session
|
||||
* discovery (site 2) may run, rather than re-deriving "is this a multi-uid fleet" a second way.
|
||||
*/
|
||||
private boolean memberHerdrSocketConfigured() {
|
||||
boolean memberHerdrSocketConfigured() {
|
||||
if (config == null) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
@@ -20,6 +21,8 @@ import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.function.BooleanSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
@@ -71,6 +74,19 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* fleetd #175: notified when {@link SessionAwareHandle} detects, on the same late-resolve
|
||||
* read that discovers the session id, that the live opencode session is running a DIFFERENT
|
||||
* model than the profile requested — opencode does not fail on an unknown {@code -m}, it
|
||||
* silently falls back to a default (potentially paid) model. Reused exactly as
|
||||
* {@code CompletionResolver}'s {@code BACKEND_EXHAUSTED} path uses it: this launcher supplies
|
||||
* only the herdr terminal id and a reason string; mapping target → session → profile →
|
||||
* credential stays entirely the sink's job (see {@code Fleetd.main}'s wiring). Defaults to
|
||||
* {@link ExhaustionSink#none()} for a caller (an older constructor, or a test not exercising
|
||||
* this) that opts out.
|
||||
*/
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
@@ -130,9 +146,25 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the live config for URI environment exclusions and the fleetd
|
||||
* #175 model-mismatch quarantine sink.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials, config);
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials, config,
|
||||
exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -180,6 +212,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = ExhaustionSink.none();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -205,17 +238,72 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
configRoot, discoveryRoot, fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the live config for URI environment exclusions and the
|
||||
* fleetd #175 model-mismatch quarantine sink — the constructor a test drives directly to
|
||||
* observe {@link ExhaustionSink#onExhausted} without going through {@code Fleetd.main}'s
|
||||
* wiring.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, null, config);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = exhaustionSink == null ? ExhaustionSink.none() : exhaustionSink;
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
/**
|
||||
* The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir) —
|
||||
* always FLEETD's OWN {@code user.home}, whichever OS user runs the daemon.
|
||||
*
|
||||
* <p><b>fleetd #219 site 2 — a decision, not a patch.</b> Under {@code memberHerdrSocket:} the
|
||||
* member pane runs as a <em>different</em> OS user, and opencode writes {@code opencode.db}
|
||||
* under <em>that</em> user's {@code $HOME}, not fleetd's. Scanning fleetd's own {@code
|
||||
* user.home} is therefore looking in the wrong place — a wrong-LOCATION failure, not a
|
||||
* wrong-PERMISSION one like site 1, and it fails quietly: {@link
|
||||
* SessionAwareHandle#agentSessionId()} would keep returning {@code null} forever, which reads
|
||||
* as "opencode does not support resume" rather than "fleetd looked in the wrong home." fleetd
|
||||
* #209 is the reason that silence is unacceptable.
|
||||
*
|
||||
* <p>Three ways to close the gap were weighed:
|
||||
* <ol>
|
||||
* <li><b>Make the member's home configurable.</b> Correct in principle, but this ticket's
|
||||
* scope is the two existing call sites, not a new config key — {@code memberHerdrSocket}
|
||||
* already carries the second herdr's socket path, not its user's home, and inventing a
|
||||
* parallel key here without also wiring it through discovery's actual callers is a
|
||||
* half-shipped feature (the exact shape CB-596/CB-611 warn against).</li>
|
||||
* <li><b>Derive it</b> (e.g. from {@code worktreeRoot}'s owner, or {@code getent passwd}).
|
||||
* Rejected: nothing in this codebase resolves a Unix username to a home directory today,
|
||||
* and guessing wrong would silently point discovery at a THIRD wrong location — worse
|
||||
* than the current gap, because it would look like it should work.</li>
|
||||
* <li><b>Declare discovery unavailable</b> under {@code memberHerdrSocket}, and say so once,
|
||||
* loudly, instead of scanning a directory that structurally cannot hold the answer.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p>Option 3 is taken — the one this ticket says to default to when unsure. {@link
|
||||
* OpenCodeLauncher#spawn} routes {@link SessionAwareHandle#agentSessionId()} through {@link
|
||||
* HerdrPeerLauncher#memberHerdrSocketConfigured()} before ever calling {@link
|
||||
* OpenCodeSessionDiscovery#sessionIdForDirectory}, so under {@code memberHerdrSocket} the
|
||||
* database at this root is never even opened, and one WARN per launcher instance names the gap
|
||||
* instead of the {@code null} return reading as "unsupported." Capability advertising is
|
||||
* unaffected: {@link #capabilities()} always includes {@code SESSION_RESUME}, since {@code
|
||||
* memberHerdrSocket} absent (today's only live mode) is unchanged by this decision.
|
||||
*/
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
@@ -331,7 +419,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "fleetd-opencode-");
|
||||
Path dir = Files.createTempDirectory(configParentDir(), "fleetd-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
@@ -402,6 +490,14 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
if (memberHerdrSocketConfigured()) {
|
||||
// fleetd #219: the same "different OS user" gap fleetd #213 closed for the ZDOTDIR
|
||||
// scrub — share read-only with worktreeGroup rather than leaving the directory under
|
||||
// fleetd's own 0700 java.io.tmpdir, where the member's OS user could not even
|
||||
// traverse it. memberGroup() cannot be null here: configParentDir() above already
|
||||
// refused this spawn if either worktreeRoot or worktreeGroup was missing.
|
||||
EnvAllowListScrub.shareWithGroup(dir, memberGroup());
|
||||
}
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
@@ -409,6 +505,57 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 1: where {@link #writeConfig} creates its per-spawn directory.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only mode): byte-identical to before this
|
||||
* fix — always {@link #configRoot} (defaults to {@code java.io.tmpdir}, fleetd's own
|
||||
* process).</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: {@code java.io.tmpdir} is fleetd's own per-user temp
|
||||
* dir (mode {@code 0700} on macOS) — the member pane runs as a DIFFERENT OS user under
|
||||
* this config key and cannot even traverse it, so the directory holding {@code
|
||||
* opencode.json} (which tells the member where the bridge MCP is) and the member charter
|
||||
* would be unreadable to the very process it is written for. The directory instead goes
|
||||
* under {@code worktreeRoot}, shared read-only with {@code worktreeGroup} via {@link
|
||||
* EnvAllowListScrub#shareWithGroup} — the SAME mechanism fleetd #213 built for the ZDOTDIR
|
||||
* scrub, reused here rather than duplicated (see {@link
|
||||
* HerdrPeerLauncher#memberScrubParentDir()}).</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><b>Unlike the ZDOTDIR scrub, a missing {@code worktreeRoot}/{@code worktreeGroup} here
|
||||
* REFUSES the spawn instead of degrading.</b> The ZDOTDIR scrub is a credential CONTROL: a
|
||||
* degraded control (CB-596's sentinel overlay) is still worth having. This config file is not a
|
||||
* control — it is the ONLY way the member learns where the bridge MCP lives. Writing it
|
||||
* somewhere the member cannot read would not degrade anything; it would spawn a member that
|
||||
* occupies a pane and never becomes deliverable, since {@code fleet_send} waits ~60s on the
|
||||
* readiness gate and then fails with nothing pointing at a temp directory as the cause. Refusing
|
||||
* up front, with a message that names the missing config key, is the honest failure — an
|
||||
* undeliverable member is not a working spawn either way, so nothing is lost by refusing loudly
|
||||
* instead of failing silently later.
|
||||
*
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* worktreeRoot} and/or {@code worktreeGroup} is not
|
||||
*/
|
||||
private Path configParentDir() {
|
||||
if (!memberHerdrSocketConfigured()) {
|
||||
return configRoot;
|
||||
}
|
||||
Path root = memberScrubParentDir();
|
||||
String group = memberGroup();
|
||||
if (root == null || group == null) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so opencode's config "
|
||||
+ "directory (opencode.json + member charter) must be placed where the member's "
|
||||
+ "OS user can read it — worktreeRoot, shared via worktreeGroup — but "
|
||||
+ (root == null ? "worktreeRoot" : "worktreeGroup") + " is not configured. "
|
||||
+ "Refusing to spawn rather than write a config the member cannot read: that "
|
||||
+ "member would occupy a pane and never become deliverable, with nothing "
|
||||
+ "pointing at the real cause. Configure both worktreeRoot and worktreeGroup to "
|
||||
+ "enable opencode member spawns under memberHerdrSocket.");
|
||||
}
|
||||
return root;
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
@@ -517,11 +664,21 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/** One WARN per launcher instance for the fleetd #219 site-2 discovery-unavailable gap. */
|
||||
private final AtomicBoolean discoveryUnavailableWarned =
|
||||
new AtomicBoolean();
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
// fleetd #175: the same profile config buildLaunch resolved for this spawn (requireProfile
|
||||
// is deterministic on req.profileName(), so re-resolving here costs a map lookup, not a
|
||||
// second decision) — SessionAwareHandle needs cfg.model() to know what THIS session should
|
||||
// be running.
|
||||
FleetConfig.Profile cfg = requireProfile(req.profileName());
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req), cfg,
|
||||
this::memberHerdrSocketConfigured, discoveryUnavailableWarned, exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -531,16 +688,37 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*
|
||||
* <p>fleetd #175: {@link #agentSessionId()} is also where the model-mismatch check lives (see
|
||||
* {@link #checkModelMatch()}) — it is the one method the real late-resolve path
|
||||
* ({@code SessionManager.resolveAgentSessionId}, via {@code get}/{@code rosterResolved}/
|
||||
* {@code release}) actually calls, and only while the session id is still unknown. Putting the
|
||||
* check anywhere else risks repeating PR #203's mistake: a check that runs before opencode has
|
||||
* written the row it needs, and so never fires.
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
private final FleetConfig.Profile cfg;
|
||||
private final BooleanSupplier discoveryUnavailable;
|
||||
private final AtomicBoolean discoveryUnavailableWarned;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
/** CAS'd true the first (and only) time a model mismatch is reported for this handle. */
|
||||
private final AtomicBoolean modelMismatchReported = new AtomicBoolean();
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd,
|
||||
FleetConfig.Profile cfg,
|
||||
BooleanSupplier discoveryUnavailable,
|
||||
AtomicBoolean discoveryUnavailableWarned,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
this.cfg = cfg;
|
||||
this.discoveryUnavailable = discoveryUnavailable;
|
||||
this.discoveryUnavailableWarned = discoveryUnavailableWarned;
|
||||
this.exhaustionSink = exhaustionSink;
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -565,11 +743,84 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// fleetd #219 site 2: under memberHerdrSocket the member pane runs as a different OS
|
||||
// user, so opencode.db lives under THAT user's $HOME, not the one discoveryRoot was
|
||||
// built from (see OpenCodeLauncher#defaultDiscoveryRoot's javadoc for the full
|
||||
// reasoning). Scanning fleetd's own $HOME under that config would only ever find "no
|
||||
// row" and read as "resume unsupported" — declare it unavailable instead, once, loudly.
|
||||
if (discoveryUnavailable.getAsBoolean()) {
|
||||
if (discoveryUnavailableWarned.compareAndSet(false, true)) {
|
||||
log.warn("opencode session discovery unavailable: memberHerdrSocket is "
|
||||
+ "configured, so opencode's on-disk session database lives under the "
|
||||
+ "MEMBER's own $HOME, not fleetd's ({}) — agentSessionId will stay null "
|
||||
+ "for every opencode member under this config, and SESSION_RESUME "
|
||||
+ "cannot be honored (fleetd #209/#219).",
|
||||
System.getProperty("user.home"));
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
String id = discovery.sessionIdForDirectory(cwd);
|
||||
// fleetd #175: check on the SAME tick — while the caller (SessionManager's late-resolve
|
||||
// step) is still re-polling because the id is unknown, the row this id came from (once
|
||||
// it exists) is exactly the row that also carries the actual model. Once id resolves,
|
||||
// the caller stops calling agentSessionId() for this session, so this is naturally a
|
||||
// once-only check that happens right when the row first appears.
|
||||
checkModelMatch();
|
||||
return id;
|
||||
}
|
||||
|
||||
/**
|
||||
* Verify the live opencode session is running the model {@link #cfg} requested (fleetd
|
||||
* #175) and, on a real mismatch, log an ERROR and quarantine through {@link
|
||||
* #exhaustionSink}. A no-op when there is nothing to compare against — no model configured,
|
||||
* already reported once for this handle, or the actual model is still UNKNOWN (no row yet,
|
||||
* unreadable database, or unparseable evidence). UNKNOWN must never be treated as a
|
||||
* mismatch: that is the single most important safety rule here — a false positive would
|
||||
* quarantine a perfectly working profile's credential.
|
||||
*/
|
||||
private void checkModelMatch() {
|
||||
if (modelMismatchReported.get() || cfg.model() == null || cfg.model().isBlank()) {
|
||||
return;
|
||||
}
|
||||
OpenCodeSessionDiscovery.ActualModel actual = discovery.actualModelForDirectory(cwd);
|
||||
if (actual == null) {
|
||||
return; // UNKNOWN evidence — never a mismatch
|
||||
}
|
||||
String[] requestedParts = splitProviderModel(cfg.model());
|
||||
String requestedProvider = requestedParts == null ? null : requestedParts[0];
|
||||
String requestedId = requestedParts == null ? cfg.model() : requestedParts[1];
|
||||
boolean idMatches = requestedId.equals(actual.id());
|
||||
// Compare the provider ONLY when BOTH sides have one. requestedProvider == null covers
|
||||
// a bare profile model with no "/" — the profile never asked for a specific provider.
|
||||
// actual.provider() == null covers opencode's model JSON having an id but no providerID
|
||||
// (a real shape parseModel accepts) — that is missing evidence, not a contradiction, and
|
||||
// acceptance rule 4 says missing evidence is UNKNOWN, never a mismatch. Narrowing the
|
||||
// provider comparison this way keeps the id comparison (the part that actually caught the
|
||||
// xf bug) fully intact — a genuine id mismatch is still caught either way (fleetd #175
|
||||
// review round 2).
|
||||
boolean providerMatches = requestedProvider == null || actual.provider() == null
|
||||
|| requestedProvider.equals(actual.provider());
|
||||
if (idMatches && providerMatches) {
|
||||
return;
|
||||
}
|
||||
if (!modelMismatchReported.compareAndSet(false, true)) {
|
||||
return; // another thread already reported this exact mismatch
|
||||
}
|
||||
String actualDisplay = actual.provider() == null
|
||||
? actual.id() : actual.provider() + "/" + actual.id();
|
||||
log.error("opencode profile '{}' requested model '{}' but the live session is actually "
|
||||
+ "running '{}' — opencode does not fail on an unknown -m, it silently "
|
||||
+ "falls back to a default model, which may be a PAID credential "
|
||||
+ "(fleetd #175); quarantining this profile's credential",
|
||||
cfg.profile(), cfg.model(), actualDisplay);
|
||||
exhaustionSink.onExhausted(delegate.terminalId(),
|
||||
"opencode model mismatch: profile '" + cfg.profile() + "' requested '"
|
||||
+ cfg.model() + "' but the live session is running '" + actualDisplay
|
||||
+ "' (fleetd #175)");
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -1,9 +1,12 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
import org.sqlite.SQLiteConfig;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.sql.Connection;
|
||||
@@ -44,6 +47,18 @@ import java.util.concurrent.atomic.AtomicBoolean;
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(OpenCodeSessionDiscovery.class);
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* The model opencode actually ran a session on, parsed from the {@code session.model} JSON
|
||||
* column (fleetd #175). {@code id} is never null/blank on a non-null {@code ActualModel} —
|
||||
* {@link #parseModel} returns {@code null} instead when {@code id} cannot be determined, so a
|
||||
* caller only ever sees a fully-known record or {@code null} (UNKNOWN). {@code provider} may
|
||||
* still be {@code null} on its own when the profile that requested the session named no
|
||||
* provider prefix, or opencode's JSON omitted {@code providerID}.
|
||||
*/
|
||||
record ActualModel(String provider, String id) {
|
||||
}
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final Path databasePath;
|
||||
@@ -117,4 +132,73 @@ final class OpenCodeSessionDiscovery {
|
||||
storageRoot, directory);
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The model opencode actually ran the {@code directory}'s most-recent session on (fleetd
|
||||
* #175), read from the same row {@link #sessionIdForDirectory} matches — but via its own
|
||||
* query and its own connection, deliberately kept independent so a database whose schema
|
||||
* predates the {@code model} column (or any other read failure on this column alone) can
|
||||
* never take {@link #sessionIdForDirectory}'s id resolution down with it. That would be a
|
||||
* regression of the id-resolution feature #209 shipped; this method degrades on its own.
|
||||
*
|
||||
* <p>Never throws, and every failure mode — no matching row, a missing/unreadable database, a
|
||||
* missing {@code model} column, a null/blank {@code model} value, or JSON that does not parse
|
||||
* into {@code {"id": "...", "providerID": "..."}} with a non-blank {@code id} — resolves to
|
||||
* {@code null}. That is UNKNOWN evidence, not a mismatch signal: the caller must never
|
||||
* quarantine a profile on the strength of a {@code null} here.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the actual model, or {@code null} when unknown
|
||||
*/
|
||||
ActualModel actualModelForDirectory(String directory) {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
if (!Files.isRegularFile(databasePath)) {
|
||||
// sessionIdForDirectory already WARNs once (shared warnedMissingDatabase) for this
|
||||
// exact condition — do not double-log it here.
|
||||
return null;
|
||||
}
|
||||
String sql = "SELECT model FROM session WHERE directory = ? ORDER BY time_updated DESC LIMIT 1";
|
||||
try (Connection connection = openReadOnly();
|
||||
PreparedStatement statement = connection.prepareStatement(sql)) {
|
||||
statement.setString(1, directory);
|
||||
try (ResultSet rows = statement.executeQuery()) {
|
||||
if (rows.next()) {
|
||||
return parseModel(rows.getString("model"));
|
||||
}
|
||||
}
|
||||
} catch (SQLException e) {
|
||||
// Locked/corrupt database, or a `model` column this schema version does not have —
|
||||
// never fatal, and never a mismatch signal. See the class doc above.
|
||||
log.debug("opencode session model unreadable at {}: {}", databasePath, e.toString());
|
||||
return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse opencode's {@code model} column — {@code {"id":"...","providerID":"..."}} — into an
|
||||
* {@link ActualModel}, or {@code null} when {@code json} is null/blank, is not valid JSON, or
|
||||
* parses without a non-blank {@code id}. {@code providerID} may be absent; that alone does not
|
||||
* make the record unknown, since a caller comparing against a profile with no provider prefix
|
||||
* never looks at it.
|
||||
*/
|
||||
private static ActualModel parseModel(String json) {
|
||||
if (json == null || json.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
JsonNode node = MAPPER.readTree(json);
|
||||
String id = node.path("id").asText(null);
|
||||
if (id == null || id.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
String provider = node.path("providerID").asText(null);
|
||||
return new ActualModel(provider, id);
|
||||
} catch (IOException e) {
|
||||
log.debug("opencode session model JSON unparseable: {}", e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -13,6 +13,7 @@ import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
@@ -161,6 +162,7 @@ public final class GitWorktrees implements Worktrees {
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
shareRootWithGroup(root);
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
removeUserInfoFromHttpsOrigin(repoRoot);
|
||||
@@ -712,6 +714,54 @@ public final class GitWorktrees implements Worktrees {
|
||||
group, repoRoot, worktreePath, touched);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #224: make {@code worktreeRoot} ITSELF group-traversable — established once, here,
|
||||
* where the root is created, never at a use site. {@link #shareWithGroup} shares each worktree
|
||||
* (and the repo's common git dir) with {@link #group}, but never the PARENT directory that
|
||||
* contains every worktree — and under {@code memberHerdrSocket:} the member pane runs as a
|
||||
* different OS user, which needs the execute bit on every ancestor directory to reach anything
|
||||
* underneath, no matter how carefully each child is shared. Without this, a member cannot read
|
||||
* the ephemeral {@code opencode.json} #219 places under this root, cannot reach its own
|
||||
* worktree, and cannot read #213's ZDOTDIR scrub when placed here either.
|
||||
*
|
||||
* <p>No-op — no process spawned — when {@link #group} is null/blank, so behaviour with
|
||||
* {@code worktreeGroup:} unset (today's only live mode) is unchanged. Only {@code root} itself
|
||||
* is touched (non-recursive): each child underneath is shared individually, either by
|
||||
* {@link #shareWithGroup} for a worktree or by the launcher that generates it (fleetd #213/#219)
|
||||
* for a scrub/config directory — sharing this level again would just duplicate that policy in
|
||||
* the wrong layer.
|
||||
*
|
||||
* <p>Fails loudly, naming {@code root}, its mode at the time of the attempt, and {@link #group}:
|
||||
* a member that starts and then cannot see its own checkout is worse than a refused spawn, since
|
||||
* nothing about that failure mode points at a directory's permission bits.
|
||||
*/
|
||||
private void shareRootWithGroup(Path root) {
|
||||
if (group == null) {
|
||||
return;
|
||||
}
|
||||
String mode = currentPosixMode(root);
|
||||
try {
|
||||
shareGroupRunner.apply(new String[]{"chgrp", group, root.toString()});
|
||||
shareGroupRunner.apply(new String[]{"chmod", "g+x", root.toString()});
|
||||
} catch (WorktreeException e) {
|
||||
throw new WorktreeException("cannot make worktree root " + root + " (mode " + mode
|
||||
+ ") group-traversable for group '" + group + "': " + e.getMessage()
|
||||
+ " — the group must exist, and the fleetd operator (" + System.getProperty("user.name")
|
||||
+ ") must be a member of it", e);
|
||||
}
|
||||
log.info("worktreeGroup={} made worktree root {} group-traversable (was mode {})", group, root, mode);
|
||||
}
|
||||
|
||||
/** {@code root}'s current POSIX permission string, or {@code "unknown"} on a filesystem that does
|
||||
* not support POSIX permissions — used only to name the mode in a refusal message. */
|
||||
private static String currentPosixMode(Path root) {
|
||||
try {
|
||||
return PosixFilePermissions.toString(Files.getPosixFilePermissions(root));
|
||||
} catch (IOException | UnsupportedOperationException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The repo's <em>common</em> git directory as an absolute path — where {@code objects},
|
||||
* {@code refs} and {@code worktrees} actually live. {@code git rev-parse --git-common-dir}
|
||||
|
||||
@@ -183,46 +183,51 @@ public final class SessionManager implements TurnListener {
|
||||
String sessionName, String resumeSessionId) {
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
requireResumeCapability(profile, resumeSessionId);
|
||||
// CB-619 / fleetd #123: an explicit profile bypasses placement (CompositePeerLauncher only
|
||||
// constrains an UNQUALIFIED spawn to the role's pool), so it is the one path that can ask
|
||||
// for a role with no slot to bind it to. Refuse before anything spawns. A blank profile is
|
||||
// left to placement, which already restricts an unqualified spawn to the role's pool.
|
||||
if (profile != null && !profile.isBlank()) {
|
||||
memberLifecycle.requireSlotFor(memberRole, profile);
|
||||
}
|
||||
MemberLifecycle.SlotReservation reservation = memberLifecycle.reserve(memberRole, profile);
|
||||
String launchProfile = reservation == null ? profile : reservation.profile();
|
||||
if (wt == null) {
|
||||
// CB-557: the role must ride on the SpawnRequest, not stay a local. The launcher needs it
|
||||
// to pick the profile out of that role's pool and to label the tab; a role kept only on
|
||||
// the MemberSession is recorded after the spawn it was supposed to steer.
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
SpawnRequest req = new SpawnRequest(launchProfile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
PeerHandle handle;
|
||||
boolean bound = false;
|
||||
try {
|
||||
handle = launcher.spawn(req);
|
||||
String resolvedProfile = resolveProfile(handle, launchProfile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
bound = reservation == null || actualRole == MemberRole.ARCHITECT;
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(), handle.terminalId(), resolvedProfile, actualRole, cwd, ownerTerminal, now, now, 0,
|
||||
MemberSession.State.SPAWNING, null, null, handle.charterReceipt(), handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={}: {}", profile, memberRole, e.getMessage());
|
||||
if (!bound) memberLifecycle.release(reservation);
|
||||
log.warn("spawn failed for profile={} role={}: {}", launchProfile, memberRole, e.getMessage());
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
null,
|
||||
null,
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
return acquireWithWorktree(profile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId);
|
||||
try {
|
||||
return acquireWithWorktree(launchProfile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId, reservation);
|
||||
} catch (RuntimeException e) {
|
||||
memberLifecycle.release(reservation);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -468,7 +473,8 @@ public final class SessionManager implements TurnListener {
|
||||
|
||||
private MemberSession acquireWithWorktree(String profile, MemberRole memberRole, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
String sessionName, String resumeSessionId,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
@@ -509,11 +515,14 @@ public final class SessionManager implements TurnListener {
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
// CB-619: see the no-worktree path above — bind before recording, and store the returned
|
||||
// actual role, so this session's role is never a lie about what it actually holds.
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
actualRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
@@ -526,13 +535,24 @@ public final class SessionManager implements TurnListener {
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
|
||||
private MemberRole acquired(MemberRole role, String profile, String terminal,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
if (reservation == null) {
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
if (memberLifecycle.bind(reservation, terminal)) {
|
||||
return role;
|
||||
}
|
||||
memberLifecycle.release(reservation);
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
|
||||
private String slug(String raw) {
|
||||
return raw == null ? "ticket" : raw.toLowerCase().replaceAll("[^a-z0-9]+", "-").replaceAll("^-+|-+$", "");
|
||||
}
|
||||
|
||||
@@ -44,10 +44,13 @@ public final class FakeHerdr implements HerdrClient {
|
||||
private String agentSendErrorCode = null;
|
||||
private boolean noPanes = false;
|
||||
private volatile String agentStatus = "idle"; // steady-state agent.get status
|
||||
private volatile String agentType = "claude"; // detected agent kind on agent.get; null = undetected
|
||||
private volatile String readText = "worker transcript tail"; // canned agent.read output
|
||||
private int pinnedStarts = 0; // how many upcoming agent.start calls report a fixed pane
|
||||
private String pinnedStartTerminal;
|
||||
private String pinnedStartPane;
|
||||
private volatile int agentGetOkCalls = Integer.MAX_VALUE; // how many agent.get calls succeed first
|
||||
private volatile String agentGetFailCode = null; // error code every agent.get call after that reports
|
||||
|
||||
public FakeHerdr healthy(boolean h) {
|
||||
this.healthy = h;
|
||||
@@ -103,6 +106,27 @@ public final class FakeHerdr implements HerdrClient {
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Set the detected agent kind ({@code "agent"} field) that {@code agent.get} reports —
|
||||
* {@code null} models herdr not (or no longer) detecting a supported backend in the pane, e.g.
|
||||
* a bare shell prompt (fleetd #176 fix 2 corroboration test).
|
||||
*/
|
||||
public FakeHerdr agentType(String type) {
|
||||
this.agentType = type;
|
||||
return this;
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code agent.get} succeed normally for its first {@code okCalls} invocations, then fail
|
||||
* every call after that with {@code code} — fleetd #176 fix 1's "backend exited mid-wait"
|
||||
* fixture. {@code okCalls == 0} fails from the very first call.
|
||||
*/
|
||||
public FakeHerdr agentGetFailsWithAfter(int okCalls, String code) {
|
||||
this.agentGetOkCalls = okCalls;
|
||||
this.agentGetFailCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** The text {@code agent.read} returns (the CB-106 completion scrape). */
|
||||
public FakeHerdr readText(String text) {
|
||||
this.readText = text;
|
||||
@@ -208,10 +232,21 @@ public final class FakeHerdr implements HerdrClient {
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
case "agent.get" -> mapper.readTree(("""
|
||||
{"type":"agent_info","agent":{"terminal_id":"term_a","agent":"claude",
|
||||
case "agent.get" -> {
|
||||
if (agentGetFailCode != null) {
|
||||
long getCalls = calls.stream().filter(c -> c.method().equals("agent.get")).count();
|
||||
if (getCalls > agentGetOkCalls) {
|
||||
throw new HerdrException(
|
||||
"herdr error [" + agentGetFailCode + "]: agent target not found",
|
||||
agentGetFailCode, null);
|
||||
}
|
||||
}
|
||||
String agentField = agentType == null ? "null" : "\"" + agentType + "\"";
|
||||
yield mapper.readTree(("""
|
||||
{"type":"agent_info","agent":{"terminal_id":"term_a","agent":%s,
|
||||
"agent_status":"%s","workspace_id":"w2","tab_id":"w2:t7","pane_id":"w2:p7"}}""")
|
||||
.formatted(agentStatus));
|
||||
.formatted(agentField, agentStatus));
|
||||
}
|
||||
case "agent.read" -> mapper.readTree(mapper.writeValueAsString(
|
||||
java.util.Map.of("type", "agent_read", "read", java.util.Map.of("text", readText))));
|
||||
case "agent.start" -> {
|
||||
|
||||
@@ -4,8 +4,11 @@ import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
@@ -633,6 +636,115 @@ class CompletionResolverTest {
|
||||
"the failure carries the rest of the pane, not only the matched line: " + reason);
|
||||
}
|
||||
|
||||
// --- fleetd#211: raw-scrape fallback classification when there is no usable assistant block ---
|
||||
|
||||
@Test
|
||||
void anExhaustionLineWithNoMarkerAndLeadingChromeIsClassifiedFromTheRawScrapeAndNotifiesTheSink() {
|
||||
// No ⏺ anywhere, and the first visible line is TUI chrome (╭). lastAssistantBlock's boundary
|
||||
// scan starts at the top of the raw screen and breaks immediately, so the trimmed block is "".
|
||||
// The fix: fall back to matching the RAW scrape so this doesn't get lost as an empty scrape.
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
The usage limit has been reached. Try again later.
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a raw-scrape match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"classified from the raw scrape even though the trimmed block was empty");
|
||||
assertEquals(1, notified.size(),
|
||||
"the sink is the whole point of this ticket — it must be notified: " + notified);
|
||||
assertTrue(notified.get(0).startsWith("term_a: "), "the sink is told which target exhausted");
|
||||
assertTrue(notified.get(0).contains("The usage limit has been reached"),
|
||||
"the sink is told the matched reason: " + notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBackendErrorLineWithNoMarkerAndLeadingChromeIsClassifiedFromTheRawScrape() {
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
API Error: 400 invalid request body
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a raw-scrape backend-error match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"classified BACKEND_ERROR from the raw scrape even though the trimmed block was empty");
|
||||
assertTrue(waiter.getNow(null).text().contains("API Error: 400 invalid request body"),
|
||||
"the failure carries the matched line: " + waiter.getNow(null).text());
|
||||
assertTrue(waiter.getNow(null).text().contains("--- pane tail ---"),
|
||||
"fleetd#164: the failure must carry the pane, not only the matched line — with an "
|
||||
+ "empty trimmed block the raw scrape is the only copy of what the member said: "
|
||||
+ waiter.getNow(null).text());
|
||||
assertTrue(waiter.getNow(null).text().contains("╭"),
|
||||
"the carried pane is the raw scrape, chrome included: " + waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void anOrdinaryPaneWithANormalAssistantBlockIsUnaffectedByTheRawScrapeFallback() {
|
||||
// Pin: on a pane that already yields a usable block, the fallback branch is never reached —
|
||||
// same outcome, same text, sink not called — even though the raw screen around the marker
|
||||
// would itself match the configured exhausted pattern.
|
||||
String block = "The usage limit has been reached, but this is a leading TUI line above the "
|
||||
+ "marker.\n⏺ complete report\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"unchanged: a usable assistant block never reaches the raw-scrape fallback");
|
||||
assertEquals("complete report", waiter.getNow(null).text(), "the reply text is unaffected");
|
||||
assertTrue(notified.isEmpty(), "the fallback never runs, so the sink is never called");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aGenuinelyEmptyScrapeStillFailsAsEmptyAndNeverNotifiesTheSink() {
|
||||
// The false-positive pin: no exhaustion or backend-error text anywhere on the pane (just
|
||||
// chrome, no marker) — the raw-scrape fallback must not manufacture a classification, and
|
||||
// the sink must stay untouched.
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
│ > │
|
||||
╰──────────────────────────────────────╯
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "an empty scrape must still resolve the send, not hang");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"no exhaustion or backend-error text anywhere ⇒ this stays the ordinary empty-scrape failure");
|
||||
assertTrue(waiter.getNow(null).text().toLowerCase().contains("empty"),
|
||||
"the failure still says the scrape was empty: " + waiter.getNow(null).text());
|
||||
assertTrue(notified.isEmpty(), "a genuinely empty pane must never quarantine a credential");
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsOffWhenNoProfileHasAPatternConfigured() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [terra])",
|
||||
@@ -650,4 +762,202 @@ class CompletionResolverTest {
|
||||
assertEquals("partial (configured: [terra]; not configured: [gx10])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra")));
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: target-keyed backend-error pattern + typed sink ----------------------
|
||||
|
||||
@Test
|
||||
void aConfiguredBackendErrorPatternClassifiesAMatchAsAFailureAndNotifiesTheSinkOnce() {
|
||||
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a configured-pattern match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals(1, notified.size(), "the sink fires exactly once for the winning classification");
|
||||
assertEquals("term_a: 503 Service Unavailable: upstream credential rejected", notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLosingBackendErrorClassificationNeverNotifiesTheSink() {
|
||||
// The waiter was already resolved (e.g. by the worker's own fleet_reply) before this scrape
|
||||
// landed — resolveFailure loses the race and must return false, so the sink must not fire.
|
||||
String block = "⏺ 503 Service Unavailable: upstream credential rejected\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied")); // fleet_reply won first
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(), "a classification that loses the race must not fire the sink");
|
||||
assertEquals("already replied", waiter.getNow(null).text(), "the earlier resolution stands untouched");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBackendErrorClassificationThatLosesARealRaceDuringTheScrapeNeverNotifiesTheSink() {
|
||||
// The test above pre-resolves the waiter BEFORE calling resolve(), so it is caught by
|
||||
// resolve()'s own top-of-method isDone() guard before ever reaching the win-gated
|
||||
// resolveFailure call this unit adds — it proves the outcome, not the specific gate. This
|
||||
// test forces the actual race window fleetd#201 must be safe in: a genuine fleet_reply lands
|
||||
// WHILE the herdr scrape for this turn is in flight, i.e. AFTER resolve() has already passed
|
||||
// the top guard (waiter not done yet) and committed to reading the pane, but BEFORE it
|
||||
// reaches the "if (rendezvous.resolveFailure(...))" line below. The side effect is attached
|
||||
// to the one real I/O step resolve() performs between those two points: the herdr
|
||||
// "agent.read" call itself.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
var waiter = rendezvous.open("term_a");
|
||||
ObjectMapper mapper = new ObjectMapper();
|
||||
HerdrClient racingDuringScrape = new HerdrClient() {
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.read".equals(method)) {
|
||||
// The real fleet_reply that wins the race, landing mid-scrape.
|
||||
assertTrue(rendezvous.resolve("term_a", "a genuine fleet_reply landed first"),
|
||||
"the racing reply must land while the waiter is still open");
|
||||
}
|
||||
try {
|
||||
return mapper.readTree(mapper.writeValueAsString(java.util.Map.of(
|
||||
"type", "agent_read",
|
||||
"read", java.util.Map.of("text",
|
||||
"⏺ 503 Service Unavailable: upstream credential rejected\n❯ "))));
|
||||
} catch (Exception e) {
|
||||
throw new RuntimeException(e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(racingDuringScrape), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var turn = new CompletionResolver.InFlight(waiter, null); // not done yet — passes the top guard
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(),
|
||||
"a classification that loses a real race during the scrape must not fire the sink");
|
||||
assertEquals(Rendezvous.Kind.REPLY, waiter.getNow(null).kind(), "the genuine fleet_reply stands");
|
||||
assertEquals("a genuine fleet_reply landed first", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void exhaustionKeepsWinningOverBackendErrorEvenWhenBothPatternsMatchTheSameLine() {
|
||||
// A line matching both a configured exhausted pattern AND a configured backend-error pattern
|
||||
// must classify BACKEND_EXHAUSTED and call only the ExhaustionSink — the ordering fleetd#578
|
||||
// already relies on must not change.
|
||||
String block = "⏺ The usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup exhausted = target -> Pattern.compile("usage limit has been reached");
|
||||
BackendErrorPatternLookup backendErrors = target -> Pattern.compile("(?i)usage limit");
|
||||
java.util.List<String> exhaustedNotified = new java.util.ArrayList<>();
|
||||
java.util.List<String> backendErrorNotified = new java.util.ArrayList<>();
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> exhaustedNotified.add(target);
|
||||
BackendErrorSink backendErrorSink = (target, matchedLine, reason) -> backendErrorNotified.add(target);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
exhausted, exhaustionSink, backendErrors, backendErrorSink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"exhaustion must keep winning over backend error");
|
||||
assertEquals(1, exhaustedNotified.size(), "the exhaustion sink fires");
|
||||
assertTrue(backendErrorNotified.isEmpty(), "the backend-error sink must never fire for this turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConfiguredPatternAlsoClassifiesTheRawScrapeFallbackAndNotifiesTheSink() {
|
||||
// No ⏺ marker and leading TUI chrome ⇒ lastAssistantBlock() yields "", so classification must
|
||||
// fall back to the raw scrape (fleetd#211) — and it must use the configured pattern too.
|
||||
String block = """
|
||||
╭──────────────────────────────────────╮
|
||||
503 Service Unavailable: upstream credential rejected
|
||||
""";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a raw-scrape configured-pattern match still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"classified from the raw scrape even though the trimmed block was empty");
|
||||
assertEquals(1, notified.size(), "the sink fires exactly once from the raw-scrape path");
|
||||
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: classification inside the fleetd#164 MIN_TURN_NANOS floor -------------
|
||||
|
||||
@Test
|
||||
void aMatchingErrorInsideTheFloorIsTypedAndNotifiesTheSink() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals(1, notified.size(), "a matching error inside the floor notifies the sink");
|
||||
assertTrue(notified.get(0).contains("503 Service Unavailable"), notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchInsideTheFloorStaysGenericAndNeverNotifiesTheSink() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ still starting up\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
long[] clock = {10_000_000_000L};
|
||||
BackendErrorPatternLookup patterns = target -> Pattern.compile("(?i)503 Service Unavailable");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
BackendErrorSink sink = (target, matchedLine, reason) -> notified.add(target + ": " + matchedLine);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous,
|
||||
ExhaustedPatternLookup.none(), ExhaustionSink.none(), patterns, sink, () -> clock[0]);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null, clock[0]); // delivered "now"
|
||||
clock[0] += CompletionResolver.MIN_TURN_NANOS - 1; // inside the floor
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone());
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind(),
|
||||
"still a failure — the floor itself, not the pattern, is why");
|
||||
assertTrue(waiter.getNow(null).text().contains("too fast to be real work"),
|
||||
"a non-match inside the floor stays the generic too-fast reason: " + waiter.getNow(null).text());
|
||||
assertTrue(notified.isEmpty(), "a non-match must never notify the typed sink");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,10 +1,13 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
@@ -1024,6 +1027,87 @@ class FleetMcpTest {
|
||||
assertTrue(textOf(res).contains("architect, dev, reviewer"), textOf(res));
|
||||
}
|
||||
|
||||
// ── CB-619 / fleetd #123: a spawn asking for a role its profile has no slot for must be
|
||||
// refused, never silently demoted with the roster still lying about it ───────────────────
|
||||
|
||||
/** Two profiles on one launcher, so a role's pool can name one and exclude the other. */
|
||||
private static ClaudeCodeLauncher architectCapableLauncher(FakeHerdr h) {
|
||||
FleetConfig.Profile opus = new FleetConfig.Profile(
|
||||
"opus", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
FleetConfig.Profile sonnet = new FleetConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("opus", opus, "sonnet", sonnet),
|
||||
"sonnet", _ -> "tok");
|
||||
}
|
||||
|
||||
/** {@code fleet.architects} carries only {@code opus} — {@code sonnet} has no matching slot. */
|
||||
private static MemberRegistry architectRegistry() {
|
||||
return new MemberRegistry(new FleetConfig.Fleet(Map.of(),
|
||||
Map.of("opus", new FleetConfig.Slot("opus")), Map.of(), Map.of(), null));
|
||||
}
|
||||
|
||||
/**
|
||||
* The literal defect (fleetd #123): {@code role=architect, profile=sonnet}, where
|
||||
* {@code fleet.architects} carries only {@code opus}. Drives the real path —
|
||||
* {@link FleetMcp#spawn} calls {@link SessionManager#acquire}, which must refuse before ever
|
||||
* reaching the real {@link ClaudeCodeLauncher} — never {@link MemberRegistry#bind} called
|
||||
* directly, which would walk around the gate under test.
|
||||
*/
|
||||
@Test
|
||||
void spawnRefusesAnArchitectWithNoMatchingSlotAndNeverTouchesTheLauncher() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(architectCapableLauncher(h));
|
||||
sessions.setMemberLifecycle(architectRegistry());
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.spawn(sessions, "sonnet", "architect",
|
||||
null, null, null, null, null, null);
|
||||
|
||||
assertEquals(Boolean.TRUE, res.isError(), textOf(res));
|
||||
String msg = textOf(res);
|
||||
assertTrue(msg.contains("architect"), "names the role asked for: " + msg);
|
||||
assertTrue(msg.contains("sonnet"), "names the profile: " + msg);
|
||||
assertTrue(msg.contains("opus"), "names the pool that does carry the role: " + msg);
|
||||
assertTrue(sessions.roster().isEmpty(), "a refused spawn must register no session");
|
||||
assertTrue(h.calls.stream().noneMatch(c -> c.method().equals("agent.start")),
|
||||
"a refused spawn must never reach the launcher — no process should ever start");
|
||||
}
|
||||
|
||||
/**
|
||||
* Positive control / parity check: when the profile DOES carry a slot, the spawn succeeds, and
|
||||
* {@code GET /members} ({@link FleetMcp#listFleet}) and {@code fleet_whoami}
|
||||
* ({@link FleetMcp#whoami}) — resolved through the SAME live {@link MemberRegistry} binding via
|
||||
* a real {@link CallerResolver}, exactly as the daemon resolves a real MCP caller — must never
|
||||
* disagree about this one live member's role.
|
||||
*/
|
||||
@Test
|
||||
void rosterAndWhoamiAgreeOnceTheArchitectSlotBinds() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
// Pin the spawn onto the one pane FakeHerdr's canned pane.process_info maps to WORKER_PID,
|
||||
// so a CallerResolver can resolve THIS session's own terminal, not a fixture double.
|
||||
h.pinNextStarts(1, "term_a", "w2:p7");
|
||||
SessionManager sessions = new SessionManager(architectCapableLauncher(h));
|
||||
MemberRegistry members = architectRegistry();
|
||||
sessions.setMemberLifecycle(members);
|
||||
|
||||
McpSchema.CallToolResult spawnRes = FleetMcp.spawn(sessions, "opus", "architect",
|
||||
null, null, null, null, null, null);
|
||||
assertNotEquals(Boolean.TRUE, spawnRes.isError(), textOf(spawnRes));
|
||||
assertTrue(textOf(spawnRes).contains("\"role\":\"architect\""), textOf(spawnRes));
|
||||
|
||||
String roster = textOf(FleetMcp.listFleet(architectCapableLauncher(h), sessions, Map.of(), ""));
|
||||
assertTrue(roster.contains("\"role\":\"architect\""), "GET /members: " + roster);
|
||||
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(h), _ -> FakeHerdr.WORKER_PID);
|
||||
CallerResolver resolver = CallerResolver.withLeadsAndMembers(identity, false, null, Map::of, members);
|
||||
Principal caller = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertTrue(caller.isArchitect(), "fleet_whoami's own resolver must agree the terminal is bound");
|
||||
String whoami = textOf(FleetMcp.whoami(caller, sessions));
|
||||
assertTrue(whoami.contains("\"role\":\"architect\""), "fleet_whoami: " + whoami);
|
||||
}
|
||||
|
||||
// ── CB-584: fleet_spawn accepts sessionName/resumeSessionId; roster shows agentSessionId ──
|
||||
|
||||
@Test
|
||||
|
||||
@@ -20,8 +20,10 @@ import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
@@ -31,6 +33,7 @@ import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
import static org.junit.jupiter.api.Assumptions.assumeTrue;
|
||||
|
||||
/** The step-4 launch-flag injection: the bridge MCP + reply charter are appended to the argv. */
|
||||
class ClaudeCodeLauncherTest {
|
||||
@@ -61,8 +64,75 @@ class ClaudeCodeLauncherTest {
|
||||
assertTrue(args.stream().noneMatch(a -> a.contains("\"bridge\"")),
|
||||
"the mount is named fleet since CB-632 — a member addresses its tools as "
|
||||
+ "mcp__fleet__*, and CLAUDE.md's role-detection ladder names that prefix");
|
||||
assertTrue(args.contains("--append-system-prompt"));
|
||||
assertTrue(args.stream().anyMatch(a -> a.contains("fleet_reply")), "reply charter present");
|
||||
// fleetd #220: the charter travels as a FILE, never inline — charter prose on the command
|
||||
// line overruns the byte cap of the pane line herdr types it into.
|
||||
assertTrue(args.contains("--append-system-prompt-file"));
|
||||
assertFalse(args.contains("--append-system-prompt"),
|
||||
"the inline flag would put ~800 bytes of prose on the pane command line");
|
||||
String charterFile = args.get(args.indexOf("--append-system-prompt-file") + 1);
|
||||
assertTrue(readFile(charterFile).contains("fleet_reply"), "reply charter present in the file");
|
||||
}
|
||||
|
||||
/** Read a charter file the launcher wrote, failing the test rather than the build on an IO error. */
|
||||
private static String readFile(String path) {
|
||||
try {
|
||||
return java.nio.file.Files.readString(java.nio.file.Path.of(path));
|
||||
} catch (java.io.IOException e) {
|
||||
throw new AssertionError("charter file " + path + " is not readable", e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220 regression. herdr TYPES the launch command into the pane, and the pty line buffer
|
||||
* holds 1024 bytes — past that the tail is dropped with no error anywhere, so the backend exits
|
||||
* on a mangled argument and the spawn dies as an unexplained readiness timeout. That is what
|
||||
* happened when #214 added --session-id to a command already 978 bytes long: --autocompact
|
||||
* 250000 arrived as --autocompact 25. This asserts the whole assembled command still fits, with
|
||||
* the flags a real spawn carries (MCP mount, charter, model, autocompact, session id).
|
||||
*/
|
||||
@Test
|
||||
void theAssembledLaunchCommandFitsThePaneLineLimit() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// Shaped like the live sonnet profile, because the bug is a SUM: the charter alone fits,
|
||||
// and so does every flag alone. Only model + autocompact + session id on top of the charter
|
||||
// crossed the cap, which is why nothing caught it until a member failed to spawn.
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"sonnet", null, "claude-sonnet-5", null, null,
|
||||
List.of("claude"), "tab", "fleetd-workers", "w #{n}", "http://127.0.0.1:8765/mcp",
|
||||
null, null, null, null, null, null, null, null, true, null, null, null, null, null,
|
||||
250000);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of()), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null)
|
||||
.spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int bytes = "claude".length();
|
||||
for (String arg : args) {
|
||||
bytes += arg.getBytes(java.nio.charset.StandardCharsets.UTF_8).length + 3;
|
||||
}
|
||||
assertTrue(bytes <= HerdrPeerLauncher.PANE_COMMAND_BYTE_LIMIT,
|
||||
"the launch command must fit the pane line: " + bytes + " bytes vs limit "
|
||||
+ HerdrPeerLauncher.PANE_COMMAND_BYTE_LIMIT + " — args " + args);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #220: the guard refuses a command that cannot fit, instead of letting the pty drop the
|
||||
* tail. The refusal must name the size and the argument to blame — a spawn that fails with
|
||||
* "did not reach injectable state" tells the operator nothing, which is the whole reason this
|
||||
* bug took a live pane scrape to find.
|
||||
*/
|
||||
@Test
|
||||
void anOverlongLaunchCommandIsRefusedWithTheSizeAndTheCulprit() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
String huge = "x".repeat(1500);
|
||||
ClaudeCodeLauncher launcher = service(herdr, List.of("claude", huge), null);
|
||||
|
||||
PeerUnreachableException refused = assertThrows(PeerUnreachableException.class, launcher::spawn);
|
||||
|
||||
assertTrue(refused.getMessage().contains("1024"), "names the limit: " + refused.getMessage());
|
||||
assertTrue(refused.getMessage().contains("1500"), "names the culprit's size: " + refused.getMessage());
|
||||
assertFalse(herdr.called("agent.start"),
|
||||
"nothing may be started — a truncated command is worse than no spawn");
|
||||
}
|
||||
|
||||
// CB-634: a profile with ideMcpUrl set mounts the IDE Index MCP as a second server and pins
|
||||
@@ -262,10 +332,16 @@ class ClaudeCodeLauncherTest {
|
||||
|
||||
Map<?, ?> start = (Map<?, ?>) herdr.lastCall("agent.start").params();
|
||||
assertEquals("claude", start.get("kind"), "herdr launches the canonical executable by kind");
|
||||
// CB-533: the shared fixture pins model "coder", so the model flag is the whole args list.
|
||||
// What this test guards is that argv[0] is NOT repeated — herdr supplies it from `kind`.
|
||||
assertEquals(List.of("--model", "coder"), start.get("args"),
|
||||
"the configured executable is not repeated in args");
|
||||
// fleetd #214: a plain spawn now also carries fleetd's own minted session id.
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("claude"), "the configured executable is not repeated in args");
|
||||
int flag = args.indexOf("--session-id");
|
||||
assertTrue(flag >= 0, "a plain spawn mints a session id: " + args);
|
||||
assertDoesNotThrow(() -> UUID.fromString(args.get(flag + 1)), "the minted id is a valid UUID");
|
||||
assertEquals(List.of("--model", "coder"),
|
||||
List.of(args.get(args.size() - 2), args.get(args.size() - 1)),
|
||||
"the CB-533 model flag still trails the launch flags: " + args);
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -278,7 +354,15 @@ class ClaudeCodeLauncherTest {
|
||||
assertFalse(args.contains("--append-system-prompt"), "no reply charter without mcpUrl");
|
||||
// CB-533: the model flag is independent of the MCP mount — pinning the model is not part of
|
||||
// "mount the bridge", so an unmounted worker still runs the model its profile names.
|
||||
assertEquals(List.of("--verbose", "--model", "coder"), args,
|
||||
// fleetd #214: a plain spawn now carries fleetd's own minted session id; drop the two
|
||||
// mint elements when checking the rest of the argv.
|
||||
int flag = args.indexOf("--session-id");
|
||||
assertTrue(flag >= 0, "a plain spawn mints a session id even without a bridge mount: " + args);
|
||||
assertDoesNotThrow(() -> UUID.fromString(args.get(flag + 1)), "the minted id is a valid UUID");
|
||||
List<String> rest = new java.util.ArrayList<>(args);
|
||||
rest.remove(flag + 1);
|
||||
rest.remove(flag);
|
||||
assertEquals(List.of("--verbose", "--model", "coder"), rest,
|
||||
"the operator's own args are preserved, in order, ahead of the model flag");
|
||||
}
|
||||
|
||||
@@ -649,7 +733,7 @@ class ClaudeCodeLauncherTest {
|
||||
assertEquals("/work/proj", cwd, "effectiveCwd via SpawnRequest must match the three-arg resolution");
|
||||
}
|
||||
|
||||
// --- CB-547a: durable session identity (mint / resume / no-identity legacy) -----------------
|
||||
// --- CB-547a / fleetd #214: durable session identity (always mint / resume) -----------------
|
||||
|
||||
@Test
|
||||
void freshSpawnMintsASessionIdAndPassesTheName() {
|
||||
@@ -684,18 +768,28 @@ class ClaudeCodeLauncherTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void noIdentitySpawnKeepsTheLegacyArgvAndCarriesNoSessionHandle() {
|
||||
void plainSpawnMintsASessionIdSoEveryMemberIsResumable() {
|
||||
// fleetd #214: a plain spawn passes no sessionName and no resumeSessionId, yet the member
|
||||
// must still be resumable — the id is minted unconditionally, and it is the ONLY resume
|
||||
// handle a claude-code member has (unlike opencode, nothing resolves it after the launch).
|
||||
// The binary requires a valid UUID (checked against claude 2.1.252: a non-UUID is refused
|
||||
// at argument parsing with "Invalid session ID. Must be a valid UUID.").
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, null));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("--session-id"), "no identity → no --session-id");
|
||||
assertFalse(args.contains("-n"), "no identity → no -n");
|
||||
assertFalse(args.contains("-r"), "no identity → no -r");
|
||||
assertNull(handle.agentSessionId(), "no identity → no resume handle");
|
||||
assertNull(handle.sessionName(), "no identity → no logical name");
|
||||
int flag = args.indexOf("--session-id");
|
||||
assertTrue(flag >= 0 && flag + 1 < args.size(),
|
||||
"--session-id is minted even when no identity is requested: " + args);
|
||||
String minted = args.get(flag + 1);
|
||||
assertDoesNotThrow(() -> UUID.fromString(minted), "--session-id is a valid UUID: " + minted);
|
||||
assertEquals(minted, handle.agentSessionId(),
|
||||
"the resume handle is the minted id, so every member is resumable from fleet_list");
|
||||
assertFalse(args.contains("-n"), "no sessionName was requested → no -n");
|
||||
assertFalse(args.contains("-r"), "no resume was requested → no -r");
|
||||
assertNull(handle.sessionName(), "no sessionName was requested → no logical name");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -869,6 +963,152 @@ class ClaudeCodeLauncherTest {
|
||||
assertDoesNotThrow(() -> UUID.fromString(handle.id()));
|
||||
}
|
||||
|
||||
// --- fleetd #176 fix 1: fail fast when the backend process exits mid-wait -------------------
|
||||
|
||||
@Test
|
||||
void spawnFailsFastWhenBackendProcessExitsMidWaitInsteadOfBurningTheTimeout() {
|
||||
// First agent.get sees UNKNOWN (one normal tick); the second reports the pane gone, exactly
|
||||
// what herdr answers when the backend process has already exited. The timeout is generous
|
||||
// (60s) so a test that wrongly falls through to the old unguarded call — and therefore
|
||||
// waits out the whole window — is unambiguously distinguishable from one that fails fast.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown");
|
||||
herdr.agentGetFailsWithAfter(1, "pane_not_found");
|
||||
long[] clock = {0};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
60_000, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(
|
||||
PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertTrue(ex.getMessage().contains("w9:pRoot_1"),
|
||||
"exception message references the paneId: " + ex.getMessage());
|
||||
assertTrue(ex.getMessage().toLowerCase().contains("exited"),
|
||||
"exception message says the backend exited, not that the pane was slow: "
|
||||
+ ex.getMessage());
|
||||
assertTrue(clock[0] < 60_000,
|
||||
"the gate must not burn the rest of the 60s timeout: clock only reached " + clock[0]);
|
||||
long getCalls = herdr.calls.stream().filter(c -> c.method().equals("agent.get")).count();
|
||||
assertEquals(2, getCalls,
|
||||
"exactly one normal poll then the not_found answer — no further polling after that: "
|
||||
+ getCalls);
|
||||
assertEquals(1, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"the pane is torn down (no orphan) even on the fail-fast path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnLetsAnUnrelatedHerdrErrorPropagateUnchanged() {
|
||||
// Fix 1 must only special-case a "*_not_found" answer. Any other herdr failure keeps
|
||||
// propagating as-is — this gate does not know how to recover from it.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown");
|
||||
herdr.agentGetFailsWithAfter(0, "internal_error");
|
||||
long[] clock = {0};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
60_000, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
dev.ltms.fleet.herdr.HerdrException ex = assertThrows(
|
||||
dev.ltms.fleet.herdr.HerdrException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertEquals("internal_error", ex.code());
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"an error this gate does not recognize is not this gate's teardown to run");
|
||||
}
|
||||
|
||||
// --- fleetd #176 fix 2: corroborated UNKNOWN refinement --------------------------------------
|
||||
|
||||
@Test
|
||||
void refinedIdleIsNotAcceptedWhenAgentTypeIsNull() {
|
||||
// The trap fix 2 must close: a pane sitting at a bare shell prompt after its backend exited
|
||||
// still contains the same "❯" glyph StatusRefiner.classify treats as "idle at the Claude
|
||||
// Code TUI prompt". Without the agentType corroboration this would be misread as injectable
|
||||
// and the gate would hand back a peer that never started. herdr reports no agentType for
|
||||
// that bare shell.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // always UNKNOWN
|
||||
herdr.agentType(null); // no supported backend detected — could be a bare shell
|
||||
herdr.readText("some-host:~ user$ ❯ "); // looks exactly like an idle Claude Code prompt
|
||||
long[] clock = {0};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
500, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(
|
||||
PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertTrue(clock[0] >= 500,
|
||||
"a null agentType must not let the '❯' prompt refine to injectable — the gate has "
|
||||
+ "to wait out the full timeout: clock only reached " + clock[0]);
|
||||
assertEquals(1, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"the pane is torn down on timeout, same as any other never-injectable spawn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void refinedIdleIsAcceptedWhenAgentTypeCorroboratesLiveness() {
|
||||
// The positive case: a genuinely live claude pane that herdr misreports as UNKNOWN (CB-115)
|
||||
// still resolves to injectable once agentType corroborates that a supported backend is
|
||||
// detected in the same sample.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // always UNKNOWN at the raw level
|
||||
herdr.agentType("claude"); // herdr still detects a live claude backend
|
||||
herdr.readText("? for shortcuts"); // StatusRefiner.classify's idle footer marker
|
||||
long[] clock = {0};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
60_000, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "a corroborated refined-IDLE sample lets the spawn succeed");
|
||||
assertTrue(clock[0] < 60_000,
|
||||
"refinement must resolve well before the timeout: clock reached " + clock[0]);
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"no pane close — the peer is genuinely injectable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void refinementNeverFiresWhenRawStatusIsAlreadyInjectable() {
|
||||
// Acceptance criterion 4: refinement must trigger only on a raw UNKNOWN sample. Seed the
|
||||
// pane content with an active-generation marker that StatusRefiner.classify would read as
|
||||
// WORKING (never injectable) if refine() were wrongly invoked here — proving that a raw
|
||||
// IDLE status short-circuits before refine() (and its extra agent.read call) ever runs.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("idle"); // already injectable at the raw level
|
||||
herdr.agentType("claude");
|
||||
herdr.readText("esc to interrupt"); // would classify as WORKING if refine() ran anyway
|
||||
|
||||
long[] clock = {0};
|
||||
ClaudeCodeLauncher gated = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
5000, () -> clock[0], () -> clock[0] += 300);
|
||||
|
||||
PeerHandle handle = gated.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "an already-injectable raw status succeeds without ever refining");
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"refine() must never run (and so never call agent.read) when the raw status is "
|
||||
+ "already injectable");
|
||||
}
|
||||
|
||||
// --- CB-511: worker environment seeding -----------------------------------------------------
|
||||
|
||||
@Test
|
||||
@@ -1660,4 +1900,184 @@ class ClaudeCodeLauncherTest {
|
||||
|
||||
assertEquals(List.of("dev: sonnet #1", "[sonnet] dev 2"), tabLabels(herdr));
|
||||
}
|
||||
|
||||
// ── fleetd #222: the charter file must not land under fleetd's own java.io.tmpdir when the ──
|
||||
// ── member pane runs as a different OS user ─────────────────────────────────────────────────
|
||||
|
||||
/** A profile that mounts the bridge MCP (so a reply charter is always generated). */
|
||||
private static FleetConfig.Profile charterCfg() {
|
||||
return new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "fleetd-workers", "worker: {profile} #{n}",
|
||||
"http://127.0.0.1:8765/mcp", null, null);
|
||||
}
|
||||
|
||||
/** A config with {@code memberHerdrSocket:} set, and optionally {@code worktreeRoot:}/{@code worktreeGroup:}. */
|
||||
private static FleetConfig configWithMemberHerdrSocket(String worktreeRoot, String worktreeGroup) {
|
||||
return new FleetConfig(
|
||||
null, // bind
|
||||
null, // herdrSocket
|
||||
"/tmp/other-user.sock", // memberHerdrSocket
|
||||
Map.of(), // profiles
|
||||
null, // guard
|
||||
worktreeRoot, // worktreeRoot
|
||||
null, // lifecycle
|
||||
null, // spawnReadyTimeoutMs
|
||||
null, // spawnReadyPollMs
|
||||
null, // broker
|
||||
null, // primary
|
||||
null, // fleet
|
||||
null, // leadHeartbeat
|
||||
null, // health
|
||||
null, // placement
|
||||
null, // auth
|
||||
null, // configReload
|
||||
null, // quarantineCooldownSeconds
|
||||
null, // memberCredentials
|
||||
null, // coordinator
|
||||
worktreeGroup, // worktreeGroup
|
||||
null // memberLoginShell
|
||||
).withDefaults();
|
||||
}
|
||||
|
||||
private static ClaudeCodeLauncher serviceWithConfig(FakeHerdr herdr, FleetConfig.Profile cfg,
|
||||
Supplier<FleetConfig> config) {
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
0, System::currentTimeMillis, () -> { }, null, null, null, config);
|
||||
}
|
||||
|
||||
/**
|
||||
* The current process's REAL primary group — resolved via {@code id -gn}, never by reading a
|
||||
* directory's owning group (fleetd #225). Those two coincide only by accident: a directory's
|
||||
* group is whichever group happened to own the path Maven was started from — {@code staff} in
|
||||
* a home checkout, {@code wheel} under {@code /private/tmp} on macOS — and the fix-up this test
|
||||
* exercises then fails for real when the operator is not a member of that borrowed group,
|
||||
* exactly the case {@code assumeTrue(view != null, ...)} never covered (it only detects a
|
||||
* filesystem with no POSIX groups at all, not a resolvable-but-wrong one). Skips (never fails)
|
||||
* when {@code id} is unavailable or its primary group cannot be resolved on this host.
|
||||
*/
|
||||
private static String currentUserGroup() {
|
||||
String out;
|
||||
boolean ok;
|
||||
try {
|
||||
Process p = new ProcessBuilder("id", "-gn").redirectErrorStream(true).start();
|
||||
try (java.io.BufferedReader r = new java.io.BufferedReader(
|
||||
new java.io.InputStreamReader(p.getInputStream(), java.nio.charset.StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(java.util.stream.Collectors.joining("\n")).trim();
|
||||
}
|
||||
ok = p.waitFor(5, java.util.concurrent.TimeUnit.SECONDS) && p.exitValue() == 0 && !out.isBlank();
|
||||
} catch (IOException e) {
|
||||
out = null;
|
||||
ok = false;
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
out = null;
|
||||
ok = false;
|
||||
}
|
||||
assumeTrue(ok, "cannot resolve this process's real primary group via `id -gn` on this host "
|
||||
+ "— skipping a POSIX-group-dependent test rather than failing it");
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #222 acceptance criterion 1: with {@code memberHerdrSocket} configured and both
|
||||
* {@code worktreeRoot}/{@code worktreeGroup} set, the charter file lives in a fresh per-spawn
|
||||
* directory under {@code worktreeRoot} — NEVER under {@code java.io.tmpdir} (fleetd's own 0700
|
||||
* temp dir, unreadable by the member's different OS user) — and that directory is shared
|
||||
* read-only with the group via the SAME mechanism (fleetd #213/#219's {@link
|
||||
* EnvAllowListScrub#shareWithGroup}) the ZDOTDIR scrub and the opencode config directory use.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithWorktreeRootAndGroupPutsCharterUnderWorktreeRootAndSharesIt(
|
||||
@TempDir Path worktreeRoot) throws Exception {
|
||||
String group = currentUserGroup();
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithConfig(herdr, charterCfg(),
|
||||
() -> configWithMemberHerdrSocket(worktreeRoot.toString(), group)).spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int fileFlag = args.indexOf("--append-system-prompt-file");
|
||||
assertTrue(fileFlag >= 0, "the charter is still mounted via file: " + args);
|
||||
Path charterFile = Path.of(args.get(fileFlag + 1));
|
||||
Path dir = charterFile.getParent();
|
||||
|
||||
// NOTE: JUnit's own @TempDir provider places worktreeRoot itself under java.io.tmpdir on this
|
||||
// host, so "not under java.io.tmpdir" is not a meaningful assertion here (it would hold by
|
||||
// accident of the fixture, not by anything this method does). What this fix actually promises
|
||||
// is that the directory is created UNDER worktreeRoot specifically — never resolved from the
|
||||
// no-argument Files.createTempFile default (java.io.tmpdir) the pre-fix code always used — so
|
||||
// that is the assertion: the parent is exactly worktreeRoot, whatever directory JUnit gave it.
|
||||
assertEquals(worktreeRoot.toAbsolutePath().normalize(), dir.getParent(),
|
||||
"the generated directory's parent must be worktreeRoot, not java.io.tmpdir — got "
|
||||
+ "parent " + dir.getParent());
|
||||
|
||||
assertEquals("rwxr-x---", PosixFilePermissions.toString(Files.getPosixFilePermissions(dir)),
|
||||
"the directory must be group-traversable+readable, owner-only writable");
|
||||
assertEquals("rw-r-----", PosixFilePermissions.toString(Files.getPosixFilePermissions(charterFile)),
|
||||
"the charter file must be group-readable, never group-writable");
|
||||
assertTrue(Files.readString(charterFile).contains("fleet_reply"),
|
||||
"the charter content itself is unaffected by where it is written");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #222 acceptance criterion 2: with {@code memberHerdrSocket} configured but NEITHER
|
||||
* {@code worktreeRoot} nor {@code worktreeGroup} set, the launcher must refuse the spawn rather
|
||||
* than hand the member a {@code --append-system-prompt-file} path under {@code java.io.tmpdir}
|
||||
* it cannot read — the member's whole turn contract would never reach it.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithoutWorktreeRootOrGroupRefusesTheSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher launcher = serviceWithConfig(herdr, charterCfg(),
|
||||
() -> configWithMemberHerdrSocket(null, null));
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class, launcher::spawn,
|
||||
"a missing worktreeRoot/worktreeGroup must refuse the spawn, not write an unreadable charter");
|
||||
assertTrue(ex.getMessage().contains("worktreeRoot"),
|
||||
"the refusal must name the missing config key — got: " + ex.getMessage());
|
||||
assertFalse(herdr.called("agent.start"),
|
||||
"the spawn must be refused BEFORE the member is ever started — got calls: " + herdr.calls);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #222 acceptance criterion 2 (the other missing half): {@code worktreeRoot} set but
|
||||
* {@code worktreeGroup} missing must ALSO refuse — either one alone is not enough to guarantee
|
||||
* the member's OS user can read the charter file.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithWorktreeRootButNoGroupRefusesTheSpawn(@TempDir Path worktreeRoot) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher launcher = serviceWithConfig(herdr, charterCfg(),
|
||||
() -> configWithMemberHerdrSocket(worktreeRoot.toString(), null));
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class, launcher::spawn);
|
||||
assertTrue(ex.getMessage().contains("worktreeGroup"),
|
||||
"worktreeRoot alone is not enough — got: " + ex.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #222 acceptance criterion 3: with {@code memberHerdrSocket} ABSENT — even when a live,
|
||||
* non-null {@code config} supplier is threaded through (not merely {@code config == null}, which
|
||||
* every other test in this file already exercises) — the charter file must still be created
|
||||
* directly under {@code java.io.tmpdir} via the same no-directory-argument
|
||||
* {@code Files.createTempFile} call as before this fix, byte-identical to today.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketAbsentStaysUnderJavaIoTmpdirEvenWithALiveConfigSupplier() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig config = new FleetConfig(null, null, null, Map.of(), null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null, null, null, null, null).withDefaults();
|
||||
serviceWithConfig(herdr, charterCfg(), () -> config).spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int fileFlag = args.indexOf("--append-system-prompt-file");
|
||||
assertTrue(fileFlag >= 0);
|
||||
Path charterFile = Path.of(args.get(fileFlag + 1));
|
||||
assertTrue(charterFile.startsWith(Path.of(System.getProperty("java.io.tmpdir"))),
|
||||
"with memberHerdrSocket absent, the charter file must still land directly under "
|
||||
+ "java.io.tmpdir, unchanged from before this fix");
|
||||
assertTrue(charterFile.getFileName().toString().startsWith("fleetd-role-charter-"),
|
||||
"same file-naming scheme as before this fix (no wrapping directory): " + charterFile);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -114,6 +114,68 @@ class EnvAllowListScrubTest {
|
||||
assertNull(EnvAllowListScrub.readReport(dir));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #224 criterion 5 (a gap the #221 reviewer flagged): {@link EnvAllowListScrub#shareWithGroup}
|
||||
* lists one FLAT level of {@code dir} and shares every file it finds there — which does cover the
|
||||
* OPTIONAL files a caller may or may not have written before calling it ({@code member-charter.md},
|
||||
* {@code ide-rules.md} — both written by {@code OpenCodeLauncher#writeConfig}), but nothing
|
||||
* pinned that down.
|
||||
* Without this test, a future change that writes a file AFTER the sharing call, or into a
|
||||
* subdirectory, would pass every existing test while quietly leaving that file unreadable to a
|
||||
* different-uid member.
|
||||
*
|
||||
* <p>This drives {@code shareWithGroup} directly against a directory holding several flat files —
|
||||
* not only {@code opencode.json}, but also the two optional ones named above plus a third,
|
||||
* unrelated file, so the assertion is "every flat file", not "the two files someone thought of".
|
||||
*/
|
||||
@Test
|
||||
void shareWithGroupCoversEveryFlatFileIncludingTheOptionalOnes(@TempDir Path dir) throws Exception {
|
||||
String group = currentUserGroup();
|
||||
Files.writeString(dir.resolve("opencode.json"), "{}\n");
|
||||
Files.writeString(dir.resolve("member-charter.md"), "# charter\n");
|
||||
Files.writeString(dir.resolve("ide-rules.md"), "# ide rules\n");
|
||||
Files.writeString(dir.resolve("another-flat-file.txt"), "unrelated\n");
|
||||
|
||||
EnvAllowListScrub.shareWithGroup(dir, group);
|
||||
|
||||
assertEquals("rwxr-x---", java.nio.file.attribute.PosixFilePermissions.toString(
|
||||
Files.getPosixFilePermissions(dir)),
|
||||
"the directory itself must be group-traversable+readable, owner-only writable");
|
||||
for (String name : List.of("opencode.json", "member-charter.md", "ide-rules.md", "another-flat-file.txt")) {
|
||||
Path file = dir.resolve(name);
|
||||
assertEquals("rw-r-----", java.nio.file.attribute.PosixFilePermissions.toString(
|
||||
Files.getPosixFilePermissions(file)),
|
||||
name + " must be group-readable, never group-writable");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The current process's REAL primary group — resolved via {@code id -gn}, never by reading a
|
||||
* directory's owning group (fleetd #225: that reads wherever Maven happened to be started from,
|
||||
* not the process's own group, and the two diverge outside a home checkout). Skips (never fails)
|
||||
* when {@code id} is unavailable or its primary group cannot be resolved on this host.
|
||||
*/
|
||||
private static String currentUserGroup() {
|
||||
String out;
|
||||
boolean ok;
|
||||
try {
|
||||
Process p = new ProcessBuilder("id", "-gn").redirectErrorStream(true).start();
|
||||
String raw = new String(p.getInputStream().readAllBytes(), StandardCharsets.UTF_8).trim();
|
||||
out = raw;
|
||||
ok = p.waitFor(5, java.util.concurrent.TimeUnit.SECONDS) && p.exitValue() == 0 && !raw.isBlank();
|
||||
} catch (IOException e) {
|
||||
out = null;
|
||||
ok = false;
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
out = null;
|
||||
ok = false;
|
||||
}
|
||||
assumeTrue(ok, "cannot resolve this process's real primary group via `id -gn` on this host "
|
||||
+ "— skipping a POSIX-group-dependent test rather than failing it");
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* The same equality, for a shell that is INTERACTIVE but NOT a login shell — the shape herdr
|
||||
* opens on Linux.
|
||||
|
||||
+31
-6
@@ -18,7 +18,6 @@ import org.slf4j.LoggerFactory;
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -513,11 +512,37 @@ class HerdrPeerLauncherAllowListWiringTest {
|
||||
+ "java.io.tmpdir, unchanged from before this fix: " + dir);
|
||||
}
|
||||
|
||||
/** The current process's own primary group — resolvable on whatever host runs this test. */
|
||||
private static String currentUserGroup() throws IOException {
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(Path.of("."), PosixFileAttributeView.class);
|
||||
assumeTrue(view != null, "this host's filesystem does not support POSIX group ownership");
|
||||
return view.readAttributes().group().getName();
|
||||
/**
|
||||
* The current process's REAL primary group — resolved via {@code id -gn}, never by reading a
|
||||
* directory's owning group (fleetd #225). Those two coincide only by accident: a directory's
|
||||
* group is whichever group happened to own the path Maven was started from — {@code staff} in
|
||||
* a home checkout, {@code wheel} under {@code /private/tmp} on macOS — and the fix-up this test
|
||||
* exercises then fails for real when the operator is not a member of that borrowed group,
|
||||
* exactly the case {@code assumeTrue(view != null, ...)} never covered (it only detects a
|
||||
* filesystem with no POSIX groups at all, not a resolvable-but-wrong one). Skips (never fails)
|
||||
* when {@code id} is unavailable or its primary group cannot be resolved on this host.
|
||||
*/
|
||||
private static String currentUserGroup() {
|
||||
String out;
|
||||
boolean ok;
|
||||
try {
|
||||
Process p = new ProcessBuilder("id", "-gn").redirectErrorStream(true).start();
|
||||
try (java.io.BufferedReader r = new java.io.BufferedReader(
|
||||
new java.io.InputStreamReader(p.getInputStream(), java.nio.charset.StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(java.util.stream.Collectors.joining("\n")).trim();
|
||||
}
|
||||
ok = p.waitFor(5, java.util.concurrent.TimeUnit.SECONDS) && p.exitValue() == 0 && !out.isBlank();
|
||||
} catch (IOException e) {
|
||||
out = null;
|
||||
ok = false;
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
out = null;
|
||||
ok = false;
|
||||
}
|
||||
assumeTrue(ok, "cannot resolve this process's real primary group via `id -gn` on this host "
|
||||
+ "— skipping a POSIX-group-dependent test rather than failing it");
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Spawn once through the real launcher path, capturing every INFO+ line this class logs. */
|
||||
|
||||
@@ -1,30 +1,43 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.Future;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
import static org.junit.jupiter.api.Assumptions.assumeTrue;
|
||||
|
||||
/**
|
||||
* The opencode adapter's launch build: a file-based MCP mount + reply-charter instructions (no
|
||||
@@ -383,6 +396,46 @@ class OpenCodeLauncherTest {
|
||||
assertNotNull(ex.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #176 fix 2's per-adapter guard: {@code StatusRefiner.classify} is written for the
|
||||
* Claude Code TUI only, and the spawn-readiness gate must never run it against an opencode
|
||||
* pane. {@code agentType("opencode")} deliberately satisfies the OTHER guard (the corroborating
|
||||
* liveness check) so it cannot be what makes this test pass — only the {@code namePrefix}
|
||||
* check can be. If that check were ever removed, this pane's {@code ❯} content would refine
|
||||
* straight to IDLE and the gate would report ready before the backend actually was.
|
||||
*/
|
||||
@Test
|
||||
void opencodePaneIsNeverRefinedEvenWhenItsContentLooksLikeAnIdleClaudePrompt(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // always UNKNOWN
|
||||
herdr.agentType("opencode"); // non-null — satisfies the liveness guard on its own
|
||||
herdr.readText("some-host:~ user$ ❯ "); // content StatusRefiner.classify reads as IDLE
|
||||
long[] clock = {0};
|
||||
OpenCodeLauncher svc = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", opencodeCfg(null, null, null)), "gemini", _ -> null,
|
||||
1000, () -> clock[0], () -> clock[0] += 50, root, root);
|
||||
|
||||
assertThrows(PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertTrue(clock[0] >= 1000,
|
||||
"an opencode pane must never refine to injectable, however its content reads — "
|
||||
+ "the gate has to wait out the full timeout: clock only reached " + clock[0]);
|
||||
// A blanket "agent.read is never called" does not hold here: the timeout path itself reads
|
||||
// the pane tail (source=recent) for its own log message, on every timeout, regardless of
|
||||
// adapter — see HerdrPeerLauncher.readPaneQuietly. So assert on the refiner's OWN probe
|
||||
// source (StatusRefiner.PROBE_SOURCE = "detection") instead — that call happens only inside
|
||||
// StatusRefiner.refine, so its absence proves the refiner itself was never reached for this
|
||||
// opencode pane, not merely that its answer was discarded.
|
||||
long detectionReads = herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("agent.read"))
|
||||
.filter(c -> "detection".equals(((Map<?, ?>) c.params()).get("source")))
|
||||
.count();
|
||||
assertEquals(0, detectionReads,
|
||||
"the refiner's own pane probe (source=detection) must never run against an "
|
||||
+ "opencode pane — the namePrefix guard has to stop it before that call");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsHandleWhenGateDisabled(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -610,4 +663,474 @@ class OpenCodeLauncherTest {
|
||||
assertTrue(json.path("mcp").path("intellij").isMissingNode(),
|
||||
"no IDE server when ideMcpUrl is unset");
|
||||
}
|
||||
|
||||
// --- fleetd #219: config root + discovery root under memberHerdrSocket ------------------------
|
||||
|
||||
/** A config with {@code memberHerdrSocket:} set, and optionally {@code worktreeRoot:}/{@code worktreeGroup:}. */
|
||||
private static FleetConfig configWithMemberHerdrSocket(String worktreeRoot, String worktreeGroup) {
|
||||
return new FleetConfig(
|
||||
null, // bind
|
||||
null, // herdrSocket
|
||||
"/tmp/other-user.sock", // memberHerdrSocket
|
||||
Map.of(), // profiles
|
||||
null, // guard
|
||||
worktreeRoot, // worktreeRoot
|
||||
null, // lifecycle
|
||||
null, // spawnReadyTimeoutMs
|
||||
null, // spawnReadyPollMs
|
||||
null, // broker
|
||||
null, // primary
|
||||
null, // fleet
|
||||
null, // leadHeartbeat
|
||||
null, // health
|
||||
null, // placement
|
||||
null, // auth
|
||||
null, // configReload
|
||||
null, // quarantineCooldownSeconds
|
||||
null, // memberCredentials
|
||||
null, // coordinator
|
||||
worktreeGroup, // worktreeGroup
|
||||
null // memberLoginShell
|
||||
).withDefaults();
|
||||
}
|
||||
|
||||
private static OpenCodeLauncher serviceWithConfig(FakeHerdr herdr, Path configRoot, Path discoveryRoot,
|
||||
FleetConfig.Profile cfg, Supplier<FleetConfig> config) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, discoveryRoot, null, null, config);
|
||||
}
|
||||
|
||||
/**
|
||||
* The current process's REAL primary group — resolved via {@code id -gn}, never by reading a
|
||||
* directory's owning group (fleetd #225). Those two coincide only by accident: a directory's
|
||||
* group is whichever group happened to own the path Maven was started from — {@code staff} in
|
||||
* a home checkout, {@code wheel} under {@code /private/tmp} on macOS — and the fix-up this test
|
||||
* exercises then fails for real when the operator is not a member of that borrowed group,
|
||||
* exactly the case {@code assumeTrue(view != null, ...)} never covered (it only detects a
|
||||
* filesystem with no POSIX groups at all, not a resolvable-but-wrong one). Skips (never fails)
|
||||
* when {@code id} is unavailable or its primary group cannot be resolved on this host.
|
||||
*/
|
||||
private static String currentUserGroup() {
|
||||
String out;
|
||||
boolean ok;
|
||||
try {
|
||||
Process p = new ProcessBuilder("id", "-gn").redirectErrorStream(true).start();
|
||||
try (java.io.BufferedReader r = new java.io.BufferedReader(
|
||||
new java.io.InputStreamReader(p.getInputStream(), java.nio.charset.StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(java.util.stream.Collectors.joining("\n")).trim();
|
||||
}
|
||||
ok = p.waitFor(5, java.util.concurrent.TimeUnit.SECONDS) && p.exitValue() == 0 && !out.isBlank();
|
||||
} catch (IOException e) {
|
||||
out = null;
|
||||
ok = false;
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
out = null;
|
||||
ok = false;
|
||||
}
|
||||
assumeTrue(ok, "cannot resolve this process's real primary group via `id -gn` on this host "
|
||||
+ "— skipping a POSIX-group-dependent test rather than failing it");
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 1, acceptance criterion 1: with {@code memberHerdrSocket} configured and both
|
||||
* {@code worktreeRoot}/{@code worktreeGroup} set, the generated {@code opencode.json} directory
|
||||
* lives under {@code worktreeRoot} — NEVER under the injected {@code configRoot} (standing in for
|
||||
* {@code java.io.tmpdir}, fleetd's own 0700 temp dir, unreadable by the member's different OS
|
||||
* user) — and is shared read-only with the group via the SAME mechanism (fleetd #213's {@link
|
||||
* EnvAllowListScrub#shareWithGroup}) the ZDOTDIR scrub uses.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithWorktreeRootAndGroupPutsConfigDirUnderWorktreeRootAndSharesIt(
|
||||
@TempDir Path configRoot, @TempDir Path worktreeRoot) throws Exception {
|
||||
String group = currentUserGroup();
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithConfig(herdr, configRoot, configRoot,
|
||||
opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null),
|
||||
() -> configWithMemberHerdrSocket(worktreeRoot.toString(), group)).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "the profile still needs a config file");
|
||||
Path cfgFile = Path.of(cfgPath);
|
||||
Path dir = cfgFile.getParent();
|
||||
assertEquals(worktreeRoot.toAbsolutePath().normalize(), dir.getParent(),
|
||||
"the generated directory's parent must be worktreeRoot, not the injected configRoot "
|
||||
+ "standing in for java.io.tmpdir — got parent " + dir.getParent());
|
||||
assertFalse(dir.startsWith(configRoot),
|
||||
"the generated directory must NOT be created under configRoot when memberHerdrSocket "
|
||||
+ "is configured: " + dir);
|
||||
|
||||
assertEquals("rwxr-x---", PosixFilePermissions.toString(Files.getPosixFilePermissions(dir)),
|
||||
"the directory must be group-traversable+readable, owner-only writable");
|
||||
assertEquals("rw-r-----", PosixFilePermissions.toString(Files.getPosixFilePermissions(cfgFile)),
|
||||
"opencode.json must be group-readable, never group-writable");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 1, acceptance criterion 2: with {@code memberHerdrSocket} configured but
|
||||
* NEITHER {@code worktreeRoot} nor {@code worktreeGroup} set, the launcher must refuse the spawn
|
||||
* rather than write a config under {@code java.io.tmpdir} the member cannot read — that member
|
||||
* would occupy a pane and never become deliverable, with nothing pointing at the real cause.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithoutWorktreeRootOrGroupRefusesTheSpawn(@TempDir Path configRoot) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = serviceWithConfig(herdr, configRoot, configRoot,
|
||||
opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null),
|
||||
() -> configWithMemberHerdrSocket(null, null));
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class,
|
||||
() -> launcher.spawn(new SpawnRequest(null, null, null)),
|
||||
"a missing worktreeRoot/worktreeGroup must refuse the spawn, not write an unreadable config");
|
||||
assertTrue(ex.getMessage().contains("worktreeRoot"),
|
||||
"the refusal must name the missing config key — got: " + ex.getMessage());
|
||||
assertFalse(herdr.called("tab.create"),
|
||||
"the spawn must be refused BEFORE any pane is created — got calls: " + herdr.calls);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 1, acceptance criterion 2 (the other missing half): {@code worktreeRoot} set
|
||||
* but {@code worktreeGroup} missing must ALSO refuse — either one alone is not enough to
|
||||
* guarantee the member's OS user can read the directory.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketWithWorktreeRootButNoGroupRefusesTheSpawn(
|
||||
@TempDir Path configRoot, @TempDir Path worktreeRoot) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = serviceWithConfig(herdr, configRoot, configRoot,
|
||||
opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null),
|
||||
() -> configWithMemberHerdrSocket(worktreeRoot.toString(), null));
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class,
|
||||
() -> launcher.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(ex.getMessage().contains("worktreeGroup"),
|
||||
"worktreeRoot alone is not enough — got: " + ex.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 1, acceptance criterion 3: with {@code memberHerdrSocket} ABSENT — even when
|
||||
* a live, non-null {@code config} supplier is threaded through (not merely {@code config == null},
|
||||
* which every other test in this file already exercises) — the generated directory must still
|
||||
* land directly under the injected {@code configRoot}, byte-identical to before this fix.
|
||||
*/
|
||||
@Test
|
||||
void memberHerdrSocketAbsentStaysUnderConfigRootEvenWithALiveConfigSupplier(@TempDir Path configRoot)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig config = new FleetConfig(null, null, null, Map.of(), null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null, null, null, null, null).withDefaults();
|
||||
serviceWithConfig(herdr, configRoot, configRoot,
|
||||
opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null), () -> config)
|
||||
.spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath);
|
||||
assertTrue(Path.of(cfgPath).startsWith(configRoot),
|
||||
"with memberHerdrSocket absent, the config directory must still be created directly "
|
||||
+ "under configRoot, unchanged from before this fix");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 2: under {@code memberHerdrSocket}, opencode session discovery must be
|
||||
* declared unavailable rather than silently scanning fleetd's own {@code discoveryRoot} — which,
|
||||
* under this config key, is NOT where the member's opencode actually writes its session
|
||||
* database. This test proves the gate is real, not merely "no record yet": a matching record IS
|
||||
* written to {@code discoveryRoot} (the exact fixture {@link
|
||||
* #theHandleDiscoversTheSessionIdForTheWorkersCwdOnlyAfterItAppears} proves discovery would
|
||||
* otherwise find), and {@code agentSessionId()} must still return {@code null} — proving the
|
||||
* gate, not a coincidental absence of data, is what produced the null. One WARN is also logged,
|
||||
* exactly once even across repeated calls.
|
||||
*/
|
||||
@Test
|
||||
void discoveryIsUnavailableUnderMemberHerdrSocketEvenWhenARecordExists(
|
||||
@TempDir Path configRoot, @TempDir Path worktreeRoot, @TempDir Path discRoot) throws Exception {
|
||||
String group = currentUserGroup();
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_should_be_hidden", "/work/dir", 1000L);
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = new OpenCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), Map.of("gemini", opencodeCfg(null, null, null)),
|
||||
"gemini", _ -> null, 0, System::currentTimeMillis, () -> { }, configRoot, discRoot,
|
||||
null, null, () -> configWithMemberHerdrSocket(worktreeRoot.toString(), group));
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(OpenCodeLauncher.class);
|
||||
Level original = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
PeerHandle handle;
|
||||
try {
|
||||
handle = launcher.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
assertNull(handle.agentSessionId(),
|
||||
"memberHerdrSocket configured: discovery must stay unavailable even though a "
|
||||
+ "matching record exists in discoveryRoot");
|
||||
assertNull(handle.agentSessionId(), "the gate must hold on a second call too");
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(original);
|
||||
}
|
||||
List<String> warnings = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.WARN)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.toList();
|
||||
assertEquals(1, warnings.size(),
|
||||
"exactly one WARN across two agentSessionId() calls — got: " + warnings);
|
||||
assertTrue(warnings.get(0).contains("memberHerdrSocket"),
|
||||
"the WARN must name memberHerdrSocket as the reason — got: " + warnings.get(0));
|
||||
}
|
||||
|
||||
// --- fleetd #175: opencode silently substitutes a model on an unknown -m flag ----------------
|
||||
|
||||
private static OpenCodeLauncher serviceWithSink(FakeHerdr herdr, Path configRoot, Path discoveryRoot,
|
||||
FleetConfig.Profile cfg, ExhaustionSink sink) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, discoveryRoot,
|
||||
null, null, null, sink);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #175 acceptance criterion 1: the check must fire on the REAL late-resolve path —
|
||||
* {@code SessionManager}'s {@code get}/{@code roster} re-polling a retained {@link PeerHandle}
|
||||
* (fleetd #209) — not on a handle built and queried directly. A handle whose row does not exist
|
||||
* yet, then does, proves the check runs exactly where production runs it: PR #203 shipped a
|
||||
* check that ran inside {@code spawn()}, before opencode had written the row, and every test
|
||||
* passed anyway because none of them drove it through this path. This test would have caught
|
||||
* that: {@code sessions.get(...)} before the row exists must show no mismatch, and the SAME
|
||||
* call, re-driven after the row appears, must be what fires the sink.
|
||||
*/
|
||||
@Test
|
||||
void theRealSessionManagerLateResolvePathCatchesAModelMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// xf's real shape (fleetd #175): weight:80, model "opencode/nemotron-3-ultra-free", no
|
||||
// credentialId — the profile that actually escaped the fleet's accounting.
|
||||
FleetConfig.Profile cfg = opencodeCfg("opencode/nemotron-3-ultra-free", null, null);
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(target + "|" + reason);
|
||||
OpenCodeLauncher launcher = serviceWithSink(herdr, configRoot, discRoot, cfg, sink);
|
||||
|
||||
SessionManager sessions = new SessionManager(launcher);
|
||||
MemberSession acquired = sessions.acquire(cfg.profile(), "/work/dir", null, null);
|
||||
|
||||
// Real late-resolve path, driven BEFORE opencode has written its session row — same shape
|
||||
// as production the instant a pane goes ready.
|
||||
Optional<MemberSession> beforeRow = sessions.get(acquired.paneId());
|
||||
assertTrue(beforeRow.isPresent());
|
||||
assertNull(beforeRow.get().agentSessionId(), "no opencode row yet");
|
||||
assertTrue(exhausted.isEmpty(), "no row yet → nothing to compare, the sink must stay silent");
|
||||
|
||||
// opencode writes its row late, running gpt-5.6-sol (a PAID credential) instead of the
|
||||
// withdrawn free model the profile actually asked for — the exact fleetd #175 scenario.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
|
||||
// Drive the SAME real late-resolve path again: sessions.get() -> resolveAgentSessionId ->
|
||||
// the retained PeerHandle's agentSessionId() -> discovery -> checkModelMatch, all in one
|
||||
// call, unmodified SessionManager code from fleetd #209.
|
||||
Optional<MemberSession> afterRow = sessions.get(acquired.paneId());
|
||||
assertEquals("ses_x", afterRow.get().agentSessionId(),
|
||||
"the id itself still resolves correctly alongside the model check");
|
||||
|
||||
assertEquals(1, exhausted.size(), "the mismatch must fire exactly once through the real path");
|
||||
assertTrue(exhausted.get(0).contains(cfg.model()), "reports the requested model: " + exhausted.get(0));
|
||||
assertTrue(exhausted.get(0).contains("gpt-5.6-sol"), "reports the actual model: " + exhausted.get(0));
|
||||
}
|
||||
|
||||
/** THE TRAP, row 1: a provider-prefixed request matching the DB's id AND provider is a match. */
|
||||
@Test
|
||||
void aProviderPrefixedModelMatchingBothIdAndProviderIsNotAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"gpt-5.6-terra\",\"providerID\":\"openai\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
assertTrue(exhausted.isEmpty(), "id and provider both match → never a mismatch: " + exhausted);
|
||||
}
|
||||
|
||||
/** THE TRAP, row 2: same shape as row 1 with a different provider/id pair (the gx profile). */
|
||||
@Test
|
||||
void aGxProviderPrefixedModelMatchingBothIdAndProviderIsNotAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("gx/deepseek-v4-flash", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"deepseek-v4-flash\",\"providerID\":\"gx\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
assertTrue(exhausted.isEmpty(), "id and provider both match → never a mismatch: " + exhausted);
|
||||
}
|
||||
|
||||
/**
|
||||
* Incomplete evidence, not THE TRAP's provider mismatch: opencode's {@code model} JSON had an
|
||||
* {@code id} but no {@code providerID} at all (a real shape {@code parseModel} accepts — see
|
||||
* {@code OpenCodeSessionDiscoveryTest}). The id matches; the provider dimension is simply
|
||||
* unknown, not contradicted. A provider-prefixed profile must NOT be quarantined on this —
|
||||
* that would quarantine on incomplete evidence, which acceptance rule 4 forbids.
|
||||
*/
|
||||
@Test
|
||||
void aMissingProviderIdInTheEvidenceIsUnknownNotAMismatchWhenTheIdMatches(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"gpt-5.6-terra\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
assertTrue(exhausted.isEmpty(),
|
||||
"id matches and provider is simply unknown (absent), never a mismatch: " + exhausted);
|
||||
}
|
||||
|
||||
/**
|
||||
* The other half of the same fix: a missing {@code providerID} must NOT blind the check to a
|
||||
* genuine id mismatch. This is what proves the fix narrows the comparison rather than switching
|
||||
* the whole check off whenever {@code providerID} happens to be absent.
|
||||
*/
|
||||
@Test
|
||||
void aMissingProviderIdInTheEvidenceStillCatchesARealIdMismatch(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
assertEquals(1, exhausted.size(),
|
||||
"the id genuinely differs, so this must still quarantine even with providerID absent: "
|
||||
+ exhausted);
|
||||
assertTrue(exhausted.get(0).contains("gpt-5.6-sol") && exhausted.get(0).contains(cfg.model()),
|
||||
"reports both the requested and actual model: " + exhausted.get(0));
|
||||
}
|
||||
|
||||
/**
|
||||
* THE TRAP, row 3: a profile that names no provider prefix (bare {@code "deepseek-v4-flash"})
|
||||
* must match on id alone — the profile never asked for a specific provider, so opencode
|
||||
* resolving it to {@code gx} is not evidence of anything wrong.
|
||||
*/
|
||||
@Test
|
||||
void aBareModelWithNoProviderPrefixMatchesOnIdAloneAndIsNotAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("deepseek-v4-flash", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"deepseek-v4-flash\",\"providerID\":\"gx\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
assertTrue(exhausted.isEmpty(),
|
||||
"no provider was requested, so opencode's own provider resolution is not a mismatch: "
|
||||
+ exhausted);
|
||||
}
|
||||
|
||||
/**
|
||||
* THE TRAP, row 4 (not a row in the table, the reason the table exists): a mismatched id, with
|
||||
* the profile's requested and the actual model both named in the ERROR and the sink's reason —
|
||||
* the exact fleetd #175 scenario (a withdrawn model silently falls back to a paid one).
|
||||
*/
|
||||
@Test
|
||||
void aRealIdMismatchLogsAnErrorNamingBothModelsAndQuarantinesThroughTheSink(
|
||||
@TempDir Path configRoot, @TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(target + "|" + reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("opencode/nemotron-3-ultra-free", null, null);
|
||||
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(OpenCodeLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
PeerHandle handle;
|
||||
try {
|
||||
handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertEquals(1, exhausted.size(), "a real id mismatch must reach the sink exactly once");
|
||||
assertTrue(exhausted.get(0).contains("opencode/nemotron-3-ultra-free"),
|
||||
"the sink reason must name the requested model: " + exhausted.get(0));
|
||||
assertTrue(exhausted.get(0).contains("gpt-5.6-sol"),
|
||||
"the sink reason must name the actual model: " + exhausted.get(0));
|
||||
|
||||
List<ILoggingEvent> errors = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.ERROR)
|
||||
.toList();
|
||||
assertEquals(1, errors.size(), "exactly one ERROR for the mismatch: " + appender.list);
|
||||
String message = errors.get(0).getFormattedMessage();
|
||||
assertTrue(message.contains("opencode/nemotron-3-ultra-free") && message.contains("gpt-5.6-sol")
|
||||
&& message.contains(cfg.profile()),
|
||||
"the ERROR must name the requested model, the actual model, AND the profile: " + message);
|
||||
|
||||
// The check runs at most once per handle even if agentSessionId() is polled again.
|
||||
handle.agentSessionId();
|
||||
assertEquals(1, exhausted.size(), "no duplicate quarantine on a repeated call");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #175's central safety rule: absent, empty, or unparseable model evidence is UNKNOWN,
|
||||
* never a mismatch — it must never quarantine a working profile. Covers every "no real
|
||||
* evidence yet" shape: no row at all, a row with a null model, and a row with unparseable JSON.
|
||||
*/
|
||||
@Test
|
||||
void unknownOrUnparseableModelEvidenceNeverQuarantines(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg("openai/gpt-5.6-terra", null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
|
||||
// No row yet at all.
|
||||
assertNull(handle.agentSessionId());
|
||||
assertTrue(exhausted.isEmpty(), "no row yet is UNKNOWN, not a mismatch: " + exhausted);
|
||||
|
||||
// A row exists (for a DIFFERENT directory) so a poll on ours still finds nothing.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_other", "/work/other", 500L,
|
||||
"{\"id\":\"gpt-5.6-sol\",\"providerID\":\"openai\"}");
|
||||
assertNull(handle.agentSessionId());
|
||||
assertTrue(exhausted.isEmpty(), "a row for a different directory is UNKNOWN, not a mismatch: "
|
||||
+ exhausted);
|
||||
}
|
||||
|
||||
/**
|
||||
* A profile with no {@code model:} configured has nothing to compare against — the check must
|
||||
* stay silent no matter what opencode actually ran, since there is no requested value to be
|
||||
* wrong about.
|
||||
*/
|
||||
@Test
|
||||
void aProfileWithNoConfiguredModelIsNeverCheckedForAMismatch(@TempDir Path configRoot,
|
||||
@TempDir Path discRoot) throws Exception {
|
||||
List<String> exhausted = new ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> exhausted.add(reason);
|
||||
FleetConfig.Profile cfg = opencodeCfg(null, null, null);
|
||||
PeerHandle handle = serviceWithSink(new FakeHerdr(), configRoot, discRoot, cfg, sink)
|
||||
.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "ses_x", "/work/dir", 1000L,
|
||||
"{\"id\":\"anything-at-all\",\"providerID\":\"anyone\"}");
|
||||
|
||||
assertEquals("ses_x", handle.agentSessionId());
|
||||
assertTrue(exhausted.isEmpty(), "no model configured → nothing to compare: " + exhausted);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -27,19 +27,30 @@ class OpenCodeSessionDiscoveryTest {
|
||||
* and insert one row. Static so {@link OpenCodeLauncherTest} can reuse it.
|
||||
*/
|
||||
static void writeRecord(Path root, String id, String directory, long timeUpdated) throws Exception {
|
||||
writeRecord(root, id, directory, timeUpdated, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Same as {@link #writeRecord(Path, String, String, long)}, plus the {@code model} column
|
||||
* fleetd #175 reads — the raw JSON opencode writes, e.g.
|
||||
* {@code {"id":"gpt-5.6-terra","providerID":"openai"}}. {@code modelJson} may be {@code null}.
|
||||
*/
|
||||
static void writeRecord(Path root, String id, String directory, long timeUpdated, String modelJson)
|
||||
throws Exception {
|
||||
Path db = root.resolve("opencode.db");
|
||||
try (Connection connection = DriverManager.getConnection("jdbc:sqlite:" + db)) {
|
||||
try (Statement statement = connection.createStatement()) {
|
||||
statement.execute("CREATE TABLE IF NOT EXISTS session ("
|
||||
+ "id TEXT PRIMARY KEY, directory TEXT, time_updated INTEGER)");
|
||||
+ "id TEXT PRIMARY KEY, directory TEXT, time_updated INTEGER, model TEXT)");
|
||||
}
|
||||
// Bound parameters, not string interpolation: the class under test uses a
|
||||
// PreparedStatement, and a hand-escaped INSERT here is a pattern someone copies out.
|
||||
try (PreparedStatement insert = connection.prepareStatement(
|
||||
"INSERT INTO session (id, directory, time_updated) VALUES (?, ?, ?)")) {
|
||||
"INSERT INTO session (id, directory, time_updated, model) VALUES (?, ?, ?, ?)")) {
|
||||
insert.setString(1, id);
|
||||
insert.setString(2, directory);
|
||||
insert.setLong(3, timeUpdated);
|
||||
insert.setString(4, modelJson);
|
||||
insert.executeUpdate();
|
||||
}
|
||||
}
|
||||
@@ -134,4 +145,96 @@ class OpenCodeSessionDiscoveryTest {
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"an unreadable database resolves to null, not an exception");
|
||||
}
|
||||
|
||||
// --- fleetd #175: actualModelForDirectory / the model JSON column ---------------------------
|
||||
|
||||
@Test
|
||||
void parsesTheModelJsonIntoProviderAndId(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "{\"id\":\"gpt-5.6-terra\",\"providerID\":\"openai\"}");
|
||||
|
||||
OpenCodeSessionDiscovery.ActualModel actual =
|
||||
new OpenCodeSessionDiscovery(root).actualModelForDirectory("/w/a");
|
||||
|
||||
assertNotNull(actual, "a well-formed model JSON parses");
|
||||
assertEquals("gpt-5.6-terra", actual.id());
|
||||
assertEquals("openai", actual.provider());
|
||||
}
|
||||
|
||||
@Test
|
||||
void prefersTheModelOfTheMostRecentlyUpdatedRow(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_old", "/w/a", 1000L, "{\"id\":\"old-model\",\"providerID\":\"openai\"}");
|
||||
writeRecord(root, "ses_new", "/w/a", 5000L, "{\"id\":\"new-model\",\"providerID\":\"openai\"}");
|
||||
|
||||
assertEquals("new-model",
|
||||
new OpenCodeSessionDiscovery(root).actualModelForDirectory("/w/a").id(),
|
||||
"the model of the row with the highest time_updated wins, same as the id");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingDirectoryYieldsUnknownModelRatherThanAMismatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "{\"id\":\"x\",\"providerID\":\"y\"}");
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForDirectory("/w/other"),
|
||||
"no row for this cwd yet → unknown, not a wrong model");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullModelColumnYieldsUnknownWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, null);
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForDirectory("/w/a"),
|
||||
"a row with no model value yet is unknown, not a mismatch");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unparseableModelJsonYieldsUnknownWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "this is not json");
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForDirectory("/w/a"),
|
||||
"JSON that fails to parse resolves to unknown, never an exception");
|
||||
}
|
||||
|
||||
@Test
|
||||
void modelJsonMissingIdYieldsUnknown(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "ses_aaa", "/w/a", 1000L, "{\"providerID\":\"openai\"}");
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForDirectory("/w/a"),
|
||||
"no id in the JSON → unknown, since id is what a caller actually compares");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingDatabaseYieldsUnknownModelWithoutThrowing(@TempDir Path root) {
|
||||
assertNull(new OpenCodeSessionDiscovery(root).actualModelForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #175 site 2: a schema that predates the {@code model} column entirely — the exact
|
||||
* shape a real opencode upgrade/downgrade could produce. This must degrade to UNKNOWN for the
|
||||
* model, and — the property that actually matters — must NOT take id resolution down with it.
|
||||
* A combined single query for both columns would fail this test; that is why
|
||||
* {@link OpenCodeSessionDiscovery#actualModelForDirectory} runs its own independent query.
|
||||
*/
|
||||
@Test
|
||||
void aMissingModelColumnYieldsUnknownButIdResolutionStillWorks(@TempDir Path root) throws Exception {
|
||||
Path db = root.resolve("opencode.db");
|
||||
try (Connection connection = DriverManager.getConnection("jdbc:sqlite:" + db)) {
|
||||
try (Statement statement = connection.createStatement()) {
|
||||
statement.execute("CREATE TABLE session (id TEXT PRIMARY KEY, directory TEXT, "
|
||||
+ "time_updated INTEGER)");
|
||||
}
|
||||
try (PreparedStatement insert = connection.prepareStatement(
|
||||
"INSERT INTO session (id, directory, time_updated) VALUES (?, ?, ?)")) {
|
||||
insert.setString(1, "ses_aaa");
|
||||
insert.setString(2, "/w/a");
|
||||
insert.setLong(3, 1000L);
|
||||
insert.executeUpdate();
|
||||
}
|
||||
}
|
||||
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertEquals("ses_aaa", discovery.sessionIdForDirectory("/w/a"),
|
||||
"id resolution must survive a database with no model column at all");
|
||||
assertNull(discovery.actualModelForDirectory("/w/a"),
|
||||
"no model column → unknown, not a throw and not a mismatch");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1099,4 +1099,82 @@ class GitWorktreesTest {
|
||||
assertTrue(e.getMessage().contains("cb185-nonexistent-group-zz"),
|
||||
"exception must name the missing/refused group: " + e.getMessage());
|
||||
}
|
||||
|
||||
// --- fleetd #224: worktreeRoot itself must be group-traversable, established in add() ---------
|
||||
|
||||
/**
|
||||
* {@code add} must make the worktree ROOT itself group-traversable when a group is configured —
|
||||
* established once here, where the root is created, never at a use site (never inside a
|
||||
* launcher). A recording runner stands in for chgrp/chmod, the same seam
|
||||
* {@link #shareWithGroupRunsConfigThenChgrpChmodSetgidPerPath} uses for the per-worktree share.
|
||||
*/
|
||||
@Test
|
||||
void addSharesWorktreeRootWithGroupWhenConfigured(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path root = tmp.resolve("wts");
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return "";
|
||||
};
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(root.toString(), "devteam", _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.add(repo.toString(), "cb-224-branch", "HEAD");
|
||||
|
||||
assertTrue(recorded.contains(List.of("chgrp", "devteam", root.toString())),
|
||||
"the worktree root itself must be chgrp'd to the configured group: " + recorded);
|
||||
assertTrue(recorded.contains(List.of("chmod", "g+x", root.toString())),
|
||||
"the worktree root itself must gain group-execute so a different-uid member can "
|
||||
+ "traverse into it: " + recorded);
|
||||
}
|
||||
|
||||
/** {@code worktreeGroup} unset (today's only live mode) ⇒ {@code add} spawns no share process
|
||||
* for the root at all — behaviour must be byte-identical to before fleetd #224. */
|
||||
@Test
|
||||
void addSharesNothingForTheRootWhenNoGroupConfigured(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path root = tmp.resolve("wts");
|
||||
List<List<String>> recorded = new java.util.ArrayList<>();
|
||||
java.util.function.Function<String[], String> recordingRunner = cmd -> {
|
||||
recorded.add(joined(cmd));
|
||||
return "";
|
||||
};
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(root.toString(), null, _ -> {}, recordingRunner);
|
||||
|
||||
gitWorktrees.add(repo.toString(), "cb-224-nogroup", "HEAD");
|
||||
|
||||
assertTrue(recorded.isEmpty(), "no group configured must spawn no share process for the "
|
||||
+ "root at all: " + recorded);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #224, acceptance criterion 2/3: when the root cannot be made group-traversable — here
|
||||
* because the configured group does not exist, the same real-failure shape
|
||||
* {@link #shareWithGroupThrowsNamingTheGroupWhenChgrpFails} drives for the per-worktree share —
|
||||
* the spawn is refused with a message naming the root, its current mode, and the group. This
|
||||
* drives the REAL {@code chgrp} (no recording runner), and the refusal happens before {@code git
|
||||
* worktree add} ever runs, so no partial worktree is left behind either.
|
||||
*/
|
||||
@Test
|
||||
void addRefusesWhenWorktreeRootCannotBeMadeGroupTraversable(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
Path root = tmp.resolve("wts");
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(root.toString(), "cb224-nonexistent-group-zz");
|
||||
|
||||
WorktreeException e = assertThrows(WorktreeException.class,
|
||||
() -> gitWorktrees.add(repo.toString(), "cb-224-refuse", "HEAD"));
|
||||
|
||||
assertTrue(e.getMessage().contains(root.toString()),
|
||||
"refusal must name the worktree root: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("cb224-nonexistent-group-zz"),
|
||||
"refusal must name the missing/refused group: " + e.getMessage());
|
||||
assertTrue(Files.isDirectory(root), "the root is created before the group check runs");
|
||||
String mode = java.nio.file.attribute.PosixFilePermissions.toString(Files.getPosixFilePermissions(root));
|
||||
assertTrue(e.getMessage().contains(mode),
|
||||
"refusal must name the root's current mode (" + mode + "): " + e.getMessage());
|
||||
try (java.util.stream.Stream<Path> children = Files.list(root)) {
|
||||
assertTrue(children.findAny().isEmpty(), "no worktree must be left behind under the root: "
|
||||
+ "the refusal must happen before `git worktree add` ever runs");
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,6 +4,8 @@ import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.MemberLifecycle;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
@@ -333,6 +335,110 @@ class SessionManagerTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #226: a contended slot is refused through the real {@link SessionManager#acquire}
|
||||
* path before the real launcher can hand an architect charter to a process.
|
||||
*/
|
||||
@Test
|
||||
void aContendedArchitectSlotRefusesBeforeTheLauncherStartsAMember() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberRegistry members = architectRegistry();
|
||||
assertTrue(members.bind("architect:opus", "term_already_bound"),
|
||||
"precondition: occupy the sole architect slot before the real spawn under test");
|
||||
sessions.setMemberLifecycle(members);
|
||||
|
||||
assertThrows(IllegalArgumentException.class, () -> sessions.acquire("ltms-local", MemberRole.ARCHITECT,
|
||||
null, "/caller", "term_primary", null));
|
||||
|
||||
assertFalse(herdr.called("agent.start"), "a refused slot must never reach the launcher");
|
||||
assertTrue(sessions.roster().isEmpty(), "no member exists to receive the wrong charter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSecondArchitectOnAnAlreadyBoundProfileIsHeldAsDevNotArchitectAndWarnsLoudly() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberRegistry members = architectRegistry();
|
||||
sessions.setMemberLifecycle(bindFailureAfterReservation(members));
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger registryLog = (ch.qos.logback.classic.Logger)
|
||||
LoggerFactory.getLogger(MemberRegistry.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
registryLog.addAppender(appender);
|
||||
registryLog.setLevel(Level.WARN);
|
||||
try {
|
||||
MemberSession session = sessions.acquire("ltms-local", MemberRole.ARCHITECT, null,
|
||||
"/caller", "term_primary", null);
|
||||
|
||||
assertEquals(MemberRole.DEV, session.role(), "a failed reservation bind must use the fallback");
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no slot-exhaustion WARN logged");
|
||||
assertTrue(warn.contains("ltms-local"), "the WARN names the profile: " + warn);
|
||||
assertTrue(warn.contains(session.terminalId()), "the WARN names the terminal: " + warn);
|
||||
} finally {
|
||||
registryLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void reservedArchitectSlotBindsTheLaunchedMember() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
sessions.setMemberLifecycle(architectRegistry());
|
||||
|
||||
MemberSession session = sessions.acquire("ltms-local", MemberRole.ARCHITECT, null,
|
||||
"/caller", "term_primary", null);
|
||||
|
||||
assertEquals(MemberRole.ARCHITECT, session.role());
|
||||
}
|
||||
|
||||
private static MemberRegistry architectRegistry() {
|
||||
return new MemberRegistry(new FleetConfig.Fleet(Map.of(), Map.of("opus", new FleetConfig.Slot("ltms-local")),
|
||||
Map.of(), Map.of(), null));
|
||||
}
|
||||
|
||||
private static MemberLifecycle bindFailureAfterReservation(MemberRegistry members) {
|
||||
return new MemberLifecycle() {
|
||||
@Override
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return members.acquired(role, profile, terminal);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
members.released(terminal);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
members.requireSlotFor(role, profile);
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return members.reserve(role, profile);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
members.release(reservation);
|
||||
assertTrue(members.bind(reservation.slot(), "term_racer"), "the racer takes the released slot");
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
members.release(reservation);
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
@Test
|
||||
void rosterReflectsAcquiredMinusReleased() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -599,6 +705,32 @@ class SessionManagerTest {
|
||||
"no session is registered when spawn times out (roster empty)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failedArchitectLaunchReleasesItsReservationForTheNextLaunch() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown");
|
||||
long[] clock = {0};
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
1, () -> clock[0], () -> clock[0] += 10);
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(), () -> 0L, 0);
|
||||
sessions.setMemberLifecycle(architectRegistry());
|
||||
|
||||
assertThrows(PeerUnreachableException.class, () -> sessions.acquire("ltms-local", MemberRole.ARCHITECT,
|
||||
null, "/caller", "term_primary", null));
|
||||
|
||||
herdr.agentStatus("idle");
|
||||
MemberSession retry = sessions.acquire("ltms-local", MemberRole.ARCHITECT, null,
|
||||
"/caller", "term_primary", null);
|
||||
assertEquals(MemberRole.ARCHITECT, retry.role(),
|
||||
"the failed launch returned its reservation instead of silently shrinking the slot pool");
|
||||
}
|
||||
|
||||
// --- CB-516: release must notify, so a blocked send can be failed --------------------------
|
||||
|
||||
@Test
|
||||
@@ -887,15 +1019,18 @@ class SessionManagerTest {
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireWithNeitherSessionFieldLeavesAgentSessionIdNull() {
|
||||
void acquireWithNeitherSessionFieldStillMintsAnAgentSessionId() {
|
||||
// fleetd #214: the claude-code launcher mints a session id for EVERY spawn, so a member is
|
||||
// resumable even when the spawn asked for no session identity.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, null, null);
|
||||
|
||||
assertNull(s.agentSessionId(), "no identity requested — unchanged from before CB-584");
|
||||
assertFalse(SessionManager.rosterView(s, null).containsKey("agentSessionId"),
|
||||
"a null id is omitted from the roster, like charterSha256 for a receipt-less session");
|
||||
assertNotNull(s.agentSessionId(),
|
||||
"fleetd #214: a plain spawn mints a session id, so every member is resumable");
|
||||
assertTrue(SessionManager.rosterView(s, null).containsKey("agentSessionId"),
|
||||
"the minted id is in the roster, so fleet_list advertises every member's resume handle");
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
@@ -95,10 +95,9 @@ class WorktreeSessionManagerTest {
|
||||
null, "/caller/proj", null, null);
|
||||
assertEquals("architect:architect", members.slotForTerminal(replacement.terminalId()));
|
||||
|
||||
MemberSession overflow = sessions.acquire("ltms-local", MemberRole.ARCHITECT,
|
||||
null, "/caller/proj", null, null);
|
||||
assertNull(members.slotForTerminal(overflow.terminalId()),
|
||||
"a full slot pool must not stop the architect spawn");
|
||||
assertThrows(IllegalArgumentException.class, () -> sessions.acquire("ltms-local", MemberRole.ARCHITECT,
|
||||
null, "/caller/proj", null, null),
|
||||
"a full slot pool must stop the spawn before it can receive an architect charter");
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
Reference in New Issue
Block a user