Compare commits
21 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 20e0e68ad7 | |||
| 17468a234a | |||
| 8a53d5bfc6 | |||
| 695da7418e | |||
| abd26c796b | |||
| 24559d81ac | |||
| 6c1c2c3994 | |||
| 01fab15713 | |||
| 0cd00e71c3 | |||
| bf0ff2adbf | |||
| ed4bbc1c56 | |||
| c456402cc5 | |||
| 4f0bf667b1 | |||
| 61af9aa574 | |||
| 3b59b34e76 | |||
| 65b38997f7 | |||
| 7510f7649c | |||
| 619792a81c | |||
| cea1183f75 | |||
| e2af4c5ae4 | |||
| dfd5f82894 |
@@ -383,7 +383,11 @@ public final class Bridged {
|
||||
}
|
||||
|
||||
BridgeMcp mcp = new BridgeMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics);
|
||||
primaryRegistry, callers, metrics, new BridgeMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime));
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
/**
|
||||
* Pure classifier. Collection and repair are deliberately outside this package.
|
||||
* {@link HealthState#ERROR_ON_SCREEN} is not decided yet because it needs a bounded pane detection
|
||||
* read and an adapter-specific fatal signature; status facts alone must not guess it.
|
||||
*/
|
||||
public final class FleetHealth {
|
||||
private FleetHealth() { }
|
||||
|
||||
public static HealthDecision decide(HealthSnapshot s, HealthPrior prior, long nowNanos) {
|
||||
if (s.controlLinkDown()) return result(HealthState.CONTROL_LINK_DOWN, false);
|
||||
if (s.targetNotFound()) return result(HealthState.GONE, false);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING && !s.present() && s.readinessGraceElapsed()) {
|
||||
return result(HealthState.NEVER_READY, false);
|
||||
}
|
||||
if (s.orphanedDelegation()) return result(HealthState.DELEGATION_ORPHANED, false);
|
||||
boolean disagreement = s.sessionState() == MemberSession.State.BUSY && s.acceptedDelivery()
|
||||
&& (s.liveStatus() == AgentStatus.IDLE || s.liveStatus() == AgentStatus.DONE);
|
||||
if (disagreement && prior.busyButDone()) return result(HealthState.TURN_BOUNDARY_LOST, true);
|
||||
if (s.stalled()) return result(HealthState.STALL_SUSPECTED, disagreement);
|
||||
if (s.replyStranded()) return result(HealthState.REPLY_STRANDED, disagreement);
|
||||
if (s.queuedDelivery() || s.inboxMessage()) return result(HealthState.WORK_PENDING, disagreement);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING) return result(HealthState.STARTING, disagreement);
|
||||
if (s.acceptedDelivery() && s.liveStatus() == AgentStatus.BLOCKED) {
|
||||
return result(HealthState.BLOCKED_AMBIGUOUS, disagreement);
|
||||
}
|
||||
// An accepted delivery remains bridge work even when herdr is late, unknown, or has already
|
||||
// reported DONE once. It cannot be IDLE until the delegation has resolved.
|
||||
if (s.acceptedDelivery()) return result(HealthState.WORKING, disagreement);
|
||||
return result(HealthState.IDLE, disagreement);
|
||||
}
|
||||
|
||||
private static HealthDecision result(HealthState state, boolean disagreement) {
|
||||
return new HealthDecision(state, new HealthPrior(disagreement));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Classification plus the private fact that the next pure decision needs. */
|
||||
public record HealthDecision(HealthState state, HealthPrior prior) { }
|
||||
@@ -0,0 +1,6 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Private cross-tick observation. It is deliberately not a reported health value. */
|
||||
public record HealthPrior(boolean busyButDone) {
|
||||
public static final HealthPrior NONE = new HealthPrior(false);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
|
||||
/** Read-only facts from one fleet collection tick. */
|
||||
public record HealthSnapshot(MemberSession.State sessionState, AgentStatus liveStatus,
|
||||
boolean acceptedDelivery, boolean queuedDelivery, boolean inboxMessage,
|
||||
boolean present, boolean targetNotFound, boolean controlLinkDown,
|
||||
boolean readinessGraceElapsed, boolean orphanedDelegation,
|
||||
boolean replyStranded, boolean stalled) { }
|
||||
@@ -0,0 +1,8 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
/** Health classifications reported for a member. */
|
||||
public enum HealthState {
|
||||
STARTING, IDLE, WORKING, WORK_PENDING, BLOCKED_AMBIGUOUS,
|
||||
NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Counts turns that ended via the completion fallback instead of {@code bridge_reply}.
|
||||
* MUTE is an observation by target and profile, not a classifier state and never suppresses faults.
|
||||
*/
|
||||
public final class MuteCounter {
|
||||
private final Map<String, Integer> byTarget = new ConcurrentHashMap<>();
|
||||
private final Map<String, Integer> byProfile = new ConcurrentHashMap<>();
|
||||
|
||||
/** Record only fallback completion; a structured reply does not make a member mute. */
|
||||
public void observe(String target, String profile, Rendezvous.Kind kind) {
|
||||
if (kind != Rendezvous.Kind.COMPLETION) return;
|
||||
byTarget.merge(target, 1, Integer::sum);
|
||||
byProfile.merge(profile, 1, Integer::sum);
|
||||
}
|
||||
|
||||
public int forTarget(String target) { return byTarget.getOrDefault(target, 0); }
|
||||
public int forProfile(String profile) { return byProfile.getOrDefault(profile, 0); }
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/** Fixed pane-probe limits. Pane content is never retained here. */
|
||||
public final class PaneBudget {
|
||||
public static final long COOLDOWN_NANOS = 60_000_000_000L;
|
||||
public static final int MAX_PER_TICK = 2;
|
||||
private final Map<String, Long> lastProbe = new HashMap<>();
|
||||
private int cursor;
|
||||
|
||||
public List<String> choose(List<String> candidates, long nowNanos, long configuredCooldownNanos) {
|
||||
long cooldown = Math.max(COOLDOWN_NANOS, configuredCooldownNanos);
|
||||
List<String> out = new ArrayList<>();
|
||||
for (int n = 0; n < candidates.size() && out.size() < MAX_PER_TICK; n++) {
|
||||
String target = candidates.get((cursor + n) % candidates.size());
|
||||
Long last = lastProbe.get(target);
|
||||
if (last == null || nowNanos - last >= cooldown) { out.add(target); lastProbe.put(target, nowNanos); }
|
||||
}
|
||||
if (!candidates.isEmpty()) cursor = (cursor + 1) % candidates.size();
|
||||
return List.copyOf(out);
|
||||
}
|
||||
}
|
||||
@@ -33,7 +33,10 @@ import jakarta.servlet.http.HttpServlet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -73,15 +76,14 @@ public final class BridgeMcp {
|
||||
private final McpSyncServer server;
|
||||
private final CallerResolver authz; // CB-501: null → authorization not enforced (legacy)
|
||||
private final Metrics metrics; // CB-502: null → auth failures not counted
|
||||
private final CapacitySource capacity;
|
||||
|
||||
/**
|
||||
* Legacy constructor — no authorization. Retained so existing tests exercise tool behaviour
|
||||
* without an auth fixture.
|
||||
*/
|
||||
public BridgeMcp(MessageService messages, PeerLauncher workers,
|
||||
SessionManager sessions, ConnectionIdentity identity, MemberPresence presence,
|
||||
PrimaryRegistry primaryRegistry) {
|
||||
this(messages, workers, sessions, identity, presence, primaryRegistry, null, null);
|
||||
/** Capacity facts used by {@code bridge_list}; production must supply the placement live count. */
|
||||
public record CapacitySource(Function<String, Integer> liveCount, Function<String, Integer> maxLoad,
|
||||
Supplier<Set<String>> configuredProfiles, LongSupplier clock) {
|
||||
/** Inert test-only source. It omits capacity rather than inventing zero live counts. */
|
||||
public static CapacitySource none() { return new CapacitySource(_ -> 0, _ -> null, Set::of, System::nanoTime); }
|
||||
boolean available() { return !configuredProfiles.get().isEmpty(); }
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -91,9 +93,10 @@ public final class BridgeMcp {
|
||||
* filter, so the REST guard does not cover it.
|
||||
* @param metrics registry for auth-failure counting; may be {@code null}
|
||||
*/
|
||||
public BridgeMcp(MessageService messages, PeerLauncher workers,
|
||||
SessionManager sessions, ConnectionIdentity identity, MemberPresence presence,
|
||||
PrimaryRegistry primaryRegistry, CallerResolver callers, Metrics metrics) {
|
||||
public BridgeMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity) {
|
||||
this.capacity = capacity;
|
||||
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
||||
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
||||
.jsonMapper(json)
|
||||
@@ -151,8 +154,8 @@ public final class BridgeMcp {
|
||||
Runnable onAccepted = () -> primaryRegistry.recordDelegation(target, caller);
|
||||
// wait defaults to true (block for the reply); wait:false is fire-and-poll.
|
||||
return Boolean.FALSE.equals(a.get("wait"))
|
||||
? sendAsync(messages, target, content, onAccepted)
|
||||
: send(messages, target, content, timeoutMs(a), onAccepted);
|
||||
? sendAsync(messages, target, content, onAccepted, workers.profiles())
|
||||
: send(messages, target, content, timeoutMs(a), onAccepted, workers.profiles());
|
||||
})
|
||||
// bridge_reply's identity is the CONNECTION, never an argument — so the authz check
|
||||
// is "is this caller a worker at all", and it can only ever reply as itself.
|
||||
@@ -207,7 +210,7 @@ public final class BridgeMcp {
|
||||
.toolCall(listTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions,
|
||||
return listFleet(workers, sessions, messages, capacity,
|
||||
callers == null ? Map.of() : callers.leads(),
|
||||
callerTerminal(exchange));
|
||||
})
|
||||
@@ -379,21 +382,22 @@ public final class BridgeMcp {
|
||||
|
||||
// --- tool logic (thin adapters over the services; unit-testable) ---------------------------
|
||||
|
||||
/** {@code bridge_send}: delegate {@code content} to a worker session and block for its reply. */
|
||||
static McpSchema.CallToolResult send(MessageService messages, String sessionId, String content, Long timeoutMs) {
|
||||
return send(messages, sessionId, content, timeoutMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(MessageService, String, String, Long)}, wiring an accepted-delivery hook
|
||||
* {@code bridge_send}: delegate {@code content} to a worker session and block for its reply.
|
||||
* The configured profiles are required so a profile name can never bypass target validation.
|
||||
*
|
||||
* (CB-548): {@code onAccepted} records delegator ownership the instant the send is accepted, so
|
||||
* a BUSY interloper never claims a turn it did not win. {@code null} disables recording.
|
||||
*/
|
||||
static McpSchema.CallToolResult send(MessageService messages, String sessionId, String content,
|
||||
Long timeoutMs, Runnable onAccepted) {
|
||||
Long timeoutMs, Runnable onAccepted, Set<String> profiles) {
|
||||
if (isBlank(sessionId) || isBlank(content)) {
|
||||
return error("sessionId and content are required");
|
||||
}
|
||||
McpSchema.CallToolResult targetError = profileTargetError(sessionId, profiles);
|
||||
if (targetError != null) {
|
||||
return targetError;
|
||||
}
|
||||
long timeout = clamp(timeoutMs == null ? DEFAULT_TIMEOUT_MS : timeoutMs);
|
||||
try {
|
||||
return formatReply(messages.send(sessionId, content, timeout, onAccepted), timeout);
|
||||
@@ -463,24 +467,33 @@ public final class BridgeMcp {
|
||||
/**
|
||||
* {@code bridge_send} with {@code wait:false}: delegate {@code content} and return a ticket
|
||||
* immediately (fire-and-poll), so a long task isn't cut off by the caller's MCP call timeout.
|
||||
*/
|
||||
static McpSchema.CallToolResult sendAsync(MessageService messages, String sessionId, String content) {
|
||||
return sendAsync(messages, sessionId, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(MessageService, String, String)}, wiring the accepted-delivery hook
|
||||
* The configured profiles are required so a profile name can never bypass target validation.
|
||||
*
|
||||
* This wires the accepted-delivery hook
|
||||
* (CB-548) so an async flooding send records delegator ownership exactly once it is accepted.
|
||||
*/
|
||||
static McpSchema.CallToolResult sendAsync(MessageService messages, String sessionId, String content,
|
||||
Runnable onAccepted) {
|
||||
Runnable onAccepted, Set<String> profiles) {
|
||||
if (isBlank(sessionId) || isBlank(content)) {
|
||||
return error("sessionId and content are required");
|
||||
}
|
||||
McpSchema.CallToolResult targetError = profileTargetError(sessionId, profiles);
|
||||
if (targetError != null) {
|
||||
return targetError;
|
||||
}
|
||||
String ticket = messages.sendAsync(sessionId, content, onAccepted);
|
||||
return text("accepted — task delegated. Poll bridge_poll with ticket=" + ticket);
|
||||
}
|
||||
|
||||
/** A configured profile is never a send target; other unknown values may be herdr-owned panes. */
|
||||
private static McpSchema.CallToolResult profileTargetError(String sessionId, Set<String> profiles) {
|
||||
if (profiles.contains(sessionId)) {
|
||||
return error("unknown send target \"" + sessionId + "\": it is a configured profile name, not a "
|
||||
+ "session id. Call bridge_list to find a member or lead sessionId.");
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** {@code bridge_poll}: check an async delegation by ticket, or drain a worker's inbox by target. */
|
||||
static McpSchema.CallToolResult poll(MessageService messages, String ticket, String target) {
|
||||
if (!isBlank(target)) {
|
||||
@@ -502,6 +515,9 @@ public final class BridgeMcp {
|
||||
? "[done — worker finished without a structured bridge_reply; transcript tail follows]\n" + v.reply()
|
||||
: v.reply());
|
||||
case PENDING -> text("[pending — " + v.detail() + "]");
|
||||
case ASKING -> text("[question — worker is waiting for your answer]\n" + v.reply()
|
||||
+ "\n\nAnswer it by calling bridge_send again with turnId=\"" + v.turnId()
|
||||
+ "\" and content set to your answer; the worker resumes the same turn.");
|
||||
case FAILED -> text("[failed — " + v.detail() + "]");
|
||||
};
|
||||
}
|
||||
@@ -708,6 +724,12 @@ public final class BridgeMcp {
|
||||
*/
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, null, CapacitySource.none(), leads, selfTerm);
|
||||
}
|
||||
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
try {
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
@@ -717,15 +739,56 @@ public final class BridgeMcp {
|
||||
.sorted(Map.Entry.comparingByValue())
|
||||
.map(e -> leadView(e.getKey(), e.getValue(), live.get(e.getKey()), selfTerm))
|
||||
.toList();
|
||||
List<Map<String, Object>> out = sessions.roster().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
List<MemberSession> roster = sessions.roster();
|
||||
List<Map<String, Object>> out = roster.stream()
|
||||
.map(s -> memberCapacityView(s, live.get(s.terminalId()), messages, capacity.clock().getAsLong()))
|
||||
.toList();
|
||||
return text(json(Map.of("leads", leadRows, "members", out)));
|
||||
Set<String> profiles = new java.util.TreeSet<>(capacity.configuredProfiles().get());
|
||||
roster.stream().map(MemberSession::profile).forEach(profiles::add);
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("leads", leadRows); result.put("members", out);
|
||||
if (capacity.available()) result.put("capacity", profiles.stream()
|
||||
.map(profile -> capacityView(profile, capacity.liveCount(), capacity.maxLoad(), roster, messages,
|
||||
capacity.clock().getAsLong())).toList());
|
||||
return text(json(result));
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error listing the fleet: " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Capacity is advisory only. {@code reclaimable} says there is no bridge work, not that bridged
|
||||
* may stop the member: the bridge has capacity facts but no work list, and choosing work needs
|
||||
* authority it does not have. {@code idleForSeconds} is derived from monotonic nanoTime and has
|
||||
* no wall-clock meaning across a daemon restart.
|
||||
*/
|
||||
private static Map<String, Object> memberCapacityView(MemberSession session, Agent live,
|
||||
MessageService messages, long nowNanos) {
|
||||
Map<String, Object> row = SessionManager.rosterView(session, live);
|
||||
boolean open = messages != null && messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean inbox = messages != null && messages.hasInboxMessage(session.terminalId());
|
||||
boolean reclaimable = (session.state() == MemberSession.State.READY || session.state() == MemberSession.State.DONE)
|
||||
&& !open && !inbox;
|
||||
row.put("reclaimable", reclaimable);
|
||||
row.put("idleForSeconds", reclaimable ? Math.max(0, (nowNanos - session.lastActivityAtNanos()) / 1_000_000_000L) : null);
|
||||
return row;
|
||||
}
|
||||
|
||||
private static Map<String, Object> capacityView(String profile, Function<String, Integer> liveCount,
|
||||
Function<String, Integer> maxLoad, List<MemberSession> roster,
|
||||
MessageService messages, long nowNanos) {
|
||||
Integer cap = maxLoad.apply(profile);
|
||||
int live = liveCount.apply(profile);
|
||||
int reclaimable = (int) roster.stream().filter(s -> profile.equals(s.profile()))
|
||||
.filter(s -> (s.state() == MemberSession.State.READY || s.state() == MemberSession.State.DONE))
|
||||
.filter(s -> messages == null || (!messages.hasAcceptedDelivery(s.terminalId()) && !messages.hasInboxMessage(s.terminalId())))
|
||||
.count();
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("profile", profile); row.put("maxLoad", cap); row.put("live", live);
|
||||
row.put("free", cap == null ? null : Math.max(0, cap - live)); row.put("reclaimable", reclaimable);
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* One lead's row: its address, its name, and whether it can be reached right now.
|
||||
*
|
||||
|
||||
@@ -76,7 +76,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
@@ -118,8 +118,8 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.guard = guard;
|
||||
@@ -181,8 +181,8 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
// id via -r and passes no --session-id (the two conflict). Both are injected before the
|
||||
// model flag so --model keeps outranking the operator's own argv.
|
||||
// mutableArgv: argvWithBridge may hand back the profile's own (immutable) List.of when it
|
||||
// has no MCP — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithBridge(cfg));
|
||||
// has neither MCP nor a charter — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithBridge(cfg, spec.charter()));
|
||||
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
|
||||
return new Launch(workerEnv, argvWithModel(argv, cfg), agentSessionId);
|
||||
}
|
||||
@@ -217,22 +217,26 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus — when {@code worker.mcpUrl} is set — inline {@code --mcp-config} for
|
||||
* the bridge server and {@code --append-system-prompt} for the {@link #REPLY_CHARTER}. Neither
|
||||
* touches the profile's config; both are pure command-line flags. This inline-flag mount is
|
||||
* Claude Code specific — other adapters mount MCP and instructions their own way.
|
||||
* The launch argv, plus an inline {@code --mcp-config} when {@code worker.mcpUrl} is set and
|
||||
* {@code --append-system-prompt} when the base composed a charter. Neither touches the profile's
|
||||
* config; both are pure command-line flags. This inline-flag mount is Claude Code specific —
|
||||
* other adapters mount MCP and instructions their own way.
|
||||
*/
|
||||
private List<String> argvWithBridge(BridgedConfig.Profile cfg) {
|
||||
if (!cfg.hasMcp()) {
|
||||
private List<String> argvWithBridge(BridgedConfig.Profile cfg, String charter) {
|
||||
if (!cfg.hasMcp() && charter == null) {
|
||||
return cfg.argv();
|
||||
}
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(REPLY_CHARTER);
|
||||
if (cfg.hasMcp()) {
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
}
|
||||
if (charter != null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(charter);
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
|
||||
@@ -155,16 +155,15 @@ public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
* As above, plus the live {@code fleet} config (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once per spawn; {@code null} ⇒ default tab label and no
|
||||
* role charter. A separate constructor rather than a new parameter on the one above,
|
||||
* so every existing call
|
||||
* site keeps the default without an edit.
|
||||
* role charter. A separate constructor rather than a new parameter on the one
|
||||
* above, so every existing call site keeps the default without an edit.
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
this.namePrefix = namePrefix;
|
||||
this.agents = agents;
|
||||
|
||||
@@ -38,7 +38,7 @@ import java.util.function.Supplier;
|
||||
* <li><strong>File-based MCP mount + instructions.</strong> opencode has no inline
|
||||
* {@code --mcp-config}/{@code --append-system-prompt}. Instead the bridge writes an ephemeral
|
||||
* {@code opencode.json} that declares the bridge as a {@code remote} MCP server and lists a
|
||||
* reply-charter file under {@code instructions}, then points the worker at it with
|
||||
* member-charter file under {@code instructions}, then points the worker at it with
|
||||
* {@code OPENCODE_CONFIG}. This is the one place the launcher touches disk — Claude never did.</li>
|
||||
* <li><strong>Model as a flag.</strong> the {@code provider/model} selector is passed as
|
||||
* {@code -m}, not an env var.</li>
|
||||
@@ -97,7 +97,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet);
|
||||
@@ -143,7 +143,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
@@ -163,17 +163,17 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Builds the opencode launch: no {@code ANTHROPIC_*} and no guard (opencode reads its own
|
||||
* provider credentials); when the profile mounts the bridge MCP, generate an ephemeral
|
||||
* {@code opencode.json} (remote MCP server + reply-charter instructions) and point the worker at
|
||||
* it via {@code OPENCODE_CONFIG}; carry the parity-neutral git-forge grant; and select the model
|
||||
* with {@code -m}.
|
||||
* provider credentials); when the profile mounts the bridge MCP or has a member charter,
|
||||
* generate an ephemeral {@code opencode.json} (remote MCP server + member-charter instructions)
|
||||
* and point the worker at it via {@code OPENCODE_CONFIG}; carry the parity-neutral git-forge
|
||||
* grant; and select the model with {@code -m}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
// A config file is needed for the bridge MCP mount, for a pinned endpoint (CB-508), or both.
|
||||
if (cfg.hasMcp() || hasCustomProvider(cfg)) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg).toString());
|
||||
// A config file is needed for the bridge MCP mount, a member charter, or a pinned endpoint (CB-508).
|
||||
if (cfg.hasMcp() || spec.charter() != null || hasCustomProvider(cfg)) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter()).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
return new Launch(workerEnv,
|
||||
@@ -236,12 +236,12 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* Write an ephemeral {@code opencode.json} (and the reply-charter file it references) into a
|
||||
* Write an ephemeral {@code opencode.json} (and the member-charter file it references) into a
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*/
|
||||
private Path writeConfig(BridgedConfig.Profile cfg) {
|
||||
private Path writeConfig(BridgedConfig.Profile cfg, String charterText) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "bridged-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
@@ -264,16 +264,19 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
// if per-profile control is ever wanted, add a profile knob rather than dropping this.
|
||||
root.putObject("compaction").put("auto", true);
|
||||
|
||||
if (cfg.hasMcp()) {
|
||||
Path charter = dir.resolve("reply-charter.md");
|
||||
Files.writeString(charter, REPLY_CHARTER);
|
||||
if (charterText != null) {
|
||||
Path charter = dir.resolve("member-charter.md");
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp()) {
|
||||
ObjectNode bridge = root.putObject("mcp").putObject("bridge");
|
||||
bridge.put("type", "remote");
|
||||
bridge.put("url", cfg.mcpUrl());
|
||||
bridge.put("enabled", true);
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
if (hasCustomProvider(cfg)) {
|
||||
addCustomProvider(root, cfg);
|
||||
|
||||
@@ -127,6 +127,8 @@ public final class MessageService {
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
PENDING,
|
||||
/** The worker is paused in {@code bridge_ask}; {@link TaskView#reply} and {@link TaskView#turnId} identify it. */
|
||||
ASKING,
|
||||
/** The worker's turn finished; {@link TaskView#reply} holds the answer. */
|
||||
DONE,
|
||||
/** The delegation could not complete (timed out, worker gone, or busy). */
|
||||
@@ -136,16 +138,28 @@ public final class MessageService {
|
||||
/**
|
||||
* A poll snapshot of an async delegation.
|
||||
*
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, else {@code null}
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, or the question when
|
||||
* {@link #phase} is {@link Phase#ASKING}; otherwise {@code null}
|
||||
* @param replySource {@code "reply"} (structured {@code bridge_reply}) or {@code "transcript"}
|
||||
* (completion scrape) when {@link Phase#DONE}, else {@code null}
|
||||
* @param detail a human note (live worker status while pending, or the failure reason)
|
||||
* @param detail a human note (live worker status while pending, ask state, or failure reason)
|
||||
* @param turnId correlation id for an {@link Phase#ASKING} ticket, else {@code null}
|
||||
*/
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail) {
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail,
|
||||
String turnId) {
|
||||
}
|
||||
|
||||
/** An in-flight or finished async delegation, keyed by its ticket. */
|
||||
private record Task(String target, CompletableFuture<Reply> future, long createdNanos) {
|
||||
private static final class Task {
|
||||
private final String target;
|
||||
private final CompletableFuture<Reply> future = new CompletableFuture<>();
|
||||
private final long createdNanos = System.nanoTime();
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
|
||||
private Task(String target) {
|
||||
this.target = target;
|
||||
}
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
@@ -156,6 +170,10 @@ public final class MessageService {
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
/** The async task that currently owns a target's send lock. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTarget = new ConcurrentHashMap<>();
|
||||
/** Async tickets paused on a specific {@code bridge_ask} turn. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTurn = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
@@ -202,6 +220,16 @@ public final class MessageService {
|
||||
return agents.status(target);
|
||||
}
|
||||
|
||||
/** Read-only delegation fact for fleet views. */
|
||||
public boolean hasAcceptedDelivery(String target) {
|
||||
return rendezvous.isWaiting(target);
|
||||
}
|
||||
|
||||
/** Read-only inbox fact for fleet views. */
|
||||
public boolean hasInboxMessage(String target) {
|
||||
return !inbox.peek(target).isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code bridge_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
@@ -333,6 +361,9 @@ public final class MessageService {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
if (hasAsyncQuestion(target)) {
|
||||
return new Reply(Outcome.BUSY, null); // the worker's current turn is paused for its lead
|
||||
}
|
||||
// Open the waiter BEFORE queueing delivery (CB-548). A fast reply — the worker already
|
||||
// injectable the instant we enqueue — otherwise arrives before the waiter is registered
|
||||
// and orphans into the inbox while this send blocks to the timeout (the enqueue-before-
|
||||
@@ -393,12 +424,14 @@ public final class MessageService {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
markAsyncQuestion(workerSession, question, ticket.turnId());
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("bridge_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
@@ -441,9 +474,12 @@ public final class MessageService {
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
clearAsyncQuestion(turnId, false);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
// turn (it never re-entered the injector), so a silent worker rides out the window.
|
||||
@@ -483,9 +519,27 @@ public final class MessageService {
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
CompletableFuture<Reply> future = CompletableFuture.supplyAsync(
|
||||
() -> send(target, content, ASYNC_TIMEOUT_MS, onAccepted), asyncExecutor);
|
||||
tasks.put(ticket, new Task(target, future, System.nanoTime()));
|
||||
Task task = new Task(target);
|
||||
tasks.put(ticket, task);
|
||||
asyncExecutor.submit(() -> {
|
||||
try {
|
||||
Runnable trackingAccepted = () -> {
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
asyncTasksByTarget.put(target, task);
|
||||
};
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, trackingAccepted);
|
||||
if (result.outcome() == Outcome.QUESTION) {
|
||||
asyncTasksByTarget.remove(target, task);
|
||||
} else {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
asyncTasksByTarget.remove(target, task);
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
log.debug("async send {} -> {}", ticket, target);
|
||||
return ticket;
|
||||
@@ -501,27 +555,32 @@ public final class MessageService {
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future();
|
||||
CompletableFuture<Reply> f = task.future;
|
||||
if (!f.isDone()) {
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target()));
|
||||
Reply question = task.question;
|
||||
if (question != null) {
|
||||
return new TaskView(ticket, Phase.ASKING, question.text(), null,
|
||||
"worker is waiting for your answer", question.turnId());
|
||||
}
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target), null);
|
||||
}
|
||||
Reply r;
|
||||
try {
|
||||
r = f.getNow(null);
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
Throwable cause = (e instanceof CompletionException ce && ce.getCause() != null) ? ce.getCause() : e;
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage());
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage(), null);
|
||||
}
|
||||
if (r.completed()) {
|
||||
String source = r.outcome() == Outcome.REPLIED ? "reply" : "transcript";
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null);
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null, null);
|
||||
}
|
||||
// A wedged worker (CB-109) carries the error context as its reason; the timeout/busy
|
||||
// outcomes carry none, so fall back to the outcome name.
|
||||
String detail = r.outcome() == Outcome.WORKER_FAILED && r.text() != null
|
||||
? r.text()
|
||||
: "no reply — " + r.outcome().name().toLowerCase();
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail);
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
@@ -536,7 +595,51 @@ public final class MessageService {
|
||||
/** Drop finished tickets older than the TTL so the registry cannot grow without bound. */
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = System.nanoTime() - TICKET_TTL_NANOS;
|
||||
tasks.values().removeIf(t -> t.future().isDone() && t.createdNanos() < cutoff);
|
||||
tasks.values().removeIf(t -> t.future.isDone() && t.createdNanos < cutoff);
|
||||
}
|
||||
|
||||
/** Record the active question for an async ticket; blocking sends have no entry and stay unchanged. */
|
||||
private void markAsyncQuestion(String target, String text, String turnId) {
|
||||
Task task = asyncTasksByTarget.get(target);
|
||||
if (task != null) {
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.question = null;
|
||||
if (forgetTurn) {
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
task.turnId = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
asyncTasksByTarget.remove(task.target, task);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete the async ticket correlated to a specific answered turn. */
|
||||
private void finishAsyncTask(String turnId, Reply result) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
private boolean hasAsyncQuestion(String target) {
|
||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
|
||||
@@ -0,0 +1,56 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
class FleetHealthTest {
|
||||
@Test void muteCountsOnlyCompletionFallbacks() {
|
||||
MuteCounter mute = new MuteCounter();
|
||||
mute.observe("target", "terra", Rendezvous.Kind.REPLY);
|
||||
mute.observe("target", "terra", Rendezvous.Kind.COMPLETION);
|
||||
assertEquals(1, mute.forTarget("target"));
|
||||
assertEquals(1, mute.forProfile("terra"));
|
||||
}
|
||||
@Test void turnBoundaryNeedsTwoSnapshots() {
|
||||
HealthSnapshot s = snapshot(MemberSession.State.BUSY, AgentStatus.DONE, true);
|
||||
HealthDecision first = FleetHealth.decide(s, HealthPrior.NONE, 1);
|
||||
assertEquals(HealthState.WORKING, first.state());
|
||||
assertEquals(new HealthPrior(true), first.prior());
|
||||
assertEquals(HealthState.TURN_BOUNDARY_LOST, FleetHealth.decide(s, first.prior(), 2).state());
|
||||
}
|
||||
|
||||
@Test void unknownLiveStatusWithAcceptedDeliveryIsNotIdle() {
|
||||
assertEquals(HealthState.WORKING, FleetHealth.decide(
|
||||
snapshot(MemberSession.State.BUSY, AgentStatus.UNKNOWN, true), HealthPrior.NONE, 1).state());
|
||||
}
|
||||
|
||||
@Test void acceptedDeliveryNeverReportsIdle() {
|
||||
for (MemberSession.State session : MemberSession.State.values()) {
|
||||
for (AgentStatus live : AgentStatus.values()) {
|
||||
HealthSnapshot s = snapshot(session, live, true);
|
||||
assertEquals(false, FleetHealth.decide(s, HealthPrior.NONE, 1).state() == HealthState.IDLE,
|
||||
() -> "accepted delivery returned IDLE for " + session + "/" + live);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Test void blockedDoesNotGuessPromptKind() {
|
||||
assertEquals(HealthState.BLOCKED_AMBIGUOUS, FleetHealth.decide(
|
||||
snapshot(MemberSession.State.BUSY, AgentStatus.BLOCKED, true), HealthPrior.NONE, 1).state());
|
||||
}
|
||||
|
||||
@Test void controlLinkOutranksMemberFault() {
|
||||
HealthSnapshot s = new HealthSnapshot(MemberSession.State.BUSY, AgentStatus.DONE, true, false,
|
||||
false, true, true, true, false, true, true, true);
|
||||
assertEquals(HealthState.CONTROL_LINK_DOWN, FleetHealth.decide(s, HealthPrior.NONE, 1).state());
|
||||
}
|
||||
|
||||
private static HealthSnapshot snapshot(MemberSession.State state, AgentStatus live, boolean accepted) {
|
||||
return new HealthSnapshot(state, live, accepted, false, false, true, false, false,
|
||||
false, false, false, false);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,14 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import java.util.List;
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
class PaneBudgetTest {
|
||||
@Test void fixedLimitsIgnoreWeakerConfig() {
|
||||
PaneBudget budget = new PaneBudget();
|
||||
List<String> targets = List.of("a", "b", "c");
|
||||
assertEquals(List.of("a", "b"), budget.choose(targets, 0, 0));
|
||||
assertEquals(List.of("c"), budget.choose(targets, 1, 0));
|
||||
}
|
||||
}
|
||||
@@ -72,7 +72,7 @@ class BridgeMcpAuthzTest {
|
||||
new PrimaryRegistry(null),
|
||||
enforce ? CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
Map::of, new MemberRegistry(null)) : null,
|
||||
metrics);
|
||||
metrics, BridgeMcp.CapacitySource.none());
|
||||
return mcp;
|
||||
}
|
||||
|
||||
|
||||
@@ -53,6 +53,18 @@ class BridgeMcpTest {
|
||||
return ((McpSchema.TextContent) r.content().getFirst()).text();
|
||||
}
|
||||
|
||||
private void assertSendRoundTrips(String target, Set<String> profiles) throws Exception {
|
||||
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.send(messages, target, "hi", 4000L, null, profiles));
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting(target) && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting(target), "send should be accepted for " + target);
|
||||
BridgeMcp.reply(messages, target, "received");
|
||||
assertEquals("received", textOf(send.get(6, TimeUnit.SECONDS)));
|
||||
}
|
||||
|
||||
private static ClaudeCodeLauncher workerService(FakeHerdr h, String baseUrl, Set<String> allow) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", baseUrl, "coder", null, "BRIDGED_WORKER_TOKEN", null,
|
||||
@@ -69,7 +81,7 @@ class BridgeMcpTest {
|
||||
void sendThenReplyRoundTrips() throws Exception {
|
||||
// bridge_send blocks; bridge_reply resolves it with the worker's structured answer.
|
||||
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.send(messages, "term_a", "review this", 4000L));
|
||||
() -> BridgeMcp.send(messages, "term_a", "review this", 4000L, null, Set.of()));
|
||||
|
||||
// Wait until the send has opened its waiter so the reply resolves it (CB-307: reply now
|
||||
// queues in the inbox if no waiter is open, which would break the round-trip).
|
||||
@@ -91,7 +103,7 @@ class BridgeMcpTest {
|
||||
@Test
|
||||
void asyncSendReturnsATicketThenPollReportsTheReply() throws Exception {
|
||||
// wait:false parity — a ticket is issued, resolved by a reply, and surfaced by bridge_poll.
|
||||
McpSchema.CallToolResult accepted = BridgeMcp.sendAsync(messages, "term_a", "do it");
|
||||
McpSchema.CallToolResult accepted = BridgeMcp.sendAsync(messages, "term_a", "do it", null, Set.of());
|
||||
assertNotEquals(Boolean.TRUE, accepted.isError());
|
||||
String out = textOf(accepted);
|
||||
assertTrue(out.contains("ticket="), out);
|
||||
@@ -120,6 +132,131 @@ class BridgeMcpTest {
|
||||
assertEquals("async LGTM", textOf(polled));
|
||||
}
|
||||
|
||||
@Test
|
||||
void asyncSendSurfacesAnAskThenKeepsTheTicketForTheFinalReply() throws Exception {
|
||||
McpSchema.CallToolResult accepted = BridgeMcp.sendAsync(messages, "term_a", "do it", null, Set.of());
|
||||
String ticket = textOf(accepted).substring(textOf(accepted).indexOf("ticket=") + "ticket=".length()).trim();
|
||||
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"));
|
||||
|
||||
CompletableFuture<McpSchema.CallToolResult> ask = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.ask(messages, "term_a", "which config?", 5000L));
|
||||
McpSchema.CallToolResult question = BridgeMcp.poll(messages, ticket, null);
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (!textOf(question).contains("[question") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
question = BridgeMcp.poll(messages, ticket, null);
|
||||
}
|
||||
assertTrue(textOf(question).contains("which config?"), textOf(question));
|
||||
String questionText = textOf(question);
|
||||
String afterTurnId = questionText.substring(questionText.indexOf("turnId=\"") + "turnId=\"".length());
|
||||
String turnId = afterTurnId.substring(0, afterTurnId.indexOf('"'));
|
||||
|
||||
CompletableFuture<McpSchema.CallToolResult> answer = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.answer(messages, turnId, "config.yaml", 5000L));
|
||||
assertEquals("config.yaml", textOf(ask.get(6, TimeUnit.SECONDS)));
|
||||
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"));
|
||||
BridgeMcp.reply(messages, "term_a", "done");
|
||||
assertEquals("done", textOf(answer.get(6, TimeUnit.SECONDS)));
|
||||
|
||||
McpSchema.CallToolResult done = BridgeMcp.poll(messages, ticket, null);
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (!"done".equals(textOf(done)) && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
done = BridgeMcp.poll(messages, ticket, null);
|
||||
}
|
||||
assertEquals("done", textOf(done));
|
||||
}
|
||||
|
||||
@Test
|
||||
void unansweredAsyncAskReturnsTheTicketToPending() throws Exception {
|
||||
McpSchema.CallToolResult accepted = BridgeMcp.sendAsync(messages, "term_a", "do it", null, Set.of());
|
||||
String ticket = textOf(accepted).substring(textOf(accepted).indexOf("ticket=") + "ticket=".length()).trim();
|
||||
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"));
|
||||
|
||||
McpSchema.CallToolResult ask = BridgeMcp.ask(messages, "term_a", "still there?", 50L);
|
||||
assertTrue(textOf(ask).contains("no answer"), textOf(ask));
|
||||
assertTrue(textOf(BridgeMcp.poll(messages, ticket, null)).startsWith("[pending"));
|
||||
|
||||
BridgeMcp.reply(messages, "term_a", "finished after timeout");
|
||||
assertEquals("finished after timeout", messages.drainReplies("term_a").getFirst().content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void asyncSendFailureDoesNotLeaveItsTicketPending() throws Exception {
|
||||
String ticket = messages.sendAsync("term_a", "do it", () -> {
|
||||
throw new IllegalStateException("accept failed");
|
||||
});
|
||||
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
MessageService.TaskView view = messages.poll(ticket);
|
||||
while (view.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
view = messages.poll(ticket);
|
||||
}
|
||||
assertEquals(MessageService.Phase.FAILED, view.phase());
|
||||
assertEquals("accept failed", view.detail());
|
||||
}
|
||||
|
||||
@Test
|
||||
void anotherAsyncTicketCannotCaptureAReplyWhileTheFirstTicketIsAsking() throws Exception {
|
||||
McpSchema.CallToolResult firstAccepted = BridgeMcp.sendAsync(messages, "term_a", "first", null, Set.of());
|
||||
String firstTicket = textOf(firstAccepted).substring(textOf(firstAccepted).indexOf("ticket=") + "ticket=".length()).trim();
|
||||
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"));
|
||||
|
||||
CompletableFuture<McpSchema.CallToolResult> ask = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.ask(messages, "term_a", "which config?", 5000L));
|
||||
MessageService.TaskView first = messages.poll(firstTicket);
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (first.phase() != MessageService.Phase.ASKING && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
first = messages.poll(firstTicket);
|
||||
}
|
||||
assertEquals(MessageService.Phase.ASKING, first.phase());
|
||||
String firstTurnId = first.turnId();
|
||||
|
||||
McpSchema.CallToolResult secondAccepted = BridgeMcp.sendAsync(messages, "term_a", "second", null, Set.of());
|
||||
String secondTicket = textOf(secondAccepted).substring(textOf(secondAccepted).indexOf("ticket=") + "ticket=".length()).trim();
|
||||
MessageService.TaskView second = messages.poll(secondTicket);
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (second.phase() == MessageService.Phase.PENDING && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
second = messages.poll(secondTicket);
|
||||
}
|
||||
assertEquals(MessageService.Phase.FAILED, second.phase());
|
||||
|
||||
BridgeMcp.reply(messages, "term_a", "late reply");
|
||||
assertEquals("late reply", messages.drainReplies("term_a").getFirst().content());
|
||||
|
||||
CompletableFuture<McpSchema.CallToolResult> answer = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.answer(messages, firstTurnId, "config.yaml", 5000L));
|
||||
assertEquals("config.yaml", textOf(ask.get(6, TimeUnit.SECONDS)));
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
Thread.sleep(5);
|
||||
}
|
||||
BridgeMcp.reply(messages, "term_a", "done");
|
||||
assertEquals("done", textOf(answer.get(6, TimeUnit.SECONDS)));
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollUnknownTicketIsAnError() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.poll(messages, "task-999", null);
|
||||
@@ -129,15 +266,40 @@ class BridgeMcpTest {
|
||||
|
||||
@Test
|
||||
void sendTimesOutWithAWorkingNote() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.send(messages, "term_a", "hi", 120L);
|
||||
McpSchema.CallToolResult res = BridgeMcp.send(messages, "term_a", "hi", 120L, null, Set.of());
|
||||
assertNotEquals(Boolean.TRUE, res.isError(), "a timeout is informational, not a tool error");
|
||||
assertTrue(textOf(res).contains("no reply"), "got: " + textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendRejectsMissingArgs() {
|
||||
assertTrue(BridgeMcp.send(messages, null, "hi", null).isError());
|
||||
assertTrue(BridgeMcp.send(messages, "term_a", " ", null).isError());
|
||||
assertTrue(BridgeMcp.send(messages, null, "hi", null, null, Set.of()).isError());
|
||||
assertTrue(BridgeMcp.send(messages, "term_a", " ", null, null, Set.of()).isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendRejectsAConfiguredProfileNameBeforeAcceptingIt() {
|
||||
McpSchema.CallToolResult blocking = BridgeMcp.send(messages, "sol", "hi", 100L, null, Set.of("sol"));
|
||||
McpSchema.CallToolResult async = BridgeMcp.sendAsync(messages, "sol", "hi", null, Set.of("sol"));
|
||||
|
||||
assertTrue(blocking.isError());
|
||||
assertTrue(async.isError());
|
||||
assertTrue(textOf(blocking).contains("sol"));
|
||||
assertTrue(textOf(blocking).contains("configured profile name"));
|
||||
assertTrue(textOf(blocking).contains("bridge_list"));
|
||||
assertFalse(textOf(async).contains("ticket="));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendAllowsPeerLeadMemberAndUnclassifiedTargets() throws Exception {
|
||||
Set<String> profiles = Set.of("sol");
|
||||
|
||||
assertSendRoundTrips("term_peer_lead", profiles);
|
||||
assertSendRoundTrips("term_live_member", profiles);
|
||||
|
||||
// A herdr-owned pane outside the bridge roster cannot be classified at accept time.
|
||||
McpSchema.CallToolResult result = BridgeMcp.send(messages, "external-pane", "hi", 10L, null, profiles);
|
||||
assertFalse(result.isError(), "an unclassified target must not be rejected at acceptance time");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -173,7 +335,7 @@ class BridgeMcpTest {
|
||||
void askThenAnswerRoundTrips() throws Exception {
|
||||
// The primary delegates and blocks; wait until its waiter is open before the worker asks.
|
||||
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.send(messages, "term_a", "do X", 5000L));
|
||||
() -> BridgeMcp.send(messages, "term_a", "do X", 5000L, null, Set.of()));
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
@@ -305,6 +467,42 @@ class BridgeMcpTest {
|
||||
assertTrue(out.contains("\"liveStatus\":\"unknown\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void capacityUsesThePlacementLiveCount() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
sessions.acquire("ltms-local", null, null, null);
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new BridgeMcp.CapacitySource(profile -> 2, profile -> 2,
|
||||
() -> Set.of("ltms-local"), () -> 0), Map.of(), "");
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"maxLoad\":2"), out);
|
||||
assertTrue(out.contains("\"live\":2"), out);
|
||||
assertTrue(out.contains("\"free\":0"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void capacityIncludesConfiguredProfileWithoutMembers() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
String out = textOf(BridgeMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
sessions, null, new BridgeMcp.CapacitySource(profile -> 0, profile -> 2,
|
||||
() -> Set.of("terra"), () -> 0), Map.of(), ""));
|
||||
assertTrue(out.contains("\"profile\":\"terra\""), out);
|
||||
assertTrue(out.contains("\"live\":0"), out);
|
||||
assertTrue(out.contains("\"free\":2"), out);
|
||||
assertTrue(out.contains("\"reclaimable\":0"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void inertCapacitySourceOmitsCapacityBlock() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
String out = textOf(BridgeMcp.listFleet(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")),
|
||||
new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw"))), null,
|
||||
BridgeMcp.CapacitySource.none(), Map.of(), ""));
|
||||
assertFalse(out.contains("\"capacity\":"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsLeadsAndFlagsTheCallersOwnRow() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -578,7 +776,7 @@ class BridgeMcpTest {
|
||||
assertTrue(out.contains("\"role\":\"dev\""), out);
|
||||
assertTrue(out.contains("\"role\":\"reviewer\""), out);
|
||||
assertEquals(2, out.split("\"profile\":\"ltms-local\"", -1).length - 1,
|
||||
"both members share one profile — that is the point: " + out);
|
||||
"inert capacity is omitted, leaving the two member rows: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
@@ -102,6 +102,53 @@ class ClaudeCodeLauncherTest {
|
||||
"the operator's own args are preserved, in order, ahead of the model flag");
|
||||
}
|
||||
|
||||
@Test
|
||||
void appendsTheBaseComposedRoleAndReplyCharter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
String roleCharter = "You review changes.";
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", null, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}",
|
||||
"http://127.0.0.1:8765/mcp", null, null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
0, 0L, () -> fleet(Map.of("reviewer", roleCharter), null));
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.REVIEWER));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int flag = args.indexOf("--append-system-prompt");
|
||||
assertEquals(1, args.stream().filter("--append-system-prompt"::equals).count(),
|
||||
"the composed charter is passed once");
|
||||
assertEquals(roleCharter + "\n\n" + HerdrPeerLauncher.REPLY_CHARTER, args.get(flag + 1),
|
||||
"the role charter comes first and the reply rule comes last");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profileWithoutMcpOrRoleCharterGetsNoSystemPrompt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = labelService(herdr, () -> fleet(Map.of(), null));
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertFalse(spawnedArgs(herdr).contains("--append-system-prompt"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void profileWithoutMcpStillGetsItsRoleCharter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
String roleCharter = "You design changes.";
|
||||
ClaudeCodeLauncher svc = labelService(herdr, () -> fleet(Map.of("architect", roleCharter), null));
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.ARCHITECT));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int flag = args.indexOf("--append-system-prompt");
|
||||
assertTrue(flag >= 0, "a role charter does not need an MCP mount");
|
||||
assertEquals(roleCharter, args.get(flag + 1));
|
||||
assertFalse(args.contains("--mcp-config"));
|
||||
}
|
||||
|
||||
private ClaudeCodeLauncher multiProfile(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile gx10 = new BridgedConfig.Profile("gx10", "http://gx10.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
|
||||
@@ -17,6 +17,10 @@ import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.Future;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -34,12 +38,19 @@ class OpenCodeLauncherTest {
|
||||
}
|
||||
|
||||
/** Gate-disabled launcher whose per-spawn config dirs land under an inspectable temp root. */
|
||||
private OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, BridgedConfig.Profile cfg) {
|
||||
private static OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, BridgedConfig.Profile cfg) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), k -> "GITEA_ACCESS_TOKEN".equals(k) ? "tok" : null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot);
|
||||
}
|
||||
|
||||
private static OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, BridgedConfig.Profile cfg,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot, fleet);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, Object> lastStart(FakeHerdr herdr) {
|
||||
return (Map<String, Object>) herdr.lastCall("agent.start").params();
|
||||
@@ -62,8 +73,10 @@ class OpenCodeLauncherTest {
|
||||
@Test
|
||||
void writesRemoteMcpConfigAndCharterInstructionsWhenMcpUrlSet(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null))
|
||||
.spawn();
|
||||
BridgedConfig.Fleet fleet = new BridgedConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null),
|
||||
() -> fleet).spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"), "opencode carries no ANTHROPIC_* / subscription boundary");
|
||||
@@ -82,13 +95,15 @@ class OpenCodeLauncherTest {
|
||||
"the profile's bridge MCP url is present");
|
||||
assertTrue(bridge.path("enabled").asBoolean(), "the bridge server is enabled");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the reply charter is mounted via instructions");
|
||||
"the member charter is mounted via instructions");
|
||||
|
||||
// The instructions entry is a real file path holding the reply charter.
|
||||
Path charter = Path.of(cfgPath).resolveSibling("reply-charter.md");
|
||||
// The instructions entry is a real file path holding the composed member charter.
|
||||
Path charter = Path.of(cfgPath).resolveSibling("member-charter.md");
|
||||
assertTrue(Files.exists(charter), "the charter file the config references was written");
|
||||
assertTrue(Files.readString(charter).contains("bridge_reply"),
|
||||
"the charter instructs the worker to answer via bridge_reply");
|
||||
assertEquals("role rule\n\n" + HerdrPeerLauncher.REPLY_CHARTER, Files.readString(charter),
|
||||
"the composed charter keeps the role rule first and the reply rule last");
|
||||
assertEquals(charter.toAbsolutePath().toString(), json.path("instructions").get(0).asText(),
|
||||
"instructions names the charter file by its absolute path");
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -100,6 +115,69 @@ class OpenCodeLauncherTest {
|
||||
"no bridge MCP url → no config file and no OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void roleCharterWithoutMcpOrCustomProviderStillWritesAConfig(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Fleet fleet = new BridgedConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
Path configRoot = Files.createDirectory(root.resolve("configs"));
|
||||
Path checkout = Files.createDirectory(root.resolve("checkout"));
|
||||
service(herdr, configRoot, opencodeCfg("google/gemini-2.5-pro", null, null), () -> fleet).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a role charter needs a config even without MCP or custom provider");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
Path charter = Path.of(json.path("instructions").get(0).asText());
|
||||
assertEquals("role rule", Files.readString(charter), "the base-composed role charter is unchanged");
|
||||
assertTrue(json.path("mcp").isMissingNode(), "a charter does not add an MCP mount");
|
||||
assertTrue(charter.startsWith(configRoot), "the charter is written under the temp config root");
|
||||
try (var files = Files.walk(checkout)) {
|
||||
assertFalse(files.anyMatch(path -> path.getFileName().toString().equals("member-charter.md")),
|
||||
"the worker checkout receives no charter file");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullCharterWritesNoCharterFileOrInstructions(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/model", "http://127.0.0.1:8000", null),
|
||||
() -> new BridgedConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(), Map.of(), null)).spawn();
|
||||
|
||||
String config = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(config, "the custom provider still needs a config");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(config).toFile());
|
||||
assertTrue(json.path("instructions").isMissingNode(), "a null charter adds no instructions entry");
|
||||
try (var files = Files.walk(root)) {
|
||||
assertFalse(files.anyMatch(path -> path.getFileName().toString().equals("member-charter.md")),
|
||||
"a null charter creates no charter file");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void concurrentSpawnsWriteSeparateCharterDirectories(@TempDir Path root) throws Exception {
|
||||
BridgedConfig.Profile cfg = opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null);
|
||||
ExecutorService executor = Executors.newFixedThreadPool(2);
|
||||
try {
|
||||
Future<String> first = executor.submit(() -> spawnConfigPath(root, cfg));
|
||||
Future<String> second = executor.submit(() -> spawnConfigPath(root, cfg));
|
||||
|
||||
Path firstCharter = Path.of(first.get()).resolveSibling("member-charter.md");
|
||||
Path secondCharter = Path.of(second.get()).resolveSibling("member-charter.md");
|
||||
assertNotEquals(firstCharter.getParent(), secondCharter.getParent(),
|
||||
"each concurrent spawn owns a separate config directory");
|
||||
assertTrue(Files.exists(firstCharter));
|
||||
assertTrue(Files.exists(secondCharter));
|
||||
} finally {
|
||||
executor.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
private static String spawnConfigPath(Path root, BridgedConfig.Profile cfg) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, cfg).spawn();
|
||||
return startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void passesTheModelAsDashMFlagAlongsideAutoApprove(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
@@ -0,0 +1,917 @@
|
||||
# M4 - Fleet health, recovery, routing, and capacity
|
||||
|
||||
**Status:** Design accepted on 2026-08-15. CB-573 part 1 has shipped the classification model and
|
||||
the `bridge_list` capacity view; the remaining M4 units are not yet shipped.
|
||||
**Scope:** Fleet evidence, safe mechanical repair, lead routing, capacity reporting, and optional
|
||||
human notification.
|
||||
**Grounded in:** `health/FleetHealth`, `health/PaneBudget`, `inject/StatusPoller`,
|
||||
`inject/StatusRefiner`, `inject/CompletionResolver`, `inject/Injector`, `session/SessionManager`,
|
||||
`msg/MessageService`, `msg/ReplyInbox`, `msg/ReplyPushLoop`, `msg/LeadHeartbeatLoop`,
|
||||
`mcp/PrimaryRegistry`, and `herdr/AgentControl`.
|
||||
|
||||
## 1. Problem and decision boundary
|
||||
|
||||
The operator asked the bridge to detect idle agents, exceptions, stopped work, and broken
|
||||
communication. The bridge may read an agent pane from time to time. It must notify a person when
|
||||
the fleet cannot move forward.
|
||||
|
||||
The four operator terms are not four equal health states. `IDLE` is a normal mode. An exception is
|
||||
sometimes visible only as pane text. Stopped work may look the same as slow work. Broken
|
||||
communication can occur on several links.
|
||||
|
||||
M4 uses this boundary:
|
||||
|
||||
- The bridge detects facts and joins evidence.
|
||||
- The bridge repairs only mechanical failures with no judgement.
|
||||
- The lead decides whether to stop, retry, replace, or reassign a member.
|
||||
- A human is notified only when no healthy lead can act.
|
||||
- n8n may route an outbound incident. It never classifies state or chooses recovery.
|
||||
|
||||
An inbound n8n decider would need bridge authority. No narrow machine-decider role exists. Giving a
|
||||
workflow engine lead authority is unsafe, while adding a new role is a separate authorization
|
||||
design. An outbound sink needs no bridge role.
|
||||
|
||||
The bridge must never replay a delivered task. That task may already have changed files, pushed a
|
||||
branch, opened a pull request, or changed external state. A replay can run those side effects twice.
|
||||
This rule must remain true even if later code stores delivered prompt text.
|
||||
|
||||
## 2. Evidence model
|
||||
|
||||
A health state is mainly a comparison between two views:
|
||||
|
||||
- **herdr view:** current agents and raw live status from one `AgentControl.list()` call.
|
||||
- **bridge view:** session FSM, MCP presence, accepted turns, tasks, inbox state, and lead ownership.
|
||||
|
||||
A strong fault often appears as a disagreement between those views. For example, `BUSY` in the
|
||||
session FSM and `DONE` in herdr means the bridge missed a turn boundary. Pane reads support this
|
||||
model, but they are not the main monitor.
|
||||
|
||||
`SessionManager.rosterView` already joins session state and live status. `AgentControl.list()`
|
||||
already gets the whole live fleet in one call. M4 makes that join persistent and adds timers,
|
||||
accepted-turn state, and incident state.
|
||||
|
||||
### 2.1 Real traces behind the design
|
||||
|
||||
The first trace was an architect that stopped making progress:
|
||||
|
||||
```text
|
||||
profile=opus role=architect state=busy liveStatus=done
|
||||
```
|
||||
|
||||
The session moved from `DONE` to `BUSY` for turn 2. Eighteen minutes later, the session still said
|
||||
`BUSY`, herdr still said `DONE`, the async task still said `PENDING`, and no completion fallback had
|
||||
run. This is `TURN_BOUNDARY_LOST`, not a general slow-turn guess.
|
||||
|
||||
The second trace had two async sends to the same pane, one second apart. The pane was then stopped.
|
||||
One ticket became failed. The other stayed `pending - worker unknown`. Current
|
||||
`MessageService.abandon` resolves only `Rendezvous.currentWaiter(target)`, while async tasks live in
|
||||
a separate ticket map. CB-568 is intended to fix that bug. M4 still keeps an independent
|
||||
post-teardown invariant so a later regression becomes `DELEGATION_ORPHANED`.
|
||||
|
||||
### 2.2 Corrections made during design
|
||||
|
||||
The first state table missed `BUSY` in bridged plus `IDLE` or `DONE` in herdr. It would have found
|
||||
the real trace only through a late, weak stall timer. The final model adds
|
||||
`TURN_BOUNDARY_LOST` as a strong disagreement state.
|
||||
|
||||
The first notification design also required a webhook before `health.enabled` could turn on. That
|
||||
removed useful local detection to avoid a narrower human-notification gap. The final design splits
|
||||
detection from notification. Missing human escalation is shown as partial coverage instead of
|
||||
disabling health.
|
||||
|
||||
## 3. Classification precedence
|
||||
|
||||
Evidence is applied in this order. A lower rule cannot hide a higher one.
|
||||
|
||||
1. **Control link:** failed fleet list plus failed ping becomes `CONTROL_LINK_DOWN`.
|
||||
2. **Definitive target loss:** `_not_found` becomes `GONE` or `LEAD_UNREACHABLE` when the control
|
||||
link is healthy.
|
||||
3. **Startup and teardown invariants:** readiness expiry becomes `NEVER_READY`; surviving tasks
|
||||
after teardown become `DELEGATION_ORPHANED`.
|
||||
4. **Bridge/live disagreement:** `BUSY` plus stable raw `IDLE` or `DONE` becomes
|
||||
`TURN_BOUNDARY_LOST`.
|
||||
5. **Known screen evidence:** a tested fatal signature becomes `ERROR_ON_SCREEN`.
|
||||
6. **Timed suspicion:** unchanged sparse pane probes may become `STALL_SUSPECTED`.
|
||||
7. **Communication quality:** completion fallback becomes `MUTE`; an old inbox entry becomes
|
||||
`REPLY_STRANDED`.
|
||||
8. **Normal mode:** `STARTING`, `IDLE`, `WORKING`, `WORK_PENDING`, or `BLOCKED_AMBIGUOUS`.
|
||||
|
||||
The member flow in Figure 1 shows lifecycle states and the main fault exits. Fault states are
|
||||
reported beside the session FSM; most are not new FSM values.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Registered["Member registered"] --> Starting["STARTING"]
|
||||
Starting -->|"MCP presence"| Idle["IDLE"]
|
||||
Starting -->|"Readiness grace expires"| NeverReady["NEVER_READY"]
|
||||
Idle -->|"Accepted delivery"| Working["WORKING"]
|
||||
Working -->|"Trusted turn boundary"| Idle
|
||||
Working -->|"Bridge BUSY and herdr IDLE or DONE"| Lost["TURN_BOUNDARY_LOST"]
|
||||
Working -->|"Known fatal screen"| Error["ERROR_ON_SCREEN"]
|
||||
Working -->|"Long age and unchanged sparse probes"| Stall["STALL_SUSPECTED"]
|
||||
Working -->|"Target not found"| Gone["GONE"]
|
||||
Idle -->|"Inbox or queued delivery exists"| Pending["WORK_PENDING"]
|
||||
Pending -->|"Delivery or collection finishes"| Idle
|
||||
Idle -->|"Raw BLOCKED with an open turn"| Blocked["BLOCKED_AMBIGUOUS"]
|
||||
Lost -->|"Strict guarded repair"| Repaired["DONE with reconciled completion"]
|
||||
Lost -->|"Repair refused"| LeadDecision["Lead decision required"]
|
||||
```
|
||||
|
||||
*Figure 1. The member lifecycle and the main health exits. Pane-based states never authorise an
|
||||
automatic retry of the task.*
|
||||
|
||||
## 4. State model
|
||||
|
||||
### 4.1 Normal and transitional member states
|
||||
|
||||
| State | Exact evidence | Meaning and certainty |
|
||||
|---|---|---|
|
||||
| `STARTING` | Session is `SPAWNING`; MCP presence is absent | Normal inside the startup grace. MCP contact is the readiness signal. |
|
||||
| `IDLE` | Session is `READY` or `DONE`; live status is `IDLE` or `DONE`; no open turn or inbox item exists | Normal. Idle is not a fault. |
|
||||
| `WORKING` | Session is `BUSY`; raw live status is `WORKING`; the accepted turn is open | Certain that herdr sees work. It does not prove useful progress. |
|
||||
| `WORK_PENDING` | Queued delivery or inbox content exists while the target is injectable | Transitional. Existing injector or push logic should move it. |
|
||||
| `BLOCKED_AMBIGUOUS` | An open turn exists and raw live status is `BLOCKED` | The bridge cannot tell whether this is permission, input, or a settled screen. |
|
||||
|
||||
Idle may drive configured resource cleanup. It never opens an incident and never pages a person.
|
||||
|
||||
### 4.2 Member fault and quality states
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `NEVER_READY` | `SPAWNING`, no MCP presence, and an accepted delivery waits through the existing readiness grace | Delivery never became possible. The exact cause is unknown. Fail the send, stop the process, and preserve a provisioned worktree. |
|
||||
| `GONE` | Per-target herdr call returns `_not_found` while fleet list or ping works | Certain target loss. Fail all target work. Do not replay it. |
|
||||
| `TURN_BOUNDARY_LOST` | Same session turn stays `BUSY`; same accepted task stays open; two raw snapshots show `IDLE` or `DONE` | Strong disagreement. Strict reconciliation may repair it. |
|
||||
| `ERROR_ON_SCREEN` | Suspicious non-working state survives grace; `detection` matches a tested adapter-specific fatal signature | Certain only for the matched signature. A bare word such as `Exception` is not enough. |
|
||||
| `STALL_SUSPECTED` | Open turn is older than the configured threshold; two normalised `recent_unwrapped` digests are unchanged; no boundary or reply occurs | Not certain. A long valid API call can look the same. Lead decides. |
|
||||
| `MUTE` | Turn resolves through completion fallback instead of `bridge_reply` | Certain that no structured reply won. It does not prove an MCP failure. A single event is a metric, not an incident. |
|
||||
| `REPLY_STRANDED` | Typed reply or health message remains after owning-lead push reaches its cap | Collection failed. This does not explain whether the lead is busy, dead, or ignoring the nudge. |
|
||||
| `DELEGATION_ORPHANED` | Target is gone, failed, or released, but one or more tasks remain `PENDING` after reconciliation grace | Certain bridge invariant failure. This is not an inbox-drain fault. |
|
||||
| `WORK_PRODUCT_AT_RISK` | Provisioned branch has commits after its recorded base; member is `DONE`, `FAILED`, or preserved after release; no turn or inbox item remains; long-idle threshold passed | A warning, not proof of loss. Work may already have an open pull request or a squash merge. |
|
||||
|
||||
`MUTE` opens an incident only after a small fixed rate threshold for one target or profile, or when
|
||||
it appears with another fault.
|
||||
|
||||
`WORK_PRODUCT_AT_RISK` must not become `WORK_PRODUCT_UNCOLLECTED`. The bridge does not know pull
|
||||
request or merge state. If committed work appears with `REPLY_STRANDED` or
|
||||
`DELEGATION_ORPHANED`, the existing incident gains `committedWorkAtRisk: true`.
|
||||
|
||||
### 4.3 Control-link state
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `CONTROL_LINK_DOWN` | Two full-fleet `agent.list` calls fail across the grace, and herdr `ping` also fails | Certain for the bridged-to-herdr link. Retry calls, record the incident, and use human escalation if no lead can be reached. |
|
||||
|
||||
A failed fleet list alone is not a dead-member claim. A single `_not_found` with a healthy global
|
||||
link is a target fault, not a control-link fault.
|
||||
|
||||
### 4.4 Lead states
|
||||
|
||||
| State | Exact evidence | Meaning and action |
|
||||
|---|---|---|
|
||||
| `LEAD_IDLE` | Expected lead is present with raw injectable status; no actionable state waits | Normal. Existing heartbeat may run under its own policy. |
|
||||
| `LEAD_WORKING` | Expected lead is present with raw `WORKING`; stall threshold is not met | Reachable and busy. Never inject into the live turn. |
|
||||
| `LEAD_STATUS_UNKNOWN` | Expected lead is present with raw `UNKNOWN` | Neither dead nor a healthy routing target. Retain evidence and retry. |
|
||||
| `LEAD_UNREACHABLE` | Expected lead is absent from two successful live-agent snapshots while ping works, or targeted lookup returns `_not_found` with a healthy control link | Route to a healthy peer. If none exists, use human escalation. |
|
||||
| `LEAD_UNRESPONSIVE` | Actionable state waits; lead stays injectable; bounded nudges exhaust; inbox remains uncollected | Route to a healthy peer or a person. |
|
||||
| `LEAD_STALL_SUSPECTED` | Lead stays `WORKING` past threshold; two sparse pane probes show no progress | Not certain. Never kill or restart automatically. Route to peer or person. |
|
||||
|
||||
The monitor retains the lead name and terminal, last successful sighting, raw status and age,
|
||||
consecutive list absences, targeted errors, pane-probe facts, pending incident age, and nudge
|
||||
outcomes. Current heartbeat and push loops discard much of this history.
|
||||
|
||||
Expected lead identity comes from the same supplier used by `CallerResolver`. It is not liveness
|
||||
evidence. `LeadTabScanner` keeps cached identity after a failed scan, so the health monitor compares
|
||||
that identity with a fresh successful agent list. A dynamic identity also survives a two-successful-
|
||||
snapshot retirement grace. This stops a dead lead from escaping health by disappearing from one map.
|
||||
|
||||
### 4.5 Evidence limits
|
||||
|
||||
M4 cannot tell these cases apart with current evidence:
|
||||
|
||||
- A valid long call and a hung call may have the same status and pane digest.
|
||||
- `BLOCKED` does not explain which input is needed.
|
||||
- An idle prompt after failure may look like an idle prompt after success.
|
||||
- A missing structured reply does not prove a broken MCP connection.
|
||||
- An undrained inbox does not explain why the lead did not collect it.
|
||||
- Arbitrary pane text cannot safely classify arbitrary exceptions.
|
||||
- A branch ahead of its base does not prove that work was not collected.
|
||||
|
||||
Logs are outputs, not classifier inputs. The monitor never parses its own logs.
|
||||
|
||||
## 5. Automatic action and lead action
|
||||
|
||||
### 5.1 Actions the bridge may take
|
||||
|
||||
The bridge may:
|
||||
|
||||
- retry transient herdr status, list, ping, and pane-read failures with bounded backoff;
|
||||
- re-submit Enter after the existing paste/submit race;
|
||||
- fail queued delivery after `NEVER_READY`;
|
||||
- stop a never-ready process while preserving its provisioned worktree;
|
||||
- fail all queued, accepted, and async tasks for a gone or released target;
|
||||
- reconcile one lost boundary when every strict gate in Section 8 passes;
|
||||
- hold typed messages, nudge the owning lead, and stop at the configured cap;
|
||||
- use the existing bounded idle-lead heartbeat;
|
||||
- deduplicate, route, update, and resolve incidents.
|
||||
|
||||
These actions do not choose new work and do not replay old work.
|
||||
|
||||
### 5.2 Decisions reserved for the lead
|
||||
|
||||
Only the lead may:
|
||||
|
||||
- stop or continue `BLOCKED_AMBIGUOUS`;
|
||||
- stop, inspect, or wait on `ERROR_ON_SCREEN`;
|
||||
- kill or continue `STALL_SUSPECTED`;
|
||||
- spawn a replacement or reassign work;
|
||||
- retry a delivered task;
|
||||
- choose how to use partial work in a worktree;
|
||||
- restart herdr or change network, model, credentials, backend, or configuration.
|
||||
|
||||
Reports include literal safe tool calls such as `bridge_status(sessionId="...")`,
|
||||
`bridge_poll(ticket="...")`, `bridge_list()`, and optional `bridge_stop(paneId="...")`. A judgement
|
||||
state never presents stop as the only action.
|
||||
|
||||
### 5.3 Release causes and worktree safety
|
||||
|
||||
| Release cause | Process action | Provisioned worktree |
|
||||
|---|---|---|
|
||||
| `SPAWN_ROLLBACK` before registration or delivery | Stop and clean up | Remove |
|
||||
| `COMPLETED` for `READY` or `DONE` without pending work, idle TTL, or successful context-cap completion | Stop | Remove under completed policy |
|
||||
| `NEVER_READY` | Stop | Preserve |
|
||||
| `GONE` | Best-effort stop | Preserve |
|
||||
| `TURN_FAILED` or lead abort while `BUSY` or `FAILED` | Stop | Preserve |
|
||||
| `RELEASE_WITH_PENDING_TASKS` | Stop | Preserve |
|
||||
| `SHUTDOWN` | Stop | Preserve |
|
||||
|
||||
Explicit stop is state-aware. `SPAWNING`, `BUSY`, `FAILED`, or any target with pending tasks uses a
|
||||
preserving cause.
|
||||
|
||||
Before abnormal release removes the live session, M4 writes an atomic manifest under the worktree
|
||||
root. It records session identity, owner, role, profile, repository, path, branch, base commit,
|
||||
release cause, release time, state, and pending task ids. `bridge_list.preservedWorktrees` loads these
|
||||
manifests after restart. Stop output and WARN logs also name the path and cause. M4 never
|
||||
auto-deletes a preserved worktree.
|
||||
|
||||
## 6. Fleet health monitor
|
||||
|
||||
Add `FleetHealthMonitor`. Do not widen `StatusPoller` into a policy loop.
|
||||
|
||||
`StatusPoller` has a 250 ms delivery cadence and samples only injector targets with outstanding
|
||||
work. Health needs all sessions, all leads, task state, inbox age, and global control evidence. One
|
||||
loop cannot serve both cadences safely.
|
||||
|
||||
Build the monitor like `LeadHeartbeatLoop`:
|
||||
|
||||
- pure `decide(snapshot, priorState, now)` logic;
|
||||
- a thin scheduler;
|
||||
- an injected clock;
|
||||
- edge-triggered state changes;
|
||||
- no network work in the pure function;
|
||||
- no sleeping in tests.
|
||||
|
||||
Each enabled fleet tick reads:
|
||||
|
||||
- one `AgentControl.list()` result for the whole fleet;
|
||||
- one in-memory `SessionManager.roster()` snapshot;
|
||||
- accepted turns and async task state;
|
||||
- typed inbox depth, kind, and age;
|
||||
- push and heartbeat outcomes;
|
||||
- configured and discovered leads.
|
||||
|
||||
Existing failure paths publish structured evidence to the monitor. The monitor does not infer events
|
||||
from log text.
|
||||
|
||||
### 6.1 Pane budget
|
||||
|
||||
Healthy idle members, recent working members, and quiet leads cause no pane reads.
|
||||
|
||||
A pane is eligible only for a stable lost boundary, sustained `BLOCKED` or `UNKNOWN`, work older
|
||||
than the suspect threshold, or one final evidence read for a confirmed fault when the pane exists.
|
||||
|
||||
Compiled brakes apply even if config asks for more:
|
||||
|
||||
- per-target pane cooldown is at least 60 seconds;
|
||||
- working age before the first progress probe is at least 300 seconds;
|
||||
- at most two pane reads occur in one fleet tick;
|
||||
- targets rotate fairly;
|
||||
- only a normalised digest and optional clipped local excerpt are stored;
|
||||
- no pane excerpt leaves bridged in a human webhook.
|
||||
|
||||
Use `detection` for tested screen signatures. Use normalised `recent_unwrapped` only for progress
|
||||
comparison.
|
||||
|
||||
## 7. Typed inbox and routing
|
||||
|
||||
### 7.1 Semantic record
|
||||
|
||||
The typed inbox record carries:
|
||||
|
||||
```text
|
||||
schemaVersion
|
||||
kind: reply | health
|
||||
msgId, target, subjectTerminal, recipientLead
|
||||
severity, state, evidence
|
||||
createdAtEpochMillis, firstSeenEpochMillis, lastSeenEpochMillis
|
||||
recoveryTried, suggestedToolCalls, content
|
||||
```
|
||||
|
||||
A health message never calls `Rendezvous.resolve`. It cannot look like the member's task result.
|
||||
|
||||
Both inbox adapters share field preservation, first-id-wins dedup, FIFO among decoded messages,
|
||||
explicit ownership, ack, and release rules. The in-memory adapter stores typed records directly. It
|
||||
does not copy AMQP migration logic.
|
||||
|
||||
### 7.2 AMQP migration
|
||||
|
||||
The reader uses AMQP `content_type`, never body sniffing:
|
||||
|
||||
```text
|
||||
Legacy v0: text/plain
|
||||
Typed family: application/vnd.ltms.bridged.inbox-message+json
|
||||
```
|
||||
|
||||
A legacy reply may begin with `{`. It remains plain text because its media type is `text/plain`.
|
||||
Legacy text becomes `kind=reply` with exact UTF-8 content and absent typed metadata.
|
||||
|
||||
Typed JSON has required integer `schemaVersion: 1`. Version 1 ignores unknown optional fields.
|
||||
Missing required fields, invalid enums, malformed UTF-8 or JSON, and property/body identity mismatch
|
||||
are invalid data.
|
||||
|
||||
An unknown schema version is not partly decoded. It remains unacknowledged on the original queue and
|
||||
creates one operator-visible `unsupported_version` failure. A newer daemon may read it later.
|
||||
|
||||
Invalid known-format data is copied byte-for-byte to durable queue
|
||||
`agent.<target>.inbox.quarantine`. A dedicated confirm-mode publisher confirms the persistent copy
|
||||
before the original is acknowledged. A failed quarantine handoff leaves the original unacknowledged.
|
||||
The raw body never enters logs.
|
||||
|
||||
Decode failure creates a redacted WARN, metric, `bridge_list` summary, and routed health incident.
|
||||
One bad entry never escapes the consumer callback and never stops later valid messages.
|
||||
|
||||
Safe downgrade is not supported. The previous build ignores `content_type` and would show typed JSON
|
||||
as ordinary reply text. If drained, it would acknowledge the message and lose typed meaning. Typed
|
||||
queues must be drained or preserved before an old jar runs.
|
||||
|
||||
The existing contract suite uses RabbitMQ. Production uses LavinMQ. The migration and lead-key
|
||||
ownership cases must run once against production LavinMQ before release, or the release must state
|
||||
that LavinMQ was not checked.
|
||||
|
||||
### 7.3 Member routing
|
||||
|
||||
A member incident first goes to the exact lead that owns its accepted delegation.
|
||||
`PrimaryRegistry` needs a no-fallback `delegatingLeadFor(memberTarget)` query. Health routing must not
|
||||
use the old singular-primary fallback when several leads exist.
|
||||
|
||||
Publish the incident under the affected member target. Trigger the existing bounded push route. The
|
||||
push waits until the owning lead is injectable, so it does not interrupt a live lead turn.
|
||||
|
||||
### 7.4 Peer lead routing
|
||||
|
||||
Figure 2 shows the route from incident to lead, peer, or person.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Incident["Open incident"] --> Member{"Member incident?"}
|
||||
Member -->|"yes"| Known{"Exact delegation owner known?"}
|
||||
Known -->|"no"| Sink{"Human webhook enabled and healthy?"}
|
||||
Known -->|"yes"| Owner{"Owner lead healthy?"}
|
||||
Owner -->|"yes"| OwnerInbox["Publish to owner lead path"]
|
||||
Owner -->|"no"| PeerSet["Build healthy peer candidate set"]
|
||||
Member -->|"no, lead incident"| PeerSet
|
||||
PeerSet --> Peer{"Healthy peer exists?"}
|
||||
Peer -->|"yes"| Select["Choose fewest assigned incidents<br/>then stable name and terminal id"]
|
||||
Select --> PeerInbox["Publish to peer lead inbox<br/>and status-gated push"]
|
||||
Peer -->|"no"| Sink
|
||||
Sink -->|"yes"| Webhook["Send classified outbound incident"]
|
||||
Sink -->|"no"| Passive["Keep incident open<br/>show partial coverage on local surfaces"]
|
||||
```
|
||||
|
||||
*Figure 2. Routing keeps delegation ownership separate from temporary peer fallback.*
|
||||
|
||||
Peer candidates exclude the incident subject, failed owner, absent leads, raw-unknown leads, and
|
||||
leads with an open unhealthy state. A reachable `WORKING` peer may be selected; its push waits for an
|
||||
injectable window.
|
||||
|
||||
Choose the candidate with the fewest assigned foreign incidents. Break ties by stable lead name,
|
||||
then terminal id. Pin the recipient. Reassign only if that peer becomes unhealthy or retires. A
|
||||
routing generation marks a reassignment, and old pending assignments become superseded.
|
||||
|
||||
`bridge_list` lead rows show health, health age, assigned foreign incident count, and a bounded list
|
||||
of incident id, subject, state, severity, age, and routing generation. The top-level view also shows
|
||||
owner, recipient, and routing reason.
|
||||
|
||||
A peer incident is published under the recipient lead's inbox key, not the failed subject's key. Its
|
||||
status-gated nudge names the failed lead and gives the exact
|
||||
`bridge_poll(target="<recipient-terminal>")` call.
|
||||
|
||||
### 7.5 Lead inbox ownership
|
||||
|
||||
Add `LeadInboxRegistry`, driven by the same expected-lead supplier as `CallerResolver`.
|
||||
|
||||
It calls `replyInbox.own(leadTerminal)` at startup for configured leads, after successful discovery,
|
||||
after config adds a lead, and before publication. Ownership is not an authorization side effect.
|
||||
|
||||
A missing lead keeps its key owned. Release happens only after confirmed retirement, all incidents
|
||||
are reassigned or resolved, typed health messages move or ack, and the queue is empty. Own a
|
||||
replacement terminal before moving messages from the old key. Never release a non-empty in-memory
|
||||
lead key, because in-memory release clears local data.
|
||||
|
||||
### 7.6 Single-lead deployment
|
||||
|
||||
One lead and no peer is a normal mode, not an edge case.
|
||||
|
||||
An idle, reachable lead may receive the existing bounded nudge. An unreachable or stalled sole lead
|
||||
has no safe in-loop recovery. The bridge must not restart or replace it. A new lead would not have the
|
||||
failed lead's plan or context, and an uncertain relaunch could create two orchestrators.
|
||||
|
||||
With no webhook, only `bridge_list`, `/healthz`, metrics, WARN logs, and the incident journal remain.
|
||||
These are passive surfaces. They are not a human notification.
|
||||
|
||||
## 8. Lost-boundary reconciliation
|
||||
|
||||
This is the only M4 path that reconstructs a result. It must prefer a visible stall over a fabricated
|
||||
reply.
|
||||
|
||||
### 8.1 Why normal completion rules are not enough
|
||||
|
||||
Current `CompletionResolver.resolve` has two fail-open rules. It resolves when the delivery baseline
|
||||
is missing. It also resolves an empty completion when the pane read fails. Those choices are valid
|
||||
after a trusted `WORKING -> IDLE` boundary because the bridge knows the turn ran. They are unsafe
|
||||
when health only guesses that a boundary was lost.
|
||||
|
||||
M4 gives each accepted send an internal `TurnToken`. It ties target, exact waiter, session turn,
|
||||
delivery baseline, and task outcome together.
|
||||
|
||||
### 8.2 Delivery baseline
|
||||
|
||||
Capture the baseline immediately after prompt send and before the delivery future completes. Store:
|
||||
|
||||
```text
|
||||
TurnToken
|
||||
exact waiter identity
|
||||
capture time and pane source
|
||||
normalised assistant block clipped to MAX_SCRAPE_CHARS
|
||||
whether a supported assistant marker was recognised
|
||||
capture result: PRESENT | READ_FAILED | UNRECOGNISED
|
||||
```
|
||||
|
||||
A failed or missing baseline never authorises repair. A late baseline is not valid evidence. After a
|
||||
daemon restart, the old waiter, task, token, and baseline are gone, so the old turn cannot be
|
||||
repaired.
|
||||
|
||||
Automatic repair is enabled only for agent kinds with tested assistant-block fixtures. Current
|
||||
extraction is Claude Code-specific and falls back to arbitrary raw text without `⏺`. That raw fallback
|
||||
cannot authorise repair. OpenCode repair stays disabled until live pane fixtures exist.
|
||||
|
||||
### 8.3 Strict gates and resolver result
|
||||
|
||||
Figure 3 shows the repair gates. Any failed gate keeps the waiter unchanged.
|
||||
|
||||
The two raw snapshots must describe the same `TurnToken` and session turn. No `WORKING`,
|
||||
`BLOCKED`, `UNKNOWN`, missing-agent, reply, failure, or new-delivery observation may occur between
|
||||
them.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Candidate["TURN_BOUNDARY_LOST candidate"] --> Stable{"Same TurnToken and BUSY turn<br/>across two raw IDLE or DONE snapshots?"}
|
||||
Stable -->|"no"| Resnapshot["Take a fresh snapshot"]
|
||||
Stable -->|"yes"| Waiter{"Exact captured waiter<br/>still open by identity?"}
|
||||
Waiter -->|"no"| Stale["STALE_TURN or ALREADY_RESOLVED"]
|
||||
Waiter -->|"yes"| Baseline{"Successful recognised<br/>delivery baseline exists?"}
|
||||
Baseline -->|"no"| Refuse["Refuse repair<br/>leave ticket pending"]
|
||||
Baseline -->|"yes"| Read{"Fresh pane read succeeds?"}
|
||||
Read -->|"no"| Refuse
|
||||
Read -->|"yes"| Output{"Recognised non-blank assistant block<br/>differs from clipped baseline?"}
|
||||
Output -->|"no"| Refuse
|
||||
Output -->|"yes"| Resolve["Shared CompletionResolver guard core<br/>resolves exact waiter"]
|
||||
Resolve -->|"won race"| Repaired["RECONCILED_COMPLETION<br/>same turn becomes DONE"]
|
||||
Resolve -->|"lost race"| Resnapshot
|
||||
```
|
||||
|
||||
*Figure 3. Repair needs stronger evidence than a normal observed turn boundary.*
|
||||
|
||||
Refactor the current resolver into one guard core with two policies:
|
||||
|
||||
```text
|
||||
resolveCaptured(target, inFlight, OBSERVED_BOUNDARY)
|
||||
resolveCaptured(target, inFlight, LOST_BOUNDARY_REPAIR)
|
||||
```
|
||||
|
||||
The health monitor calls only:
|
||||
|
||||
```text
|
||||
CompletionResolver.reconcileLostBoundary(target, expectedTurnToken)
|
||||
```
|
||||
|
||||
It returns `REPAIRED`, `ALREADY_RESOLVED`, `REFUSED_NO_CAPTURE`, `REFUSED_NO_BASELINE`,
|
||||
`REFUSED_UNREADABLE`, `REFUSED_UNCHANGED`, `REFUSED_AMBIGUOUS_OUTPUT`, `STALE_TURN`, or
|
||||
`RACE_LOST`.
|
||||
|
||||
Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move that turn from `BUSY` to `DONE`. A
|
||||
per-target reconciliation gate stops a queued second send from being accepted between waiter
|
||||
resolution and the FSM transition.
|
||||
|
||||
### 8.4 Lead-visible marker and refusal
|
||||
|
||||
A repaired result uses distinct `RECONCILED_COMPLETION` values in `Rendezvous`, `MessageService`,
|
||||
task poll source, and metrics. The lead sees:
|
||||
|
||||
```text
|
||||
[repaired completion - bridged detected a lost turn boundary. The member did not call
|
||||
bridge_reply; pane-derived text follows and may be partial]
|
||||
```
|
||||
|
||||
Clipped text also keeps the existing clipped-tail marker.
|
||||
|
||||
A refused repair leaves `TURN_BOUNDARY_LOST` open and the ticket pending. The report states that no
|
||||
reply was reconstructed and no task was replayed. `UNCHANGED`, `UNREADABLE`, and
|
||||
`AMBIGUOUS_OUTPUT` get at most one delayed retry for the same token. Missing capture or baseline gets
|
||||
no retry. After two refused scrapes, automatic repair stops for that token.
|
||||
|
||||
### 8.5 Target-wide teardown invariant
|
||||
|
||||
CB-568 owns the multi-ticket cancellation mechanism. M4 routes every terminal cause through that one
|
||||
idempotent operation and checks this independent invariant after teardown:
|
||||
|
||||
- no injector entry exists for the target;
|
||||
- no accepted turn or completion record exists;
|
||||
- no rendezvous waiter or ask exists;
|
||||
- every async task is terminal or was already terminal;
|
||||
- no thread waiting for the target send lock can later accept it;
|
||||
- new sends fail immediately;
|
||||
- each old task has one terminal outcome and one metric count.
|
||||
|
||||
A violation becomes `DELEGATION_ORPHANED`. The monitor may call the same idempotent target-wide
|
||||
failure operation once. It never recreates the task.
|
||||
|
||||
## 9. Capacity and utilisation
|
||||
|
||||
Capacity is a view, not a health state.
|
||||
|
||||
`bridge_list` adds one block per profile:
|
||||
|
||||
```text
|
||||
profile, maxLoad, live, free, reclaimable
|
||||
```
|
||||
|
||||
For an unlimited profile, `maxLoad` and `free` are null. `free` is
|
||||
`max(0, maxLoad - live)` for a capped profile.
|
||||
|
||||
The view must use the exact live-count function used by placement. A second calculation could show a
|
||||
free slot that placement then refuses. Member rows add `idleForSeconds` only when state is `READY` or
|
||||
`DONE`, no accepted turn exists, and the inbox is empty. `reclaimable` means only that the member
|
||||
holds capacity without open bridge work.
|
||||
|
||||
The existing idle-lead nudge gains a bounded capacity summary. It lists per-profile live, cap, free,
|
||||
and reclaimable counts, plus at most three long-idle members. Capacity does not make
|
||||
`FleetState.hasPending()` true. A changed capacity fingerprint may re-arm one capped heartbeat
|
||||
sequence. The fingerprint excludes changing idle durations, so a static idle fleet cannot reset the
|
||||
cap forever. Reply-push stand-down remains first.
|
||||
|
||||
The bridge must never:
|
||||
|
||||
- spawn a member because a slot is free;
|
||||
- generate a task or acceptance criteria;
|
||||
- move queued work to another member or profile;
|
||||
- treat a free slot or idle member as an incident;
|
||||
- stop an idle member only to improve utilisation.
|
||||
|
||||
The bridge knows capacity facts but has no work list. Only the lead has the plan, task context,
|
||||
side-effect history, and acceptance criteria.
|
||||
|
||||
Capacity calculation is in memory and adds no pane reads. Work-product checks run on a terminal
|
||||
session edge, not every fleet tick.
|
||||
|
||||
This capacity design adds no automatic stop. The accepted `NEVER_READY` cleanup can still stop a
|
||||
very slow startup after the existing grace, which is a known risk. Free capacity and long idle time
|
||||
never trigger that path.
|
||||
|
||||
## 10. Human escalation and notification
|
||||
|
||||
### 10.1 Escalation rule
|
||||
|
||||
Notify a person only when no healthy lead can act:
|
||||
|
||||
- `CONTROL_LINK_DOWN` survives grace;
|
||||
- a lead is unhealthy and no healthy peer can receive the incident;
|
||||
- a member incident has no known owning lead;
|
||||
- the only owning lead becomes unreachable, unresponsive, or stalled;
|
||||
- incident publication or routing itself fails.
|
||||
|
||||
Do not page a person for a member fault while a healthy owning lead exists. An uncollected member
|
||||
incident feeds lead-health evidence. If the lead then becomes unhealthy, peer or human routing starts.
|
||||
|
||||
### 10.2 Detection and notification switches
|
||||
|
||||
`health.enabled` controls detection and bridge-local reporting. It does not require a webhook.
|
||||
|
||||
`health.notifications.mode` is `disabled` or `webhook`. Disabled is valid and is the default.
|
||||
Webhook mode requires a resolved environment variable. Turning notification off stops outbound
|
||||
attempts but keeps incidents. Turning it back on resumes still-open human incidents.
|
||||
|
||||
Without a sink, `bridge_list.healthCoverage` states that human escalation is unavailable. `/healthz`
|
||||
keeps its existing HTTP liveness result and adds a nested `fleetHealth.status=partial` component.
|
||||
Metrics and one startup or reload WARN expose the same limit.
|
||||
|
||||
### 10.3 Incident and delivery deduplication
|
||||
|
||||
One open incident uses this key:
|
||||
|
||||
```text
|
||||
(scope, subjectStableId, state, causeFingerprint)
|
||||
```
|
||||
|
||||
The cause fingerprint includes stable error codes, dependency names, signature ids, or invariant
|
||||
names. It excludes times, ages, retry counts, pane text, and changing digests. A later recurrence
|
||||
after resolution gets a new generation and incident id.
|
||||
|
||||
Each outbound event uses:
|
||||
|
||||
```text
|
||||
Idempotency-Key = hash(incidentId, eventType, eventRevision)
|
||||
```
|
||||
|
||||
Event types are `open`, `severity_changed`, `reminder`, and `resolved`. Transport retries keep the
|
||||
same key.
|
||||
|
||||
An atomic owner-only journal beside the active config stores open incidents, routing, delivered
|
||||
revisions, retry state, and resolution state. It stores no pane or task content. Journal failure does
|
||||
not stop detection, but notification coverage becomes degraded.
|
||||
|
||||
### 10.4 Retry, reminder, and resolve
|
||||
|
||||
Send the first event immediately. Retry network errors, timeouts, HTTP 408, HTTP 429, and HTTP 5xx
|
||||
with full-jitter exponential backoff:
|
||||
|
||||
```text
|
||||
base: 5 seconds
|
||||
factor: 3
|
||||
maximum delay: 15 minutes
|
||||
one outstanding attempt per event
|
||||
```
|
||||
|
||||
Respect `Retry-After` up to 15 minutes. Other HTTP 4xx responses are permanent for that event until
|
||||
config changes or a person requests replay.
|
||||
|
||||
Transport retry is not an incident reminder. `humanRepeatSeconds` creates a new reminder revision
|
||||
for an unresolved critical incident after the last successful human event. Disabled mode does not
|
||||
build an unbounded reminder queue.
|
||||
|
||||
Send `resolved` only if at least one human event for that incident was delivered. If an incident
|
||||
resolves before its first successful delivery, cancel the pending open event and record local
|
||||
resolution.
|
||||
|
||||
### 10.5 Outbound payload boundary
|
||||
|
||||
An outbound payload may contain incident id and event type, severity, state, scope, stable bridge
|
||||
ids, role or profile, times, duration, structured evidence type and counts, recovery attempted,
|
||||
routing reason, coverage, and safe tool calls.
|
||||
|
||||
It must never contain:
|
||||
|
||||
- raw pane text, pane excerpts, or pane digests;
|
||||
- task briefs, prompts, or member reply content;
|
||||
- source files, diffs, or worktree file content;
|
||||
- worktree paths;
|
||||
- environment values, tokens, credentials, headers, or webhook URL;
|
||||
- raw exception messages or stack traces;
|
||||
- arbitrary model output.
|
||||
|
||||
The sink response body is ignored. A webhook cannot direct recovery. n8n remains outbound-only.
|
||||
|
||||
### 10.6 Metrics
|
||||
|
||||
M4 adds bounded-label series:
|
||||
|
||||
```text
|
||||
bridged_health_incidents{scope,state,severity}
|
||||
bridged_health_incidents_total{event}
|
||||
bridged_health_notifications_total{event,outcome}
|
||||
bridged_health_notification_queue_depth
|
||||
bridged_health_notification_last_success_seconds
|
||||
bridged_health_notification_capability{mode,status}
|
||||
bridged_lead_health{lead,state}
|
||||
bridged_lead_assigned_incidents{lead}
|
||||
```
|
||||
|
||||
Metric labels never include terminal ids, incident ids, URLs, or error text.
|
||||
|
||||
## 11. Configuration
|
||||
|
||||
The optional `health:` block is absent or disabled by default. The dormant monitor scheduler does no
|
||||
herdr or pane work while disabled. Every listed key is hot because the monitor reads `ConfigRef` on
|
||||
each tick or notification.
|
||||
|
||||
| Key | Class | Default and hard bound | Purpose |
|
||||
|---|---|---|---|
|
||||
| `health.enabled` | Hot | `false` | Enable detection and bridge-local reporting. |
|
||||
| `health.snapshotIntervalSeconds` | Hot | default 30, minimum 15 | Whole-fleet comparison cadence. |
|
||||
| `health.workingSuspectAfterSeconds` | Hot | default 600, minimum 300 | Age before working-pane probes. |
|
||||
| `health.paneProbeIntervalSeconds` | Hot | default 60, minimum 60 | Per-target pane cooldown. |
|
||||
| `health.leadUnresponsiveAfterSeconds` | Hot | default 300, minimum 120 | Delay after exhausted actionable nudges before lead fault. |
|
||||
| `health.humanRepeatSeconds` | Hot | default 3600, minimum 900 | Minimum repeat period for one open human incident. |
|
||||
| `health.capacityLongIdleAfterSeconds` | Hot | default 900, minimum 300 | Long-idle threshold for capacity summaries. |
|
||||
| `health.includePaneExcerpt` | Hot | `false` | Allow a clipped excerpt in local lead reports only. Human payloads still exclude it. |
|
||||
| `health.notifications.mode` | Hot | `disabled` | Select `disabled` or `webhook`. |
|
||||
| `health.notifications.webhookUrlEnv` | Hot | required in webhook mode | Name of the environment variable that holds the sink URL. |
|
||||
| `health.notifications.requestTimeoutMs` | Hot | default 10000, range 1000-30000 | Whole webhook request limit. |
|
||||
|
||||
Two consecutive snapshots are compiled floors for lost boundary, lead disappearance, and control
|
||||
link failure. The two-pane-reads-per-tick limit is also compiled and cannot be weakened by config.
|
||||
|
||||
## 12. Delivery units and acceptance
|
||||
|
||||
### Unit 1 - Evidence model and fleet snapshot
|
||||
|
||||
Scope: health state model, fleet join, clocks, evidence retention, and pane budget.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. One `agent.list` call covers one enabled fleet tick.
|
||||
2. Pure decision tests cover every state and every evidence limit in Section 4.
|
||||
3. `BUSY` plus stable raw `DONE` opens `TURN_BOUNDARY_LOST` after two snapshots.
|
||||
4. Healthy fleet snapshots perform zero pane reads.
|
||||
5. Pane cooldown, two-read fleet budget, and fair rotation cannot be disabled by config.
|
||||
6. Logs are outputs only; no log parsing exists.
|
||||
7. Fleet snapshots expose the same profile live-count calculation that placement uses.
|
||||
8. Capacity rows report cap, live, free, and reclaimable values without opening incidents.
|
||||
|
||||
### Unit 2 - Lost boundary and task reconciliation
|
||||
|
||||
Scope: accepted-turn identity, guarded repair, target-wide teardown, release causes, and preserved
|
||||
worktree discovery.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Every accepted send receives a stable `TurnToken` tied to target, exact waiter, session turn,
|
||||
delivery baseline, and task outcome.
|
||||
2. Repair requires the same `BUSY` token, two raw `IDLE` or `DONE` snapshots, no conflicting
|
||||
observation, exact open waiter, successful baseline, and new recognised assistant output.
|
||||
3. Missing, failed, late, or post-restart baseline never authorises repair.
|
||||
4. Repair is enabled only for agent kinds with tested assistant-block extraction. Raw-text fallback
|
||||
without a recognised marker refuses repair.
|
||||
5. Normal completion and repair use one resolver guard core. Waiter, scrape, clipping, unchanged, and
|
||||
exact-turn guards are not duplicated.
|
||||
6. `reconcileLostBoundary` returns every typed result named in Section 8.3.
|
||||
7. Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move the same turn to `DONE`.
|
||||
8. A per-target reconciliation gate blocks a queued second send during repair and FSM update.
|
||||
9. Repaired completion has distinct rendezvous kind, message outcome, poll source, lead marker, and
|
||||
metric. Clipping keeps its extra marker.
|
||||
10. Unchanged, unreadable, or ambiguous evidence gets at most one delayed retry. Missing capture or
|
||||
baseline gets none.
|
||||
11. Refusal leaves the ticket pending and tells the lead that no result was rebuilt or replayed.
|
||||
12. Release, gone, never-ready, and abnormal stop use CB-568's one idempotent target-wide failure
|
||||
operation.
|
||||
13. The post-teardown invariant in Section 8.5 is tested independently of CB-568 internals.
|
||||
14. A violated teardown invariant creates `DELEGATION_ORPHANED` and retries only the idempotent
|
||||
failure operation.
|
||||
15. `SPAWN_ROLLBACK` and normal `COMPLETED` remove worktrees. Abnormal and shutdown causes preserve
|
||||
them.
|
||||
16. Explicit stop is state-aware. Any pending task or non-terminal state preserves the worktree.
|
||||
17. Atomic preserved-worktree manifests reload after restart and appear in lead-only
|
||||
`bridge_list.preservedWorktrees`.
|
||||
18. Manifest failure preserves the worktree and opens an operator-visible health failure.
|
||||
19. Provision records the base commit. Terminal, long-idle worktrees report
|
||||
`WORK_PRODUCT_AT_RISK` only under the evidence in Section 4.2 and never auto-delete work.
|
||||
20. No path replays a delivered task, rebuilds its brief, or retargets it, even when prompt text is
|
||||
available.
|
||||
21. Tests cover both real traces, all repair refusals, clipping, explicit-reply and next-turn races,
|
||||
restart without capture, concurrent send and release, and preserved discovery after restart.
|
||||
|
||||
### Unit 3 - Typed inbox and member routing
|
||||
|
||||
Scope: semantic record, AMQP migration, both adapters, member routing, polling, and member health in
|
||||
`bridge_list`.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. AMQP selects legacy or typed decoding only from `content_type`; it never sniffs the body.
|
||||
2. Persistent `text/plain` from the old build becomes `kind=reply` with exact UTF-8 content,
|
||||
including content beginning with `{`.
|
||||
3. New entries use the vendor media type, `schemaVersion: 1`, UTF-8, persistent delivery, and AMQP
|
||||
message ids.
|
||||
4. Version 1 ignores unknown optional fields but rejects missing fields and identity mismatch.
|
||||
5. Unknown versions are not decoded or acked. They remain on the original queue and create one
|
||||
deduplicated failure.
|
||||
6. Invalid known data never escapes the callback, appears as a reply, or blocks later valid messages.
|
||||
7. Invalid data reaches durable per-target quarantine before original ack. Failed handoff leaves the
|
||||
original unacked.
|
||||
8. Decode failures create redacted WARN, metric, `bridge_list` summary, and routed incident without
|
||||
raw content.
|
||||
9. Both adapters pass one semantic contract for fields, FIFO, dedup, ownership, ack, and release.
|
||||
10. Lead keys require explicit ownership. Publication never claims a queue.
|
||||
11. Unit codec tests cover legacy `{`, Unicode, malformed UTF-8, typed round trip, additive fields,
|
||||
malformed JSON, missing fields, identity mismatch, media type, version, and dedup.
|
||||
12. A live broker contract writes old wire data and reads it with the new adapter after reconnect.
|
||||
13. Live contract tests cover mixed entries, quarantine confirm-before-ack, unsupported redelivery,
|
||||
later progress past poison, property persistence, lead ownership, and ack removal.
|
||||
14. Safe downgrade is documented as unsupported.
|
||||
15. RabbitMQ contract tests pass with `mvn test -Pcontract`. The same cases run once on production
|
||||
LavinMQ, or the release states that LavinMQ was not checked.
|
||||
16. Member incidents route to the exact delegating lead and never resolve a task rendezvous.
|
||||
17. `bridge_list` shows compact member health and capacity without pane content. Member
|
||||
`idleForSeconds` is present only when no accepted turn or inbox item exists.
|
||||
|
||||
### Unit 4 - Lead health and peer routing
|
||||
|
||||
Scope: lead evidence, exact ownership, peer selection, explicit-recipient push, and lead inbox
|
||||
lifecycle.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Lead identity uses the `CallerResolver` supplier. Liveness uses successful current agent data.
|
||||
2. Two successful-list absences with healthy ping become `LEAD_UNREACHABLE`; global link failure does
|
||||
not.
|
||||
3. Raw `WORKING`, raw `UNKNOWN`, first-seen time, failures, last success, and error class persist
|
||||
across ticks.
|
||||
4. Heartbeat and push publish status and nudge outcomes before safe no-injection decisions.
|
||||
5. Dynamic lead identity survives a two-successful-snapshot retirement grace.
|
||||
6. Member incidents first use exact delegation ownership with no singular-primary fallback.
|
||||
7. Peer selection follows the exclusions, load rule, and stable tie break in Section 7.4.
|
||||
8. A selected working peer is not interrupted. Its push waits for an injectable window.
|
||||
9. Recipient assignment stays pinned. Reassignment increments generation and supersedes old pending
|
||||
assignment.
|
||||
10. `bridge_list` shows bounded foreign assignments, recipient, reason, and generation without pane
|
||||
content.
|
||||
11. `LeadInboxRegistry` owns configured and discovered lead keys before publication.
|
||||
12. Missing leads keep ownership. Retirement needs an empty queue and handled incidents.
|
||||
13. Replacement owns the new key before messages move. Non-empty in-memory keys are not released.
|
||||
14. Tests cover dead versus busy, unknown, global failure, stale scan cache, disappearance, one peer,
|
||||
several peers, reassignment, and no peer.
|
||||
15. Adapter tests cover lead ownership, restart re-ownership, retirement, and terminal replacement.
|
||||
LavinMQ is checked or named as unchecked.
|
||||
16. A sole unreachable or stalled lead is never restarted or replaced. Without a sink, only passive
|
||||
evidence remains and every coverage surface says so.
|
||||
|
||||
### Unit 5 - Human sink, hot config, metrics, and operator coverage
|
||||
|
||||
Scope: generic webhook, config split, incident journal, retry, resolve, metrics, example config, and
|
||||
operator documentation.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. `health.enabled` works without a human sink.
|
||||
2. Notification mode is hot, defaults to disabled, and supports disabled or webhook.
|
||||
3. Webhook mode requires a resolved environment value. Bad notification config does not disable an
|
||||
already valid detector.
|
||||
4. Mode changes keep open incidents. Re-enable resumes eligible incidents.
|
||||
5. `bridge_list`, `/healthz`, metrics, and one WARN show partial coverage without a sink. HTTP
|
||||
liveness behavior stays unchanged.
|
||||
6. One-lead, no-sink coverage states that lead failure has no active notification or recovery.
|
||||
7. Incident and outbound dedupe use the stable keys in Section 10.3.
|
||||
8. The owner-only local journal survives restart and contains no pane or task content.
|
||||
9. Journal failure keeps detection running but marks notification coverage degraded.
|
||||
10. Retry tests cover network failure, timeout, 408, 429, `Retry-After`, 5xx, permanent 4xx, jitter,
|
||||
delay cap, config re-arm, and one outstanding attempt.
|
||||
11. Reminders and transport retries remain separate. Disabled mode does not build an unbounded queue.
|
||||
12. Resolve sends only after an earlier human event succeeded. Resolve-before-delivery cancels stale
|
||||
open delivery.
|
||||
13. Metrics use bounded labels and exclude ids, URLs, and error text.
|
||||
14. Payload tests reject every content type forbidden in Section 10.5.
|
||||
15. Webhook response bodies are ignored and cannot direct recovery.
|
||||
16. Tests cover disabled mode, one lead without sink, open/update/reminder/resolve, restart, dedup,
|
||||
reassignment, disable/re-enable, and sink failure while local health continues.
|
||||
17. `bridged.example.yaml` documents all hot keys and compiled floors.
|
||||
18. The operator Features wiki is updated separately. The portable `CLAUDE.md` block is checked and
|
||||
changed only if shipped tool or inbox semantics make it untrue.
|
||||
19. `mvn clean install` passes.
|
||||
|
||||
## 13. Not checked and release gates
|
||||
|
||||
These limits are part of the design, not optional follow-up notes.
|
||||
|
||||
- **OpenCode pane status and assistant markers were not checked.** OpenCode lost-boundary repair is
|
||||
disabled until live fixtures exist.
|
||||
- **Permission-prompt status was not checked** for Claude Code or OpenCode. `BLOCKED` remains
|
||||
ambiguous and has no automatic action.
|
||||
- **`recent_unwrapped` stability was not checked** across all supported agent kinds. If normalisation
|
||||
is not stable, `STALL_SUSPECTED` must say its evidence is weaker.
|
||||
- **The real `BUSY + DONE` trace was not replayed against live herdr.** The design uses the observed
|
||||
production trace and current poller behavior.
|
||||
- **CB-568 was not present when Unit 2 was designed.** Unit 2 must inspect the landed API and keep its
|
||||
independent teardown invariant.
|
||||
- **Production LavinMQ was not checked.** Existing durable-inbox contracts use RabbitMQ. Migration,
|
||||
quarantine, redelivery, lead ownership, and reassignment must run on LavinMQ before release or be
|
||||
recorded as unchecked.
|
||||
- **Live multi-lead routing was not checked.** Peer choice and reassignment are design rules backed by
|
||||
fake-clock and adapter tests until a live exercise runs.
|
||||
- **A live sole-lead failure with a webhook was not checked.** The no-peer path is a design result,
|
||||
not a tested recovery.
|
||||
- **No n8n, Slack, PagerDuty, or other receiver was checked.** The webhook remains generic and
|
||||
outbound-only.
|
||||
- **Deployment supervisor behavior for nested `/healthz` fields was not checked.** HTTP liveness
|
||||
status stays unchanged to reduce this risk.
|
||||
- **Incident-journal crash behavior was not checked** because the journal does not exist yet. Unit 5
|
||||
must test atomic replacement and restart recovery.
|
||||
- **Worktree merge state cannot be checked reliably** without forge or explicit collection evidence.
|
||||
`WORK_PRODUCT_AT_RISK` stays a warning.
|
||||
|
||||
## 14. Locked exclusions
|
||||
|
||||
M4 does not expose `agent.read` as a bridge tool. It does not add a workflow engine, inbound n8n
|
||||
authority, automatic task assignment, task replay, automatic lead replacement, or automatic member
|
||||
spawn for free capacity.
|
||||
|
||||
The bridge remains a message bus with evidence and bounded mechanical repair. The lead remains the
|
||||
place where judgement and work planning happen.
|
||||
Reference in New Issue
Block a user