Merge remote-tracking branch 'origin/main' into fix-charter
# Conflicts: # bridged/src/test/java/dev/ltms/bridged/session/SessionManagerTest.java
This commit is contained in:
@@ -23,6 +23,7 @@ import dev.ltms.bridged.mcp.BridgeMcp;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.health.FleetHealthMonitor;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.bridged.mcp.LsofProcessCwdLookup;
|
||||
@@ -192,34 +193,34 @@ public final class Bridged {
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the static registry, discover leads by the tab labels the operator
|
||||
// writes. CB-557 moved the settings onto the lead they describe, so scanning is on whenever
|
||||
// a `fleet.leaders:` entry exists — with no leads configured the supplier is a constant and
|
||||
// never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
// One scanner, so one prefix and one interval. Distinct per-lead prefixes would need a
|
||||
// scanner each; until a config actually wants that, take the first entry's settings and
|
||||
// say so, rather than silently honouring one lead's prefix and dropping another's.
|
||||
var scan = leaders.values().iterator().next();
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(Objects::nonNull)
|
||||
.collect(Collectors.toSet());
|
||||
leads = new LeadTabScanner(herdr, scan.tabPrefix(), memberSpaces, leadTerminals,
|
||||
TimeUnit.SECONDS.toNanos(scan.scanIntervalSeconds()), System::nanoTime);
|
||||
log.info("lead scan: tabs labelled '{}…' host a lead (rescan every {}s, member spaces {} "
|
||||
+ "excluded)",
|
||||
scan.tabPrefix(), scan.scanIntervalSeconds(), memberSpaces);
|
||||
long distinctPrefixes = leaders.values().stream()
|
||||
.map(BridgedConfig.Leader::tabPrefix).distinct().count();
|
||||
if (distinctPrefixes > 1) {
|
||||
log.warn("fleet.leaders declares {} different tabPrefix values; only '{}' is scanned "
|
||||
+ "for. Give every lead the same tabPrefix, or leads under the others "
|
||||
+ "will not be discovered.",
|
||||
distinctPrefixes, scan.tabPrefix());
|
||||
}
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
leads = new LeadTabScanner(herdr, tabToName, memberSpaces,
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, member spaces {} excluded)",
|
||||
tabToName.keySet(), scanIntervalSeconds, memberSpaces);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
@@ -275,9 +276,9 @@ public final class Bridged {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
completion.onDelivered(target);
|
||||
sessions.onDelivered(target);
|
||||
public void onDelivered(String target, dev.ltms.bridged.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -285,6 +286,12 @@ public final class Bridged {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Injector injector = new Injector(agents, turnListener, deliverableTo(presence, leads),
|
||||
presence::forget);
|
||||
@@ -348,6 +355,28 @@ public final class Bridged {
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
@@ -387,7 +416,12 @@ public final class Bridged {
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime));
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
new BridgeMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}));
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
@@ -408,6 +442,7 @@ public final class Bridged {
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
|
||||
@@ -73,6 +73,7 @@ public record BridgedConfig(
|
||||
Primary primary,
|
||||
Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat,
|
||||
Health health,
|
||||
String placement,
|
||||
Auth auth,
|
||||
ConfigReload configReload) {
|
||||
@@ -83,7 +84,16 @@ public record BridgedConfig(
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, String placement, Auth auth) {
|
||||
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, placement, auth, null);
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, null, placement, auth, null);
|
||||
}
|
||||
|
||||
/** Back-compat form before the optional {@code health:} block was added. */
|
||||
public BridgedConfig(Bind bind, String herdrSocket, Map<String, Profile> profiles, Guard guard,
|
||||
String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
|
||||
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
|
||||
LeadHeartbeat leadHeartbeat, String placement, Auth auth, ConfigReload configReload) {
|
||||
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
|
||||
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, null, placement, auth, configReload);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -377,6 +387,17 @@ public record BridgedConfig(
|
||||
boolean clearAfterTurn) {
|
||||
}
|
||||
|
||||
/** Optional fleet detection. A missing block stays dormant. */
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Health(Boolean enabled, Integer intervalSeconds, Integer workingSuspectAfterSeconds,
|
||||
Integer paneProbeIntervalSeconds, Notifications notifications) {
|
||||
public boolean isEnabled() { return Boolean.TRUE.equals(enabled); }
|
||||
public int intervalOrDefault() { return Math.max(15, intervalSeconds == null ? 30 : intervalSeconds); }
|
||||
public record Notifications(String mode) {
|
||||
public boolean configured() { return "webhook".equalsIgnoreCase(mode); }
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* External AMQP broker for durable, cross-restart reply delivery (CB-307 Stage 2). Its mere
|
||||
* presence swaps the in-memory {@code ReplyInbox} for the AMQP-backed adapter; absent, bridged
|
||||
@@ -437,24 +458,32 @@ public record BridgedConfig(
|
||||
* pre-existed, which is why it had to be recognised by configuration rather than created. With
|
||||
* {@code profile} and {@code instances} the daemon may stand one up when none is live, so the
|
||||
* pane no longer has to exist before the daemon does. Recognition still comes first: a lead
|
||||
* already running under {@code tabPrefix} is adopted, and only the shortfall is launched.
|
||||
* already running in its configured {@code tab} is adopted, and only the shortfall is launched.
|
||||
*
|
||||
* <p><b>{@code tab} replaced {@code terminal} (CB-579).</b> A herdr {@code terminal_id} changes
|
||||
* every time the lead's session restarts, so pinning one cost a config edit and a daemon restart
|
||||
* per restart. A tab is stable: a human opens it once, it holds exactly one pane, and its label
|
||||
* survives restarts of the agent inside it — so identity is now the tab label alone.
|
||||
*
|
||||
* @param profile the {@code profiles:} entry to launch this lead on when one must
|
||||
* be created; {@code null} ⇒ recognise-only, never create
|
||||
* @param terminal the lead's herdr {@code terminal_id} when pinned by hand; the only
|
||||
* field identity depends on. {@code null} ⇒ found by {@code tabPrefix}
|
||||
* @param tab the exact tab label hosting this lead, matched case-insensitively;
|
||||
* the only field identity depends on. Required — a lead with no
|
||||
* {@code tab} can never be discovered, launched or not
|
||||
* @param instances how many of this lead should be live (default 1). The daemon
|
||||
* launches only the shortfall, so a restart adopts rather than doubles
|
||||
* @param tabPrefix label prefix marking this lead's tab, matched case-insensitively;
|
||||
* the remainder is the lead's name ({@code "lead: opus"} →
|
||||
* {@code opus}). Default {@code "lead:"}
|
||||
* @param tabPrefix no longer used to find a lead's tab — {@code tab} is matched
|
||||
* exactly. Its only remaining job is the startup collision guard
|
||||
* ({@link #validateLeadTabPrefixes()}), which still uses it to refuse
|
||||
* a worker {@code tabLabel} template that could be misread as a lead.
|
||||
* Default {@code "lead:"}
|
||||
* @param scanIntervalSeconds how long a tab scan is cached before herdr is asked again; also the
|
||||
* worst case before a newly-labelled tab is recognised. Default 10
|
||||
* @param kind which agent runs there ({@code claude}, {@code opencode}, …)
|
||||
* @param model the model or selector it runs, for operators reading the roster
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Leader(String profile, String terminal, Integer instances, String tabPrefix,
|
||||
public record Leader(String profile, String tab, Integer instances, String tabPrefix,
|
||||
Integer scanIntervalSeconds, String kind, String model,
|
||||
String workspace, String cwd) {
|
||||
|
||||
@@ -472,12 +501,13 @@ public record BridgedConfig(
|
||||
(scanIntervalSeconds == null || scanIntervalSeconds <= 0) ? 10 : scanIntervalSeconds;
|
||||
workspace = (workspace == null || workspace.isBlank())
|
||||
? DEFAULT_WORKSPACE : workspace.strip();
|
||||
tab = (tab == null || tab.isBlank()) ? null : tab.strip();
|
||||
}
|
||||
|
||||
/** Back-compat 7-arg form — no workspace or cwd, so both take their defaults. */
|
||||
public Leader(String profile, String terminal, Integer instances, String tabPrefix,
|
||||
public Leader(String profile, String tab, Integer instances, String tabPrefix,
|
||||
Integer scanIntervalSeconds, String kind, String model) {
|
||||
this(profile, terminal, instances, tabPrefix, scanIntervalSeconds, kind, model, null, null);
|
||||
this(profile, tab, instances, tabPrefix, scanIntervalSeconds, kind, model, null, null);
|
||||
}
|
||||
|
||||
/** True when this lead may be launched by the daemon rather than only recognised. */
|
||||
@@ -485,9 +515,9 @@ public record BridgedConfig(
|
||||
return profile != null && !profile.isBlank() && instances > 0;
|
||||
}
|
||||
|
||||
/** The tab label an auto-launched instance of this lead gets — what the scanner reads back. */
|
||||
public String tabLabel(String name) {
|
||||
return tabPrefix + " " + name;
|
||||
/** The tab label an auto-launched instance of this lead gets — its configured {@code tab}. */
|
||||
public String tabLabel() {
|
||||
return tab;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -686,28 +716,20 @@ public record BridgedConfig(
|
||||
}
|
||||
|
||||
/**
|
||||
* The terminal → lead-name map that {@link dev.ltms.bridged.auth.CallerResolver} resolves
|
||||
* against, merging the {@code leaders:} registry with the legacy singular {@code primary:} pin.
|
||||
* The terminal → lead-name map seeded from the legacy singular {@code primary:} pin (CB-530).
|
||||
*
|
||||
* <p>Precedence: an explicit {@code leaders:} entry wins over the {@code primary:} pin for the
|
||||
* same terminal. The pin is the older, less expressive spelling of the same fact, so when both
|
||||
* name a pane the named entry is the one an operator meant. The pin is still honoured on its
|
||||
* own — a config carrying only {@code primary:} behaves exactly as it did before CB-530.
|
||||
* <p>{@code fleet.leaders} no longer carries a per-entry terminal pin (CB-579): a lead's identity
|
||||
* comes from its {@code tab} alone, resolved live by {@code LeadTabScanner}. This method now
|
||||
* exists only for the {@code primary.terminal} fallback — a config that never migrated off it
|
||||
* still resolves that one pane as a lead named {@code "primary"}, exactly as before CB-530.
|
||||
*
|
||||
* @return an unmodifiable map, empty when neither block is configured (nothing is pinned, and
|
||||
* every pane therefore resolves as a worker — the pre-CB-307 behaviour)
|
||||
* @return an unmodifiable map, empty when {@code primary.terminal} is not configured (nothing is
|
||||
* pinned, and every pane therefore resolves as a worker — the pre-CB-307 behaviour)
|
||||
*/
|
||||
public Map<String, String> leaderTerminals() {
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
fleet.leaders().forEach((name, leader) -> {
|
||||
if (leader != null && leader.terminal() != null && !leader.terminal().isBlank()) {
|
||||
byTerminal.put(leader.terminal(), name);
|
||||
}
|
||||
});
|
||||
}
|
||||
if (primary != null && primary.terminal() != null && !primary.terminal().isBlank()) {
|
||||
byTerminal.putIfAbsent(primary.terminal(), "primary");
|
||||
byTerminal.put(primary.terminal(), "primary");
|
||||
}
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
@@ -812,13 +834,14 @@ public record BridgedConfig(
|
||||
private static final Set<String> KNOWN_TOP_LEVEL_KEYS = Set.of(
|
||||
"bind", "herdrSocket", "profiles", "guard", "worktreeRoot",
|
||||
"lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs", "broker", "primary", "fleet",
|
||||
"leadHeartbeat", "placement", "auth", "configReload");
|
||||
"leadHeartbeat", "health", "placement", "auth", "configReload");
|
||||
|
||||
/** Load and validate config from {@code path}. */
|
||||
public static BridgedConfig load(Path path) {
|
||||
try {
|
||||
String yaml = Files.readString(path);
|
||||
rejectRenamedTopLevelKeys(yaml);
|
||||
rejectLeaderTerminalKey(yaml);
|
||||
warnUnknownTopLevelKeys(yaml, path);
|
||||
rejectDuplicateMemberSlots(yaml);
|
||||
BridgedConfig cfg = YAML.readValue(yaml, BridgedConfig.class);
|
||||
@@ -1043,6 +1066,43 @@ public record BridgedConfig(
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Reject a config whose {@code fleet.leaders.<name>} still carries the retired {@code terminal:}
|
||||
* pin (CB-579), naming {@code tab:} as its replacement.
|
||||
*
|
||||
* <p>{@code Leader} is {@code @JsonIgnoreProperties(ignoreUnknown = true)}, so simply dropping
|
||||
* the record component would make a leftover {@code terminal:} key silently no-op — the daemon
|
||||
* would start, the pin would never take effect, and nothing would say why. Fatal and specific
|
||||
* instead, exactly like {@link #rejectRenamedTopLevelKeys}, which this mirrors for a key one
|
||||
* level deeper than the ones that method covers.
|
||||
*
|
||||
* @param yaml the raw config text
|
||||
* @throws IllegalStateException when any {@code fleet.leaders.<name>.terminal} key is present
|
||||
*/
|
||||
static void rejectLeaderTerminalKey(String yaml) {
|
||||
Map<?, ?> raw;
|
||||
try {
|
||||
raw = YAML.readValue(yaml, Map.class);
|
||||
} catch (IOException | IllegalArgumentException e) {
|
||||
return; // a malformed file is reported by the real parse, not here
|
||||
}
|
||||
if (raw == null || !(raw.get("fleet") instanceof Map<?, ?> fleet)
|
||||
|| !(fleet.get("leaders") instanceof Map<?, ?> leaders)) {
|
||||
return;
|
||||
}
|
||||
List<String> bad = leaders.entrySet().stream()
|
||||
.filter(e -> e.getValue() instanceof Map<?, ?> leader && leader.containsKey("terminal"))
|
||||
.map(e -> String.valueOf(e.getKey()))
|
||||
.sorted()
|
||||
.toList();
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: fleet.leaders entries ["
|
||||
+ String.join(", ", bad) + "] still use the retired 'terminal:' key — replace it "
|
||||
+ "with 'tab:', the exact tab label hosting the lead. A terminal_id changes on "
|
||||
+ "every restart of the lead's session; a tab label does not.");
|
||||
}
|
||||
}
|
||||
|
||||
static List<String> unknownTopLevelKeys(String yaml) {
|
||||
Map<?, ?> raw;
|
||||
try {
|
||||
@@ -1082,7 +1142,7 @@ public record BridgedConfig(
|
||||
// defaults the fields of a block that IS present. Defaulting it here would start watching
|
||||
// the file for every config that never asked to be watched.
|
||||
return new BridgedConfig(b, herdrSocket, profiles, g, worktreeRoot, l, timeout, pollMs,
|
||||
broker, primary, f, leadHeartbeat, placementOrDefault, a, configReload);
|
||||
broker, primary, f, leadHeartbeat, health, placementOrDefault, a, configReload);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -1292,10 +1352,9 @@ public record BridgedConfig(
|
||||
+ "', which is not a configured profiles: entry (have: " + profiles.keySet()
|
||||
+ ").");
|
||||
}
|
||||
if (!leader.isCreatable() && (leader.terminal() == null || leader.terminal().isBlank())) {
|
||||
bad.add("fleet.leaders." + name + " can neither be found nor created — it pins no "
|
||||
+ "terminal: and names no profile: to launch one on. Give it one or the "
|
||||
+ "other, or drop the entry.");
|
||||
if (leader.tab() == null || leader.tab().isBlank()) {
|
||||
bad.add("fleet.leaders." + name + " has no tab: — a lead is now found (and, if "
|
||||
+ "auto-launched, labelled) purely by its tab, so every entry must name one.");
|
||||
}
|
||||
});
|
||||
if (!bad.isEmpty()) {
|
||||
|
||||
@@ -30,7 +30,11 @@ import java.util.function.Supplier;
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config.</li>
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Bridged.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code guard:},
|
||||
|
||||
@@ -0,0 +1,151 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long intervalSeconds;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
|
||||
// These facts need the evidence publishers introduced by later M4 units. They are not negatives.
|
||||
private static final boolean NOT_YET_OBSERVED = false;
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code BridgeMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<Agent> agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, NOT_YET_OBSERVED,
|
||||
messages.hasInboxMessage(session.terminalId()), agent != null, NOT_YET_OBSERVED,
|
||||
NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
clock.getAsLong());
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// A list failure is health evidence, and must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
@@ -6,6 +6,7 @@ import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -23,6 +24,15 @@ import java.util.function.Supplier;
|
||||
* by first starting the session and asking it. Scanning closes that loop: label the tab, and the
|
||||
* pane is recognised on the next resolve.
|
||||
*
|
||||
* <p><strong>CB-579 — matched by name, not prefix.</strong> This used to strip one shared
|
||||
* {@code tabPrefix} off a label to derive the lead's name, and merged a config-supplied
|
||||
* {@code terminal_id} pin over every scan result so the pin could never expire. Both are gone: each
|
||||
* lead now configures its own exact {@code tab} label ({@code fleet.leaders.<name>.tab}), so this
|
||||
* class is handed a {@code tab → name} map up front and matches labels against it exactly
|
||||
* (case-insensitively). There is no merge step — a scan result is the whole answer. That is the
|
||||
* fix for the bug this replaces: a {@code terminal_id} pin surviving in config after the pane it
|
||||
* named was gone, so the daemon kept treating a dead session as a live lead forever.
|
||||
*
|
||||
* <p><strong>Direction of trust.</strong> The label names the lead; it never <em>grants</em>
|
||||
* anything a pane could take for itself. Three properties keep that honest:
|
||||
* <ol>
|
||||
@@ -48,7 +58,7 @@ import java.util.function.Supplier;
|
||||
* ever make a decision that <em>removes</em> something based on this map, add the same check.
|
||||
* The remaining hazard is an <em>operator</em> one — a worker {@code tabLabel} template that
|
||||
* happens to start with the same prefix would promote the whole fleet — and that is refused at
|
||||
* startup by {@code BridgedConfig.validateLeadScan} rather than documented here.
|
||||
* startup by {@code BridgedConfig.validateLeadTabPrefixes} rather than documented here.
|
||||
*
|
||||
* <p><strong>Caching.</strong> {@link #get()} is on the request path (every resolve), so the scan
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan keeps
|
||||
@@ -60,38 +70,46 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadTabScanner.class);
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final String tabPrefix;
|
||||
private final Map<String, String> tabToName;
|
||||
private final Set<String> excludedWorkspaceLabels;
|
||||
private final Map<String, String> configuredLeads;
|
||||
private final long ttlNanos;
|
||||
private final LongSupplier clock;
|
||||
|
||||
private Map<String, String> cached;
|
||||
private Map<String, String> cached = Map.of();
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
* @param tabPrefix a tab whose label starts with this (case-insensitively) hosts a
|
||||
* lead; the rest of the label, trimmed, is the lead's name
|
||||
* @param tabToName every configured lead's exact tab label → its name
|
||||
* ({@code fleet.leaders.<name>.tab}), matched case-insensitively
|
||||
* @param excludedWorkspaceLabels workspaces never scanned — the configured worker spaces
|
||||
* @param configuredLeads the static {@code leaders:}/{@code primary:} registry, merged
|
||||
* over every scan result. Explicit config outranks the
|
||||
* convention, and survives a scan that cannot run at all
|
||||
* @param ttlNanos how long a scan result is reused before the next one
|
||||
* @param clock nanosecond time source ({@code System::nanoTime} in production)
|
||||
*/
|
||||
public LeadTabScanner(HerdrClient herdr, String tabPrefix, Set<String> excludedWorkspaceLabels,
|
||||
Map<String, String> configuredLeads, long ttlNanos, LongSupplier clock) {
|
||||
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
||||
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||
this.herdr = herdr;
|
||||
this.tabPrefix = tabPrefix == null || tabPrefix.isBlank() ? "lead:" : tabPrefix.strip();
|
||||
this.tabToName = normalize(tabToName);
|
||||
this.excludedWorkspaceLabels = excludedWorkspaceLabels == null
|
||||
? Set.of() : Set.copyOf(excludedWorkspaceLabels);
|
||||
this.configuredLeads = configuredLeads == null ? Map.of() : Map.copyOf(configuredLeads);
|
||||
this.ttlNanos = ttlNanos;
|
||||
this.clock = clock;
|
||||
this.cached = this.configuredLeads;
|
||||
}
|
||||
|
||||
/** Keys stripped and lower-cased once, so every lookup is a plain map hit. */
|
||||
private static Map<String, String> normalize(Map<String, String> tabToName) {
|
||||
if (tabToName == null || tabToName.isEmpty()) {
|
||||
return Map.of();
|
||||
}
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
tabToName.forEach((tab, name) -> {
|
||||
if (tab != null && !tab.isBlank() && name != null && !name.isBlank()) {
|
||||
out.put(tab.strip().toLowerCase(Locale.ROOT), name);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -151,25 +169,20 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
}
|
||||
}
|
||||
}
|
||||
byTerminal.putAll(configuredLeads); // an explicit pin outranks a label
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead name a tab label declares, or {@code null} if it declares none.
|
||||
* The lead name a tab label declares, or {@code null} if it names none of the configured leads.
|
||||
*
|
||||
* <p>{@code "lead: opus-5.0"} → {@code "opus-5.0"}. A bare {@code "lead:"} names nobody and is
|
||||
* rejected: an unnamed lead would resolve as {@code PRIMARY} with nothing to attribute it to.
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String l = label.strip();
|
||||
if (!l.regionMatches(true, 0, tabPrefix, 0, tabPrefix.length())) {
|
||||
return null;
|
||||
}
|
||||
String name = l.substring(tabPrefix.length()).strip();
|
||||
return name.isEmpty() ? null : name;
|
||||
return tabToName.get(label.strip().toLowerCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -83,17 +84,17 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target);
|
||||
captureBaseline(target, token);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
void captureBaseline(String target, TurnToken token) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
@@ -137,7 +138,13 @@ public final class CompletionResolver implements TurnListener {
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn));
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, null));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, reason));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
@@ -192,6 +199,11 @@ public final class CompletionResolver implements TurnListener {
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
}
|
||||
|
||||
/** Synchronous fail with an optional reason supplied by a dropped worker queue. */
|
||||
void fail(String target, InFlight turn, String explicitReason) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
@@ -201,16 +213,18 @@ public final class CompletionResolver implements TurnListener {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason;
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
String reason = explicitReason;
|
||||
if (reason == null || reason.isBlank()) {
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -129,7 +130,7 @@ public final class Injector {
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, CompletableFuture<Void> delivered) {
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
@@ -159,9 +160,9 @@ public final class Injector {
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text) {
|
||||
public CompletableFuture<Void> enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, delivered);
|
||||
Pending p = new Pending(text, token, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
@@ -356,7 +357,7 @@ public final class Injector {
|
||||
} else {
|
||||
// Baseline the pane's pre-turn content so a misattributed completion (no new output)
|
||||
// can't resolve this send with the previous turn's stale answer (CB-115).
|
||||
turnListener.onDelivered(target);
|
||||
turnListener.onDelivered(target, sent.token());
|
||||
sent.delivered().complete(null);
|
||||
}
|
||||
}
|
||||
@@ -410,8 +411,8 @@ public final class Injector {
|
||||
for (Pending p : pending) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
if (hadDeliveredTurn) {
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
// A queued send has no in-flight record, while a delivered turn does. CompletionResolver
|
||||
// handles both forms and resolves its waiter at most once.
|
||||
turnListener.onTurnFailed(target, cause.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
|
||||
/**
|
||||
* Notified when a worker's delegated turn is observed to complete — a confirmed
|
||||
* {@code WORKING → IDLE} transition after a delivery. This is the CB-106 completion signal the
|
||||
@@ -41,6 +43,14 @@ public interface TurnListener {
|
||||
default void onTurnFailed(String target) {
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #onTurnFailed(String)}, carrying the reason a worker became unreachable. The default
|
||||
* keeps existing listeners working while allowing the completion resolver to report a useful cause.
|
||||
*/
|
||||
default void onTurnFailed(String target, String reason) {
|
||||
onTurnFailed(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* A message was just delivered into {@code target}'s pane (CB-115). Fired so the completion
|
||||
* resolver can snapshot the pane's pre-turn content: a later {@link #onTurnComplete} whose
|
||||
@@ -49,7 +59,7 @@ public interface TurnListener {
|
||||
* resolve the send with the previous turn's stale answer. A default no-op keeps the interface
|
||||
* functional for callers that don't scrape.
|
||||
*/
|
||||
default void onDelivered(String target) {
|
||||
default void onDelivered(String target, TurnToken token) {
|
||||
}
|
||||
|
||||
/** No-op default for callers that only need delivery, not completion signalling. */
|
||||
|
||||
@@ -59,7 +59,7 @@ public final class LeadLauncher {
|
||||
/**
|
||||
* @param agents herdr agent control (start, list)
|
||||
* @param spaces workspace / tab control (ensure, create, label, list)
|
||||
* @param cfg the loaded config — {@code fleet.leaders}, {@code profiles} and the lead pins
|
||||
* @param cfg the loaded config — {@code fleet.leaders}, {@code profiles} and each lead's tab
|
||||
*/
|
||||
public LeadLauncher(AgentControl agents, WorkspaceControl spaces, BridgedConfig cfg) {
|
||||
this.agents = agents;
|
||||
@@ -102,8 +102,8 @@ public final class LeadLauncher {
|
||||
continue;
|
||||
}
|
||||
if (!lead.isCreatable()) {
|
||||
// A lead with a `terminal:` pin and no `profile:` is recognise-only by design: the
|
||||
// operator opens it by hand. Say so once rather than looking like a silent failure.
|
||||
// A lead with a `tab:` but no `profile:` is recognise-only by design: the operator
|
||||
// opens it by hand. Say so once rather than looking like a silent failure.
|
||||
log.info("lead '{}' is not live, and names no profile — it can be recognised but not "
|
||||
+ "launched. Add `profile:` under fleet.leaders.{} to have bridged start it.",
|
||||
name, name);
|
||||
@@ -127,18 +127,16 @@ public final class LeadLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* How many live leads exist per configured name.
|
||||
* How many live leads exist per configured name: a running agent in a tab labelled with that
|
||||
* lead's exact {@code tab} (CB-579). Member workspaces are excluded, exactly as the scanner
|
||||
* excludes them: a member must not be counted as a lead because it happens to sit in a matching
|
||||
* tab.
|
||||
*
|
||||
* <p>Two independent pieces of evidence, because either alone double-spawns:
|
||||
* <ul>
|
||||
* <li>a running agent in a tab labelled {@code "<tabPrefix> <name>"} — how an auto-launched
|
||||
* lead, or an operator following the labelling convention, is found;</li>
|
||||
* <li>a running agent on a terminal the config pins in {@code fleet.leaders.<name>.terminal} —
|
||||
* how a lead the operator opened and pinned by hand is found. Without this, a pinned lead
|
||||
* whose tab carries no matching label would be relaunched on every boot.</li>
|
||||
* </ul>
|
||||
* Member workspaces are excluded, exactly as the scanner excludes them: a member must not be
|
||||
* counted as a lead because it happens to sit in a matching tab.
|
||||
* <p>There used to be a second path here — a running agent on the terminal a
|
||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||
* pin is retired: {@code tab} is now the only field identity depends on, and {@link Agent}
|
||||
* already carries {@link Agent#tabId()} directly, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is — by labelling its tab to match.
|
||||
*/
|
||||
private Map<String, Integer> liveLeads(Map<String, BridgedConfig.Leader> leaders) {
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
@@ -160,20 +158,9 @@ public final class LeadLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
// terminalId → the lead name the config pins it to.
|
||||
Map<String, String> nameByPinnedTerminal = new LinkedHashMap<>();
|
||||
leaders.forEach((name, lead) -> {
|
||||
if (lead.terminal() != null && !lead.terminal().isBlank()) {
|
||||
nameByPinnedTerminal.put(lead.terminal().strip(), name);
|
||||
}
|
||||
});
|
||||
|
||||
Map<String, Integer> counts = new LinkedHashMap<>();
|
||||
for (Agent a : agents.list()) {
|
||||
String name = nameByTab.get(a.tabId());
|
||||
if (name == null) {
|
||||
name = nameByPinnedTerminal.get(a.terminalId());
|
||||
}
|
||||
if (name != null) {
|
||||
counts.merge(name, 1, Integer::sum);
|
||||
}
|
||||
@@ -184,7 +171,7 @@ public final class LeadLauncher {
|
||||
/**
|
||||
* The configured lead a tab label names, or {@code null} for a label that names none.
|
||||
*
|
||||
* <p>Matched against the declared lead names rather than by splitting on the prefix, so an
|
||||
* <p>Matched exactly (case-insensitively) against each lead's configured {@code tab}, so an
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead.
|
||||
*/
|
||||
private String leadNameOf(String label, Map<String, BridgedConfig.Leader> leaders) {
|
||||
@@ -193,7 +180,8 @@ public final class LeadLauncher {
|
||||
}
|
||||
String l = label.strip();
|
||||
for (Map.Entry<String, BridgedConfig.Leader> e : leaders.entrySet()) {
|
||||
if (l.equalsIgnoreCase(e.getValue().tabLabel(e.getKey()).strip())) {
|
||||
String tab = e.getValue().tabLabel();
|
||||
if (tab != null && l.equalsIgnoreCase(tab.strip())) {
|
||||
return e.getKey();
|
||||
}
|
||||
}
|
||||
@@ -202,7 +190,7 @@ public final class LeadLauncher {
|
||||
|
||||
/** Start one lead. Returns false (having logged) rather than throwing on any failure. */
|
||||
private boolean launch(String name, BridgedConfig.Leader lead, BridgedConfig.Profile profile) {
|
||||
String label = lead.tabLabel(name);
|
||||
String label = lead.tabLabel();
|
||||
String cwd = (lead.cwd() == null || lead.cwd().isBlank())
|
||||
? System.getProperty("user.dir") : lead.cwd();
|
||||
|
||||
|
||||
@@ -77,6 +77,7 @@ public final class BridgeMcp {
|
||||
private final CallerResolver authz; // CB-501: null → authorization not enforced (legacy)
|
||||
private final Metrics metrics; // CB-502: null → auth failures not counted
|
||||
private final CapacitySource capacity;
|
||||
private final HealthCoverageSource healthCoverage;
|
||||
|
||||
/** Capacity facts used by {@code bridge_list}; production must supply the placement live count. */
|
||||
public record CapacitySource(Function<String, Integer> liveCount, Function<String, Integer> maxLoad,
|
||||
@@ -86,6 +87,9 @@ public final class BridgeMcp {
|
||||
boolean available() { return !configuredProfiles.get().isEmpty(); }
|
||||
}
|
||||
|
||||
/** Coverage is supplied by the health wiring, not inferred from a missing dependency. */
|
||||
public record HealthCoverageSource(Supplier<String> value) { }
|
||||
|
||||
/**
|
||||
* @param callers resolves each call's {@link Principal}; {@code null} disables authorization.
|
||||
* This surface needs its own enforcement: {@code /mcp} is a raw servlet on
|
||||
@@ -95,8 +99,9 @@ public final class BridgeMcp {
|
||||
*/
|
||||
public BridgeMcp(MessageService messages, PeerLauncher workers, SessionManager sessions,
|
||||
ConnectionIdentity identity, MemberPresence presence, PrimaryRegistry primaryRegistry,
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity) {
|
||||
CallerResolver callers, Metrics metrics, CapacitySource capacity, HealthCoverageSource healthCoverage) {
|
||||
this.capacity = capacity;
|
||||
this.healthCoverage = healthCoverage;
|
||||
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
||||
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
||||
.jsonMapper(json)
|
||||
@@ -210,7 +215,7 @@ public final class BridgeMcp {
|
||||
.toolCall(listTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions, messages, capacity,
|
||||
return listFleet(workers, sessions, messages, capacity, healthCoverage,
|
||||
callers == null ? Map.of() : callers.leads(),
|
||||
callerTerminal(exchange));
|
||||
})
|
||||
@@ -724,11 +729,11 @@ public final class BridgeMcp {
|
||||
*/
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
return listFleet(workers, sessions, null, CapacitySource.none(), leads, selfTerm);
|
||||
return listFleet(workers, sessions, null, CapacitySource.none(), new HealthCoverageSource(() -> "off"), leads, selfTerm);
|
||||
}
|
||||
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions, MessageService messages,
|
||||
CapacitySource capacity,
|
||||
CapacitySource capacity, HealthCoverageSource healthCoverage,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
try {
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
@@ -747,6 +752,7 @@ public final class BridgeMcp {
|
||||
roster.stream().map(MemberSession::profile).forEach(profiles::add);
|
||||
Map<String, Object> result = new LinkedHashMap<>();
|
||||
result.put("leads", leadRows); result.put("members", out);
|
||||
result.put("healthCoverage", healthCoverage.value().get());
|
||||
if (capacity.available()) result.put("capacity", profiles.stream()
|
||||
.map(profile -> capacityView(profile, capacity.liveCount(), capacity.maxLoad(), roster, messages,
|
||||
capacity.clock().getAsLong())).toList());
|
||||
|
||||
@@ -170,8 +170,9 @@ public final class MessageService {
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
/** The async task that currently owns a target's send lock. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTarget = new ConcurrentHashMap<>();
|
||||
/** Async task that owns each exact forward rendezvous waiter. */
|
||||
private final ConcurrentHashMap<CompletableFuture<Rendezvous.Resolution>, Task> asyncTasksByWaiter =
|
||||
new ConcurrentHashMap<>();
|
||||
/** Async tickets paused on a specific {@code bridge_ask} turn. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTurn = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
@@ -300,14 +301,18 @@ public final class MessageService {
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
return false; // nobody is blocked on this worker — nothing to abandon
|
||||
boolean failed = waiter != null && !waiter.isDone() && rendezvous.resolveFailure(waiter, reason);
|
||||
boolean asyncFailed = false;
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null
|
||||
&& task.future.complete(new Reply(Outcome.WORKER_FAILED, reason))) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
}
|
||||
boolean failed = rendezvous.resolveFailure(waiter, reason);
|
||||
if (failed) {
|
||||
log.warn("abandoning the blocked send to {}: {}", target, reason);
|
||||
}
|
||||
return failed;
|
||||
return failed || asyncFailed;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -354,6 +359,11 @@ public final class MessageService {
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
return send(target, content, timeoutMillis, onAccepted, null);
|
||||
}
|
||||
|
||||
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
@@ -361,6 +371,9 @@ public final class MessageService {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
if (task != null && task.future.isDone()) {
|
||||
return task.future.getNow(null);
|
||||
}
|
||||
if (hasAsyncQuestion(target)) {
|
||||
return new Reply(Outcome.BUSY, null); // the worker's current turn is paused for its lead
|
||||
}
|
||||
@@ -372,13 +385,17 @@ public final class MessageService {
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
try {
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
TurnToken token = new TurnToken(target, reply);
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content);
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content, token);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
@@ -395,6 +412,7 @@ public final class MessageService {
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
}
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(target, reply);
|
||||
}
|
||||
} finally {
|
||||
@@ -420,11 +438,15 @@ public final class MessageService {
|
||||
if (ticket.fresh()) {
|
||||
// Register the reverse waiter first, then surface the question — so the answer, which can
|
||||
// arrive the instant the primary reacts, always finds an open waiter to resolve.
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(workerSession);
|
||||
Task task = markAsyncQuestion(waiter, question, ticket.turnId());
|
||||
if (!rendezvous.resolveQuestion(workerSession, question, ticket.turnId())) {
|
||||
if (task != null) {
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
}
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
markAsyncQuestion(workerSession, question, ticket.turnId());
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
@@ -523,21 +545,15 @@ public final class MessageService {
|
||||
tasks.put(ticket, task);
|
||||
asyncExecutor.submit(() -> {
|
||||
try {
|
||||
Runnable trackingAccepted = () -> {
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
asyncTasksByTarget.put(target, task);
|
||||
};
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, trackingAccepted);
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task);
|
||||
if (result.outcome() == Outcome.QUESTION) {
|
||||
asyncTasksByTarget.remove(target, task);
|
||||
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
||||
// just after resolveQuestion wakes this thread.
|
||||
} else {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
asyncTasksByTarget.remove(target, task);
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
@@ -599,13 +615,14 @@ public final class MessageService {
|
||||
}
|
||||
|
||||
/** Record the active question for an async ticket; blocking sends have no entry and stay unchanged. */
|
||||
private void markAsyncQuestion(String target, String text, String turnId) {
|
||||
Task task = asyncTasksByTarget.get(target);
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
@@ -623,7 +640,6 @@ public final class MessageService {
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
asyncTasksByTarget.remove(task.target, task);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,20 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
|
||||
/**
|
||||
* Identity for one accepted send. The session turn is deliberately absent: CompletionResolver's
|
||||
* delivery callback runs before SessionManager.onDelivered, so binding it needs a later ordering design.
|
||||
*/
|
||||
public final class TurnToken {
|
||||
private final String target;
|
||||
private final CompletableFuture<Rendezvous.Resolution> waiter;
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter) {
|
||||
this.target = target;
|
||||
this.waiter = waiter;
|
||||
}
|
||||
|
||||
public String target() { return target; }
|
||||
public CompletableFuture<Rendezvous.Resolution> waiter() { return waiter; }
|
||||
}
|
||||
@@ -167,6 +167,22 @@ public final class GitWorktrees implements Worktrees {
|
||||
exec("git", "-C", repoRoot, "worktree", "remove", "--force", worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// A worktree that is already gone holds no work to lose, and it must not break teardown:
|
||||
// git -C <missing-dir> status exits non-zero and would throw where release() is mid-way
|
||||
// through stopping a pane. Mirror remove()'s already-gone tolerance by treating it as clean.
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing can be uncommitted", worktreePath);
|
||||
return false;
|
||||
}
|
||||
// No --untracked-files=no: the exact shape of the work lost in CB-576 was a new file
|
||||
// that was never added, so an untracked-only worktree is still dirty.
|
||||
String out = exec("git", "-C", worktreePath, "status", "--porcelain");
|
||||
return !out.isBlank();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlay == null || overlay.isEmpty()) {
|
||||
|
||||
@@ -4,6 +4,7 @@ import dev.ltms.bridged.auth.MemberLifecycle;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
@@ -201,6 +202,16 @@ public final class SessionManager implements TurnListener {
|
||||
removed.paneId(), removed.terminalId(), removed.state(), cause);
|
||||
if (preserveWorktree && removed.worktree() != null) {
|
||||
logPreservedForShutdown(removed);
|
||||
} else if (removed.worktree() != null && worktrees.hasUncommitted(removed.worktree())) {
|
||||
// CB-576: a release that would otherwise remove the worktree finds it holding
|
||||
// uncommitted work the bridge cannot see. A worker that ends a turn without
|
||||
// committing (normally because it stopped to ask a question or refused the turn)
|
||||
// has its only copy of that work in the worktree. Remove would --force-delete it,
|
||||
// so preserve the directory and tell an operator where to find it.
|
||||
preserveWorktree = true;
|
||||
log.warn("release {} preserves dirty worktree {} for pane={} terminal={}: "
|
||||
+ "the worktree holds uncommitted changes that --force remove would destroy",
|
||||
cause, removed.worktree(), removed.paneId(), removed.terminalId());
|
||||
}
|
||||
// CB-516: a send still waiting on this worker can never be answered now. Tell the
|
||||
// listener BEFORE the pane is torn down, so a blocked caller fails fast with a real
|
||||
@@ -437,7 +448,7 @@ public final class SessionManager implements TurnListener {
|
||||
* can be re-delivered for multi-turn reuse until it is released.
|
||||
*/
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) return;
|
||||
if (current.state() != MemberSession.State.READY && current.state() != MemberSession.State.DONE) {
|
||||
|
||||
@@ -10,6 +10,18 @@ public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/**
|
||||
* True when the worktree holds uncommitted changes the bridge cannot see: tracked
|
||||
* modifications, staged files, or untracked files. {@code git status --porcelain} is the
|
||||
* test; an empty result means clean. Callers use this to decide whether removing the
|
||||
* worktree would silently destroy a worker's only copy of its work.
|
||||
*
|
||||
* <p>An already-gone worktree is reported as clean (no throw), matching {@link #remove}'s
|
||||
* idempotent contract: a path that does not exist holds no work to lose, and must not break
|
||||
* a teardown that is mid-way through stopping the pane.
|
||||
*/
|
||||
boolean hasUncommitted(String worktreePath);
|
||||
|
||||
/** Copy each existing overlay path repoRoot→worktree; mark tracked ones --skip-worktree. */
|
||||
void overlayParity(String repoRoot, String worktreePath, List<String> overlay);
|
||||
|
||||
|
||||
Reference in New Issue
Block a user