Compare commits

..

1 Commits

Author SHA1 Message Date
Dai Ha 2757bc7185 autoCompactWindow: per-profile bounded auto-compaction, both backends
CI / contract (pull_request) Successful in 41s
CI / build (pull_request) Failing after 1m37s
Add opt-in Integer autoCompactWindow to FleetConfig.Profile (last field,
null/unset = today's behaviour). Validated at config load to [100000,
1000000] — the band Claude Code's own --autocompact flag accepts.

Claude Code: appends --autocompact <window> to argv (mirrors --model),
so it survives the ccs <profile> wrapper.

opencode: has no absolute compact-at-N knob (only compaction.auto/prune/
reserved/tail_turns/preserve_recent_tokens), so the window is applied as
the resolved model's own limit.context (+ a required limit.output:16384
default) in the generated opencode.json, merged via get-or-create nodes
so it does not clobber a custom-provider block. Only applies when model:
resolves to "provider/model"; otherwise logs a WARN naming the profile
rather than silently doing nothing.

Docs added to fleetd.example.yaml explaining the cross-backend semantics
difference (compacts AT the window vs. WITHIN it).
2026-08-24 16:54:40 +02:00
10 changed files with 343 additions and 804 deletions
+13 -17
View File
@@ -140,6 +140,18 @@ herdrSocket: ~/.config/herdr/herdr.sock
# e.g. `env DISPLAY=:10.0 idea {dir}`. Best-effort: a failure is logged, never
# fails the spawn. Omit to open the member's module by hand. There is no close
# half yet — an opened module stays open until the operator closes it.
# autoCompactWindow → opt-in, default off. A bounded token window that forces a spawned member to
# compact its context instead of running on the backend's own default and dying
# mid-turn (losing its fleet_reply — the whole point of the turn — with it).
# Validated at config load to [100000, 1000000] — the band Claude Code's own
# --autocompact flag accepts.
# CROSS-BACKEND SEMANTICS DIFFER: on claude-code this is a launch-time
# `--autocompact <tokens>` flag — the member compacts AT this window. opencode
# has no equivalent flag (it only forces `compaction.auto: true`, unconditionally,
# already), so this is instead applied as the model's `limit.context` in the
# generated opencode.json — the member compacts WITHIN this window, not exactly
# at it — and only when this profile's `model:` is in `provider/model` form; if it
# isn't, bridged logs a WARN naming the profile rather than silently doing nothing.
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
# omit for a backend that needs no token (e.g. a local ollama).
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
@@ -256,6 +268,7 @@ profiles:
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
# ideProjectDir: bridged # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in bridged/)
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
# autoCompactWindow: 250000 # opt-in: bound member context; claude-code compacts AT this, opencode within it (model limit.context)
gx11: # a second backend, so `placement: weighted` has a choice
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
placement: tab
@@ -654,23 +667,6 @@ guard:
# uriEnv: LAVINMQ_URI
# prefetch: 32
# Shared cross-host LEADER coordination broker. OMIT this block to leave lead-to-lead messaging
# off entirely (config-only in this ticket — nothing here wires it into a live LeadMailbox yet).
# This is a SEPARATE AMQP vhost from `broker:` above: member/worker inboxes always stay on the
# per-fleet `broker:` vhost, and this vhost carries only leader-to-leader traffic, so two fleets
# whose members must never see each other can still share one coordination vhost for their leads.
# uriEnv → name of a host env var holding the coordination AMQP URI, same convention as
# broker.uriEnv (keeps the credential out of fleetd.yaml). Wins over `uri` when set.
# selfId → this daemon's own lead coord-id — the name its mailbox is owned under
# (lead.<selfId>.inbox), e.g. "mac-opus" or "fleet01-lead". Must be globally unique
# across every daemon sharing this vhost.
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
# Default 32 when omitted.
# coordinator:
# uriEnv: LEAD_COORD_URI
# selfId: mac-opus
# prefetch: 32
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
@@ -6,7 +6,6 @@ import com.fasterxml.jackson.core.JsonToken;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.fasterxml.jackson.dataformat.yaml.YAMLFactory;
import dev.ltms.fleet.msg.AmqpReplyInbox;
import dev.ltms.fleet.msg.LeadMailbox;
import dev.ltms.fleet.peer.MemberRole;
import dev.ltms.fleet.placement.PlacementPolicies;
import org.slf4j.Logger;
@@ -70,11 +69,6 @@ import java.util.Set;
* @param memberCredentials deny-by-default policy (CB-596) for which of the operator's own host
* credentials a spawned member's pane inherits. {@code null} (the block
* omitted) blocks nothing — see {@link MemberCredentials}.
* @param coordinator shared cross-host leader coordination broker: a SEPARATE AMQP vhost from
* {@link #broker} used only for lead-to-lead traffic (member/worker inboxes
* stay on {@code broker}'s vhost). {@code null} → no lead mailbox is opened.
* Config parsing + accessors only — nothing here wires it into a live
* {@code LeadMailbox}; that is a separate ticket. See {@link Coordinator}.
*/
@JsonIgnoreProperties(ignoreUnknown = true)
public record FleetConfig(
@@ -95,20 +89,7 @@ public record FleetConfig(
Auth auth,
ConfigReload configReload,
Integer quarantineCooldownSeconds,
MemberCredentials memberCredentials,
Coordinator coordinator) {
/** Back-compat form before the {@code coordinator:} block was added. */
public FleetConfig(Bind bind, String herdrSocket, Map<String, Profile> profiles, Guard guard,
String worktreeRoot, Lifecycle lifecycle, Integer spawnReadyTimeoutMs,
Integer spawnReadyPollMs, Broker broker, Primary primary, Fleet fleet,
LeadHeartbeat leadHeartbeat, Health health, String placement, Auth auth,
ConfigReload configReload, Integer quarantineCooldownSeconds,
MemberCredentials memberCredentials) {
this(bind, herdrSocket, profiles, guard, worktreeRoot, lifecycle, spawnReadyTimeoutMs,
spawnReadyPollMs, broker, primary, fleet, leadHeartbeat, health, placement, auth,
configReload, quarantineCooldownSeconds, memberCredentials, null);
}
MemberCredentials memberCredentials) {
/** Back-compat form before the CB-596 {@code memberCredentials:} block was added. */
public FleetConfig(Bind bind, String herdrSocket, Map<String, Profile> profiles, Guard guard,
@@ -294,6 +275,19 @@ public record FleetConfig(
* profile that does not opt in. Read live off the current config, so it is
* HOT: a change takes effect on the next exhaustion classification / spawn,
* no restart needed.
* @param autoCompactWindow opt-in per-profile token window that forces a spawned member to
* auto-compact its context at (Claude Code) or within (opencode) a bound the
* operator chooses, instead of the backend's own default. {@code null} (the
* default) leaves today's behaviour exactly — opencode already forces
* {@code compaction.auto: true} unconditionally (CB-523) but has no absolute
* window, and Claude Code has neither. When set, validated at config load to
* {@code [100000, 1000000]} — the band Claude Code's own {@code --autocompact
* <tokens>} flag accepts. The two backends honour it differently: Claude Code
* compacts AT this window (a launch-time {@code --autocompact} flag);
* opencode has no such knob, so this is applied as the model's
* {@code limit.context} instead, which bounds the window opencode compacts
* <em>within</em>, and only when {@code model} resolves to a
* {@code provider/model} pair.
*/
@JsonIgnoreProperties(ignoreUnknown = true)
public record Profile(String profile, String baseUrl, String model,
@@ -311,7 +305,8 @@ public record FleetConfig(
String credentialId,
String ideMcpUrl,
String ideProjectDir,
String ideOpenCommand) {
String ideOpenCommand,
Integer autoCompactWindow) {
/** Peer kind spawned by {@link dev.ltms.fleet.member.ClaudeCodeLauncher} (the default). */
public static final String KIND_CLAUDE_CODE = "claude-code";
@@ -381,6 +376,10 @@ public record FleetConfig(
// substituted; blank ⇒ no auto-open (the operator opens the module by hand).
ideProjectDir = (ideProjectDir == null || ideProjectDir.isBlank()) ? null : ideProjectDir;
ideOpenCommand = (ideOpenCommand == null || ideOpenCommand.isBlank()) ? null : ideOpenCommand;
// Opt-in per profile, default off (null). No clamping here — unlike ideMcpUrl/ideProjectDir
// there is no blank-string form to normalize (it's an Integer), and the [100000, 1000000]
// range is enforced eagerly at config load (rejectAutoCompactWindowOutOfRange), naming the
// profile, rather than silently clamped here. A profile that never sets it keeps null.
}
/**
@@ -425,7 +424,29 @@ public record FleetConfig(
public Profile withProfile(String p) {
return new Profile(p, baseUrl, model, configDir, tokenEnv, argv, placement, workspace, tabLabel,
mcpUrl, cwd, parityOverlay, gitTokenEnv, gitHostEnv, kind, env, weight, maxLoad, subscription,
exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand);
exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand, autoCompactWindow);
}
/**
* Backward-compatible constructor without the {@code autoCompactWindow} field — the profile
* leaves auto-compaction at the backend's own default (opencode's unconditional
* {@code compaction.auto: true}, or Claude Code's built-in threshold), exactly as before this
* key existed. This is the shape the canonical constructor had before the field was added —
* every pre-existing Java call site (and any YAML that omits the key) keeps compiling and
* behaving identically; Jackson binds the canonical (longest) constructor, so YAML omitting
* {@code autoCompactWindow:} still lands here as {@code null} via that path, not this one.
*/
public Profile(String profile, String baseUrl, String model,
String configDir, String tokenEnv, List<String> argv,
String placement, String workspace, String tabLabel, String mcpUrl,
String cwd, List<String> parityOverlay, String gitTokenEnv, String gitHostEnv,
String kind, Map<String, String> env, Float weight, Integer maxLoad,
Boolean subscription, String exhaustedPattern, String credentialId,
String ideMcpUrl, String ideProjectDir, String ideOpenCommand) {
this(profile, baseUrl, model, configDir, tokenEnv, argv, placement, workspace, tabLabel,
mcpUrl, cwd, parityOverlay, gitTokenEnv, gitHostEnv, kind, env, weight, maxLoad,
subscription, exhaustedPattern, credentialId, ideMcpUrl, ideProjectDir, ideOpenCommand,
null);
}
/**
@@ -669,82 +690,6 @@ public record FleetConfig(
}
}
/**
* Shared cross-host leader coordination broker: a {@link LeadMailbox} lets two leads on
* different daemons — possibly different hosts — exchange durable messages, which a herdr pane
* injection (how {@code fleet_send} reaches a lead today) cannot do at all. Its mere presence is
* config only in this ticket: nothing here opens a live {@code LeadMailbox} yet, that wiring is
* a separate ticket.
*
* <p>Deliberately a SEPARATE vhost from {@link Broker}, not a reuse of it. {@link Broker} is
* per-fleet — its queues are named by worker session id, and two fleets sharing one broker stay
* isolated by vhost (see {@code Two fleets share one LavinMQ}). Leader coordination is meant to
* cross exactly that boundary: two independently-owned fleets' leads talking to each other. Using
* the same vhost would either leak member traffic across the fleet boundary this is meant to
* cross, or force every fleet's members onto one shared vhost to get leader coordination — a
* second, dedicated vhost keeps "member inboxes stay fleet-local" true while still letting leads
* reach across fleets.
*
* @param uri AMQP connection URI for the coordination vhost, e.g.
* {@code amqp://guest:guest@127.0.0.1:5672/coord}. Blank/{@code null} ⇒ the
* coordinator block is treated as unconfigured. Ignored when {@code uriEnv} is set.
* @param uriEnv name of a host env var holding the AMQP URI, same convention as
* {@link Broker#uriEnv()} — keeps the credential out of the config file. Wins
* over {@code uri} whenever set. Blank/{@code null} ⇒ ignored.
* @param selfId this daemon's own lead coord-id — the name its {@code LeadMailbox} is owned
* under ({@code lead.<selfId>.inbox}), e.g. {@code "mac-opus"}. Blank/
* {@code null} ⇒ kept as {@code null} (no self id configured).
* @param prefetch the consumer's {@code basicQos} prefetch count. {@code null}/non-positive ⇒
* {@link LeadMailbox#DEFAULT_PREFETCH}.
*/
@JsonIgnoreProperties(ignoreUnknown = true)
public record Coordinator(String uri, String uriEnv, String selfId, Integer prefetch) {
public Coordinator {
selfId = (selfId == null || selfId.isBlank()) ? null : selfId;
}
/** True when a {@code uriEnv} is configured by name, whether or not its variable resolves. */
public boolean hasUriEnv() {
return uriEnv != null && !uriEnv.isBlank();
}
/**
* True when a usable coordination broker URI is configured (an empty block does not enable
* it). Honors {@code uriEnv} first: if it names a variable that is unset or blank, the
* coordinator is <em>not</em> configured — a bare {@code uri} is only consulted when no
* {@code uriEnv} is set.
*/
public boolean isConfigured() {
return effectiveUri() != null;
}
/**
* The effective AMQP URI to connect with. {@code uriEnv} wins when set (both over
* {@code uri} and alone). When {@code uriEnv} names a variable that is unset or blank,
* returns {@code null} rather than falling back to {@code uri} — an operator who moved to
* the secret store must not silently drop back onto a stale clear-text URI. Returns the
* literal {@code uri} when no {@code uriEnv} is configured.
*/
public String effectiveUri() {
return effectiveUri(System.getenv());
}
/** As {@link #effectiveUri()}, reading the variable value from {@code env} (the injection seam). */
public String effectiveUri(Map<String, String> env) {
if (hasUriEnv()) {
String value = env.get(uriEnv);
return (value != null && !value.isBlank()) ? value : null;
}
return (uri != null && !uri.isBlank()) ? uri : null;
}
/** The prefetch to use, defaulting to {@link LeadMailbox#DEFAULT_PREFETCH} when unset. */
public int prefetchOrDefault() {
return (prefetch != null && prefetch > 0) ? prefetch : LeadMailbox.DEFAULT_PREFETCH;
}
}
/**
* Optional pinned primary terminal config (CB-307). When present with a non-blank
* {@code terminal}, the bridge uses this as the primary's herdr identity instead of
@@ -1279,7 +1224,7 @@ public record FleetConfig(
"bind", "herdrSocket", "profiles", "guard", "worktreeRoot",
"lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs", "broker", "primary", "fleet",
"leadHeartbeat", "health", "placement", "auth", "configReload", "quarantineCooldownSeconds",
"memberCredentials", "coordinator");
"memberCredentials");
/** Load and validate config from {@code path}. */
public static FleetConfig load(Path path) {
@@ -1290,6 +1235,7 @@ public record FleetConfig(
warnUnknownTopLevelKeys(yaml, path);
rejectDuplicateMemberSlots(yaml);
rejectNegativeMaxLoad(yaml);
rejectAutoCompactWindowOutOfRange(yaml);
rejectUnknownKind(yaml);
rejectUnknownAuthMode(yaml);
rejectUnknownPlacement(yaml);
@@ -1592,6 +1538,52 @@ public record FleetConfig(
}
}
/** Lowest {@code autoCompactWindow} Claude Code's {@code --autocompact <tokens>} flag accepts. */
static final int AUTO_COMPACT_WINDOW_MIN = 100_000;
/** Highest {@code autoCompactWindow} Claude Code's {@code --autocompact <tokens>} flag accepts. */
static final int AUTO_COMPACT_WINDOW_MAX = 1_000_000;
/**
* Reject a profile whose {@code autoCompactWindow:} is set but outside the token band Claude
* Code's own {@code --autocompact <tokens>} flag accepts (100k–1M), naming both the profile and
* the value.
*
* <p>Unset/{@code null} means "off" and passes silently — today's behaviour for every profile
* that does not opt in (see {@link Profile#autoCompactWindow()}). A profile that DOES set the key
* is validated eagerly, at config load, rather than failing later when Claude Code itself refuses
* the launch flag on spawn — the same "fail loud at load, not lazily at first spawn" reasoning as
* {@link #rejectNegativeMaxLoad} and {@link #rejectUnknownPlacementPolicy}.
*
* @param yaml the raw config text
* @throws IllegalStateException when any profile's {@code autoCompactWindow} is set and outside
* {@code [100000, 1000000]}
*/
static void rejectAutoCompactWindowOutOfRange(String yaml) {
Map<?, ?> raw;
try {
raw = YAML.readValue(yaml, Map.class);
} catch (IOException | IllegalArgumentException e) {
return; // a malformed file is reported by the real parse, not here
}
if (raw == null || !(raw.get("profiles") instanceof Map<?, ?> profiles)) {
return;
}
List<String> bad = profiles.entrySet().stream()
.filter(e -> e.getValue() instanceof Map<?, ?> p
&& p.get("autoCompactWindow") instanceof Number n
&& (n.doubleValue() < AUTO_COMPACT_WINDOW_MIN || n.doubleValue() > AUTO_COMPACT_WINDOW_MAX))
.map(e -> String.valueOf(e.getKey()))
.sorted()
.toList();
if (!bad.isEmpty()) {
throw new IllegalStateException("refusing to start: profile(s) [" + String.join(", ", bad)
+ "] set autoCompactWindow outside [" + AUTO_COMPACT_WINDOW_MIN + ", "
+ AUTO_COMPACT_WINDOW_MAX + "] — Claude Code's --autocompact flag accepts only "
+ "that band of tokens; omit the key to leave auto-compaction at each backend's "
+ "own default.");
}
}
/** The peer kinds this build has an adapter for — {@link Profile#kind()}'s only valid values. */
private static final Set<String> KNOWN_KINDS = Set.of(Profile.KIND_CLAUDE_CODE, Profile.KIND_OPENCODE);
@@ -1848,11 +1840,9 @@ public record FleetConfig(
// and an upgrade must not change what a running deployment's members inherit.
MemberCredentials mc = memberCredentials != null ? memberCredentials
: new MemberCredentials(null, List.of(), List.of());
// coordinator is left as-is, like broker/primary above: null keeps no LeadMailbox opened,
// and this ticket's Coordinator is config-only anyway (nothing yet reads it at startup).
return new FleetConfig(b, herdrSocket, profiles, g, worktreeRoot, l, timeout, pollMs,
broker, primary, f, leadHeartbeat, health, placementOrDefault, a, configReload,
quarantineCooldown, mc, coordinator);
quarantineCooldown, mc);
}
/**
@@ -248,7 +248,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
// has neither MCP nor a charter — session flags must be added into a list we own.
List<String> argv = mutableArgv(argvWithFleet(cfg, spec));
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
return new Launch(workerEnv, argvWithModel(argv, cfg), agentSessionId);
return new Launch(workerEnv, argvWithAutoCompact(argvWithModel(argv, cfg), cfg), agentSessionId);
}
/**
@@ -463,6 +463,30 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
return withModel;
}
/**
* Pin a bounded auto-compaction window on the command line via {@code --autocompact <tokens>},
* opt-in per profile (CB-634's sibling ticket: a member that runs out of context dies mid-turn
* and its {@code fleet_reply} — the whole point of the turn — is lost with it; opencode already
* forces {@code compaction.auto: true} unconditionally, CB-523, but Claude Code has no equivalent
* and runs at the backend's own default window).
*
* <p>Mirrors {@link #argvWithModel}: appended after it, so it survives the {@code ccs <profile>}
* wrapper the same way {@code --model} does, and outranks env/settings and the operator's own
* {@code argv}. Verified: {@code claude 2.1.241 --help} lists {@code --autocompact <auto|tokens>}
* (either the literal {@code auto}, or an integer 100k–1M) — {@link FleetConfig#load} rejects a
* configured value outside that band before this ever runs, so the flag Claude Code receives here
* is always in range.
*/
private static List<String> argvWithAutoCompact(List<String> argv, FleetConfig.Profile cfg) {
if (cfg.autoCompactWindow() == null) {
return argv;
}
List<String> withAutoCompact = mutableArgv(argv);
withAutoCompact.add("--autocompact");
withAutoCompact.add(String.valueOf(cfg.autoCompactWindow()));
return withAutoCompact;
}
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
/** Spawn a worker for the default profile in the resolved default cwd. */
@@ -209,9 +209,19 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
@Override
protected Launch buildLaunch(FleetConfig.Profile cfg, LaunchSpec spec) {
Map<String, String> workerEnv = baseEnv(cfg);
// autoCompactWindow's opencode lever (limit.context) only targets a specific provider/model
// entry, so it needs model: in "provider/model" form. A profile that opts in without that
// shape gets no silent no-op — log it, once, here, whether or not writeConfig ends up running.
boolean wantsContextLimit = cfg.autoCompactWindow() != null && splitProviderModel(cfg.model()) != null;
if (cfg.autoCompactWindow() != null && !wantsContextLimit) {
log.warn("profile '{}' sets autoCompactWindow but model '{}' is not \"<provider>/<model>\" "
+ "form — opencode's per-model context limit could not be applied for this profile",
cfg.profile(), cfg.model());
}
// A config file is needed for the bridge MCP mount, a member charter, the IDE MCP (+ its
// guidance overlay, CB-634), or a pinned endpoint (CB-508).
if (cfg.hasMcp() || cfg.hasIdeMcp() || spec.charter() != null || hasCustomProvider(cfg)) {
// guidance overlay, CB-634), a pinned endpoint (CB-508), or a resolvable autoCompactWindow.
if (cfg.hasMcp() || cfg.hasIdeMcp() || spec.charter() != null || hasCustomProvider(cfg)
|| wantsContextLimit) {
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter(), spec.cwd()).toString());
}
applyGitToken(workerEnv, cfg);
@@ -360,6 +370,9 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
if (hasCustomProvider(cfg)) {
addCustomProvider(root, cfg);
}
if (cfg.autoCompactWindow() != null) {
applyContextLimit(root, cfg);
}
Path cfgFile = dir.resolve("opencode.json");
// Built with Jackson rather than string concatenation: the provider block is nested and
@@ -404,18 +417,68 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
* default gateway — a worker quietly talking to the wrong endpoint is the failure this avoids.
*/
private static String[] splitModelSelector(FleetConfig.Profile cfg) {
String model = cfg.model();
int slash = model == null ? -1 : model.indexOf('/');
if (model == null || model.isBlank() || slash <= 0 || slash == model.length() - 1) {
String[] parts = splitProviderModel(cfg.model());
if (parts == null) {
throw new IllegalArgumentException(
"profile " + cfg.profile() + " sets baseUrl (a pinned opencode endpoint) so"
+ " model: must be \"<provider>/<model>\", e.g."
+ " \"local-vllm/deepseek-v4-flash\"; got "
+ (model == null ? "null" : '"' + model + '"'));
+ (cfg.model() == null ? "null" : '"' + cfg.model() + '"'));
}
return parts;
}
/**
* Split {@code model} into its {@code provider} and {@code model} halves, or {@code null} when
* it is not in that shape (unset/blank, or no non-trailing {@code /}). Unlike
* {@link #splitModelSelector}, non-throwing — callers that only *optionally* need the split
* (autoCompactWindow's context-limit application) use this to fall back to a WARN rather than an
* exception, since a profile without {@code baseUrl} is not required to name a provider/model.
*/
private static String[] splitProviderModel(String model) {
int slash = model == null ? -1 : model.indexOf('/');
if (model == null || model.isBlank() || slash <= 0 || slash == model.length() - 1) {
return null;
}
return new String[]{model.substring(0, slash), model.substring(slash + 1)};
}
/**
* Apply the per-profile {@code autoCompactWindow} as opencode's per-model context limit.
*
* <p>opencode has no absolute "compact at N tokens" knob — its {@code compaction} block only
* exposes {@code auto}/{@code prune}/{@code reserved}/{@code tail_turns}/
* {@code preserve_recent_tokens} — so the real lever is the model's own
* {@code provider.<p>.models.<m>.limit.context}, which bounds the window opencode compacts
* <em>within</em> rather than compacting exactly AT it the way Claude Code's {@code --autocompact}
* does.
*
* <p>Uses get-or-create nodes ({@code withObject}) at every level so this MERGES with any provider
* block {@link #addCustomProvider} already wrote for a custom-provider (pinned-endpoint) profile —
* it must never overwrite that block's {@code npm}/{@code name}/{@code options}. For a gateway
* profile (no {@code baseUrl}, so no prior provider block) this writes a partial
* {@code provider.<p>.models.<m>.limit} override, which opencode merges over its own built-in
* provider definition.
*
* <p>opencode's {@code limit} schema requires both {@code context} and {@code output}; there is no
* independent signal for the latter here, so 16384 is written as a safe default (documented in
* {@code fleetd.example.yaml}).
*
* <p>Silently does nothing when {@code model:} is not in {@code provider/model} form — a warning
* for that case is already logged once in {@code buildLaunch}, so this stays quiet rather than
* duplicating it.
*/
private static void applyContextLimit(ObjectNode root, FleetConfig.Profile cfg) {
String[] parts = splitProviderModel(cfg.model());
if (parts == null) {
return;
}
ObjectNode limit = root.withObject("provider").withObject(parts[0])
.withObject("models").withObject(parts[1]).withObject("limit");
limit.put("context", cfg.autoCompactWindow());
limit.put("output", 16384);
}
/**
* The OpenAI-compatible base URL for {@code baseUrl}. A bare {@code host:port} gets {@code /v1}
* appended (where these servers put the API); a URL that already carries a path is taken as-is,
@@ -1,422 +0,0 @@
package dev.ltms.fleet.msg;
import com.fasterxml.jackson.core.JsonProcessingException;
import com.fasterxml.jackson.databind.ObjectMapper;
import com.rabbitmq.client.AMQP;
import com.rabbitmq.client.Channel;
import com.rabbitmq.client.Connection;
import com.rabbitmq.client.ConnectionFactory;
import com.rabbitmq.client.DeliverCallback;
import com.rabbitmq.client.Recoverable;
import com.rabbitmq.client.RecoveryListener;
import com.rabbitmq.client.Return;
import org.slf4j.Logger;
import org.slf4j.LoggerFactory;
import java.io.IOException;
import java.util.LinkedHashMap;
import java.util.List;
import java.util.NavigableMap;
import java.util.concurrent.CompletableFuture;
import java.util.concurrent.ConcurrentHashMap;
import java.util.concurrent.ConcurrentSkipListMap;
import java.util.concurrent.ExecutionException;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.TimeoutException;
/**
* Durable, AMQP-backed mailbox for lead-to-lead messages across daemons — including daemons on
* different hosts, where a herdr pane injection (how {@code fleet_send} reaches a lead today)
* cannot reach at all. The broker is the only medium two independently-owned daemons share, which
* is exactly why {@link AmqpReplyInbox}'s javadoc already calls out "one gateway may publish to an
* agent owned by another gateway" (CB-308 federation) as the reason {@code publish} and
* {@code own}/consume are separate operations there — this class leans on the same split.
*
* <p><strong>Single-target, unlike {@link AmqpReplyInbox}.</strong> {@code AmqpReplyInbox}
* multiplexes many workers' reply queues under one gateway connection. A {@code LeadMailbox}
* instance is simpler: it owns exactly <em>one</em> queue — this daemon's own
* {@code lead.<selfCoordId>.inbox} — declared and consumed the moment it is constructed. There is
* no {@code own}/{@code release} pair to call separately; a daemon either runs a {@code LeadMailbox}
* for its own coord-id, or it does not run one at all.
*
* <p><strong>Consume-and-hold with deferred manual ack</strong> — same mapping as
* {@code AmqpReplyInbox}. The constructor declares the durable queue and starts a manual-ack
* consumer that pulls persistent messages into an in-memory {@code held} map (keyed by
* {@link LeadMessage#msgId()}) but does not ack them. {@link #peek} returns a non-destructive
* snapshot; {@link #ack} acks the broker delivery-tag and drops the entry. A message that is never
* acked (a crash, a bounce) survives — the broker redelivers it to the next connection that owns
* the queue.
*
* <p><strong>Publishing does not imply owning.</strong> {@link #publish} sends to
* {@code lead.<toCoordId>.inbox} over a dedicated confirm-mode channel; it never declares that
* queue as owned and never attaches a consumer to it. A sender that has never opened its own
* {@code LeadMailbox} for {@code toCoordId} can still publish to it, exactly as CB-308 federation
* requires. Publish blocks for the broker's publisher confirm (persistent delivery, {@code
* mandatory=true}) and throws {@link IllegalStateException} on an unroutable return, a nack, or a
* timeout — the caller must not report success for a black-holed message.
*
* <p><strong>Recovery.</strong> The connection is opened with automatic + topology recovery
* enabled, mirroring {@code AmqpReplyInbox}: on reconnect the broker hands out fresh delivery tags,
* so the held snapshot is cleared (dedup by {@code msgId} still prevents any double-queue on
* redelivery) and any publish still awaiting its confirm is failed rather than left to idle out
* the confirm timeout against a sequence number that means nothing on the new channel.
*/
public final class LeadMailbox implements AutoCloseable {
private static final Logger log = LoggerFactory.getLogger(LeadMailbox.class);
private static final String QUEUE_PREFIX = "lead.";
private static final String QUEUE_SUFFIX = ".inbox";
/** The prefetch used when a caller does not pass an explicit value to {@link #open(String, String, int)}. */
public static final int DEFAULT_PREFETCH = 32;
/** How long {@link #publish} waits for its publisher confirm before failing the call. */
private static final long CONFIRM_TIMEOUT_MS = 10_000L;
private static final ObjectMapper MAPPER = new ObjectMapper();
private final Connection connection;
private final String selfCoordId;
private final Channel channel;
/** All consume-channel operations (declare/consume/ack) serialize on this — a Channel is not thread-safe. */
private final Object channelLock = new Object();
/** msgId → held delivery, for this mailbox's own queue only (there is exactly one). */
private final LinkedHashMap<String, Held> held = new LinkedHashMap<>();
/**
* A dedicated channel for {@link #publish}, kept separate from {@link #channel} (consume + ack)
* so a publish confirm round trip never blocks under {@link #channelLock} and stalls an ack.
*/
private final Channel publishChannel;
private final Object publishChannelLock = new Object();
/** In-flight publishes awaiting their confirm, keyed by the publish channel's sequence number. */
private final ConcurrentSkipListMap<Long, Pending> pendingBySeq = new ConcurrentSkipListMap<>();
/** The same in-flight publishes, keyed by {@code msgId} — a broker {@code Return} carries no delivery tag. */
private final ConcurrentHashMap<String, Pending> pendingByMsgId = new ConcurrentHashMap<>();
/** A message pulled off the broker but not yet acked: its delivery-tag plus the deserialized envelope. */
private record Held(long deliveryTag, LeadMessage message) {}
/** A publish awaiting its confirm; {@link #returned} records whether the broker already returned it. */
private static final class Pending {
final String msgId;
final CompletableFuture<Void> confirmed = new CompletableFuture<>();
volatile boolean returned;
Pending(String msgId) {
this.msgId = msgId;
}
}
/**
* Connect to {@code uri} (the shared cross-host coordination vhost, e.g.
* {@code amqp://guest:guest@127.0.0.1:5672/coord}) and own {@code selfCoordId}'s mailbox, with
* {@link #DEFAULT_PREFETCH}.
*/
public static LeadMailbox open(String uri, String selfCoordId) {
return open(uri, selfCoordId, DEFAULT_PREFETCH);
}
/** As {@link #open(String, String)}, with an explicit consumer prefetch. */
public static LeadMailbox open(String uri, String selfCoordId, int prefetch) {
try {
ConnectionFactory factory = new ConnectionFactory();
factory.setUri(uri);
// Self-heal transient blips; topology recovery re-declares the queue and re-attaches the consumer.
factory.setAutomaticRecoveryEnabled(true);
factory.setTopologyRecoveryEnabled(true);
return new LeadMailbox(factory.newConnection("bridged-lead-mailbox"), selfCoordId, prefetch);
} catch (Exception e) {
throw new IllegalStateException("cannot connect to AMQP coordination broker at " + uri, e);
}
}
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for tests). */
LeadMailbox(Connection connection, String selfCoordId) {
this(connection, selfCoordId, DEFAULT_PREFETCH);
}
/** As above, with an explicit prefetch (injection seam for tests). */
LeadMailbox(Connection connection, String selfCoordId, int prefetch) {
this.connection = connection;
this.selfCoordId = selfCoordId;
try {
this.channel = connection.createChannel();
// Bound the held backlog — must be set before basicConsume.
this.channel.basicQos(prefetch);
this.publishChannel = connection.createChannel();
this.publishChannel.confirmSelect();
this.publishChannel.addReturnListener(this::onReturn);
this.publishChannel.addConfirmListener(this::onAck, this::onNack);
own();
} catch (IOException e) {
throw new IllegalStateException("cannot open AMQP channel", e);
}
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). Any publish
// confirm still in flight when the connection dropped is equally stale — fail it now rather
// than let it silently ride out CONFIRM_TIMEOUT_MS.
if (connection instanceof Recoverable recoverable) {
recoverable.addRecoveryListener(new RecoveryListener() {
@Override
public void handleRecovery(Recoverable recoverable) {
synchronized (held) {
held.clear();
}
failPendingPublishesOnRecovery();
log.info("AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery");
}
@Override
public void handleRecoveryStarted(Recoverable recoverable) {
// no-op: we act once recovery completes
}
});
}
}
/** Declare + consume this daemon's own {@code lead.<selfCoordId>.inbox}. Called once, at construction. */
private void own() throws IOException {
String queue = queueName(selfCoordId);
synchronized (channelLock) {
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
channel.basicConsume(queue, false, deliverCallback(), _ -> { }); // autoAck=false: manual ack
}
log.debug("lead mailbox owns queue {} for coord-id {}", queue, selfCoordId);
}
/**
* Publish {@code msg} to {@code toCoordId}'s mailbox and block until the broker's publisher
* confirm for it lands. Does <em>not</em> imply owning or consuming {@code toCoordId}'s queue.
* Throws {@link IllegalStateException} if the message is returned as unroutable, nacked, or not
* confirmed within {@link #CONFIRM_TIMEOUT_MS} — the caller must treat that as a failed publish,
* not a lost-and-forgotten one.
*/
public void publish(String toCoordId, LeadMessage msg) {
byte[] body;
try {
body = MAPPER.writeValueAsBytes(msg);
} catch (JsonProcessingException e) {
throw new IllegalStateException("cannot serialize lead message " + msg.msgId(), e);
}
AMQP.BasicProperties props = new AMQP.BasicProperties.Builder()
.messageId(msg.msgId())
.deliveryMode(2) // persistent — survives a broker restart
.contentType("application/json")
.build();
Pending pending = new Pending(msg.msgId());
long seq;
synchronized (publishChannelLock) {
seq = publishChannel.getNextPublishSeqNo();
pendingBySeq.put(seq, pending);
pendingByMsgId.put(msg.msgId(), pending);
try {
publishChannel.basicPublish("", queueName(toCoordId), true, props, body);
} catch (IOException e) {
pendingBySeq.remove(seq, pending);
pendingByMsgId.remove(msg.msgId(), pending);
throw new IllegalStateException("cannot publish lead message to " + queueName(toCoordId), e);
}
}
try {
pending.confirmed.get(CONFIRM_TIMEOUT_MS, TimeUnit.MILLISECONDS);
} catch (ExecutionException e) {
Throwable cause = e.getCause();
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
} catch (TimeoutException e) {
throw new IllegalStateException("publish confirm for lead message " + msg.msgId() + " to "
+ queueName(toCoordId) + " timed out after " + CONFIRM_TIMEOUT_MS
+ "ms — broker may be unreachable or overloaded", e);
} catch (InterruptedException e) {
Thread.currentThread().interrupt();
throw new IllegalStateException("interrupted awaiting publish confirm for " + msg.msgId(), e);
} finally {
pendingBySeq.remove(seq, pending);
pendingByMsgId.remove(msg.msgId(), pending);
}
}
/** Non-destructive FIFO snapshot of this mailbox's currently-held messages. */
public List<LeadMessage> peek() {
synchronized (held) {
return held.values().stream().map(Held::message).toList();
}
}
/** Convenience: {@link #peek} the current snapshot, then {@link #ack} every message in it. */
public List<LeadMessage> drain() {
List<LeadMessage> snapshot = peek();
snapshot.forEach(m -> ack(m.msgId()));
return snapshot;
}
/** Remove the held message {@code msgId} and ack it on the broker. No-op if not held. */
public void ack(String msgId) {
Held h;
synchronized (held) {
h = held.remove(msgId);
}
if (h == null) {
return; // never held (or already acked) — no-op
}
try {
synchronized (channelLock) {
channel.basicAck(h.deliveryTag(), false);
}
} catch (IOException e) {
// Ack didn't reach the broker: restore the entry so a later ack (or a redelivery after
// reconnect) can retry. Keeps the at-least-once contract — a message is never silently lost.
synchronized (held) {
held.putIfAbsent(msgId, h);
}
throw new IllegalStateException("cannot ack lead message " + msgId, e);
}
}
private DeliverCallback deliverCallback() {
return (_, delivery) -> {
long tag = delivery.getEnvelope().getDeliveryTag();
LeadMessage msg;
try {
msg = MAPPER.readValue(delivery.getBody(), LeadMessage.class);
} catch (IOException e) {
// A malformed body can never be dedup-keyed or handed to a caller; ack it so the
// broker does not redeliver it forever, and log loudly since this should never happen
// for a producer that only ever calls publish(String, LeadMessage).
log.warn("dropping malformed lead-mailbox delivery (tag {}): {}", tag, e.toString());
synchronized (channelLock) {
channel.basicAck(tag, false);
}
return;
}
boolean duplicate;
synchronized (held) {
if (held.containsKey(msg.msgId())) {
duplicate = true;
} else {
held.put(msg.msgId(), new Held(tag, msg));
duplicate = false;
}
}
if (duplicate) {
// Redelivered duplicate: ack the new tag and drop it so the broker stops resending.
synchronized (channelLock) {
channel.basicAck(tag, false);
}
}
};
}
/** Broker return for an unroutable {@code mandatory} publish — arrives BEFORE its confirm. */
private void onReturn(Return r) {
String msgId = r.getProperties() == null ? null : r.getProperties().getMessageId();
Pending pending = msgId == null ? null : pendingByMsgId.get(msgId);
if (pending != null) {
pending.returned = true;
} else {
log.warn("AMQP return for lead message {} (routingKey={}, {} {}) with no matching in-flight publish"
+ " — already resolved by a prior confirm", msgId, r.getRoutingKey(), r.getReplyCode(),
r.getReplyText());
}
}
private void onAck(long seq, boolean multiple) {
resolveConfirm(seq, multiple, true);
}
private void onNack(long seq, boolean multiple) {
resolveConfirm(seq, multiple, false);
}
/**
* Resolve every pending publish covered by this confirm (a single seq, or — {@code multiple} —
* every seq up to and including it). Checks {@link Pending#returned} at confirm time: since the
* broker's return for an unroutable message always precedes its confirm, an ack that arrives after
* a return means "confirmed but never routed", not "durably queued".
*/
private void resolveConfirm(long seq, boolean multiple, boolean ack) {
NavigableMap<Long, Pending> covered = multiple
? pendingBySeq.headMap(seq, true)
: pendingBySeq.subMap(seq, true, seq, true);
for (var it = covered.entrySet().iterator(); it.hasNext(); ) {
Pending pending = it.next().getValue();
it.remove();
pendingByMsgId.remove(pending.msgId, pending);
if (ack && !pending.returned) {
pending.confirmed.complete(null);
} else if (ack) {
pending.confirmed.completeExceptionally(new IllegalStateException(
"lead message " + pending.msgId + " was returned as unroutable (mailbox not owned)"));
} else {
pending.confirmed.completeExceptionally(new IllegalStateException(
"broker nacked publish of lead message " + pending.msgId));
}
}
}
/**
* Fail every publish still awaiting its confirm — their sequence numbers are stale after
* recovery. Guarded by {@link #publishChannelLock}, the same lock {@link #publish} holds while
* it takes its sequence number and registers its {@link Pending} — see
* {@code AmqpReplyInbox.failPendingPublishesOnRecovery}'s javadoc for the full race analysis this
* mirrors. Package-private only so a unit test can drive it directly without a live broker
* reconnect.
*/
void failPendingPublishesOnRecovery() {
synchronized (publishChannelLock) {
for (var it = pendingBySeq.entrySet().iterator(); it.hasNext(); ) {
Pending pending = it.next().getValue();
it.remove();
pendingByMsgId.remove(pending.msgId, pending);
pending.confirmed.completeExceptionally(new IllegalStateException(
"AMQP connection recovered mid-publish; confirm status of lead message "
+ pending.msgId + " is unknown"));
}
}
}
/**
* Fail every publish still awaiting its confirm with a clear, immediate error instead of leaving
* it to time out after {@link #CONFIRM_TIMEOUT_MS} once the channels are closed underneath it.
*/
private void failPendingPublishesOnClose() {
synchronized (publishChannelLock) {
for (var it = pendingBySeq.entrySet().iterator(); it.hasNext(); ) {
Pending pending = it.next().getValue();
it.remove();
pendingByMsgId.remove(pending.msgId, pending);
pending.confirmed.completeExceptionally(new IllegalStateException(
"lead mailbox closed while publish of lead message " + pending.msgId
+ " was still awaiting its confirm"));
}
}
}
/** The durable queue name a coord-id's mailbox lives on: {@code lead.<coordId>.inbox}. */
public static String queueName(String coordId) {
return QUEUE_PREFIX + coordId + QUEUE_SUFFIX;
}
@Override
public void close() {
failPendingPublishesOnClose();
try {
channel.close();
} catch (Exception e) {
log.debug("AMQP lead mailbox channel close: {}", e.toString());
}
try {
publishChannel.close();
} catch (Exception e) {
log.debug("AMQP lead mailbox publish channel close: {}", e.toString());
}
try {
connection.close();
} catch (Exception e) {
log.debug("AMQP lead mailbox connection close: {}", e.toString());
}
}
}
@@ -1,22 +0,0 @@
package dev.ltms.fleet.msg;
/**
* Wire envelope for a lead-to-lead message carried over {@link LeadMailbox}.
*
* <p>Unlike {@link ReplyInbox.InboxMessage} (a worker→primary reply, addressed only by the single
* gateway that owns the worker), a lead message crosses independently-owned daemons — possibly on
* different hosts — so it carries an explicit sender ({@code from}) as well as the recipient
* ({@code to}): the recipient needs the sender's coord-id to reply back.
*
* <p>{@code from} and {@code to} are globally-unique lead coordination ids (e.g. {@code "mac-opus"},
* {@code "fleet01-lead"}) — NOT herdr terminal ids. A herdr terminal id is meaningful only on the
* host that owns it, so it cannot address a lead running on another daemon; a coord-id is chosen
* by configuration ({@code coordinator.selfId}) precisely so it means the same thing everywhere.
*
* @param msgId idempotency id; a redelivered duplicate (at-least-once delivery) is deduped on this
* @param from the sending lead's coord-id
* @param to the receiving lead's coord-id — identifies the mailbox this message is held on
* @param content the message text
*/
public record LeadMessage(String msgId, String from, String to, String content) {
}
@@ -1,7 +1,6 @@
package dev.ltms.fleet.config;
import dev.ltms.fleet.auth.MemberRegistry;
import dev.ltms.fleet.msg.LeadMailbox;
import dev.ltms.fleet.peer.MemberRole;
import org.junit.jupiter.api.Test;
import org.junit.jupiter.api.io.TempDir;
@@ -43,6 +42,49 @@ class FleetConfigTest {
assertTrue(cfg.guard().hostSet().contains("ollama.ltms.dev"));
}
@Test
void aProfileWithAnAutoCompactWindowBelowTheAcceptedRangeIsRejectedAtLoad(@TempDir Path dir)
throws Exception {
Path f = dir.resolve("low-window.yaml");
Files.writeString(f, """
profiles:
ltms-local:
baseUrl: http://gx00.gw:8000
autoCompactWindow: 50000
""");
IllegalStateException e = assertThrows(IllegalStateException.class, () -> FleetConfig.load(f));
assertTrue(e.getMessage().contains("ltms-local"), "the offending profile is named");
assertTrue(e.getMessage().contains("autoCompactWindow"));
}
@Test
void aProfileWithAnAutoCompactWindowInRangeLoadsFine(@TempDir Path dir) throws Exception {
Path f = dir.resolve("in-range-window.yaml");
Files.writeString(f, """
profiles:
ltms-local:
baseUrl: http://gx00.gw:8000
autoCompactWindow: 250000
""");
FleetConfig cfg = FleetConfig.load(f);
assertEquals(250_000, cfg.profiles().get("ltms-local").autoCompactWindow());
}
@Test
void aProfileWithNoAutoCompactWindowLeavesItNull(@TempDir Path dir) throws Exception {
Path f = dir.resolve("no-window.yaml");
Files.writeString(f, """
profiles:
ltms-local:
baseUrl: http://gx00.gw:8000
""");
assertNull(FleetConfig.load(f).profiles().get("ltms-local").autoCompactWindow(),
"unset means off — today's behaviour, unchanged");
}
@Test
void appliesDefaultsForMissingSections(@TempDir Path dir) throws Exception {
Path f = dir.resolve("minimal.yaml");
@@ -933,66 +975,6 @@ class FleetConfigTest {
assertFalse(cfg.broker().isConfigured(), "an empty uri must not enable AMQP");
}
@Test
void absentCoordinatorBlockLeavesCoordinatorNull(@TempDir Path dir) throws Exception {
Path f = dir.resolve("no-coordinator.yaml");
Files.writeString(f, "bind:\n port: 8080\n");
FleetConfig cfg = FleetConfig.load(f);
assertNull(cfg.coordinator(), "no coordinator: block → null → no lead mailbox is opened");
}
@Test
void coordinatorBlockParses(@TempDir Path dir) throws Exception {
Path f = dir.resolve("coordinator.yaml");
Files.writeString(f, """
bind:
port: 8080
coordinator:
uri: amqp://guest:guest@127.0.0.1:5672/coord
selfId: mac-opus
prefetch: 16
""");
FleetConfig cfg = FleetConfig.load(f);
assertNotNull(cfg.coordinator());
assertTrue(cfg.coordinator().isConfigured(), "a non-blank uri enables the coordinator");
assertEquals("amqp://guest:guest@127.0.0.1:5672/coord", cfg.coordinator().uri());
assertEquals("mac-opus", cfg.coordinator().selfId());
assertEquals(16, cfg.coordinator().prefetchOrDefault());
}
@Test
void coordinatorBlockWithBlankUriStaysUnconfigured(@TempDir Path dir) throws Exception {
Path f = dir.resolve("coordinator-blank.yaml");
Files.writeString(f, "bind:\n port: 8080\ncoordinator:\n uri: \"\"\n");
FleetConfig cfg = FleetConfig.load(f);
assertNotNull(cfg.coordinator());
assertFalse(cfg.coordinator().isConfigured(), "an empty uri must not enable the coordinator");
assertNull(cfg.coordinator().selfId(), "a blank/absent selfId stays null, never coerced to empty");
}
@Test
void coordinatorEffectiveUriHonorsUriEnv() {
FleetConfig.Coordinator withEnv = new FleetConfig.Coordinator(
"amqp://stale-clear-text@127.0.0.1:5672/coord", "LEAD_COORD_URI", "fleet01-lead", null);
assertEquals("amqp://from-env@127.0.0.1:5672/coord",
withEnv.effectiveUri(Map.of("LEAD_COORD_URI", "amqp://from-env@127.0.0.1:5672/coord")),
"uriEnv wins over a literal uri when its variable resolves");
assertNull(withEnv.effectiveUri(Map.of()),
"an unset uriEnv variable must not fall back to the literal uri");
assertNull(withEnv.effectiveUri(Map.of("LEAD_COORD_URI", " ")),
"a blank uriEnv variable must not fall back to the literal uri");
FleetConfig.Coordinator noEnv = new FleetConfig.Coordinator(
"amqp://guest:guest@127.0.0.1:5672/coord", null, null, null);
assertEquals("amqp://guest:guest@127.0.0.1:5672/coord", noEnv.effectiveUri(Map.of()),
"the literal uri is used when no uriEnv is configured");
assertEquals(LeadMailbox.DEFAULT_PREFETCH, noEnv.prefetchOrDefault());
}
@Test
void absentPrimaryBlockLeavesPrimaryNull(@TempDir Path dir) throws Exception {
Path f = dir.resolve("no-primary.yaml");
@@ -1138,6 +1138,42 @@ class ClaudeCodeLauncherTest {
assertNull(startEnv(herdr).get("ANTHROPIC_MODEL"));
}
// ── autoCompactWindow: --autocompact is pinned on the command line, opt-in per profile ───────
/** A launcher for a profile identical but for its {@code autoCompactWindow:} — the only variable. */
private ClaudeCodeLauncher serviceWithAutoCompactWindow(FakeHerdr herdr, Integer window) {
FleetConfig.Profile cfg = profileWithAutoCompactWindow(window);
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
_ -> null);
}
private static FleetConfig.Profile profileWithAutoCompactWindow(Integer window) {
return new FleetConfig.Profile("sonnet", "http://gx00.gw:8000", "claude-sonnet-5", null,
"BRIDGED_WORKER_TOKEN", List.of("ccs", "sonnet"), "tab", "bridged-workers",
"w #{n}", "http://127.0.0.1:8765/mcp", null, null, null, null, null, Map.of(),
null, null, null, null, null, null, null, null, window);
}
@Test
void aConfiguredAutoCompactWindowIsPassedAsAnAutocompactFlag() {
FakeHerdr herdr = new FakeHerdr();
serviceWithAutoCompactWindow(herdr, 250_000).spawn("sonnet", null, null);
List<String> args = spawnedArgs(herdr);
int flag = args.indexOf("--autocompact");
assertTrue(flag >= 0, "the flag is what survives a wrapper argv like [ccs, sonnet]");
assertEquals("250000", args.get(flag + 1));
}
@Test
void aProfileWithNoAutoCompactWindowGetsNoAutocompactFlag() {
FakeHerdr herdr = new FakeHerdr();
serviceWithAutoCompactWindow(herdr, null).spawn("sonnet", null, null);
assertFalse(spawnedArgs(herdr).contains("--autocompact"));
}
// --- CB-539: subscription-profile opt-in ----------------------------------------------------
/** A claude-code profile on the subscription: no baseUrl (by design), no off-sub endpoint. */
@@ -495,6 +495,71 @@ class OpenCodeLauncherTest {
"without a baseUrl opencode resolves its own provider as before");
}
// --- autoCompactWindow: opencode has no absolute compact-at-N knob, so this is applied as the
// model's own limit.context, only when model: resolves to "provider/model" -----------------
private static FleetConfig.Profile opencodeCfgWithAutoCompactWindow(String model, String baseUrl,
Integer window) {
return new FleetConfig.Profile("gemini", baseUrl, model, null, "BRIDGED_WORKER_TOKEN",
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", null,
null, null, null, null, FleetConfig.Profile.KIND_OPENCODE, Map.of(), null, null,
null, null, null, null, null, null, window);
}
@Test
void autoCompactWindowIsAppliedAsThePerModelContextLimit(@TempDir Path root) throws Exception {
FakeHerdr herdr = new FakeHerdr();
service(herdr, root, opencodeCfgWithAutoCompactWindow("openai/gpt-5", null, 250_000)).spawn();
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
assertNotNull(cfgPath, "autoCompactWindow alone must trigger config generation, with no MCP"
+ " and no custom provider set");
JsonNode limit = new ObjectMapper().readTree(Path.of(cfgPath).toFile())
.path("provider").path("openai").path("models").path("gpt-5").path("limit");
assertEquals(250_000, limit.path("context").asInt());
assertEquals(16384, limit.path("output").asInt(),
"opencode's limit schema requires both keys; output gets a safe documented default");
}
@Test
void autoCompactWindowMergesIntoACustomProviderRatherThanOverwritingIt(@TempDir Path root)
throws Exception {
FakeHerdr herdr = new FakeHerdr();
service(herdr, root, opencodeCfgWithAutoCompactWindow(
"local-vllm/deepseek-v4-flash", "http://127.0.0.1:8000", 300_000)).spawn();
JsonNode provider = new ObjectMapper()
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile())
.path("provider").path("local-vllm");
assertEquals("@ai-sdk/openai-compatible", provider.path("npm").asText(),
"addCustomProvider's own fields must survive the later limit merge");
assertEquals(300_000, provider.path("models").path("deepseek-v4-flash")
.path("limit").path("context").asInt());
assertEquals("deepseek-v4-flash", provider.path("models").path("deepseek-v4-flash")
.path("name").asText(),
"the model's pre-existing 'name' field must survive the limit merge too");
}
@Test
void autoCompactWindowWithNoProviderSlashInModelGetsNoLimitAndAWarn(@TempDir Path root)
throws Exception {
FakeHerdr herdr = new FakeHerdr();
// mcpUrl set too, only so a config file gets written at all to inspect; a bare model name
// with no other config-triggering knob would leave OPENCODE_CONFIG unset entirely, which is
// also correct (nothing to write) but not what this test is asserting.
service(herdr, root, new FleetConfig.Profile("gemini", null, "some-free-model", null,
"BRIDGED_WORKER_TOKEN", List.of("opencode"), "tab", "bridged-workers",
"opencode: {model} #{n}", "http://127.0.0.1:8765/mcp", null, null, null, null,
FleetConfig.Profile.KIND_OPENCODE, Map.of(), null, null, null, null, null, null,
null, null, 250_000)).spawn();
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
assertTrue(json.path("provider").isMissingNode(),
"a bare model name cannot be targeted at a specific provider/model limit entry — "
+ "no silent no-op, but also no broken partial write");
}
// --- CB-634: IDE Index MCP + guidance overlay (opencode does not read CLAUDE.local.md) -------
private static FleetConfig.Profile opencodeIdeCfg(String mcpUrl, String ideUrl, String cwd) {
@@ -1,173 +0,0 @@
package dev.ltms.fleet.msg;
import org.junit.jupiter.api.BeforeAll;
import org.junit.jupiter.api.Tag;
import org.junit.jupiter.api.Test;
import org.testcontainers.containers.RabbitMQContainer;
import org.testcontainers.junit.jupiter.Testcontainers;
import org.testcontainers.utility.DockerImageName;
import java.util.List;
import java.util.concurrent.TimeUnit;
import java.util.concurrent.atomic.AtomicLong;
import static org.junit.jupiter.api.Assertions.assertEquals;
import static org.junit.jupiter.api.Assertions.assertThrows;
import static org.junit.jupiter.api.Assertions.assertTrue;
/**
* Contract test for {@link LeadMailbox} against a REAL broker — same approach as
* {@code AmqpReplyInboxContractTest}, which this mirrors: a Testcontainers RabbitMQ locally, or an
* externally-provisioned broker in CI via {@code AMQP_URI}. Tagged {@code contract} so it is
* excluded from {@code mvn test}/{@code mvn clean install} (which stay hermetic and need no
* Docker); run it with Docker present via {@code mvn test -Pcontract}.
*
* <p>Proves the mechanism this ticket adds: a {@link LeadMessage} published to
* {@code lead.<to>.inbox} is received with {@code from}/{@code to}/{@code content} intact, and
* {@link LeadMailbox#ack} removes it — the same publish→peek→ack roundtrip
* {@code AmqpReplyInboxContractTest} proves for {@link AmqpReplyInbox}, adapted to this class's
* single-owned-mailbox shape (no {@code own}/{@code release} — the mailbox for {@code selfCoordId}
* is owned the moment {@link LeadMailbox#open} returns).
*/
@Tag("contract")
// disabledWithoutDocker=false: on the CI path (AMQP_URI set) no container is started and the class
// must still run against the external broker even though the runner has no Docker.
@Testcontainers(disabledWithoutDocker = false)
class LeadMailboxTest {
private static final String EXTERNAL_URI = System.getenv("AMQP_URI");
private static final RabbitMQContainer BROKER =
new RabbitMQContainer(DockerImageName.parse("rabbitmq:3.13-management"));
private static final AtomicLong SEQ = new AtomicLong();
// No @Container: the JUnit 5 extension would force-start it even when AMQP_URI is set. Start it
// manually only on the local (no-external-broker) path; Ryuk reaps it on JVM exit.
@BeforeAll
static void startBrokerUnlessExternal() {
if (EXTERNAL_URI == null) {
BROKER.start();
}
}
private static String uri() {
if (EXTERNAL_URI != null) {
return EXTERNAL_URI;
}
// No trailing slash: an empty path is vhost "", which does not exist — omitting it selects
// the default vhost "/".
return "amqp://guest:guest@" + BROKER.getHost() + ":" + BROKER.getAmqpPort();
}
/** A fresh coord-id per test run so parallel/repeat runs never collide on the same queue. */
private static String coordId(String prefix) {
return prefix + "-" + System.nanoTime() + "-" + SEQ.incrementAndGet();
}
@Test
void publishThenPeekThenAckRoundTrip() throws Exception {
String to = coordId("lead-to");
String from = "lead-from";
try (LeadMailbox inbox = LeadMailbox.open(uri(), to)) {
LeadMessage sent = new LeadMessage("m1", from, to, "hello peer lead");
inbox.publish(to, sent);
List<LeadMessage> got = awaitPeek(inbox);
assertEquals(1, got.size(), "the published message should be held for drain");
assertEquals("m1", got.getFirst().msgId());
assertEquals(from, got.getFirst().from());
assertEquals(to, got.getFirst().to());
assertEquals("hello peer lead", got.getFirst().content());
inbox.ack("m1");
assertTrue(inbox.peek().isEmpty(), "an acked message is dropped");
}
}
@Test
void duplicateMsgIdIsNotDoubleQueued() throws Exception {
String to = coordId("lead-dedup");
try (LeadMailbox inbox = LeadMailbox.open(uri(), to)) {
inbox.publish(to, new LeadMessage("dup", "lead-from", to, "first"));
awaitPeek(inbox);
inbox.publish(to, new LeadMessage("dup", "lead-from", to, "second")); // same msgId — no-op
Thread.sleep(500); // give any erroneous second delivery time to land
List<LeadMessage> got = inbox.peek();
assertEquals(1, got.size(), "a repeated msgId must not double-queue");
assertEquals("first", got.getFirst().content(), "the first payload wins");
}
}
@Test
void unackedMessageSurvivesRestartAndIsRedelivered() throws Exception {
String to = coordId("lead-durable");
// First "process life": publish, see it held, but crash before acking.
try (LeadMailbox first = LeadMailbox.open(uri(), to)) {
first.publish(to, new LeadMessage("persist-1", "lead-from", to, "survive me"));
assertEquals(1, awaitPeek(first).size());
// no ack — simulate a java -jar bounce with the message still pending
}
// Second "process life": a fresh connection owning the same mailbox must be redelivered it.
try (LeadMailbox second = LeadMailbox.open(uri(), to)) {
List<LeadMessage> got = awaitPeek(second);
assertEquals(1, got.size(), "an unacked persistent message is redelivered after restart");
assertEquals("persist-1", got.getFirst().msgId());
assertEquals("survive me", got.getFirst().content());
second.ack("persist-1");
}
// Third life: once acked, it is gone for good.
try (LeadMailbox third = LeadMailbox.open(uri(), to)) {
Thread.sleep(500);
assertTrue(third.peek().isEmpty(), "an acked message does not come back on the next restart");
}
}
@Test
void publishDoesNotRequireTheSenderToOwnTheTargetMailbox() throws Exception {
// CB-308 federation: a sender that never opened its own LeadMailbox for `to` can still
// publish to it — publish must not imply ownership. Only the owner ever consumes here.
String to = coordId("lead-federated");
String senderId = coordId("lead-sender");
try (LeadMailbox owner = LeadMailbox.open(uri(), to);
LeadMailbox sender = LeadMailbox.open(uri(), senderId)) {
sender.publish(to, new LeadMessage("m1", senderId, to, "from a federated peer"));
List<LeadMessage> got = awaitPeek(owner);
assertEquals(1, got.size(), "only the owner's mailbox should receive the message");
assertEquals(senderId, got.getFirst().from());
assertTrue(sender.peek().isEmpty(), "the sender must not also hold a copy — it never owns `to`");
}
}
@Test
void unroutablePublishReportsFailureNotSilentSuccess() throws Exception {
// Publish to a coord-id whose mailbox was never opened by anyone: the queue is never
// declared, so the default-exchange route to lead.<to>.inbox does not exist and the broker
// must return the publish.
String to = coordId("lead-nobody-home");
try (LeadMailbox sender = LeadMailbox.open(uri(), coordId("lead-sender"))) {
IllegalStateException ex = assertThrows(IllegalStateException.class,
() -> sender.publish(to, new LeadMessage("m1", "lead-from", to, "nobody home")));
assertTrue(ex.getMessage() != null && ex.getMessage().toLowerCase().contains("unroutable"),
"expected an unroutable-publish failure, got: " + ex.getMessage());
}
}
/** Poll peek until at least one message is held, or ~10s elapse (broker delivery is async). */
@SuppressWarnings("BusyWait")
private static List<LeadMessage> awaitPeek(LeadMailbox inbox) throws InterruptedException {
long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(10);
List<LeadMessage> msgs = inbox.peek();
while (msgs.isEmpty() && System.nanoTime() < deadline) {
Thread.sleep(50);
msgs = inbox.peek();
}
return msgs;
}
}