Compare commits
37 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 5289eb509f | |||
| 703a05db41 | |||
| 82fae94c55 | |||
| b1f34c2e6b | |||
| 3f807d9f1b | |||
| e70263062c | |||
| 1a1e586b62 | |||
| c16d118f09 | |||
| 4b10d02207 | |||
| c5fbfdbf4a | |||
| 12cff28abb | |||
| 77a6a7142e | |||
| 9f3671b801 | |||
| 84034b34d1 | |||
| 1e9b2c9b7e | |||
| 5d422f85fa | |||
| ed2027b202 | |||
| 6b0a99b2b7 | |||
| 6b7caba248 | |||
| f8b0d42a5c | |||
| d1e7d71eee | |||
| 7fd914df1a | |||
| b066eb1903 | |||
| a196d34455 | |||
| 051d320ea0 | |||
| 766772763f | |||
| eab8185d7b | |||
| 7f672f0fb8 | |||
| e2801b9bbc | |||
| ea02c7b248 | |||
| ce74e164c6 | |||
| e60f892efd | |||
| ce05886831 | |||
| be123d0ac7 | |||
| 2d09c8b027 | |||
| ed54f0224e | |||
| 8d5bc3ee89 |
@@ -138,6 +138,7 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_send{coordId}` — or `{sessionId}` if they are on this host. **Not** `fleet_reply`: it has no peer route and the publish is refused |
|
||||
| Read your own held lead-to-lead mail (no ack) | `fleet_poll{coordId: <your own coord-id, from fleet_list's coordinator.selfId>}` — primary-only; never acks, so `fleet_list`'s `held[]` still shows it after. `fleet_list`'s `held[]` gives only a truncated preview — this is the only way to read the full body |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
|
||||
|
||||
@@ -69,6 +69,7 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
@@ -231,6 +232,13 @@ public final class Fleetd {
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine,
|
||||
outagePolicy);
|
||||
// fleetd #422 follow-up: say which of the three model-gate states the daemon booted into —
|
||||
// no models: block at all, a block armed with nothing off, or a block with N off — the same
|
||||
// way exhaustedPatternCoverageLine/errorPatternCoverageLine report CB-578 stage A/fleetd
|
||||
// #201 Unit 5 coverage just below. Read from workers.modelGateState() (never a separate
|
||||
// config.get().models() here) so this line and fleet_profiles' modelGateArmed can never
|
||||
// disagree about what CompositePeerLauncher's spawn gate actually enforces.
|
||||
log.info("model gate (fleetd #422): {}", modelGateCoverageLine(workers.modelGateState()));
|
||||
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
@@ -352,7 +360,11 @@ public final class Fleetd {
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
// fleetd #424: MemberRegistry.live re-reads fleet.architects through `config` on every
|
||||
// reserve/requireSlotFor call, so a reload that removes or adds an architect slot governs
|
||||
// the next spawn with no restart — the frozen `new MemberRegistry(cfg.fleet())` this used
|
||||
// to be let a "revoked" slot keep granting new architect spawns forever.
|
||||
MemberRegistry members = MemberRegistry.live(() -> config.get().fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
@@ -380,8 +392,7 @@ public final class Fleetd {
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage("exhaustedPattern", cfg.profiles().keySet(),
|
||||
exhaustedPatternsByProfile.keySet()));
|
||||
exhaustedPatternCoverageLine(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
// fleetd #201 Unit 5: classify a completion-fallback scrape that matches a profile's
|
||||
// configured backend-error refusal (a credential outage, a provider 5xx) as a backend error
|
||||
// rather than handing it back as a real answer. Compiled once at startup, keyed by profile
|
||||
@@ -401,8 +412,7 @@ public final class Fleetd {
|
||||
BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,
|
||||
errorPatternsByProfile);
|
||||
log.info("backend-error classification (fleetd #201 Unit 5): {}",
|
||||
CompletionResolver.coverage("errorPattern", cfg.profiles().keySet(),
|
||||
errorPatternsByProfile.keySet()));
|
||||
errorPatternCoverageLine(cfg.profiles().keySet(), errorPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
@@ -670,11 +680,8 @@ public final class Fleetd {
|
||||
}, outagePolicy);
|
||||
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new FleetMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
primaryRegistry, callers, metrics,
|
||||
capacitySource(config, cfg, profile -> liveCountRef.get().apply(profile)),
|
||||
new FleetMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
@@ -810,6 +817,93 @@ public final class Fleetd {
|
||||
}, quarantine, profile -> startupExhaustedPatterns.containsKey(profile));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): package-private factory for the CB-578 stage A {@code
|
||||
* exhaustedPattern} startup coverage line, paired explicitly with {@link
|
||||
* CompletionResolver.UnsetMeaning#OFF} — {@code exhaustedPattern} has no fallback, so a
|
||||
* profile with none configured really does have the classification off.
|
||||
*
|
||||
* <p>Extracted out of {@code main} for the same reason {@link #capacitySource} and {@link
|
||||
* #worktreeBranchLookup} were: {@code coverage()}'s own tests ({@code CompletionResolverTest})
|
||||
* prove it words {@code OFF} and {@link CompletionResolver.UnsetMeaning#BUILT_IN_DEFAULT}
|
||||
* correctly when a test supplies the meaning itself — they cannot prove {@code main} pairs the
|
||||
* right meaning with the right key, which is the actual fleetd #415 defect. <b>Measured:</b>
|
||||
* swapping the {@code UnsetMeaning} arguments between this method and {@link
|
||||
* #errorPatternCoverageLine} — recreating #415's defect with the two keys exchanged — compiled
|
||||
* with 0 errors and left all 1506 existing tests green before {@code
|
||||
* FleetdPatternCoverageLineTest} was added to catch exactly that swap.
|
||||
*/
|
||||
static String exhaustedPatternCoverageLine(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
return CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
allProfiles, configuredProfiles);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): the {@code errorPattern} counterpart of {@link
|
||||
* #exhaustedPatternCoverageLine}, paired explicitly with {@link
|
||||
* CompletionResolver.UnsetMeaning#BUILT_IN_DEFAULT} — an unset {@code errorPattern} still runs
|
||||
* backend-error classification against {@code CompletionResolver}'s built-in {@code
|
||||
* BACKEND_ERROR} pattern, so the empty case is not "off". See {@link
|
||||
* #exhaustedPatternCoverageLine}'s javadoc for the measured swap mutation this pairing guards
|
||||
* against.
|
||||
*/
|
||||
static String errorPatternCoverageLine(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
return CompletionResolver.coverage("errorPattern", CompletionResolver.UnsetMeaning.BUILT_IN_DEFAULT,
|
||||
allProfiles, configuredProfiles);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: package-private factory for the startup line reporting which of the
|
||||
* three central {@code models.allow:} gate states the daemon booted into. Extracted the same
|
||||
* way {@link #exhaustedPatternCoverageLine}/{@link #errorPatternCoverageLine} are, so a
|
||||
* dedicated test can call it directly rather than parsing log output, and so {@code main}'s
|
||||
* only source for this line is {@link PeerLauncher#modelGateState()} — never a second,
|
||||
* independently-derived read of {@code cfg.models()} that could disagree with what {@code
|
||||
* CompositePeerLauncher}'s spawn gate actually enforces (the fleetd #404 lesson).
|
||||
*
|
||||
* <p>Unlike the two pattern-key lines above, there is no {@code UnsetMeaning} choice to make
|
||||
* here: {@link PeerLauncher.ModelGateState#configured()} already states, unambiguously, whether
|
||||
* an empty {@link PeerLauncher.ModelGateState#off()} means "no {@code models:} block to gate
|
||||
* with" or "a block armed and currently reporting zero off" — the exact two states a bare
|
||||
* {@code disabledModels()} read could not tell apart before this ticket.
|
||||
*/
|
||||
static String modelGateCoverageLine(PeerLauncher.ModelGateState state) {
|
||||
if (!state.configured()) {
|
||||
return "not configured (no models: block — nothing is gated, and nothing can be)";
|
||||
}
|
||||
Set<String> off = state.off();
|
||||
return off.isEmpty()
|
||||
? "armed (models: block present; 0 models currently turned off)"
|
||||
: "armed (" + off.size() + " model(s) turned off: " + new TreeSet<>(off) + ")";
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #416: production source for {@code fleet_list}'s per-profile capacity facts.
|
||||
*
|
||||
* <p>The profile <em>set</em> ({@code configuredProfiles}) must come from {@code cfg} — the
|
||||
* startup snapshot — not the live {@code config.get()}. {@code profiles} as a whole is a
|
||||
* {@code DEFERRED} key ({@link ConfigRef#DEFERRED_KEYS}): {@code HerdrPeerLauncher} takes
|
||||
* {@code Map.copyOf(profiles)} once at construction and a profile only added to the
|
||||
* hot-reloaded map can never actually be spawned, so enumerating it live made {@code fleet_list}
|
||||
* report a profile as available when {@code fleet_spawn} on that same profile fails with
|
||||
* {@code unknown worker profile}. {@code fleet_list}'s own contract for {@code free} is "the
|
||||
* same check the spawn gate itself runs" — the set the spawn gate can see is the startup one,
|
||||
* so this must enumerate that one too, the same shape as {@code coordinator.peers} above.
|
||||
*
|
||||
* <p>{@code maxLoad} stays live on purpose: it is read off {@code config.get()} exactly like
|
||||
* {@code credentialId} ({@link ConfigRef} documents both as hot), so an existing profile's
|
||||
* {@code maxLoad} edit must still change what {@code fleet_list} reports without a restart.
|
||||
*/
|
||||
static FleetMcp.CapacitySource capacitySource(ConfigRef config, FleetConfig cfg,
|
||||
Function<String, Integer> liveCount) {
|
||||
return new FleetMcp.CapacitySource(liveCount,
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
},
|
||||
cfg.profiles()::keySet, System::nanoTime);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #248: package-private factory for the member worktree/branch lookup {@link
|
||||
* CompletionResolver} uses to name a fallback report's worktree and branch (fleetd#241).
|
||||
|
||||
@@ -30,6 +30,16 @@ public final class Authz {
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
READ,
|
||||
/**
|
||||
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
||||
*
|
||||
* <p>Deliberately <strong>not</strong> folded into {@link #READ}. {@code READ}'s grant
|
||||
* rests on "the roster carries no secrets" (see its case below) — a lead-to-lead body is
|
||||
* not the roster; it is where leads discuss host shapes, credentials and unmerged work.
|
||||
* Mapping this to {@code READ} would let any worker read every peer lead's mail in full
|
||||
* and would silently falsify that comment for every other {@code READ} caller.
|
||||
*/
|
||||
COORD_READ,
|
||||
/** Scrape the metrics endpoint. */
|
||||
METRICS
|
||||
}
|
||||
@@ -68,6 +78,11 @@ public final class Authz {
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
|
||||
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
||||
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
||||
// is coordination between leads, not observation of the roster.
|
||||
case COORD_READ -> caller.isPrimary();
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -11,6 +11,7 @@ import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
@@ -19,16 +20,43 @@ import java.util.Objects;
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/{@code reviewers}
|
||||
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
||||
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
||||
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
||||
* {@code CompositePeerLauncher}'s (see {@code ConfigRef}'s class doc) — so a config reload
|
||||
* that removes or adds an architect slot governs the <em>next</em> spawn with no restart.
|
||||
* Only {@link #MemberRegistry(FleetConfig.Fleet)} freezes the pool at construction, and that
|
||||
* constructor exists for tests and for the (rare) case of wiring a fixed, code-built config.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry, initially <em>empty</em>, and
|
||||
* <strong>never</strong> touched by a reload. Config declares no architect terminal, so at
|
||||
* startup every slot is idle and nothing resolves to an architect; a session only becomes one
|
||||
* when the spawn lifecycle {@linkplain #bind(String, String) binds} its terminal to a slot.
|
||||
* {@link CallerResolver} reads this through {@link #snapshot()} to turn a pane into an
|
||||
* {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The binding rule (fleetd #424): config governs what a bound slot still grants, as
|
||||
* well as what may be bound next.</strong> Removing a slot from config revokes it — that is the
|
||||
* ticket's entire point ("Revoking an architect slot does not revoke it"). Revoking it means an
|
||||
* architect already bound to that slot loses the ARCHITECT privilege on its very next request:
|
||||
* {@link #roleForSlot} and {@link #nameForSlot} read {@link #slots()} directly, with no cache, so
|
||||
* the moment a slot drops out of config, {@link CallerResolver#resolve} (which calls both on every
|
||||
* request from a bound pane, {@code CallerResolver.java:220}) can no longer confirm the pane's slot
|
||||
* is an architect slot, and the pane falls through to {@code Principal.worker(...)}. What does
|
||||
* <em>not</em> change is the {@code terminalToSlot} <em>occupancy</em> — the binding created by
|
||||
* {@link #bind} is untouched by a reload, on purpose: unbinding it here would double-book the slot
|
||||
* key (a second terminal could then bind to the "freed" key while the first is still the terminal
|
||||
* the operator actually meant to demote) and would silently break {@link #unbind}'s compare-safe
|
||||
* contract, which needs the original {@code terminal → slot} pair intact to remove it cleanly. So
|
||||
* the demoted session keeps occupying its slot — {@link #slotForTerminal} and {@link #snapshot()}
|
||||
* still name it — it just no longer resolves as an architect through that occupancy, and a fresh
|
||||
* spawn still cannot bind to the same key while it is occupied ({@link #reserve}/
|
||||
* {@link #requireSlotFor} refuse it anyway, since it is gone from {@link #slots()}). The demoted
|
||||
* session's own turn is unaffected: {@code fleet_reply}'s authorization
|
||||
* ({@code Authz.Action.REPLY}) is {@code caller.ownsSession(targetSession)} — identity by terminal,
|
||||
* not by role — so a demoted architect can still end its own turn normally.
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
@@ -55,14 +83,42 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
private final Supplier<FleetConfig.Fleet> fleet;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
/**
|
||||
* Freeze the pool at construction — for tests, and for the rare case of wiring a fixed,
|
||||
* code-built config. Production wiring should prefer {@link #live}, which re-reads
|
||||
* {@code fleet:} on every call.
|
||||
*/
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
this(() -> fleet);
|
||||
}
|
||||
|
||||
private MemberRegistry(Supplier<FleetConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
}
|
||||
|
||||
/**
|
||||
* Live variant (fleetd #424): {@code fleet} is read fresh on every {@link #slots()} call — pass
|
||||
* {@code () -> config.get().fleet()}, the same supplier shape {@code CompositePeerLauncher}
|
||||
* already uses for placement — so a reload that adds or removes an architect slot governs the
|
||||
* next spawn's {@link #reserve}/{@link #requireSlotFor} check with no restart. A separate,
|
||||
* private constructor rather than a same-arity public overload of
|
||||
* {@link #MemberRegistry(FleetConfig.Fleet)}: a {@code FleetConfig.Fleet} and a
|
||||
* {@code Supplier<FleetConfig.Fleet>} overload are ambiguous for a literal {@code null} — the
|
||||
* same reason {@code CallerResolver.withLeads} is a static factory rather than a fourth
|
||||
* constructor overload.
|
||||
*/
|
||||
public static MemberRegistry live(Supplier<FleetConfig.Fleet> fleet) {
|
||||
return new MemberRegistry(Objects.requireNonNull(fleet, "fleet"));
|
||||
}
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one map. Leaders are not members. */
|
||||
private static Map<String, Entry> flatten(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
@@ -74,18 +130,22 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
return Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
/**
|
||||
* The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot of
|
||||
* {@code fleet:} <em>as of this call</em> — see the class doc for which constructor makes that
|
||||
* live versus frozen.
|
||||
*/
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
return flatten(fleet.get());
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
/** The slots belonging to {@code role}, in definition order, as of this call. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
slots().forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
@@ -117,31 +177,49 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under — what the spawn lifecycle reads.
|
||||
* The strong-model profile a slot runs under, as of this call.
|
||||
*
|
||||
* <p>Nothing in {@code src/main} calls this (fleetd #431 — grepped both the {@code
|
||||
* .profileForSlot(} and the {@code ::profileForSlot} form). This javadoc used to say "what the
|
||||
* spawn lifecycle reads", and that seam does not exist: the spawn lifecycle takes its profile
|
||||
* from the {@link MemberLifecycle.SlotReservation} that {@code reserve} returns, never from
|
||||
* here. Kept and pinned rather than deleted because it is the natural accessor for that seam
|
||||
* if one is added; live for the same reason as {@link #roleForSlot}, so a reload cannot leave
|
||||
* it answering for the old config.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
/**
|
||||
* The role a qualified slot key belongs to, or {@code null} when the key is not currently
|
||||
* configured. Deliberately live, with no cache (fleetd #424, see the class doc's binding rule):
|
||||
* removing a slot from config must make {@link CallerResolver#resolve} stop granting the
|
||||
* ARCHITECT role for it on the very next request from a terminal that was bound to it, which is
|
||||
* the ticket's whole point — revoking a slot must actually revoke it, not just refuse the next
|
||||
* spawn.
|
||||
*/
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
/**
|
||||
* The unqualified configured name for a slot, or {@code null} if it is not currently configured.
|
||||
* Live for the same reason as {@link #roleForSlot} — see the class doc's binding rule.
|
||||
*/
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
return slots().containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -32,9 +32,26 @@ import java.util.function.Supplier;
|
||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||
* ({@code () -> config.get().fleet()}). <strong>But {@code fleet:} as a whole is NOT in this
|
||||
* class</strong>: {@code fleet.leaders} inside the same key is frozen, which is exactly what
|
||||
* makes {@code fleet:} split rather than hot — see below.</li>
|
||||
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
||||
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
||||
* reads it live for placement (which profile an unqualified architect spawn may land on), and
|
||||
* {@code MemberRegistry} separately reads it live, through its own instance of the same
|
||||
* supplier shape, for identity — both which slot a spawn may bind to <em>and</em> what a slot
|
||||
* already bound still grants. Removing an architect slot from config therefore revokes the
|
||||
* {@link dev.ltms.fleet.auth.Role#ARCHITECT} role on the bound pane's very next request; only
|
||||
* the slot <em>occupancy</em> survives, so the demoted session still holds its slot key until
|
||||
* it unbinds. See {@code MemberRegistry}'s class doc for that binding rule.
|
||||
* <strong>But {@code fleet:} as a whole is NOT in this class</strong>: {@code fleet.leaders}
|
||||
* inside the same key is frozen, which is exactly what makes {@code fleet:} split rather than
|
||||
* hot — see below. {@code models:} (fleetd #422) joined this class whole: {@link
|
||||
* FleetConfig#validateModels()} re-runs fully against the fresh config on every {@link
|
||||
* #reload()} (via {@link FleetConfig#validateAll()}), refusing a bad edit outright rather than
|
||||
* caching a stale copy anywhere, and the on/off half added by fleetd #422 is read live both by
|
||||
* {@code CompositePeerLauncher}'s spawn gate ({@code enforceModelEnabled} and its candidate
|
||||
* filter) and by {@code fleet_profiles}/{@code GET /profiles} (via
|
||||
* {@code PeerLauncher.disabledModels()}). Nothing about {@code models:} is baked into an
|
||||
* object built at startup, so — unlike the deferred keys below — there is no frozen half left
|
||||
* to report; it moved here from deferred rather than joining split.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code idleSleepGuard:} ({@code Fleetd.java} reads it once, at startup, to decide whether
|
||||
@@ -43,12 +60,6 @@ import java.util.function.Supplier;
|
||||
* running daemon keeps whatever this was at startup regardless of a later edit),
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code models:} (fleetd ticket "central allow-list of usable models" — {@link
|
||||
* FleetConfig#validateModels()} re-runs against the fresh config in {@link #reload()}
|
||||
* (via {@link FleetConfig#validateAll()}), so a
|
||||
* models.allow: edit that would refuse to boot still refuses the reload; a change that
|
||||
* passes has nothing built at startup to rebuild, so it is reported deferred rather than
|
||||
* silently accepted with no report at all),
|
||||
* {@code guard:}, {@code worktreeRoot:}, {@code worktreeGroup:} and {@code memberSkills:}
|
||||
* (all three of the latter baked once into the {@code GitWorktrees} built at
|
||||
* {@code Fleetd.java:251} and never rebuilt — fleetd #323 instance 2 found
|
||||
@@ -141,16 +152,18 @@ import java.util.function.Supplier;
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The denominator, measured on 2026-09-04 (fleetd #330; recounted for fleetd #333);
|
||||
* recounted again for fleetd #362, again after {@code idleSleepGuard:} was added, and again after
|
||||
* {@code models:} was added.</strong>
|
||||
* {@code FleetConfig} has 25 top-level record components: 5 cold, 14 deferred, 3 split, 3
|
||||
* hot-excluded. Three of them are named nowhere in this file, and the reason is the same for all
|
||||
* three: {@code placement}, {@code memberCredentials} and {@code memberLoginShell} are
|
||||
* <strong>hot</strong> and correctly absent — all three are read live off {@code config.get()}
|
||||
* recounted again for fleetd #362, again after {@code idleSleepGuard:} was added, again after
|
||||
* {@code models:} was added as deferred, and again for fleetd #422, which moved {@code models:}
|
||||
* from deferred to hot-excluded once its on/off half was read live everywhere.</strong>
|
||||
* {@code FleetConfig} has 25 top-level record components: 5 cold, 13 deferred, 3 split, 4
|
||||
* hot-excluded. Four of them are named nowhere in this file, and the reason is the same for all
|
||||
* four: {@code placement}, {@code memberCredentials}, {@code memberLoginShell} and {@code models}
|
||||
* are <strong>hot</strong> and correctly absent — all four are read live off {@code config.get()}
|
||||
* (placement through the {@code CompositePeerLauncher} supplier the Hot bullet names;
|
||||
* {@code memberCredentials}/{@code memberLoginShell} at spawn time, {@code Fleetd.java:198, 205, 729}
|
||||
* and {@code HerdrPeerLauncher#configuredMemberLoginShell}), so a reload takes effect on the next
|
||||
* spawn with no entry needed here.
|
||||
* and {@code HerdrPeerLauncher#configuredMemberLoginShell}; {@code models} the same way, through the
|
||||
* Hot bullet's {@code models:} paragraph), so a reload takes effect on the next spawn (or, for
|
||||
* {@code models}, the next reported status) with no entry needed here.
|
||||
* {@code health} and {@code coordinator} used to be a third kind — <strong>undecided</strong>, not
|
||||
* hot — until fleetd #330 added the <strong>split</strong> class above and gave them a home. A
|
||||
* reload touching either used to report a bare "config reloaded", which under-claimed; now it names
|
||||
@@ -226,7 +239,7 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
static final Set<String> DEFERRED_KEYS = Set.of(
|
||||
"guard", "worktreeRoot", "worktreeGroup", "memberSkills", "primary", "configReload",
|
||||
"leadHeartbeat", "lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs",
|
||||
"quarantineCooldownSeconds", "profiles", "idleSleepGuard", "models");
|
||||
"quarantineCooldownSeconds", "profiles", "idleSleepGuard");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
@@ -446,15 +459,6 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
if (!Objects.equals(old.idleSleepGuard(), fresh.idleSleepGuard())) {
|
||||
changed.add("idleSleepGuard");
|
||||
}
|
||||
// fleetd ticket "central allow-list of usable models": validateModels() runs again in
|
||||
// reload() above (via validateAll()), so a bad edit is already refused as cold-adjacent
|
||||
// (the whole reload is refused via the catch block, never partially applied). A GOOD edit
|
||||
// to the allow-list
|
||||
// itself has nothing built at startup to rebuild — it only ever mattered to the validation
|
||||
// call that already ran — so report it deferred rather than silently swallowing the change.
|
||||
if (!Objects.equals(old.models(), fresh.models())) {
|
||||
changed.add("models");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
@@ -529,26 +533,35 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||
+ "member's environment is read live on every spawn and already applied");
|
||||
}
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (architects, developers,
|
||||
// reviewers, charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, reviewers,
|
||||
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||
// with no restart note. Only fleet.leaders is frozen (Fleetd.java:281 reads
|
||||
// cfg.fleet().leaders() off the startup snapshot to build both the LeadTabScanner's
|
||||
// tab-label-to-name map, wired into CallerResolver.withLeadsAndMembers at Fleetd.java:620/624,
|
||||
// and — when herdr answered — LeadLauncher(...).ensureLeads() at Fleetd.java:315, which
|
||||
// auto-launches each lead up to its `instances` count; neither is rebuilt on reload). So this
|
||||
// compares fleet.leaders alone, not the whole Fleet record: comparing the whole record would
|
||||
// report "split" for a tabLabel-only or charters-only change that is actually fully hot,
|
||||
// which is the over-claim mirror of the under-claim bug this class exists to prevent.
|
||||
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
||||
// its consumers, not just the one this comment used to name: CompositePeerLauncher reads it
|
||||
// live for PLACEMENT through the () -> config.get().fleet() supplier named in the class doc's
|
||||
// Hot bullet, and MemberRegistry separately reads it live for IDENTITY (which slot a spawn
|
||||
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
||||
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
||||
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds. Only
|
||||
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
||||
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
||||
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
||||
// LeadLauncher(...).ensureLeads() at Fleetd.java:315, which auto-launches each lead up to its
|
||||
// `instances` count; neither is rebuilt on reload). So this compares fleet.leaders alone, not
|
||||
// the whole Fleet record: comparing the whole record would report "split" for a tabLabel-only
|
||||
// or architects-only change that is actually fully hot, which is the over-claim mirror of the
|
||||
// under-claim bug this class exists to prevent.
|
||||
if (!Objects.equals(leadersOf(old), leadersOf(fresh))) {
|
||||
changed.add("fleet: fleet.leaders (each lead's tab, workspace, cwd, profile and "
|
||||
+ "instances count) is read once at startup to build the LeadTabScanner's "
|
||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||
+ "lead; the rest of fleet: (architects, developers, reviewers, charters, "
|
||||
+ "tabLabel) is read live through the supplier on CompositePeerLauncher and "
|
||||
+ "already applied");
|
||||
+ "lead; the rest of fleet: (developers, reviewers, charters, tabLabel) is read "
|
||||
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
||||
+ "live through that same supplier for placement AND through a separate supplier "
|
||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
|
||||
@@ -133,8 +133,10 @@ import java.util.regex.PatternSyntaxException;
|
||||
* nothing and every existing config keeps working exactly as it does today.
|
||||
* When non-empty, a profile whose {@code model:} is not one of {@link
|
||||
* Models#ids()} fails config load, naming both the model and the profile —
|
||||
* see {@link #validateModels()}. This block only decides what may be
|
||||
* CONFIGURED; nothing here enforces it at spawn time. See {@link Models}.
|
||||
* see {@link #validateModels()}. This block decides what may be CONFIGURED;
|
||||
* fleetd #422 added the separate on/off question — whether a configured model
|
||||
* may be spawned onto RIGHT NOW ({@link Models.ModelEntry#enabled}) — enforced
|
||||
* live at spawn by {@code CompositePeerLauncher}, not here. See {@link Models}.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record FleetConfig(
|
||||
@@ -1362,9 +1364,14 @@ public record FleetConfig(
|
||||
* the set of permitted models — only editing {@code models.allow:} itself can. This is the
|
||||
* invariant the ticket asked for: the two blocks are validated in one direction only.
|
||||
*
|
||||
* <p><b>Out of scope here, deliberately:</b> nothing in this block is read at spawn time —
|
||||
* enforcing it against a live spawn, an on/off runtime switch, and any interaction with {@code
|
||||
* BackendQuarantine} are separate units. This block is config-load validation only.
|
||||
* <p><b>Spawn-time enforcement (fleetd #422, units 2+3) lives outside this record</b> —
|
||||
* {@code CompositePeerLauncher.enforceModelEnabled} and its candidate-set filter read this
|
||||
* block LIVE (through the same kind of supplier {@code weight}/{@code maxLoad} already use),
|
||||
* so the on/off state below is hot: no restart needed. This block itself still only decides
|
||||
* what may be CONFIGURED (membership in {@link #allow}); {@link ModelEntry#enabled} decides
|
||||
* whether a member of that list is currently spawnable. The two questions are deliberately
|
||||
* separate — see {@link ModelEntry}'s javadoc for why turning a model off must never mean
|
||||
* removing it from {@link #allow}.
|
||||
*
|
||||
* @param allow the permitted models, each its own {@link ModelEntry} rather than a bare
|
||||
* string — see that record's javadoc for why. {@code null}/empty ⇒ the block is
|
||||
@@ -1378,24 +1385,56 @@ public record FleetConfig(
|
||||
}
|
||||
|
||||
/**
|
||||
* One permitted model, named as a record rather than a bare string on purpose: a later unit
|
||||
* needs to hang an on/off state and a load-limit state off each entry, and a bare {@code
|
||||
* List<String>} cannot grow those fields without changing the YAML shape underneath every
|
||||
* operator who already wrote one. {@link #model()} is intentionally a single flat,
|
||||
* opaque-string namespace — a bare Claude id ({@code claude-sonnet-5}) and an opencode
|
||||
* provider-prefixed id ({@code openai/gpt-5.6-terra}) both fit it unchanged, because
|
||||
* {@link FleetConfig#validateModels()} only ever compares a profile's {@code model:} value
|
||||
* against this string for exact equality; it never parses a provider prefix or branches on
|
||||
* a profile's {@code kind:}.
|
||||
* One permitted model, named as a record rather than a bare string on purpose: this ticket
|
||||
* (fleetd #422) is the "later unit" the original comment here predicted — it hangs an on/off
|
||||
* state ({@link #enabled}) off each entry, and a bare {@code List<String>} could not have
|
||||
* grown that field without changing the YAML shape underneath every operator who already
|
||||
* wrote one. {@link #model()} is intentionally a single flat, opaque-string namespace — a
|
||||
* bare Claude id ({@code claude-sonnet-5}) and an opencode provider-prefixed id
|
||||
* ({@code openai/gpt-5.6-terra}) both fit it unchanged, because {@link
|
||||
* FleetConfig#validateModels()} only ever compares a profile's {@code model:} value against
|
||||
* this string for exact equality; it never parses a provider prefix or branches on a
|
||||
* profile's {@code kind:}.
|
||||
*
|
||||
* @param model the model id exactly as a {@code profiles:} entry's {@code model:} field
|
||||
* would name it
|
||||
* @param model the model id exactly as a {@code profiles:} entry's {@code model:} field
|
||||
* would name it
|
||||
* @param enabled {@code false} turns spawning onto this model off; {@code null} (the field
|
||||
* omitted — every config written before fleetd #422 is this shape) or
|
||||
* {@code true} leaves it on. Turning a model off must NEVER remove it from
|
||||
* {@link Models#allow} — {@link FleetConfig#validateModels()} checks
|
||||
* <em>membership</em> only, never the on/off state, so an off entry stays a
|
||||
* valid thing for a {@code profiles:} entry to name; only
|
||||
* {@code CompositePeerLauncher}'s spawn-time gate reads {@link #enabled}.
|
||||
* Collapsing the two — turning a model off by deleting its {@code allow:}
|
||||
* entry — would make {@link FleetConfig#validateModels()} refuse the whole
|
||||
* config reload the moment a still-configured profile names it, which is
|
||||
* exactly the restart-to-flip-a-switch problem this field exists to avoid.
|
||||
* <p>A model id can be named by more than one {@code profiles:} entry (e.g.
|
||||
* {@code deepseek-v4-flash} backs both {@code local} and {@code
|
||||
* local-direct} in the live config) — turning it off disables every profile
|
||||
* that names it, on purpose: the model is what a subscription's rate limit
|
||||
* actually constrains, not any one profile alias for it.
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record ModelEntry(String model) {
|
||||
public record ModelEntry(String model, Boolean enabled) {
|
||||
public ModelEntry {
|
||||
model = (model == null || model.isBlank()) ? null : model.trim();
|
||||
}
|
||||
|
||||
/**
|
||||
* Back-compat form before {@link #enabled} was added (fleetd #422) — the model is
|
||||
* unconditionally on, exactly as every {@code ModelEntry} behaved before this field
|
||||
* existed. Keeps pre-#422 call sites (and any YAML that omits {@code enabled:})
|
||||
* compiling and behaving identically.
|
||||
*/
|
||||
public ModelEntry(String model) {
|
||||
this(model, null);
|
||||
}
|
||||
|
||||
/** {@code true} unless {@link #enabled} is explicitly {@code false} — absent means on. */
|
||||
public boolean isEnabled() {
|
||||
return !Boolean.FALSE.equals(enabled);
|
||||
}
|
||||
}
|
||||
|
||||
/** {@link #allow}'s model ids, as a set for membership checks. Blank/null entries are dropped. */
|
||||
@@ -1408,6 +1447,23 @@ public record FleetConfig(
|
||||
}
|
||||
return Collections.unmodifiableSet(ids);
|
||||
}
|
||||
|
||||
/**
|
||||
* Model ids currently turned off (fleetd #422: {@link ModelEntry#isEnabled()} {@code
|
||||
* false}). Read live by {@code CompositePeerLauncher}'s spawn gate and by {@code
|
||||
* fleet_profiles}/{@code GET /profiles} — both must read this same accessor off the same
|
||||
* live config so the two surfaces cannot disagree about which model is off (the fleetd
|
||||
* #404 lesson: a status field must read the source the behaviour reads).
|
||||
*/
|
||||
public Set<String> offIds() {
|
||||
Set<String> off = new java.util.LinkedHashSet<>();
|
||||
for (ModelEntry e : allow) {
|
||||
if (e != null && e.model() != null && !e.isEnabled()) {
|
||||
off.add(e.model());
|
||||
}
|
||||
}
|
||||
return Collections.unmodifiableSet(off);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -698,17 +698,57 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* What an unset pattern key means for the classification it configures (fleetd#415).
|
||||
* {@code coverage()} cannot infer this from the key's name — the two keys it currently
|
||||
* describes disagree on it, and a string comparison on the name would just move the same bug
|
||||
* to a new spot — so every caller must state it explicitly.
|
||||
*
|
||||
* <p><strong>This alone does not prove a caller passes the right one for its key.</strong> A
|
||||
* test that calls {@code coverage()} directly and supplies the meaning itself only proves this
|
||||
* enum is worded correctly, never that {@code Fleetd}'s two call sites pair each key with its
|
||||
* true meaning — that pairing is #415's actual defect. Measured on review: swapping the two
|
||||
* {@code UnsetMeaning} arguments at those call sites (giving {@code exhaustedPattern} the
|
||||
* built-in-default wording and {@code errorPattern} the off wording — #415's exact defect with
|
||||
* the keys exchanged) compiled with 0 errors and left all 1506 existing tests green. See
|
||||
* {@code dev.ltms.fleet.Fleetd#exhaustedPatternCoverageLine}/{@code #errorPatternCoverageLine}
|
||||
* and {@code FleetdPatternCoverageLineTest}, which exists specifically to catch that swap.
|
||||
*/
|
||||
public enum UnsetMeaning {
|
||||
/** No fallback exists: a profile with no configured pattern truly has this classification off. */
|
||||
OFF,
|
||||
/** A built-in pattern applies when unset: the classification still runs for that profile. */
|
||||
BUILT_IN_DEFAULT
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for a fleetd#201/CB-578-style pattern-key classification, logged at startup
|
||||
* the way {@link dev.ltms.fleet.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* <p>fleetd#415: this method measures <em>pattern coverage</em> — how many profiles set the
|
||||
* key — which is not the same thing as <em>feature state</em> for a key with a fallback. For
|
||||
* {@code errorPattern}, an empty {@code configuredProfiles} still runs the classification
|
||||
* against {@code CompletionResolver}'s built-in compatibility pattern ({@link #BACKEND_ERROR}
|
||||
* at line ~84); for {@code exhaustedPattern} there is no fallback, so empty really does mean
|
||||
* off. {@code unsetMeaning} is the single, required source of that fact — see
|
||||
* {@link dev.ltms.fleet.config.FleetConfig#rejectMalformedProfilePatterns} lines ~2029-2032 for
|
||||
* where it is documented for config authors. It is a required parameter, not a defaulted
|
||||
* overload: a third pattern key added later must supply one to compile at all, rather than
|
||||
* silently inheriting whichever wording this method happened to default to.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry the pattern
|
||||
*/
|
||||
public static String coverage(String patternKey, Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
public static String coverage(String patternKey, UnsetMeaning unsetMeaning, Set<String> allProfiles,
|
||||
Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an " + patternKey + " configured; profiles: " + sorted(allProfiles) + ")";
|
||||
return switch (unsetMeaning) {
|
||||
case OFF -> "off (no profile has an " + patternKey + " configured; profiles: "
|
||||
+ sorted(allProfiles) + ")";
|
||||
case BUILT_IN_DEFAULT -> "built-in default for all profiles (no profile customises "
|
||||
+ patternKey + "; profiles: " + sorted(allProfiles) + ")";
|
||||
};
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
|
||||
@@ -37,6 +37,7 @@ import io.modelcontextprotocol.spec.McpSchema;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import jakarta.servlet.http.HttpServlet;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -385,10 +386,11 @@ public final class FleetMcp {
|
||||
(exchange, req) -> {
|
||||
Map<String, Object> a = req.arguments();
|
||||
String target = str(a, "target");
|
||||
String coordId = str(a, "coordId");
|
||||
// The action depends on the ARGUMENTS, not on the tool name -- see pollAction.
|
||||
McpSchema.CallToolResult denied = deny(exchange, toolAction("fleet_poll", a), target);
|
||||
if (denied != null) return denied;
|
||||
return poll(messages, str(a, "ticket"), target);
|
||||
return poll(messages, leadChannel, str(a, "ticket"), target, coordId);
|
||||
};
|
||||
// CB-307 Increment 3: per-msgId ack (not needed in v1 but supported by the inbox).
|
||||
// Acking removes a reply from the inbox, so it is a drain, not a read.
|
||||
@@ -810,29 +812,42 @@ public final class FleetMcp {
|
||||
|
||||
/**
|
||||
* Which authorization action a {@code fleet_poll} call needs, decided by its arguments
|
||||
* (fleetd #272).
|
||||
* (fleetd #272, widened by fleetd #421).
|
||||
*
|
||||
* <p>{@code fleet_poll} is <strong>two operations behind one tool name</strong>. With {@code
|
||||
* ticket} it observes an async delegation and changes nothing, which is a {@link
|
||||
* <p>{@code fleet_poll} is now <strong>three operations behind one tool name</strong>. With
|
||||
* {@code ticket} it observes an async delegation and changes nothing, which is a {@link
|
||||
* Authz.Action#READ}. With {@code target} it calls {@link MessageService#drainReplies} on that
|
||||
* session -- the replies are removed from the inbox and a second call returns nothing -- so it
|
||||
* is a {@link Authz.Action#DRAIN}, the same gate {@code fleet_ack} already uses for removing a
|
||||
* single message, and the same one the REST path uses at {@code FleetApp.drainReplies}.
|
||||
* single message, and the same one the REST path uses at {@code FleetApp.drainReplies}. With
|
||||
* {@code coordId} it reads (never acks) this daemon's own held lead-to-lead mail, which is a
|
||||
* {@link Authz.Action#COORD_READ} -- <strong>not</strong> {@code READ}, even though nothing is
|
||||
* consumed: {@code READ}'s grant is open to every authenticated role on the premise that the
|
||||
* roster carries no secrets, and a lead-to-lead body is not the roster. Mapping a non-destructive
|
||||
* peer-mail read to {@code READ} would let any worker read every peer lead's mail in full.
|
||||
*
|
||||
* <p>Until this method existed the handler passed a constant {@code READ} for both branches.
|
||||
* {@code READ} is open to every authenticated role, so any worker could read a peer's id out of
|
||||
* {@code fleet_list} and destroy the replies that peer had queued for the primary. The gate
|
||||
* failed open, and it did so because the required action is a function of the arguments while
|
||||
* the handler chose it before looking at them.
|
||||
* <p>Before this method existed (fleetd #272) the handler passed a constant {@code READ} for
|
||||
* both of the original branches. {@code READ} is open to every authenticated role, so any
|
||||
* worker could read a peer's id out of {@code fleet_list} and destroy the replies that peer had
|
||||
* queued for the primary. The gate failed open, and it did so because the required action is a
|
||||
* function of the arguments while the handler chose it before looking at them.
|
||||
*
|
||||
* <p>The choice lives in this method, and not inline in the handler, so that a test can assert
|
||||
* the mapping the handler actually uses. {@code FleetMcpAuthzTest} already checked every
|
||||
* {@link Authz.Action} against every {@link Role} and passed throughout -- it tested the policy
|
||||
* table, which was correct, while the defect was in which action the caller handed it.
|
||||
*
|
||||
* @param target the {@code target} argument of the call, or {@code null}/blank when absent
|
||||
* <p>Checked first, and exclusively of {@code target}: a call naming {@code coordId} is reading
|
||||
* a different inbox entirely (this daemon's own lead channel, never a worker's), so it takes
|
||||
* priority over whatever {@code target} might also say.
|
||||
*
|
||||
* @param target the {@code target} argument of the call, or {@code null}/blank when absent
|
||||
* @param coordId the {@code coordId} argument of the call, or {@code null}/blank when absent
|
||||
*/
|
||||
static Authz.Action pollAction(String target) {
|
||||
static Authz.Action pollAction(String target, String coordId) {
|
||||
if (!isBlank(coordId)) {
|
||||
return Authz.Action.COORD_READ;
|
||||
}
|
||||
return isBlank(target) ? Authz.Action.READ : Authz.Action.DRAIN;
|
||||
}
|
||||
|
||||
@@ -847,7 +862,7 @@ public final class FleetMcp {
|
||||
case "fleet_reply" -> Authz.Action.REPLY;
|
||||
case "fleet_ask" -> Authz.Action.ASK;
|
||||
case "fleet_status", "fleet_list", "fleet_profiles", "fleet_whoami" -> Authz.Action.READ;
|
||||
case "fleet_poll" -> pollAction(str(arguments, "target"));
|
||||
case "fleet_poll" -> pollAction(str(arguments, "target"), str(arguments, "coordId"));
|
||||
case "fleet_ack" -> Authz.Action.DRAIN;
|
||||
case "fleet_spawn" -> Authz.Action.SPAWN;
|
||||
case "fleet_stop" -> Authz.Action.STOP;
|
||||
@@ -857,6 +872,21 @@ public final class FleetMcp {
|
||||
|
||||
/** {@code fleet_poll}: check an async delegation by ticket, or drain a worker's inbox by target. */
|
||||
static McpSchema.CallToolResult poll(MessageService messages, String ticket, String target) {
|
||||
return poll(messages, null, ticket, target, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus (fleetd #421) a held-peer-mail read when {@code coordId} is present: returns
|
||||
* this daemon's own held lead-to-lead messages, in full, without acking them. Checked first and
|
||||
* exclusively of {@code ticket}/{@code target} — see {@link #pollAction}'s javadoc for why this
|
||||
* is a different inbox (this daemon's own {@link LeadChannel}) that authorizes differently
|
||||
* ({@link Authz.Action#COORD_READ}, primary-only) from either of the original two branches.
|
||||
*/
|
||||
static McpSchema.CallToolResult poll(MessageService messages, LeadChannel leadChannel, String ticket,
|
||||
String target, String coordId) {
|
||||
if (!isBlank(coordId)) {
|
||||
return pollHeldPeerMail(leadChannel, coordId);
|
||||
}
|
||||
if (!isBlank(target)) {
|
||||
var replies = messages.drainReplies(target);
|
||||
if (replies.isEmpty()) {
|
||||
@@ -884,6 +914,46 @@ public final class FleetMcp {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #421: a lead's own held lead-to-lead mail, read without consuming it.
|
||||
*
|
||||
* <p>{@link LeadChannel#peek} is non-destructive, so calling this twice returns the same
|
||||
* bodies, and {@code fleet_list}'s {@code coordinator.held[]} is unaffected — this adds a
|
||||
* read, it never acks. Never let this short-circuit the delivery contract: {@code
|
||||
* LeadCoordLoop}'s javadoc explains why a message must stay unacked until actually delivered,
|
||||
* and that is unchanged here.
|
||||
*
|
||||
* <p>{@code coordId} must be THIS daemon's own coord-id ({@code fleet_list}'s
|
||||
* {@code coordinator.selfId}) — there is no route here to read a PEER's outbound mail, only
|
||||
* your own inbound mail. Requiring the caller to echo its own id catches the exact confusion
|
||||
* that opened fleetd #421: the original failed attempt passed a PEER's id ("fleet01") as
|
||||
* {@code fleet_poll}'s {@code target}, expecting to read that peer's messages, and got the same
|
||||
* silent {@code []} as a genuinely empty worker inbox. This refuses the same mistake here with a
|
||||
* reason, instead of a second silent wrong answer.
|
||||
*/
|
||||
static McpSchema.CallToolResult pollHeldPeerMail(LeadChannel leadChannel, String coordId) {
|
||||
if (leadChannel == null) {
|
||||
return error("lead coordination is not configured (no coordinator: block) — there is "
|
||||
+ "no held peer mail to read.");
|
||||
}
|
||||
String selfId = leadChannel.selfCoordId();
|
||||
if (!coordId.equals(selfId)) {
|
||||
return error("coordId \"" + coordId + "\" is not this daemon's own coord-id (\"" + selfId
|
||||
+ "\"). fleet_poll reads only YOUR OWN held mail — pass your own coordId "
|
||||
+ "(fleet_list's coordinator.selfId), not a peer's.");
|
||||
}
|
||||
return text(json(leadChannel.peek().stream().map(FleetMcp::heldMailView).toList()));
|
||||
}
|
||||
|
||||
/** The full body of one held lead-to-lead message — never truncated, unlike {@link #heldView}. */
|
||||
private static Map<String, Object> heldMailView(LeadMessage m) {
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("msgId", m.msgId());
|
||||
row.put("from", m.from());
|
||||
row.put("content", m.content());
|
||||
return row;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code fleet_reply}: the worker returns its structured answer, resolving the awaiting send
|
||||
* or — when no send is open — queueing the reply in the inbox for later drain (CB-307).
|
||||
@@ -924,7 +994,11 @@ public final class FleetMcp {
|
||||
if (isBlank(target) || isBlank(msgId)) {
|
||||
return error("target and msgId are required");
|
||||
}
|
||||
messages.ackReply(target, msgId);
|
||||
if (!messages.ackReply(target, msgId)) {
|
||||
return error(msgId + " is not in " + target + "'s reply inbox (wrong id, wrong target, "
|
||||
+ "or already acked). Held lead-to-lead (peer) mail cannot be acked this way — "
|
||||
+ "read it with fleet_poll{coordId}.");
|
||||
}
|
||||
return text("acknowledged " + msgId);
|
||||
}
|
||||
|
||||
@@ -1179,6 +1253,23 @@ public final class FleetMcp {
|
||||
if (!coolingOff.isEmpty()) {
|
||||
result.put("coolingOff", coolingOff);
|
||||
}
|
||||
// fleetd #422: read the exact same accessor CompositePeerLauncher's spawn gate reads
|
||||
// (PeerLauncher.modelGateState(), which for the composite is models0() read live) — never a
|
||||
// separately-derived answer, so this status can never overstate or understate what the gate
|
||||
// actually enforces (the fleetd #404 lesson).
|
||||
//
|
||||
// fleetd #422 follow-up: "armed" and "off" come from the ONE modelGateState() call below,
|
||||
// never two independent reads of the gate — a reload landing between two separate reads
|
||||
// could otherwise make them disagree. modelGateArmed is reported unconditionally (never
|
||||
// omitted like quarantined/coolingOff above) precisely so a lead can tell "no models: block
|
||||
// at all" (false) apart from "a models: block with nothing currently off" (true, with
|
||||
// modelsOff simply absent below) — the two states PeerLauncher.disabledModels() alone
|
||||
// cannot distinguish, both reporting an empty set.
|
||||
PeerLauncher.ModelGateState modelGate = workers.modelGateState();
|
||||
result.put("modelGateArmed", modelGate.configured());
|
||||
if (!modelGate.off().isEmpty()) {
|
||||
result.put("modelsOff", new ArrayList<>(modelGate.off()));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -1305,6 +1396,16 @@ public final class FleetMcp {
|
||||
* counts, it can never make {@code fleet_list} itself slow or fail. {@code held} comes from
|
||||
* {@link LeadChannel#peek}, a pure in-memory read with no broker round trip, so it is never
|
||||
* subject to that bound.
|
||||
*
|
||||
* <p><strong>fleetd #421: {@code heldCount}/{@code heldDurable} fix the "pending: 0" trap.</strong>
|
||||
* {@code mailbox.pending} counts only broker-<em>ready</em> messages; a held message is already
|
||||
* an unacked delivery sitting with this consumer, so the normal, healthy state of a blocked lead
|
||||
* is {@code "pending": 0} next to a non-empty {@code held[]} — which invites the false reading
|
||||
* "these are only in memory, a restart will lose them". {@code heldCount} is the honest second
|
||||
* number beside {@code pending} ({@code held.size()}, not left for the reader to count the
|
||||
* array). {@code heldDurable} comes straight from {@link LeadChannel#heldDurable}, which the
|
||||
* channel implementation derives from what it actually did when it declared and consumed its own
|
||||
* queue (fleetd #440) — this method never asserts the fact itself.
|
||||
*/
|
||||
private static Map<String, Object> coordinatorView(CoordinationSource coordination) {
|
||||
LeadChannel channel = coordination.leadChannel();
|
||||
@@ -1312,11 +1413,14 @@ public final class FleetMcp {
|
||||
return null;
|
||||
}
|
||||
String selfId = channel.selfCoordId();
|
||||
List<LeadMessage> held = channel.peek();
|
||||
Map<String, Object> row = new LinkedHashMap<>();
|
||||
row.put("selfId", selfId);
|
||||
row.put("configured", true);
|
||||
row.put("mailbox", mailboxView(probe(channel, selfId)));
|
||||
row.put("held", channel.peek().stream().map(FleetMcp::heldView).toList());
|
||||
row.put("heldCount", held.size());
|
||||
row.put("heldDurable", channel.heldDurable());
|
||||
row.put("held", held.stream().map(FleetMcp::heldView).toList());
|
||||
row.put("peers", coordination.peers().stream().map(p -> peerView(channel, p)).toList());
|
||||
return row;
|
||||
}
|
||||
@@ -1645,10 +1749,18 @@ public final class FleetMcp {
|
||||
"Check an async delegation (a fleet_send with wait:false) by its ticket: "
|
||||
+ "pending, done (with the worker's reply), or failed. When target (a worker "
|
||||
+ "session id) is present instead of ticket, drain that worker's inbox of "
|
||||
+ "replies delivered when no send was open.",
|
||||
+ "replies delivered when no send was open. When coordId is present instead, "
|
||||
+ "read (never consume) your own held lead-to-lead mail — primary-only.",
|
||||
objectSchema(Map.of(
|
||||
"ticket", stringProp("The ticket returned by fleet_send wait:false"),
|
||||
"target", stringProp("Worker session id to drain pending replies from (optional)")),
|
||||
"target", stringProp("Worker session id to drain pending replies from (optional)"),
|
||||
"coordId", stringProp("Your own coord-id (fleet_list's coordinator.selfId) — "
|
||||
+ "reads every message currently held[] for you in full, without "
|
||||
+ "acking. Read twice, get the same bodies both times; fleet_list's "
|
||||
+ "held[] still reports them afterward. Primary-only, and this can "
|
||||
+ "read only YOUR OWN mailbox — there is no route to a peer's outbound "
|
||||
+ "mail, so passing a peer's coordId here is refused rather than "
|
||||
+ "silently returning the wrong thing (or nothing).")),
|
||||
List.of()));
|
||||
}
|
||||
|
||||
@@ -1658,7 +1770,10 @@ public final class FleetMcp {
|
||||
+ "has processed a reply and wants to confirm it, leaving other pending replies "
|
||||
+ "in the inbox for later drain.",
|
||||
objectSchema(Map.of(
|
||||
"target", stringProp("Worker session id whose inbox to ack from"),
|
||||
"target", stringProp("Worker session id whose inbox to ack from. Must name a "
|
||||
+ "reply actually queued for it — an id in no inbox, or a coord-id "
|
||||
+ "(peer held mail, read with fleet_poll{coordId} instead), errors "
|
||||
+ "rather than reporting a false success"),
|
||||
"msgId", stringProp("The message id to acknowledge")),
|
||||
List.of("target", "msgId")));
|
||||
}
|
||||
|
||||
@@ -14,6 +14,7 @@ import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementCandidate;
|
||||
import dev.ltms.fleet.placement.PlacementContext;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.placement.PlacementPolicy;
|
||||
@@ -90,6 +91,19 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
private final Supplier<Map<String, FleetConfig.Profile>> profileConfigs;
|
||||
private final Supplier<PlacementPolicy> placementPolicy;
|
||||
|
||||
/**
|
||||
* fleetd #422: the central model allow-list's on/off state, read live per spawn — same reason
|
||||
* {@link #profileConfigs} is a supplier rather than a captured map (see the class doc above and
|
||||
* {@link #enforceModelEnabled}). A caller with no {@code models:} block to read from (the
|
||||
* simpler, map-based constructors used throughout this class's own tests) wires this to a
|
||||
* constant {@code null}, which {@link #models0} treats as "nothing configured, gate never
|
||||
* fires" — the pre-#422 behaviour.
|
||||
*/
|
||||
private final Supplier<FleetConfig.Models> models;
|
||||
|
||||
/** The value {@link #models0} normalizes a {@code null} supplier result to. */
|
||||
private static final FleetConfig.Models NO_MODELS_CONFIGURED = new FleetConfig.Models(List.of());
|
||||
|
||||
/** CB-578 stage B: credential cooldown, checked before an explicit spawn and filtered into placement. */
|
||||
private final BackendQuarantine quarantine;
|
||||
|
||||
@@ -205,12 +219,44 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
// No models: block to read from a plain profiles Map — fleetd #422's gate is wired to a
|
||||
// constant null (see enforceModelEnabled/models0), the pre-#422 behaviour for every caller
|
||||
// of this overload. Use the 9-arg overload below to test the gate against a LIVE supplier.
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, quarantine,
|
||||
outagePolicy, constant(null));
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus a LIVE model-gate source (fleetd #422). The 8-arg overload above wires
|
||||
* {@code models} to a constant {@code null} because it has only a static {@code Map<String,
|
||||
* Profile>}, never a full {@code FleetConfig}, to read one from; this overload exists so a test
|
||||
* can prove {@link #enforceModelEnabled} and its candidate filter re-read {@link
|
||||
* FleetConfig.Models} on every call rather than a value captured once at construction — the
|
||||
* exact distinction fleetd #422 exists to get right (see the class doc's CB-559 note on {@link
|
||||
* #profileConfigs}, which this follows). Production wiring uses the
|
||||
* {@code Supplier<FleetConfig>} constructor below instead, which already threads a live
|
||||
* {@code config.get().models()} through.
|
||||
*
|
||||
* @param models required — pass a supplier returning {@code null} for a caller that has no
|
||||
* {@code models:} block to gate against, never a defaulting overload (the same
|
||||
* "explicit opt-out, never a silent default" rule {@code quarantine}/
|
||||
* {@code outagePolicy} already follow).
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy,
|
||||
Supplier<FleetConfig.Models> models) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine, outagePolicy);
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine, outagePolicy, models);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -250,7 +296,10 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
liveCount,
|
||||
() -> config.get().fleet(),
|
||||
quarantine,
|
||||
outagePolicy);
|
||||
outagePolicy,
|
||||
// fleetd #422: read live, same as profiles/placement/fleet above — a models.allow
|
||||
// edit (on/off or otherwise) is visible to the very next spawn, no restart needed.
|
||||
() -> config.get().models());
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
@@ -261,7 +310,8 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
BackendOutagePolicy outagePolicy,
|
||||
Supplier<FleetConfig.Models> models) {
|
||||
this.fleet = fleet;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
this.outagePolicy = Objects.requireNonNull(outagePolicy, "outagePolicy");
|
||||
@@ -273,6 +323,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
this.profileConfigs = profileConfigs;
|
||||
this.placementPolicy = placementPolicy;
|
||||
this.liveCount = liveCount;
|
||||
this.models = Objects.requireNonNull(models, "models");
|
||||
Map<String, HerdrPeerLauncher> index = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : this.delegates) {
|
||||
for (String profile : d.profiles()) {
|
||||
@@ -303,6 +354,16 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return m == null ? Map.of() : m;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-configured {@code models:} block, never null (fleetd #422). Read fresh on
|
||||
* every call, the same reason {@link #profiles0} is — a config reload's on/off edit must reach
|
||||
* the very next spawn.
|
||||
*/
|
||||
private FleetConfig.Models models0() {
|
||||
FleetConfig.Models m = models.get();
|
||||
return m == null ? NO_MODELS_CONFIGURED : m;
|
||||
}
|
||||
|
||||
/** The adapter owning {@code profileName} (null/blank → the default). Throws on an unknown profile. */
|
||||
private HerdrPeerLauncher route(String profileName) {
|
||||
String resolved = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
@@ -332,6 +393,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
enforceNotQuarantined(requestedProfile);
|
||||
enforceNotCoolingOff(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
enforceModelEnabled(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
@@ -341,18 +403,10 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// the whole profile list. An EXPLICIT profile (above) is left alone on purpose — it is the
|
||||
// operator overriding, and refusing it would break `fleet_spawn{profile:"opus"}`, which
|
||||
// carries no role and so would be judged against the dev pool it was never meant for.
|
||||
List<PlacementCandidate> candidates = candidates(req.role());
|
||||
String roleDefault = defaultProfileFor(req.role());
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
// fleetd #201 Unit 5: a distinct set from quarantined — see PlacementContext.coolingOff.
|
||||
Set<String> coolingOff = coolingOffProfiles(candidates);
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable,
|
||||
quarantined, coolingOff);
|
||||
PlacementContext ctx = placementContextFor(req.role(), unreachable);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
int maxAttempts = ctx.candidates().isEmpty() ? 1 : ctx.candidates().size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
// Deliberately uncaught: when no candidate is left (all at cap, or all unreachable) the
|
||||
// policy already throws a clear message. Catching it to rethrow a generic
|
||||
@@ -379,8 +433,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable,
|
||||
quarantined, coolingOff);
|
||||
ctx = placementContextFor(req.role(), unreachable);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -492,6 +545,49 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose {@code model:} the operator has turned off in the
|
||||
* central {@code models.allow:} list (fleetd #422). Deliberately worded apart from {@link
|
||||
* #enforceNotQuarantined} and {@link #enforceNotCoolingOff}: those two report a BACKEND-reported
|
||||
* outage (exhaustion, repeated errors); this one reports an OPERATOR decision, so the message
|
||||
* says "turned off" and names the model, never "quarantined" or "cooling off". A FOURTH,
|
||||
* independent reason to refuse a spawn — never layered onto {@code BackendQuarantine} or {@code
|
||||
* BackendOutagePolicy}, which would misattribute an operator's own choice to the backend.
|
||||
*
|
||||
* <p>Reads {@link #models0()} fresh on every call — the same liveness {@link #profiles0()}
|
||||
* already has — so flipping {@code enabled: false} and reloading takes effect on the very next
|
||||
* spawn, no restart (criterion 4). A profile naming no model, or a model absent from {@code
|
||||
* models.allow:} entirely (nothing to gate against), is never refused here.
|
||||
*
|
||||
* @throws PlacementException naming the model and the profile, distinct from quarantine/cool-off
|
||||
*/
|
||||
private void enforceModelEnabled(String profile) {
|
||||
FleetConfig.Profile cfg = profiles0().get(profile);
|
||||
String model = (cfg == null) ? null : cfg.model();
|
||||
if (model == null) {
|
||||
return;
|
||||
}
|
||||
if (models0().offIds().contains(model)) {
|
||||
throw new PlacementException("worker profile '" + profile + "' names model '" + model
|
||||
+ "', which the operator has turned off in models.allow — refusing spawn");
|
||||
}
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose {@code model:} is currently turned off (fleetd #422). */
|
||||
private Set<String> modelOffProfiles(List<PlacementCandidate> candidates) {
|
||||
Set<String> off = models0().offIds();
|
||||
if (off.isEmpty()) {
|
||||
return Set.of();
|
||||
}
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> {
|
||||
FleetConfig.Profile cfg = profiles0().get(p);
|
||||
return cfg != null && cfg.model() != null && off.contains(cfg.model());
|
||||
})
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile names {@code role} may be placed on, in definition order.
|
||||
*
|
||||
@@ -508,8 +604,23 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return known.isEmpty() ? List.copyOf(configured.keySet()) : known;
|
||||
}
|
||||
|
||||
/** The profile an unqualified spawn for {@code role} falls back to under {@code fixed} placement. */
|
||||
private String defaultProfileFor(MemberRole role) {
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Live: reads {@link #poolFor}, which reads {@link #profileConfigs} and {@link #fleet} fresh
|
||||
* on every call, so a config reload is visible without a restart (fleetd #425) — unlike {@link
|
||||
* #defaultProfile}, the field captured once at construction, which this falls back to only when
|
||||
* {@link #poolFor} has nothing to offer at all (no profiles configured for this composite).
|
||||
*
|
||||
* <p>Exact only under the {@code fixed} placement policy — the one that reads this value
|
||||
* ({@code FixedPlacementPolicy}, package-private, hence not linked) as its first, preferred
|
||||
* candidate. {@code weighted}/{@code round-robin} placement can choose a different candidate
|
||||
* from {@code role}'s pool even on the very first spawn; this method does not simulate that
|
||||
* choice, matching what the {@code defaultProfile:}-derived reporting this replaces has always
|
||||
* done.
|
||||
*/
|
||||
@Override
|
||||
public String defaultProfileFor(MemberRole role) {
|
||||
List<String> pool = poolFor(role);
|
||||
return pool.isEmpty() ? defaultProfile : pool.getFirst();
|
||||
}
|
||||
@@ -526,6 +637,142 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the {@link PlacementContext} an unqualified spawn of {@code role} would be judged
|
||||
* against right now — the single source both {@link #spawn} and {@link #place} read, so the two
|
||||
* can never disagree about which conditions (quarantine, cool-off, model-off) apply to which
|
||||
* candidate (fleetd #425 rework: round 1 duplicated this into a second, blind resolver —
|
||||
* {@link #defaultProfileFor} — which is why it regressed; round 2 found that even a single
|
||||
* shared resolver is not enough on its own if the CALLER re-resolves through an explicit
|
||||
* profile afterwards — see {@link PlacementDecision}).
|
||||
*
|
||||
* @param unreachable the caller's mutable unreachable set; {@link #spawn} grows this across
|
||||
* retries and rebuilds the context from it, {@link #place} passes a fresh
|
||||
* empty one since it never retries
|
||||
*/
|
||||
private PlacementContext placementContextFor(MemberRole role, Set<String> unreachable) {
|
||||
List<PlacementCandidate> candidates = candidates(role);
|
||||
String roleDefault = defaultProfileFor(role);
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
// fleetd #201 Unit 5: a distinct set from quarantined — see PlacementContext.coolingOff.
|
||||
Set<String> coolingOff = coolingOffProfiles(candidates);
|
||||
// fleetd #422: read live per spawn, same as quarantined/coolingOff above — a config reload
|
||||
// that flips a model's enabled state is visible to the very next unqualified spawn.
|
||||
Set<String> modelOff = modelOffProfiles(candidates);
|
||||
return new PlacementContext(roleDefault, candidates, liveCount, unreachable,
|
||||
quarantined, coolingOff, modelOff);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #425 rework, round 2: runs the exact same selection {@link #spawn} uses for a
|
||||
* blank-profile request — {@link #placementContextFor} plus one {@link PlacementPolicy#select}
|
||||
* — rather than {@link #defaultProfileFor}'s blind "pool's first entry", so a quarantined,
|
||||
* cooling-off, or model-off pool-first candidate is routed around here exactly as it would be
|
||||
* by a real spawn. Unlike {@link #spawn}, this never retries on {@link
|
||||
* PeerUnreachableException}: there is no spawn attempt to fail, so "unreachable" never grows
|
||||
* past the empty set it starts with, and a single {@link PlacementPolicy#select} call already
|
||||
* reflects the live quarantine/cool-off/model-off state.
|
||||
*
|
||||
* <p>Deliberately does <em>not</em> apply {@link #enforceMaxLoad} (or any of the other three
|
||||
* {@code enforce*} checks): those belong to {@link #spawn}'s EXPLICIT-profile branch, the
|
||||
* operator-override path, and this method answers a different question — "where would an
|
||||
* UNQUALIFIED spawn land". That is not the same as {@code select} ignoring these conditions —
|
||||
* every condition {@code select} filters on (quarantine, cooling off, {@code maxLoad} under
|
||||
* every placement policy including the default {@code fixed}, since fleetd #435, model-off,
|
||||
* unreachable, weight-0) is already reflected in the {@link PlacementDecision} this method
|
||||
* returns, because {@code select} walked past every excluded candidate to find it. What this
|
||||
* method's caller must not do is take that resolved name and hand it back to {@link
|
||||
* #spawn(SpawnRequest)} as an explicit profile: the explicit-profile branch treats the same
|
||||
* exclusion conditions as a reason to REFUSE, where {@code select} had already treated them as
|
||||
* a reason to fall through — round 1 of this fix did exactly that, turning a fall-through this
|
||||
* method had already resolved around into a refusal one call later. Round 2 fixes that at the
|
||||
* caller: {@link #spawn(SpawnRequest, PlacementDecision)} carries this exact decision to the
|
||||
* spawn without re-resolving or re-checking it, through the same routing path {@code select}
|
||||
* itself was consulted from.
|
||||
*
|
||||
* @throws PlacementException if no candidate in {@code role}'s pool is currently placeable
|
||||
* (mirrors what an actual unqualified spawn would throw)
|
||||
*/
|
||||
@Override
|
||||
public PlacementDecision place(MemberRole role) {
|
||||
PlacementContext ctx = placementContextFor(role, new HashSet<>());
|
||||
return new PlacementDecision(placementPolicy.get().select(ctx).profile());
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to {@link #place}, so the two can never disagree about the answer for the same
|
||||
* {@code role} at the same instant — kept as a convenience for a caller that only wants the
|
||||
* resolved name (a status report, a log line), never for a caller that will act on it by
|
||||
* spawning: that caller must hold the {@link PlacementDecision} itself and pass it to {@link
|
||||
* #spawn(SpawnRequest, PlacementDecision)} — see {@link PlacementDecision}'s javadoc for why
|
||||
* resolving here and spawning separately, with the name fed back in as an explicit profile,
|
||||
* regressed fleetd #425 twice.
|
||||
*/
|
||||
@Override
|
||||
public String routedProfileFor(MemberRole role) {
|
||||
return place(role).profile();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Routes {@code decision.profile()} directly to its owning delegate — the identical
|
||||
* {@code d.spawn(routedReq)} call {@link #spawn(SpawnRequest)}'s blank-profile branch makes for
|
||||
* its first pick — WITHOUT re-running {@link #enforceNotQuarantined}, {@link
|
||||
* #enforceNotCoolingOff}, {@link #enforceMaxLoad}, or {@link #enforceModelEnabled}: those are
|
||||
* the EXPLICIT-profile branch's checks, and {@code decision} did not come from an operator
|
||||
* naming a profile — it came from {@link #place}, which already applied whichever of these
|
||||
* conditions {@link PlacementPolicy#select} actually filters on (fleetd #425 rework, round 2).
|
||||
*
|
||||
* <p>The two branches disagree on purpose about what an excluded profile means, and that
|
||||
* disagreement is not what this method removes. The blank-profile routing branch (and
|
||||
* {@link #place}) treats a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0 profile
|
||||
* as a reason to fall through to the next candidate; the EXPLICIT-profile branch treats naming
|
||||
* that same profile as a reason to refuse outright — someone who names a profile should get a
|
||||
* refusal, not a silent substitution onto a different backend. That is still correct after
|
||||
* fleetd #435. What round 1 got wrong, and what this method exists to stop happening again, is
|
||||
* turning a fall-through into a refusal by accident: resolving a name via {@link #place} and
|
||||
* then handing that same name back to {@link #spawn(SpawnRequest)} as an explicit profile takes
|
||||
* the refusing branch on a decision the routing branch had already approved by falling through
|
||||
* past everything else.
|
||||
*
|
||||
* <p>Before fleetd #435, this exact accident was reachable through {@code maxLoad} specifically:
|
||||
* {@code FixedPlacementPolicy} — the default policy — did not evaluate {@code maxLoad} at all
|
||||
* for automatic selection, so {@link #place} could approve an at-cap profile that {@link
|
||||
* #enforceMaxLoad} would then refuse one call later. fleetd #435 closed that: {@code
|
||||
* FixedPlacementPolicy} now walks past an at-cap candidate exactly like {@code weighted}/
|
||||
* {@code round-robin} already did, so {@link #place} can no longer return one, and this specific
|
||||
* failure — an approved placement dying at {@code enforceMaxLoad} — cannot happen any more.
|
||||
* What this method still buys, now that {@code maxLoad} can no longer cause it: it never
|
||||
* re-evaluates a condition {@link #place} already decided, and it closes the window between
|
||||
* that decision and the spawn in which the underlying state (another spawn landing on the same
|
||||
* profile, a config reload) could otherwise move and make a stale explicit re-check wrong.
|
||||
*
|
||||
* <p>Deliberately does not retry on {@link PeerUnreachableException} across candidates the way
|
||||
* {@link #spawn(SpawnRequest)}'s blank-profile branch does: retrying here would silently
|
||||
* re-place the caller onto a different profile than the one {@code decision} named, behind the
|
||||
* back of a caller that may already have provisioned something (a worktree's {@code repoRoot},
|
||||
* parity overlay) specifically for that name. A caller that wants the composite's own failover
|
||||
* should call {@link #spawn(SpawnRequest)} with a blank profile directly, not resolve through
|
||||
* {@link #place} first. Losing that retry on a resolve-then-spawn path is an accepted, unrelated
|
||||
* cost — see {@code SessionManager.acquireWithWorktree}'s own comment on it — never widened by
|
||||
* this round to include {@code maxLoad}, which is what round 1 actually lost.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||
HerdrPeerLauncher d = route(decision.profile());
|
||||
SpawnRequest routedReq = req.withProfile(decision.profile());
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return route(req.profileName()).effectiveCwd(req);
|
||||
@@ -536,6 +783,32 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return route(profileName).parityOverlay(profileName);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: the single live read that answers both "is the models.allow: gate
|
||||
* armed" and "which models are off", off the exact same accessor ({@link #models0()}) {@link
|
||||
* #enforceModelEnabled} and {@link #modelOffProfiles} read — so {@code fleet_profiles}/{@code
|
||||
* GET /profiles} (via {@link PeerLauncher#disabledModels()}, which now delegates here) can
|
||||
* never report a different answer than the gate enforces (the fleetd #404 lesson), and the
|
||||
* startup log line built from this can never disagree with either.
|
||||
*
|
||||
* <p>{@link #models0()} itself normalizes a {@code null} {@link #models} read to the shared
|
||||
* {@link #NO_MODELS_CONFIGURED} sentinel — deliberately the one object no config-supplied
|
||||
* {@code Models} instance can ever be identical to, since it is private to this class — so
|
||||
* comparing by reference here recovers exactly the fact {@code models0()}'s normalization
|
||||
* would otherwise erase: whether the live source was {@code null} (no {@code models:} block,
|
||||
* armed = false) or a real, config-supplied block (armed = true, even one whose {@code allow:}
|
||||
* is itself empty or absent — {@link FleetConfig.Models}'s "absent or empty allow: is off"
|
||||
* wording governs config-load validation, a distinct question from whether this gate is armed
|
||||
* for reporting).
|
||||
*/
|
||||
@Override
|
||||
public PeerLauncher.ModelGateState modelGateState() {
|
||||
FleetConfig.Models m = models0();
|
||||
return m == NO_MODELS_CONFIGURED
|
||||
? PeerLauncher.ModelGateState.notConfigured()
|
||||
: PeerLauncher.ModelGateState.armed(m.offIds());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.get(id);
|
||||
@@ -647,9 +920,21 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return byProfile.keySet();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #425: reports the <em>live</em> {@code dev} pool's first entry — the same value
|
||||
* {@link #defaultProfileFor} computes for {@link MemberRole#DEV} — not the {@link
|
||||
* #defaultProfile} field captured at construction. An unqualified {@code fleet_spawn} defaults
|
||||
* to {@code MemberRole#DEV} (see {@link dev.ltms.fleet.peer.SpawnRequest}), so "the dev pool's
|
||||
* live first entry" is exactly the profile such a spawn actually lands on right now — the
|
||||
* question {@code fleet_profiles}' {@code "default"} field exists to answer. The frozen field is
|
||||
* a role-agnostic fallback used only when {@link #poolFor} has nothing to report at all (no
|
||||
* profiles configured), which {@link #defaultProfileFor} already handles.
|
||||
*/
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
return defaultProfileFor(MemberRole.DEV);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -394,17 +394,17 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
public boolean ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
Held h;
|
||||
synchronized (perTarget) {
|
||||
h = perTarget.remove(msgId);
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
return false; // never held (or already acked) — no-op
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
@@ -418,6 +418,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
throw new IllegalStateException("cannot ack reply " + msgId + " on " + queueName(target), e);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback(String target) {
|
||||
|
||||
@@ -59,16 +59,17 @@ public final class InMemoryReplyInbox implements ReplyInbox {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
public boolean ack(String target, String msgId) {
|
||||
if (!owned.contains(target)) {
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
var perTarget = store.get(target);
|
||||
if (perTarget != null) {
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
perTarget.remove(msgId);
|
||||
}
|
||||
if (perTarget == null) {
|
||||
return false;
|
||||
}
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
return perTarget.remove(msgId) != null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -42,6 +42,18 @@ public interface LeadChannel {
|
||||
/** This daemon's own lead coordination id — the mailbox it owns, and the {@code from} it sends as. */
|
||||
String selfCoordId();
|
||||
|
||||
/**
|
||||
* Whether a message sitting in {@link #peek}'s held set (fetched but not yet {@link #ack}ed) is
|
||||
* still safe if this daemon crashes or restarts right now — the conclusion of two independent
|
||||
* facts about how this channel owns its own queue: the queue was declared <em>durable</em>, and
|
||||
* the consumer that filled {@code held} uses <em>manual ack</em>, so an unacked delivery is still
|
||||
* owned by the broker rather than only in this process's memory. Both must hold for {@code true};
|
||||
* an implementation must derive this from what it actually did when it declared and consumed its
|
||||
* queue, never return a literal — fleetd #440 found {@code FleetMcp}'s {@code heldDurable} field
|
||||
* doing exactly that, unable to ever report {@code false} even after the fact stopped being true.
|
||||
*/
|
||||
boolean heldDurable();
|
||||
|
||||
/**
|
||||
* A non-destructive look at {@code coordId}'s mailbox — does it exist, how many messages are
|
||||
* waiting on it, and how many consumers are attached — without owning, consuming, or otherwise
|
||||
|
||||
@@ -86,6 +86,12 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
private final Object channelLock = new Object();
|
||||
/** msgId → held delivery, for this mailbox's own queue only (there is exactly one). */
|
||||
private final LinkedHashMap<String, Held> held = new LinkedHashMap<>();
|
||||
/**
|
||||
* fleetd #440: the answer to {@link #heldDurable()}, set once by {@link #own()} from the exact
|
||||
* booleans it passed to {@code queueDeclare}/{@code basicConsume} — never a separate literal that
|
||||
* could drift from what those calls actually did.
|
||||
*/
|
||||
private boolean heldDurable;
|
||||
/** Successful broker acks on this connection, retained only to make a repeated caller ack quiet. */
|
||||
private final LinkedHashMap<String, Boolean> recentlyAcked = new LinkedHashMap<>();
|
||||
/** Bounds {@link #recentlyAcked}: it is only an idempotency aid, never delivery state. */
|
||||
@@ -191,13 +197,22 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
/** Declare + consume this daemon's own {@code lead.<selfCoordId>.inbox}. Called once, at construction. */
|
||||
private void own() throws IOException {
|
||||
String queue = queueName(selfCoordId);
|
||||
boolean durableQueue = true; // durable, non-exclusive, keep on idle
|
||||
boolean autoAck = false; // manual ack
|
||||
synchronized (channelLock) {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
channel.basicConsume(queue, false, deliverCallback(), _ -> { }); // autoAck=false: manual ack
|
||||
channel.queueDeclare(queue, durableQueue, false, false, null);
|
||||
channel.basicConsume(queue, autoAck, deliverCallback(), _ -> { });
|
||||
}
|
||||
// fleetd #440: held mail is durable only while both hold — a durable queue AND manual ack.
|
||||
this.heldDurable = durableQueue && !autoAck;
|
||||
log.debug("lead mailbox owns queue {} for coord-id {}", queue, selfCoordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return heldDurable;
|
||||
}
|
||||
|
||||
/**
|
||||
* Publish {@code msg} to {@code toCoordId}'s mailbox and block until the broker's publisher
|
||||
* confirm for it lands. Does <em>not</em> imply owning or consuming {@code toCoordId}'s queue.
|
||||
|
||||
@@ -826,9 +826,13 @@ public final class MessageService {
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*
|
||||
* @return {@code true} if an entry was actually removed, {@code false} if {@code msgId} was not
|
||||
* in {@code target}'s inbox (wrong id, wrong target, or already acked). The caller —
|
||||
* {@link dev.ltms.fleet.mcp.FleetMcp#ack} — must not report success on {@code false}.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
public boolean ackReply(String target, String msgId) {
|
||||
return inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -46,6 +46,13 @@ public interface ReplyInbox {
|
||||
/** Non-destructive snapshot of pending replies for {@code target} (FIFO), empty list if none. */
|
||||
List<InboxMessage> peek(String target);
|
||||
|
||||
/** Remove the reply {@code msgId} for {@code target} once the primary has taken it. No-op if absent. */
|
||||
void ack(String target, String msgId);
|
||||
/**
|
||||
* Remove the reply {@code msgId} for {@code target} once the primary has taken it.
|
||||
*
|
||||
* @return {@code true} if an entry was actually removed, {@code false} if there was nothing to
|
||||
* remove (unknown {@code target}, unowned {@code target}, or a {@code msgId} not held for
|
||||
* it). A {@code false} is not an error — acking a {@code target} this daemon does not own is
|
||||
* part of the normal contract, not a failure.
|
||||
*/
|
||||
boolean ack(String target, String msgId);
|
||||
}
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.fleet.peer;
|
||||
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
@@ -143,9 +145,124 @@ public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*
|
||||
* <p>fleetd #425: for an implementation with role pools (a no-argument spawn is read as {@link
|
||||
* MemberRole#DEV}, see {@link SpawnRequest}), this must be the profile a live spawn of that role
|
||||
* would actually be placed on right now, not a value captured once at startup — a caller such as
|
||||
* {@code fleet_profiles} relies on this to report a live, not frozen, fact.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* The profile an unqualified spawn of {@code role} would resolve to right now — the role-aware,
|
||||
* live counterpart of {@link #defaultProfile()} (fleetd #425).
|
||||
*
|
||||
* <p>A caller that must provision something profile-specific (working directory, parity overlay
|
||||
* files) <em>before</em> the actual spawn — {@code SessionManager.acquireWithWorktree} is the one
|
||||
* that exists today — needs the exact profile that spawn will use, for the caller's real role,
|
||||
* not a role-agnostic guess. Calling {@link #defaultProfile()} for that purpose reads {@code
|
||||
* MemberRole#DEV}'s answer regardless of the caller's actual role, which is wrong for any other
|
||||
* role and can provision for a profile the spawn never lands on.
|
||||
*
|
||||
* <p>Default implementation returns {@link #defaultProfile()}, ignoring {@code role} — the right
|
||||
* answer for a launcher with no role-pool concept of its own (e.g. a single {@code
|
||||
* HerdrPeerLauncher} adapter, which is never reached this way in production: {@code
|
||||
* CompositePeerLauncher} always fronts it and resolves roles itself).
|
||||
*/
|
||||
default String defaultProfileFor(MemberRole role) {
|
||||
return defaultProfile();
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile an <em>unqualified</em> spawn of {@code role} would actually be routed to right
|
||||
* now — the same candidate list, the same {@code quarantined}/{@code coolingOff}/{@code
|
||||
* modelOff} filtering, and the same {@code PlacementPolicy} that {@link #spawn} itself
|
||||
* consults for a blank-profile request (fleetd #425 rework).
|
||||
*
|
||||
* <p>This is <em>not</em> {@link #defaultProfileFor}: that method answers "what is first in
|
||||
* {@code role}'s pool", blind to quarantine, cool-off, and the model on/off gate — the right
|
||||
* answer for a role-agnostic, best-effort report ({@code fleet_profiles}' {@code "default"}
|
||||
* field), but the wrong one for a caller that needs the profile a spawn will actually land on.
|
||||
* A quarantined or model-off pool-first profile makes {@link #defaultProfileFor} return a name
|
||||
* an unqualified spawn will never be routed to.
|
||||
*
|
||||
* <p>Just the resolved name, not the full {@link PlacementDecision} — a caller that only wants
|
||||
* to know the answer (a status report, a log line) can call this; a caller that will later
|
||||
* <em>act</em> on the answer by spawning — provisioning a worktree for a specific profile
|
||||
* before the peer exists is the one that matters — must call {@link #place} and carry the
|
||||
* {@link PlacementDecision} itself through to {@link #spawn(SpawnRequest, PlacementDecision)}
|
||||
* instead of calling this method and feeding the string back in as an explicit profile. Doing
|
||||
* that re-enters {@link #spawn(SpawnRequest)}'s explicit-profile branch, which disagrees with
|
||||
* the routing branch on purpose about what an excluded profile means: the routing branch (and
|
||||
* {@link #place}) falls through a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0
|
||||
* profile to the next candidate, while the explicit branch refuses outright — correct for an
|
||||
* operator who named that profile on purpose, wrong for a name that only ever came from placement
|
||||
* itself. That accidental refusal is exactly the regression fleetd #425 rework round 2 fixes:
|
||||
* the default implementation below delegates to {@link #place}, so the two can never drift apart,
|
||||
* but a caller that resolves through this method alone and spawns separately can still recreate
|
||||
* the round-1 defect for itself. (Before fleetd #435, this accident was also reachable through
|
||||
* {@code maxLoad} specifically, because {@code FixedPlacementPolicy} — the default policy — did
|
||||
* not evaluate it at all for automatic selection; #435 closed that gap, so a placement decision
|
||||
* can no longer be at cap in the first place. The refusal-vs-fall-through disagreement above is
|
||||
* the part that was never about {@code maxLoad} and is still real.)
|
||||
*
|
||||
* @throws RuntimeException (implementation-specific, typically a placement exception) if no
|
||||
* candidate in {@code role}'s pool is currently placeable
|
||||
*/
|
||||
default String routedProfileFor(MemberRole role) {
|
||||
return place(role).profile();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve, <em>without spawning</em>, the {@link PlacementDecision} an unqualified spawn of
|
||||
* {@code role} would make right now — the same candidate list, the same {@code
|
||||
* quarantined}/{@code coolingOff}/{@code modelOff} filtering, and the same {@code
|
||||
* PlacementPolicy} {@link #spawn(SpawnRequest)}'s blank-profile branch itself consults (fleetd
|
||||
* #425 rework).
|
||||
*
|
||||
* <p>Pair this with {@link #spawn(SpawnRequest, PlacementDecision)}, never with {@link
|
||||
* #spawn(SpawnRequest)} fed the decision's profile as an explicit name — see {@link
|
||||
* PlacementDecision}'s own javadoc for why that second form regressed.
|
||||
*
|
||||
* <p>Default implementation wraps {@link #defaultProfile()}, ignoring {@code role} and every
|
||||
* placement condition — the right answer for a launcher with no pool or placement-policy
|
||||
* concept of its own, matching {@link #defaultProfileFor}'s own default.
|
||||
*
|
||||
* @throws RuntimeException (implementation-specific, typically a placement exception) if no
|
||||
* candidate in {@code role}'s pool is currently placeable
|
||||
*/
|
||||
default PlacementDecision place(MemberRole role) {
|
||||
return new PlacementDecision(defaultProfile());
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn against an already-resolved {@link PlacementDecision} from {@link #place}, honoring it
|
||||
* completely: none of the conditions {@link #place} already applied — quarantine, cooling off,
|
||||
* {@code maxLoad} (evaluated by every placement policy including the default {@code fixed},
|
||||
* since fleetd #435), model-off — are re-evaluated here; {@code decision} already reflects them.
|
||||
* This is not skipping a check {@code place} left undone; it is not repeating one {@code place}
|
||||
* already did, and not re-opening the window between that decision and this spawn in which the
|
||||
* underlying state could otherwise move. This is what lets a resolve-then-spawn caller
|
||||
* ({@code SessionManager.acquireWithWorktree}, which must know the profile before it can
|
||||
* provision a worktree for it) and a plain blank-profile {@link #spawn(SpawnRequest)} caller
|
||||
* land on the exact same outcome for the exact same placement state (fleetd #425 rework,
|
||||
* round 2).
|
||||
*
|
||||
* <p>{@code req}'s own {@link SpawnRequest#profileName()} is ignored in favor of {@code
|
||||
* decision.profile()} — the caller is expected to have built {@code req} with a blank or
|
||||
* matching profile; passing a request that names a <em>different</em>, explicit profile than
|
||||
* the decision it is paired with is a caller bug this method does not attempt to detect.
|
||||
*
|
||||
* <p>Default implementation for a launcher with no placement concept of its own: delegates to
|
||||
* {@link #spawn(SpawnRequest)} with the decision's profile named explicitly — its only spawn
|
||||
* contract, since there is no separate routing path to honor.
|
||||
*
|
||||
* @throws IllegalArgumentException if the decision names an unknown profile
|
||||
*/
|
||||
default PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||
return spawn(req.withProfile(decision.profile()));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
@@ -192,4 +309,66 @@ public interface PeerLauncher {
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
|
||||
/**
|
||||
* Model ids the operator has currently turned off in the central {@code models.allow:} list
|
||||
* (fleetd #422) — empty for a launcher with nothing to gate against. {@code fleet_profiles}/
|
||||
* {@code GET /profiles} (via {@code FleetMcp.profilesView}) call this to report which models
|
||||
* are off, and MUST read this exact accessor rather than deriving their own answer: the fleetd
|
||||
* #404 lesson is that a status field reading a different source than the behaviour it describes
|
||||
* can drift from what the gate ({@code CompositePeerLauncher.enforceModelEnabled} and its
|
||||
* candidate filter) actually enforces. A default of {@code Set.of()} keeps every other {@link
|
||||
* PeerLauncher} implementer (the herdr adapters, and the two test-fake implementers) unchanged.
|
||||
*
|
||||
* <p>fleetd #422 follow-up: this alone cannot tell "no {@code models:} block at all" from "a
|
||||
* {@code models:} block where nothing is currently off" — both report an empty set here. Delegates
|
||||
* to {@link #modelGateState()} so the two facts always come from the one read {@link
|
||||
* #modelGateState()}'s implementer makes; do not override this method separately from that one.
|
||||
*/
|
||||
default Set<String> disabledModels() {
|
||||
return modelGateState().off();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the central {@code models.allow:} gate (fleetd #422) is armed at all, together with
|
||||
* which model ids are currently off — fleetd #422 follow-up. {@link #disabledModels()} alone
|
||||
* cannot distinguish two states that both report an empty set: a host with no {@code models:}
|
||||
* block (nothing is gated, and nothing can be) and a host WITH a {@code models:} block where
|
||||
* nothing is currently turned off (the gate is armed and reporting zero). This method exists so
|
||||
* a caller — the startup log, {@code fleet_profiles}/{@code GET /profiles} — can tell the two
|
||||
* apart, the same reason {@code CompletionResolver.UnsetMeaning} exists: an accessor that can
|
||||
* legitimately report "empty" must never let a caller guess why.
|
||||
*
|
||||
* <p>Default {@link ModelGateState#notConfigured()} — every launcher without a {@code models:}
|
||||
* block to read from (the herdr adapters, and the two test-fake implementers), matching {@link
|
||||
* #disabledModels()}'s own default of an empty set.
|
||||
*/
|
||||
default ModelGateState modelGateState() {
|
||||
return ModelGateState.notConfigured();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: the result of {@link #modelGateState()} — see that method's javadoc
|
||||
* for why "armed" and "off" must be reported together from one read rather than as two
|
||||
* separately-derived facts that a reload landing between them could make disagree.
|
||||
*
|
||||
* @param configured {@code true} when a {@code models:} block exists at all (armed), regardless
|
||||
* of whether anything in it is currently turned off; {@code false} when there
|
||||
* is no block to gate against
|
||||
* @param off the model ids currently turned off; always empty when {@code configured} is
|
||||
* {@code false}
|
||||
*/
|
||||
record ModelGateState(boolean configured, Set<String> off) {
|
||||
public ModelGateState {
|
||||
off = Set.copyOf(off);
|
||||
}
|
||||
|
||||
public static ModelGateState notConfigured() {
|
||||
return new ModelGateState(false, Set.of());
|
||||
}
|
||||
|
||||
public static ModelGateState armed(Set<String> off) {
|
||||
return new ModelGateState(true, off);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,14 +5,12 @@ import java.util.List;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps
|
||||
* ({@code maxLoad}) so that a pre-existing config behaves identically after upgrade — capacity
|
||||
* gating for automatic placement is deliberately out of scope for {@code fixed}, exactly as it
|
||||
* always has been. Reachability is a narrower exception (fleetd #315, below): a profile is never
|
||||
* checked for reachability up front, only skipped once it has already failed in <em>this same</em>
|
||||
* spawn call's retry loop — see the unreachable case below.
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. Reachability is a narrower
|
||||
* exception (fleetd #315, below): a profile is never checked for reachability up front, only
|
||||
* skipped once it has already failed in <em>this same</em> spawn call's retry loop — see the
|
||||
* unreachable case below.
|
||||
*
|
||||
* <p>Four exceptions walk past the default instead of returning it unconditionally:
|
||||
* <p>Six exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
@@ -20,6 +18,24 @@ import java.util.List;
|
||||
* ({@code BackendOutagePolicy}) — a separate, shorter-lived source from quarantine. When a
|
||||
* profile is both quarantined and cooling off, only the quarantine reason is reported
|
||||
* (exhaustion takes priority), matching {@code CompositePeerLauncher}'s explicit-spawn order.
|
||||
* <li>At cap (fleetd #435): a profile whose live count has reached its {@code maxLoad}
|
||||
* ({@link PlacementPolicyUtil#atCap}) — a documented, unconditional capacity limit (see
|
||||
* {@code FleetConfig.Profile#maxLoad}), so {@code fixed} must gate on it exactly as {@code
|
||||
* weighted}/{@code round-robin} already do via {@link PlacementPolicyUtil#available}. Before
|
||||
* this fix {@code fixed} built its own {@link PlacementCandidate} for the default with {@code
|
||||
* maxLoad} forced to {@code null}, so a capped default was chosen anyway on every unqualified
|
||||
* spawn — the cap was advisory, not enforced, for the one placement policy every config uses
|
||||
* by default. Reported only when quarantine and cooling off are both absent, matching {@code
|
||||
* CompositePeerLauncher}'s explicit-spawn check order (quarantine, then cooling off, then max
|
||||
* load, then model-off).
|
||||
* <li>Model off (fleetd #422): a profile whose {@code model:} the operator has turned off in
|
||||
* {@code models.allow:} — an operator decision, never a backend-reported outage, so it is a
|
||||
* fifth, independent source from quarantine, cooling off, and at-cap (never merged with any
|
||||
* of them), exactly as {@code CompositePeerLauncher.enforceModelEnabled} and {@link
|
||||
* PlacementPolicyUtil#available} treat it. When a profile is model-off <em>and</em> quarantined,
|
||||
* cooling off, or at cap, only the higher-priority reason is reported, matching {@code
|
||||
* CompositePeerLauncher}'s explicit-spawn check order (quarantine, then cooling off, then max
|
||||
* load, then model-off).
|
||||
* <li>Unreachable (fleetd #315): {@code CompositePeerLauncher.spawn} retries a failed candidate
|
||||
* on the next one and rebuilds the {@link PlacementContext} so {@code ctx.unreachable()}
|
||||
* names every profile that already failed with {@code PeerUnreachableException} in this same
|
||||
@@ -32,9 +48,10 @@ import java.util.List;
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code fleet_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined, cooling off, unreachable, or weight-0 never exercises
|
||||
* any of these paths, so today's behaviour is unchanged — in particular, the very first selection
|
||||
* of a spawn call always sees an empty {@code unreachable} set, so the first choice is untouched.
|
||||
* A fleet where nothing is ever quarantined, cooling off, at cap, model-off, unreachable, or
|
||||
* weight-0 never exercises any of these paths, so today's behaviour is unchanged — in particular,
|
||||
* the very first selection of a spawn call always sees an empty {@code unreachable} set, so the
|
||||
* first choice is untouched.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@@ -42,12 +59,14 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !ctx.coolingOff().contains(d)
|
||||
&& !ctx.unreachable().contains(d) && !weightExcluded(ctx, d)) {
|
||||
&& !ctx.modelOff().contains(d) && !ctx.unreachable().contains(d) && !weightExcluded(ctx, d)
|
||||
&& !capExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !ctx.coolingOff().contains(c.profile())
|
||||
&& !ctx.unreachable().contains(c.profile()) && !c.excluded()) {
|
||||
&& !ctx.modelOff().contains(c.profile()) && !ctx.unreachable().contains(c.profile())
|
||||
&& !c.excluded() && !PlacementPolicyUtil.atCap(ctx, c)) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
@@ -56,9 +75,18 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
// Exhaustion quarantine takes priority: reported only when quarantine is absent, so the
|
||||
// message never claims "cooling off" for a profile that is really backend-exhausted.
|
||||
boolean dCoolingOff = !dQuarantined && ctx.coolingOff().contains(d);
|
||||
// fleetd #435: at-cap sits between cooling off and model-off, matching
|
||||
// CompositePeerLauncher's explicit-spawn check order (quarantine, cooling off, max load,
|
||||
// then model-off) — reported only when quarantine/cooling-off are both absent.
|
||||
boolean dAtCap = !dQuarantined && !dCoolingOff && capExcluded(ctx, d);
|
||||
// fleetd #422: model-off is a fifth, independent source (an operator decision) — but
|
||||
// quarantine/cooling-off/at-cap still take priority when more than one applies, matching
|
||||
// CompositePeerLauncher's explicit-spawn check order (quarantine, cooling off, max load,
|
||||
// then model-off).
|
||||
boolean dModelOff = !dQuarantined && !dCoolingOff && !dAtCap && ctx.modelOff().contains(d);
|
||||
boolean dUnreachable = ctx.unreachable().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined || dCoolingOff || dUnreachable || dWeightExcluded) {
|
||||
if (dQuarantined || dCoolingOff || dAtCap || dModelOff || dUnreachable || dWeightExcluded) {
|
||||
List<String> reasons = new ArrayList<>();
|
||||
if (dQuarantined) {
|
||||
reasons.add("is quarantined (backend exhausted)");
|
||||
@@ -66,6 +94,14 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
if (dCoolingOff) {
|
||||
reasons.add("is cooling off after repeated backend errors");
|
||||
}
|
||||
if (dAtCap) {
|
||||
PlacementCandidate c = candidateFor(ctx, d);
|
||||
int live = ctx.liveCount().apply(d);
|
||||
reasons.add("is at maxLoad (" + live + " live >= " + c.maxLoad() + " cap)");
|
||||
}
|
||||
if (dModelOff) {
|
||||
reasons.add("names a model the operator has turned off in models.allow");
|
||||
}
|
||||
if (dUnreachable) {
|
||||
reasons.add("is unreachable");
|
||||
}
|
||||
@@ -78,18 +114,38 @@ final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException("all worker profiles are excluded from automatic "
|
||||
+ "selection (quarantined, cooling off, unreachable, or weight-0)");
|
||||
+ "selection (quarantined, cooling off, at cap, model-off, unreachable, or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
/** Whether {@code profile} carries {@code weight <= 0} (CB-554) among {@code ctx}'s candidates. */
|
||||
private static boolean weightExcluded(PlacementContext ctx, String profile) {
|
||||
PlacementCandidate c = candidateFor(ctx, profile);
|
||||
return c != null && c.excluded();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code profile} has reached its {@code maxLoad} cap (fleetd #435), using the shared
|
||||
* {@link PlacementPolicyUtil#atCap} definition — the same one {@code weighted}/{@code
|
||||
* round-robin} already consult via {@link PlacementPolicyUtil#available}. Looked up by name,
|
||||
* the same way {@link #weightExcluded} is: the default fast path above builds its own {@link
|
||||
* PlacementCandidate} with {@code maxLoad} forced to {@code null} (it carries no cap of its
|
||||
* own), so the candidate actually configured for {@code profile} has to be found in {@code
|
||||
* ctx.candidates()} first.
|
||||
*/
|
||||
private static boolean capExcluded(PlacementContext ctx, String profile) {
|
||||
PlacementCandidate c = candidateFor(ctx, profile);
|
||||
return c != null && PlacementPolicyUtil.atCap(ctx, c);
|
||||
}
|
||||
|
||||
/** The configured candidate named {@code profile} in {@code ctx}, or {@code null} if none. */
|
||||
private static PlacementCandidate candidateFor(PlacementContext ctx, String profile) {
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.profile().equals(profile)) {
|
||||
return c.excluded();
|
||||
return c;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -22,11 +22,29 @@ import java.util.function.Function;
|
||||
* the two apart so its refusal message says "cooling off", not "exhausted",
|
||||
* when only this one is active. A profile can be in both sets at once; when it
|
||||
* is, exhaustion quarantine is reported (it takes priority).
|
||||
* @param modelOff profiles whose {@code model:} is currently turned off in {@code
|
||||
* models.allow:} (fleetd #422) — an operator decision, not a backend-reported
|
||||
* outage, so a SEPARATE, independent source from both {@code quarantined} and
|
||||
* {@code coolingOff}. A profile can be in this set together with either (or
|
||||
* both) of the others; {@link PlacementPolicyUtil} counts it into its own
|
||||
* bucket rather than merging it into theirs, the same reason
|
||||
* {@code coolingOff} is kept apart from {@code quarantined}.
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable,
|
||||
Set<String> quarantined,
|
||||
Set<String> coolingOff) {
|
||||
Set<String> coolingOff,
|
||||
Set<String> modelOff) {
|
||||
|
||||
/**
|
||||
* Back-compat form before the fleetd #422 model on/off gate was added — no candidate's model
|
||||
* is off. Keeps pre-#422 call sites (tests included) compiling and behaving identically.
|
||||
*/
|
||||
public PlacementContext(String defaultProfile, List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount, Set<String> unreachable,
|
||||
Set<String> quarantined, Set<String> coolingOff) {
|
||||
this(defaultProfile, candidates, liveCount, unreachable, quarantined, coolingOff, Set.of());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
|
||||
/**
|
||||
* An already-completed placement choice — the outcome of one {@link PeerLauncher#place} call,
|
||||
* carried forward so a later {@link PeerLauncher#spawn(SpawnRequest, PlacementDecision)} can honor
|
||||
* it directly instead of re-resolving the profile a second time (fleetd #425 rework, round 2).
|
||||
*
|
||||
* <p>The problem this exists to close: a caller that must know the profile <em>before</em> it can
|
||||
* spawn — {@code SessionManager.acquireWithWorktree} provisions a worktree's {@code repoRoot} and
|
||||
* parity overlay for a specific profile before the peer process exists — used to resolve that name
|
||||
* with {@code PeerLauncher.routedProfileFor(role)} and then hand the SAME string back to {@link
|
||||
* PeerLauncher#spawn(SpawnRequest)} as an EXPLICIT profile. That re-resolution is not free: naming
|
||||
* a profile explicitly makes {@code CompositePeerLauncher.spawn} take its THROWING branch
|
||||
* ({@code enforceNotQuarantined}/{@code enforceNotCoolingOff}/{@code enforceMaxLoad}/{@code
|
||||
* enforceModelEnabled}), while an unqualified spawn's ROUTING branch never runs those checks at
|
||||
* all — it instead FALLS THROUGH to the next candidate on exactly the same conditions the throwing
|
||||
* branch refuses on. That disagreement is deliberate: an operator who names a profile should get a
|
||||
* refusal, not a silent substitution. The bug is turning the fall-through into a refusal by
|
||||
* accident — resolving a name through the routing side and then re-entering the refusing side with
|
||||
* it, for a decision the routing side had already approved by walking past everything else.
|
||||
* Before fleetd #435, this accident was also reachable through {@code maxLoad} specifically: the
|
||||
* default {@code fixed} placement policy did not evaluate {@code maxLoad} at all for automatic
|
||||
* selection, so a profile placement itself just approved could still die at {@code enforceMaxLoad}
|
||||
* one call later, purely because the caller's route to the spawn passed through an explicit
|
||||
* profile name instead of the routing branch — a failure a worktree-less unqualified spawn would
|
||||
* never hit. fleetd #435 closed that specific gap ({@code fixed} now evaluates {@code maxLoad}
|
||||
* exactly like every other placement policy), so a {@link PlacementDecision} can no longer be
|
||||
* at-cap in the first place — but the refusal-vs-fall-through disagreement above was never about
|
||||
* {@code maxLoad}, and resolving a name and re-entering the refusing branch with it is still wrong
|
||||
* for every OTHER condition placement filters on.
|
||||
*
|
||||
* <p>{@link PeerLauncher#spawn(SpawnRequest, PlacementDecision)} closes that by spawning through
|
||||
* the identical code path the routing branch itself uses, keyed off the SAME decision {@link
|
||||
* PeerLauncher#place} returned — no re-checking of any condition placement already evaluated. A
|
||||
* resolve-then-spawn caller and a blank-profile {@link PeerLauncher#spawn(SpawnRequest)} caller can
|
||||
* then never disagree about which conditions apply to the same placement state, and neither one
|
||||
* re-opens the window between the placement decision and the spawn in which the underlying state
|
||||
* could otherwise move.
|
||||
*
|
||||
* @param profile the profile this decision resolved to (may be {@code null} only when no profile is
|
||||
* configured at all — the same corner case {@link PeerLauncher#defaultProfile()}
|
||||
* already tolerates)
|
||||
*/
|
||||
public record PlacementDecision(String profile) {
|
||||
}
|
||||
@@ -11,28 +11,40 @@ final class PlacementPolicyUtil {
|
||||
private PlacementPolicyUtil() {
|
||||
}
|
||||
|
||||
/**
|
||||
* True when {@code c} has reached its {@code maxLoad} cap: {@code liveCount(c.profile()) >=
|
||||
* c.maxLoad()}. A {@code null} maxLoad means unlimited, so it is never at cap.
|
||||
*
|
||||
* <p>Extracted as the single shared definition of "at cap" (fleetd #435): before this fix it
|
||||
* was computed inline in both {@link #available} and {@link #emptyException}, and {@code
|
||||
* FixedPlacementPolicy} — not a caller of either — quietly kept its own {@code select} free of
|
||||
* any cap check at all, so a capped default profile was chosen anyway under the default
|
||||
* placement policy. Every automatic policy must call this, not re-derive it.
|
||||
*/
|
||||
static boolean atCap(PlacementContext ctx, PlacementCandidate c) {
|
||||
Integer cap = c.maxLoad();
|
||||
return cap != null && ctx.liveCount().apply(c.profile()) >= cap;
|
||||
}
|
||||
|
||||
/**
|
||||
* Candidates that are not weight-excluded (CB-554: explicit {@code weight <= 0}, checked
|
||||
* first because it is a static config choice rather than transient state), not
|
||||
* known-unreachable, not quarantined (CB-578 stage B), not cooling off after repeated backend
|
||||
* errors (fleetd #201 Unit 5 — a separate, shorter-lived source from quarantine), and have not
|
||||
* reached their maxLoad. A {@code null} maxLoad means unlimited.
|
||||
* errors (fleetd #201 Unit 5 — a separate, shorter-lived source from quarantine), not naming a
|
||||
* model the operator has turned off (fleetd #422 — a third, independent source: an operator
|
||||
* decision, never a backend-reported outage), and have not reached their maxLoad (see {@link
|
||||
* #atCap}). A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.excluded() || ctx.unreachable().contains(c.profile())
|
||||
|| ctx.quarantined().contains(c.profile())
|
||||
|| ctx.coolingOff().contains(c.profile())) {
|
||||
|| ctx.coolingOff().contains(c.profile())
|
||||
|| ctx.modelOff().contains(c.profile())
|
||||
|| atCap(ctx, c)) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
if (cap != null) {
|
||||
int live = ctx.liveCount().apply(c.profile());
|
||||
if (live >= cap) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.add(c);
|
||||
}
|
||||
return out;
|
||||
@@ -40,12 +52,13 @@ final class PlacementPolicyUtil {
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all weight-0, all
|
||||
* quarantined, all cooling off, all at capacity, all unreachable, or a mix. Each candidate is
|
||||
* counted into exactly one bucket (weight-excluded first, then quarantined, then cooling off)
|
||||
* so a candidate excluded for more than one reason is never double-counted — a candidate that is
|
||||
* both quarantined (CB-578 stage B, backend exhausted) and cooling off (fleetd #201 Unit 5,
|
||||
* repeated backend errors) counts only as quarantined, matching {@code CompositePeerLauncher}'s
|
||||
* explicit-spawn ordering: exhaustion quarantine takes priority when both are active.
|
||||
* quarantined, all cooling off, all model-off, all at capacity, all unreachable, or a mix. Each
|
||||
* candidate is counted into exactly one bucket (weight-excluded first, then quarantined, then
|
||||
* cooling off, then model-off) so a candidate excluded for more than one reason is never
|
||||
* double-counted — a candidate that is both quarantined (CB-578 stage B, backend exhausted) and
|
||||
* cooling off (fleetd #201 Unit 5, repeated backend errors) counts only as quarantined, matching
|
||||
* {@code CompositePeerLauncher}'s explicit-spawn ordering: exhaustion quarantine takes priority
|
||||
* when more than one applies.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int weightExcluded = 0;
|
||||
@@ -53,17 +66,19 @@ final class PlacementPolicyUtil {
|
||||
int unreachable = 0;
|
||||
int quarantined = 0;
|
||||
int coolingOff = 0;
|
||||
int modelOff = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (c.excluded()) {
|
||||
weightExcluded++;
|
||||
} else if (ctx.quarantined().contains(c.profile())) {
|
||||
quarantined++;
|
||||
} else if (ctx.coolingOff().contains(c.profile())) {
|
||||
coolingOff++;
|
||||
} else if (ctx.modelOff().contains(c.profile())) {
|
||||
modelOff++;
|
||||
} else if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
} else if (atCap(ctx, c)) {
|
||||
atCap++;
|
||||
}
|
||||
}
|
||||
@@ -83,6 +98,10 @@ final class PlacementPolicyUtil {
|
||||
return new PlacementException(
|
||||
"all worker profiles are cooling off after repeated backend errors");
|
||||
}
|
||||
if (modelOff == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles name a model the operator has turned off");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
@@ -92,8 +111,9 @@ final class PlacementPolicyUtil {
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + quarantined + " quarantined, "
|
||||
+ coolingOff + " cooling off, "
|
||||
+ modelOff + " model-off, "
|
||||
+ weightExcluded + " weight-0, "
|
||||
+ (total - atCap - unreachable - quarantined - coolingOff - weightExcluded)
|
||||
+ (total - atCap - unreachable - quarantined - coolingOff - modelOff - weightExcluded)
|
||||
+ " remaining");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -584,8 +585,65 @@ public final class SessionManager implements TurnListener {
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
// fleetd #425 rework (round 2): resolved through launcher.place(memberRole) — the same
|
||||
// candidate list, quarantine/cool-off/model-off filtering, and PlacementPolicy an unqualified
|
||||
// spawn of this role is actually judged against right now — never launcher.defaultProfile()
|
||||
// (MemberRole.DEV only, wrong for any other role) and never launcher.defaultProfileFor()
|
||||
// (the role's pool FIRST entry, blind to quarantine/cool-off/model-off: a first-round fix
|
||||
// used exactly this and regressed fleetd #429's "the fleet keeps working when a model is
|
||||
// turned off" guarantee — a quarantined or model-off pool-first profile made this throw
|
||||
// instead of routing around it, which an unqualified spawn is supposed to do). This same
|
||||
// resolved decision is reused below for repoRoot, parityOverlay, AND the spawn itself so the
|
||||
// worktree is always provisioned for the profile the member actually runs on — the two could
|
||||
// disagree before fleetd #425: this name picked repoRoot/overlay, but the spawn below passed
|
||||
// the ORIGINAL (blank) profile through to placement, which re-resolves live and can pick a
|
||||
// different profile if the pool changed between the two reads, or a genuinely different one
|
||||
// under weighted/round-robin placement.
|
||||
//
|
||||
// Round 1 of this rework fed the resolved name back into launcher.spawn(SpawnRequest) as an
|
||||
// EXPLICIT profile. That was a mistake this round corrects, and the mistake is not that the
|
||||
// two branches apply different checks — they are SUPPOSED to disagree: the routing branch a
|
||||
// blank spawn takes treats a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0
|
||||
// profile as a reason to fall through to the next candidate, while CompositePeerLauncher's
|
||||
// THROWING branch (enforceNotQuarantined/enforceNotCoolingOff/enforceMaxLoad/
|
||||
// enforceModelEnabled) treats naming that same profile explicitly as a reason to refuse
|
||||
// outright. That is correct: an operator who names a profile should get a refusal, not a
|
||||
// silent substitution onto a different backend. The mistake was turning a fall-through into
|
||||
// a refusal by accident — resolving a name via the routing side and then re-entering the
|
||||
// refusing side with it, for a placement the routing side had already approved by walking
|
||||
// past everything else.
|
||||
//
|
||||
// Before fleetd #435, this accident was reachable through maxLoad specifically: the default
|
||||
// `fixed` placement policy did not evaluate maxLoad at all for automatic selection, so an
|
||||
// at-cap pool-first profile that placement itself would have picked for a plain unqualified
|
||||
// spawn could die at enforceMaxLoad one call later, purely because this method's route to
|
||||
// the spawn passed through an explicit profile name — a failure a worktree-less unqualified
|
||||
// spawn never hit. fleetd #435 closed that gap (`fixed` now evaluates maxLoad exactly like
|
||||
// every other placement policy), so that specific failure can no longer happen — a
|
||||
// PlacementDecision this method resolves can no longer be at-cap in the first place. What
|
||||
// this round's fix still buys, now that maxLoad can no longer cause the accident: it keeps
|
||||
// the PlacementDecision from place() and hands it to launcher.spawn(SpawnRequest,
|
||||
// PlacementDecision) for an unqualified request, which spawns through the SAME routing
|
||||
// branch a blank spawn uses — no enforce* check is newly applied, and the window between the
|
||||
// placement decision and the spawn (in which the pool, a config reload, or another spawn
|
||||
// landing on the same profile could otherwise move the state) never reopens. An
|
||||
// explicitly-named profile still goes through launcher.spawn(SpawnRequest) and its throwing
|
||||
// branch, unchanged — that caller asked for one profile by name and still gets everything
|
||||
// enforceNotQuarantined/enforceNotCoolingOff/enforceMaxLoad/enforceModelEnabled decide about
|
||||
// it, refusal included.
|
||||
//
|
||||
// The one cost that remains, unchanged from round 1: an unqualified worktree-provisioned
|
||||
// spawn does not get CompositePeerLauncher's cross-candidate retry on a live
|
||||
// PeerUnreachableException raised by the backend itself at spawn time (a transport-level
|
||||
// failure placement cannot see in advance) — spawn(req, decision) commits to the one profile
|
||||
// place() already chose, the same way an explicit-profile spawn commits to its one name. That
|
||||
// trade is deliberate: a worktree provisioned for the wrong backend (the #425 hazard) is worse
|
||||
// than a spawn that fails cleanly and can be retried by the caller. Nothing else is lost:
|
||||
// maxLoad, quarantine, cool-off and model-off all behave identically whether or not a
|
||||
// worktree was requested — that agreement is the invariant this rework exists to hold.
|
||||
boolean unqualifiedProfile = profile == null || profile.isBlank();
|
||||
PlacementDecision decision = unqualifiedProfile ? launcher.place(memberRole) : new PlacementDecision(profile);
|
||||
String preResolvedProfile = decision.profile();
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
// daemon cwd → "."), never the raw args. A plain REST spawn supplies neither a requested
|
||||
// nor a caller cwd, so taking the first non-blank of those two yielded null and put
|
||||
@@ -608,7 +666,15 @@ public final class SessionManager implements TurnListener {
|
||||
// copies more files into the worktree after add() returns, so sharing the group any earlier
|
||||
// leaves those overlay files operator-owned and read-only for a different-uid member.
|
||||
worktrees.shareWithGroup(repoRoot, path);
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd, sessionName, resumeSessionId, memberRole));
|
||||
// fleetd #425: preResolvedProfile, not the original (possibly blank) profile — see the
|
||||
// comment above where it is resolved. The overlay/repoRoot above and the spawn here must
|
||||
// name the same profile. An unqualified request stays unqualified here and is honored via
|
||||
// the PlacementDecision already captured above (spawn(req, decision) — the routing branch,
|
||||
// no enforce* re-check); an explicitly-named profile still goes through the single-arg
|
||||
// spawn(req) and its throwing branch, exactly as before this rework.
|
||||
SpawnRequest spawnReq = new SpawnRequest(unqualifiedProfile ? null : preResolvedProfile,
|
||||
path, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
handle = unqualifiedProfile ? launcher.spawn(spawnReq, decision) : launcher.spawn(spawnReq);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={} branch={} path={}: {}",
|
||||
preResolvedProfile, memberRole, branch, path, e.getMessage());
|
||||
@@ -636,7 +702,7 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String resolvedProfile = resolveProfile(handle, preResolvedProfile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
// CB-619: see the no-worktree path above — bind before recording, and store the returned
|
||||
|
||||
@@ -0,0 +1,155 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #416: {@code fleet_list}'s {@code CapacitySource.configuredProfiles} must enumerate the
|
||||
* <em>startup</em> profile set, not the live, hot-reloaded one.
|
||||
*
|
||||
* <p>{@code profiles} as a whole is a {@code DEFERRED} key ({@link ConfigRef#DEFERRED_KEYS}):
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} once at construction, so a profile
|
||||
* only added to the hot-reloaded map can never actually be spawned. Before this fix, {@code
|
||||
* Fleetd.main} built {@code CapacitySource} with {@code () -> config.get().profiles().keySet()} —
|
||||
* the live map — so {@code fleet_list} would report a freshly hot-reloaded profile as available
|
||||
* ({@code free > 0}) while {@code fleet_spawn} on that same profile failed with
|
||||
* {@code unknown worker profile}. Measured on another host: adding a throwaway profile and letting
|
||||
* it hot-reload gave {@code fleet_list} -> {@code free: 3} and {@code fleet_spawn} ->
|
||||
* {@code error: unknown worker profile}.
|
||||
*
|
||||
* <p>This test needs a reload, the same reason {@link FleetdExhaustionDetectionArmedWiringTest}
|
||||
* does: at startup the two snapshots agree, so a test of only a newly started daemon would not
|
||||
* detect a live {@code config.get()} lookup for the set.
|
||||
*
|
||||
* <p><b>maxLoad must stay hot.</b> It is read off {@code config.get()} exactly like
|
||||
* {@code credentialId} ({@link ConfigRef} documents both as hot, "read live off the config
|
||||
* supplier ... exactly like weight/maxLoad"), so a reload that only changes an existing profile's
|
||||
* {@code maxLoad} — no add/remove — must still change what {@code fleet_list} reports without a
|
||||
* restart. A fix that freezes the whole {@code CapacitySource} against {@code cfg} (rather than
|
||||
* only its {@code configuredProfiles} set) would trade this bug for its mirror image and is pinned
|
||||
* wrong by {@link #reloadedMaxLoadStillChangesWhatFleetListReports}.
|
||||
*/
|
||||
class FleetdCapacitySourceWiringTest {
|
||||
|
||||
private static final String STARTUP = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 3
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_NEW_PROFILE = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 3
|
||||
ghost404:
|
||||
baseUrl: http://gx00.gw:8001
|
||||
model: ghost404
|
||||
maxLoad: 3
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_CHANGED_MAX_LOAD = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 9
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile present only in the live (hot-reloaded) config is NOT listed")
|
||||
void liveOnlyProfileIsNotListed(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
Files.writeString(file, WITH_NEW_PROFILE);
|
||||
assertTrue(config.reload().applied());
|
||||
// The live snapshot now has the new profile — proves the reload really happened and this
|
||||
// test is not accidentally passing because nothing changed.
|
||||
assertTrue(config.get().profiles().containsKey("ghost404"));
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
|
||||
assertFalse(source.configuredProfiles().get().contains("ghost404"),
|
||||
"a profile added only to the hot-reloaded config must not be listed by fleet_list — "
|
||||
+ "HerdrPeerLauncher never learns about it until a restart, so fleet_spawn on it "
|
||||
+ "would fail with 'unknown worker profile' while fleet_list claimed it free");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile present in the startup set IS listed")
|
||||
void startupProfileIsListed(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
|
||||
// fleetd #416, both-directions requirement: a test that only ever passes an empty/absent
|
||||
// startup set (the case above) cannot tell a correct lookup from one that is permanently
|
||||
// empty (e.g. a mutation replacing the supplier with Set::of). This is the direction that
|
||||
// fails if the fix regresses to reporting nothing at all.
|
||||
assertTrue(source.configuredProfiles().get().contains("terra"),
|
||||
"a profile present in the startup snapshot must still be listed by fleet_list");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a hot maxLoad edit still changes what fleet_list reports")
|
||||
void reloadedMaxLoadStillChangesWhatFleetListReports(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
assertEquals(3, source.maxLoad().apply("terra"),
|
||||
"sanity: maxLoad reads 3 from the startup config before any reload");
|
||||
|
||||
Files.writeString(file, WITH_CHANGED_MAX_LOAD);
|
||||
assertTrue(config.reload().applied());
|
||||
|
||||
assertEquals(9, source.maxLoad().apply("terra"),
|
||||
"maxLoad must stay hot — the SAME CapacitySource instance must reflect a reloaded "
|
||||
+ "maxLoad without a restart, exactly like credentialId. Freezing the whole "
|
||||
+ "CapacitySource against the startup snapshot (rather than only its "
|
||||
+ "configuredProfiles set) would trade fleetd #416 for its mirror image.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,62 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: {@code Fleetd.modelGateCoverageLine} is the startup-log counterpart of
|
||||
* {@code exhaustedPatternCoverageLine}/{@code errorPatternCoverageLine} — see {@code
|
||||
* FleetdPatternCoverageLineTest} for the identical shape this follows — except here there is no
|
||||
* {@code UnsetMeaning} choice for a caller to get backwards: {@link
|
||||
* PeerLauncher.ModelGateState#configured()} already states, unambiguously, whether an empty
|
||||
* {@link PeerLauncher.ModelGateState#off()} means "no {@code models:} block to gate with at all"
|
||||
* or "a block armed and currently reporting zero off". This class proves {@code
|
||||
* modelGateCoverageLine} words those two states — plus the third, N off — distinctly, so a
|
||||
* mutation that made it ignore {@code configured()} either way is caught here.
|
||||
*/
|
||||
class FleetdModelGateCoverageLineTest {
|
||||
|
||||
@Test
|
||||
@DisplayName("no models: block reports not configured, distinct from armed-with-zero")
|
||||
void noModelsBlockReportsNotConfigured() {
|
||||
String line = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.notConfigured());
|
||||
assertEquals("not configured (no models: block — nothing is gated, and nothing can be)", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a models: block armed with nothing off reports armed, distinct from not configured")
|
||||
void armedWithNothingOffReportsArmed() {
|
||||
String line = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of()));
|
||||
assertEquals("armed (models: block present; 0 models currently turned off)", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a models: block with N off names the off models")
|
||||
void armedWithModelsOffNamesThem() {
|
||||
String line = Fleetd.modelGateCoverageLine(
|
||||
PeerLauncher.ModelGateState.armed(Set.of("deepseek-v4-flash", "claude-opus-9000")));
|
||||
assertEquals("armed (2 model(s) turned off: [claude-opus-9000, deepseek-v4-flash])", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the three states produce pairwise-distinct wording for the same empty-looking input")
|
||||
void theThreeStatesProduceDistinctWording() {
|
||||
String notConfigured = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.notConfigured());
|
||||
String armedZero = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of()));
|
||||
String armedOne = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of("x")));
|
||||
|
||||
// Pinned individually above; restated here so this test alone still catches a regression
|
||||
// even if one of the three tests above were ever deleted — the exact FleetdPatternCoverageLineTest
|
||||
// pattern, adapted from "two keys" to "three states of one gate".
|
||||
assertNotEquals(notConfigured, armedZero,
|
||||
"collapsing 'no models: block' into 'armed, zero off' is fleetd #422 follow-up's exact defect");
|
||||
assertNotEquals(armedZero, armedOne);
|
||||
assertNotEquals(notConfigured, armedOne);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): {@code CompletionResolverTest} proves {@code coverage()} words
|
||||
* {@code UnsetMeaning.OFF} and {@code UnsetMeaning.BUILT_IN_DEFAULT} correctly — but every one of
|
||||
* those tests supplies the meaning itself. That proves the enum's wording, never that {@code
|
||||
* Fleetd} pairs the right meaning with the right pattern key. That pairing is #415's actual
|
||||
* defect: {@code coverage()} had no way to know what unset meant for its key, so the fix moved
|
||||
* the fact to the caller — and nothing yet proved the caller states it correctly.
|
||||
*
|
||||
* <p><b>Measured or it didn't happen:</b> swapping the two {@code UnsetMeaning} arguments at
|
||||
* {@code Fleetd}'s two coverage call sites — giving {@code exhaustedPattern} the built-in-default
|
||||
* wording and {@code errorPattern} the off wording, #415's exact defect with the keys exchanged —
|
||||
* compiled with 0 errors and left all 1506 existing tests green. This class exists to turn that
|
||||
* swap red.
|
||||
*
|
||||
* <p>It calls {@link Fleetd#exhaustedPatternCoverageLine} and {@link Fleetd#errorPatternCoverageLine}
|
||||
* directly rather than reading {@code Fleetd.java} as source text (the shape {@code
|
||||
* FleetdCompletionResolverWiringTest} uses for a different wiring gap): those two methods are the
|
||||
* extracted call sites {@code main} actually invokes, following the same {@code static} factory +
|
||||
* dedicated-test pattern as {@link Fleetd#capacitySource} and {@link Fleetd#worktreeBranchLookup}.
|
||||
*/
|
||||
class FleetdPatternCoverageLineTest {
|
||||
|
||||
private static final Set<String> PROFILES = Set.of("terra", "gx10");
|
||||
|
||||
@Test
|
||||
@DisplayName("exhaustedPatternCoverageLine says off when no profile configures exhaustedPattern")
|
||||
void exhaustedPatternCoverageLineSaysOffWhenNoProfileConfiguresIt() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
Fleetd.exhaustedPatternCoverageLine(PROFILES, Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("errorPatternCoverageLine says built-in default when no profile configures errorPattern")
|
||||
void errorPatternCoverageLineSaysBuiltInDefaultWhenNoProfileConfiguresIt() {
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])",
|
||||
Fleetd.errorPatternCoverageLine(PROFILES, Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the two keys produce different wording for the identical empty-coverage input")
|
||||
void theTwoKeysProduceDifferentWordingForTheSameEmptyInput() {
|
||||
String exhaustedLine = Fleetd.exhaustedPatternCoverageLine(PROFILES, Set.of());
|
||||
String errorLine = Fleetd.errorPatternCoverageLine(PROFILES, Set.of());
|
||||
|
||||
// Pinned individually above; restated here so this test alone still catches a swap even
|
||||
// if one of the two tests above were ever deleted.
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
exhaustedLine);
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])", errorLine);
|
||||
assertNotEquals(exhaustedLine, errorLine,
|
||||
"swapping which UnsetMeaning pairs with which pattern key at Fleetd's call sites "
|
||||
+ "must be caught here — that pairing, not coverage()'s own wording in isolation, "
|
||||
+ "is fleetd #415's actual defect");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,79 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Proves {@link Fleetd#main(String[])} calls every startup report before validation aborts startup.
|
||||
* The invalid non-loopback bind makes {@link FleetConfig#validateAll()} throw before {@code main}
|
||||
* can open the herdr socket or bind a port. The fixture also triggers every report, so removing any
|
||||
* one call from {@code main} leaves its expected log line absent.
|
||||
*/
|
||||
class FleetdStartupReportTest {
|
||||
|
||||
private static Level originalLevel;
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
originalLevel = logger.getLevel();
|
||||
logger.setLevel(Level.INFO);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
logger.detachAppender(appender);
|
||||
logger.setLevel(originalLevel);
|
||||
}
|
||||
|
||||
private static boolean contains(ListAppender<ILoggingEvent> appender, String fragment) {
|
||||
return appender.list.stream()
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.anyMatch(message -> message.contains(fragment));
|
||||
}
|
||||
|
||||
@Test
|
||||
void mainReportsEveryStartupGapBeforeValidationAborts(@TempDir Path dir) throws Exception {
|
||||
Path config = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(config, """
|
||||
bind:
|
||||
host: 0.0.0.0
|
||||
port: 8765
|
||||
profiles:
|
||||
worker:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
gitTokenEnv: GITEA_TOKEN
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
assertThrows(IllegalStateException.class, () -> Fleetd.main(new String[]{config.toString()}));
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(contains(appender, "startup git host GITEA_HOST:"),
|
||||
"Fleetd.main must report the git host shape");
|
||||
assertTrue(contains(appender, "member trust model: members run as the same OS user"),
|
||||
"Fleetd.main must report the member trust model");
|
||||
assertTrue(contains(appender, "memberCredentials: absent or empty"),
|
||||
"Fleetd.main must report an absent memberCredentials policy");
|
||||
assertTrue(contains(appender, "exhaustedPattern: profile(s) [worker] have no exhaustedPattern configured"),
|
||||
"Fleetd.main must report profiles without exhaustedPattern");
|
||||
}
|
||||
}
|
||||
@@ -113,6 +113,20 @@ class AuthzTest {
|
||||
assertTrue(Authz.permits(WORKER_A, METRICS, null));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #421 — unlike READ (the case above), COORD_READ (a lead's own held peer mail) is the
|
||||
* primary's alone. A worker or an architect reading it would disclose lead-to-lead
|
||||
* coordination bodies, not the secret-free roster READ is open about.
|
||||
*/
|
||||
@Test
|
||||
void coordReadIsThePrimarysAloneNotAWidenedRead() {
|
||||
assertTrue(Authz.permits(PRIMARY, COORD_READ, null));
|
||||
assertFalse(Authz.permits(WORKER_A, COORD_READ, null),
|
||||
"a worker must not read held lead-to-lead mail");
|
||||
assertFalse(Authz.permits(ARCH_DESIGN, COORD_READ, null),
|
||||
"an architect holds READ today, but not-primary must mean not-architect here too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unauthenticatedIsDistinguishedFromMerelyForbidden() {
|
||||
// Drives the 401-vs-403 split: a missing credential is fixable by the caller, a wrong role
|
||||
|
||||
@@ -0,0 +1,360 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* fleetd #424 — revoking (or granting) an architect slot must take effect on the next spawn with
|
||||
* no restart. A session already bound to a slot keeps its <em>binding</em> (the {@code
|
||||
* terminalToSlot} occupancy) even after that slot drops out of config, but NOT the ARCHITECT
|
||||
* <em>privilege</em> the slot used to grant — that is revoked on the bound session's very next
|
||||
* request. See {@link MemberRegistry}'s class doc for the exact rule: "config governs what a
|
||||
* bound slot still grants, as well as what may be bound next."
|
||||
*
|
||||
* <p>Every test here drives a REAL {@link ConfigRef#reload()} against a {@code @TempDir} file and
|
||||
* asserts {@link ConfigRef.Outcome#applied()}, rather than comparing two frozen
|
||||
* {@code MemberRegistry} instances in memory — the defect this ticket fixes is specifically that
|
||||
* {@link MemberRegistry} used to ignore a live reload, so a test that never reloads cannot tell the
|
||||
* fixed registry from the broken one. {@code requireSlotFor} and {@code reserve} are pinned in
|
||||
* separate tests, in both directions (removed and added), so a registry that simply refuses (or
|
||||
* simply allows) everything cannot pass by accident — see {@link MemberRegistryTest} for the
|
||||
* registry's other invariants (bind/unbind cardinality, thread-safety), which are unaffected by
|
||||
* this ticket and still exercised against the frozen constructor.
|
||||
*/
|
||||
class MemberRegistryLiveTest {
|
||||
|
||||
private static String yaml(String fleetBlock) {
|
||||
return """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
opus:
|
||||
baseUrl: http://gx00.gw:8001
|
||||
model: opus
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""" + fleetBlock;
|
||||
}
|
||||
|
||||
private static final String WITH_SONNET_SLOT = """
|
||||
fleet:
|
||||
architects:
|
||||
designer:
|
||||
profile: sonnet
|
||||
""";
|
||||
|
||||
/** No architect pool at all — developers is unrelated dead data for this registry (#424 out of scope). */
|
||||
private static final String WITHOUT_ARCHITECT_SLOTS = """
|
||||
fleet:
|
||||
developers:
|
||||
dev1:
|
||||
profile: sonnet
|
||||
""";
|
||||
|
||||
/** Same slot name ({@code designer}) as {@link #WITH_SONNET_SLOT}, repointed to a different profile. */
|
||||
private static final String WITH_OPUS_SLOT = """
|
||||
fleet:
|
||||
architects:
|
||||
designer:
|
||||
profile: opus
|
||||
""";
|
||||
|
||||
/** Same profile ({@code sonnet}) as {@link #WITH_SONNET_SLOT}, but the pool key is renamed. */
|
||||
private static final String WITH_RENAMED_SLOT = """
|
||||
fleet:
|
||||
architects:
|
||||
architect-lead:
|
||||
profile: sonnet
|
||||
""";
|
||||
|
||||
private static ConfigRef refFor(Path f) {
|
||||
return new ConfigRef(f, FleetConfig.load(f));
|
||||
}
|
||||
|
||||
// ── requireSlotFor is live (criteria 1, 2, 4) ──────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void requireSlotForRefusesAProfileWhoseSlotWasRemovedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertDoesNotThrow(() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"the slot is configured before the reload");
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"revoking the slot must refuse the NEXT spawn that names it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void requireSlotForAllowsAProfileWhoseSlotWasAddedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"no architect slot is configured yet");
|
||||
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertDoesNotThrow(() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"a slot added by reload must be usable with no restart");
|
||||
}
|
||||
|
||||
// ── reserve is live too — tested separately from requireSlotFor (criteria 1, 2, 4) ────────
|
||||
|
||||
@Test
|
||||
void reserveRefusesAProfileWhoseSlotWasRemovedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation before = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertEquals("architect:designer", before.slot());
|
||||
registry.release(before); // free it back up so the reload-side reserve below starts clean
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.reserve(MemberRole.ARCHITECT, "sonnet"),
|
||||
"revoking the slot must refuse the NEXT reservation for it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reserveAllowsAProfileWhoseSlotWasAddedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.reserve(MemberRole.ARCHITECT, "sonnet"),
|
||||
"no architect slot is configured yet");
|
||||
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
MemberLifecycle.SlotReservation after = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertEquals("architect:designer", after.slot(),
|
||||
"a slot added by reload must be reservable with no restart");
|
||||
}
|
||||
|
||||
// ── profileForSlot, isSlot and nameForSlot are live too (fleetd #431) ─────────────────────────
|
||||
// #424 pinned roleForSlot against a live reload but left these three untested — proved by
|
||||
// mutating each to read a snapshot flattened once at construction: the full suite stayed green
|
||||
// for all three.
|
||||
//
|
||||
// The three differ in how much production behaviour depends on them, and the ticket first got
|
||||
// this ranking wrong. nameForSlot is wired: CallerResolver passes members::nameForSlot, next to
|
||||
// members::roleForSlot. isSlot is reached through bind, which calls it to refuse an unknown
|
||||
// slot. profileForSlot has NO caller in src/main at all — grepped both the ".profileForSlot("
|
||||
// and the "::profileForSlot" form — so there is no seam to drive its test through beyond the
|
||||
// accessor itself, and its own javadoc ("what the spawn lifecycle reads") describes a caller
|
||||
// that does not exist. These tests pin the accessors as they are; whether profileForSlot should
|
||||
// be wired or deleted is a separate question.
|
||||
|
||||
@Test
|
||||
void profileForSlotReflectsAProfileChangedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertEquals("sonnet", registry.profileForSlot("architect:designer"),
|
||||
"the slot's profile before the reload");
|
||||
|
||||
Files.writeString(f, yaml(WITH_OPUS_SLOT));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertEquals("opus", registry.profileForSlot("architect:designer"),
|
||||
"repointing the slot to a different profile must take effect with no restart");
|
||||
}
|
||||
|
||||
@Test
|
||||
void isSlotStopsReportingASlotRemovedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertTrue(registry.isSlot("architect:designer"), "the slot is configured before the reload");
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertFalse(registry.isSlot("architect:designer"),
|
||||
"removing the slot from config must make isSlot say so on the very next call, "
|
||||
+ "with no restart");
|
||||
}
|
||||
|
||||
@Test
|
||||
void isSlotStartsReportingASlotAddedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertFalse(registry.isSlot("architect:designer"), "no architect slot is configured yet");
|
||||
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertTrue(registry.isSlot("architect:designer"),
|
||||
"a slot added by reload must be visible to isSlot with no restart");
|
||||
}
|
||||
|
||||
@Test
|
||||
void nameForSlotReflectsANameChangedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertEquals("designer", registry.nameForSlot("architect:designer"),
|
||||
"the configured name before the reload");
|
||||
|
||||
Files.writeString(f, yaml(WITH_RENAMED_SLOT));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertNull(registry.nameForSlot("architect:designer"),
|
||||
"the old key no longer names a configured slot — it was renamed away by the reload");
|
||||
assertEquals("architect-lead", registry.nameForSlot("architect:architect-lead"),
|
||||
"the new name must be visible under its new qualified key with no restart");
|
||||
}
|
||||
|
||||
// ── a bound architect is demoted, but the binding itself is not touched (fleetd #424) ───────
|
||||
// The lead's corrected ruling: the PRIVILEGE a slot grants is revoked on the bound session's
|
||||
// very next request, but the terminalToSlot BINDING itself is untouched by a reload — dropping
|
||||
// it would double-book the slot key and break unbind's compare-safe contract. See the class
|
||||
// doc's binding rule.
|
||||
|
||||
/** A caller identity resolving the one canned pane (terminal {@code term_a}) in {@link FakeHerdr}. */
|
||||
private static ConnectionIdentity boundPaneIdentity() {
|
||||
return new ConnectionIdentity(new PaneLocator(new FakeHerdr()), _ -> FakeHerdr.WORKER_PID);
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectAlreadyBoundToASlotIsDemotedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation reservation = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertTrue(registry.bind(reservation, "term_a"));
|
||||
assertEquals("architect:designer", registry.slotForTerminal("term_a"));
|
||||
|
||||
// Drive the real caller path, not the roleForSlot seam directly: CallerResolver.resolve is
|
||||
// what a live request actually goes through (CallerResolver.java:220), and a resolver that
|
||||
// ignored roleForSlot entirely would still pass a test that only checked the seam.
|
||||
CallerResolver resolver = CallerResolver.withLeadsAndMembers(
|
||||
boundPaneIdentity(), false, null, Map::of, registry);
|
||||
|
||||
Principal before = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.ARCHITECT, before.role(), "sanity check: the harness binds term_a as an architect");
|
||||
assertEquals("designer", before.name());
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
Principal after = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.WORKER, after.role(),
|
||||
"removing the slot from config must demote the bound session to worker on its "
|
||||
+ "NEXT request — this is the ticket's whole point");
|
||||
assertEquals("term_a", after.terminal(), "same pane, same terminal — only the role changed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theOriginalBindingStillOccupiesTheRemovedSlotSoASecondTerminalCannotClaimIt(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation reservation = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertTrue(registry.bind(reservation, "term_a"));
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome removed = ref.reload();
|
||||
assertTrue(removed.applied(), "the reload must actually take effect: " + removed.summary());
|
||||
|
||||
// The binding survives the removal untouched.
|
||||
assertEquals("architect:designer", registry.slotForTerminal("term_a"),
|
||||
"a live binding must never be retroactively unbound by a config edit");
|
||||
assertEquals(Map.of("term_a", "architect:designer"), registry.snapshot());
|
||||
|
||||
// Bring the slot back into config. If the binding had been silently dropped by the removal
|
||||
// (rather than merely losing the privilege it grants), a second terminal could now claim
|
||||
// the "freed" key — the exact double-booking the class doc's binding rule rules out.
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef.Outcome restored = ref.reload();
|
||||
assertTrue(restored.applied(), "the reload must actually take effect: " + restored.summary());
|
||||
|
||||
assertFalse(registry.bind("architect:designer", "term_b"),
|
||||
"the slot is still occupied by term_a — a second terminal must not bind to it");
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.reserve(MemberRole.ARCHITECT, "sonnet"),
|
||||
"the slot is still occupied by term_a — a fresh reservation must not find it free");
|
||||
assertEquals("architect:designer", registry.slotForTerminal("term_a"),
|
||||
"the original binding is unchanged throughout");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unbindStillSucceedsForTheOriginalTerminalAfterItsSlotIsRemoved(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation reservation = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertTrue(registry.bind(reservation, "term_a"));
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertTrue(registry.unbind("architect:designer", "term_a"),
|
||||
"unbind must still work for a slot that config has since removed, or a session "
|
||||
+ "that outlives its slot's removal could never release it");
|
||||
assertNull(registry.slotForTerminal("term_a"));
|
||||
assertEquals(Map.of(), registry.snapshot());
|
||||
}
|
||||
}
|
||||
@@ -79,6 +79,14 @@ class ConfigRefTopLevelCoverageTest {
|
||||
* <li>{@code memberCredentials} — read live at {@code Fleetd.java:198, 205, 729}.</li>
|
||||
* <li>{@code memberLoginShell} — read live at
|
||||
* {@code HerdrPeerLauncher#configuredMemberLoginShell}.</li>
|
||||
* <li>{@code models} (fleetd #422) — membership is re-validated in full against the fresh
|
||||
* config on every {@code ConfigRef.reload()} (via {@code FleetConfig#validateAll()}), so
|
||||
* a bad edit is refused, never cached stale; the on/off half is read live by {@code
|
||||
* CompositePeerLauncher.enforceModelEnabled} and its candidate filter, and by {@code
|
||||
* fleet_profiles}/{@code GET /profiles} through {@code PeerLauncher.disabledModels()}.
|
||||
* Unlike {@code fleet} below, nothing about {@code models} is baked into a startup-built
|
||||
* object anywhere — there is no frozen half, so it belongs here whole rather than in
|
||||
* {@code SPLIT_KEYS}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>{@code fleet} used to sit here too, on the strength of most of it (role pools, charters,
|
||||
@@ -93,7 +101,7 @@ class ConfigRefTopLevelCoverageTest {
|
||||
* while this test stayed green throughout.</p>
|
||||
*/
|
||||
private static final Set<String> HOT_EXCLUDED_TOP_LEVEL_KEYS =
|
||||
Set.of("placement", "memberCredentials", "memberLoginShell");
|
||||
Set.of("placement", "memberCredentials", "memberLoginShell", "models");
|
||||
|
||||
@Test
|
||||
void everyTopLevelComponentIsAccountedForInExactlyOneClass() {
|
||||
@@ -120,7 +128,7 @@ class ConfigRefTopLevelCoverageTest {
|
||||
|
||||
// The escape hatch is pinned. Growing it requires editing this line — a visible, deliberate
|
||||
// diff, not a quiet one. See the field javadoc above for what "belongs here" actually means.
|
||||
assertEquals(Set.of("placement", "memberCredentials", "memberLoginShell"), hot,
|
||||
assertEquals(Set.of("placement", "memberCredentials", "memberLoginShell", "models"), hot,
|
||||
"HOT_EXCLUDED_TOP_LEVEL_KEYS changed. A component belongs here ONLY if it is read "
|
||||
+ "live off the config supplier, never because adding it makes this test "
|
||||
+ "pass. If you are adding one to silence this test, that is fleetd #323 "
|
||||
|
||||
@@ -2901,4 +2901,119 @@ class FleetConfigTest {
|
||||
void modelsIsAKnownTopLevelKey() {
|
||||
assertTrue(FleetConfig.KNOWN_TOP_LEVEL_KEYS.contains("models"));
|
||||
}
|
||||
|
||||
// --- fleetd #422: model on/off (spawn-time gate + hot reload) -----------------------------
|
||||
|
||||
/**
|
||||
* Criterion 3: an {@code allow:} entry written before {@code enabled:} existed — no such key in
|
||||
* the YAML at all — must behave exactly as it always did: on, and absent from {@link
|
||||
* FleetConfig.Models#offIds()}. This is the old-style fixture the ticket asks for, proven
|
||||
* through a real YAML load rather than only through the {@code ModelEntry(String)} back-compat
|
||||
* constructor.
|
||||
*/
|
||||
@Test
|
||||
void anOldStyleAllowEntryWithNoEnabledFieldStaysOn(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx10.gw:8000
|
||||
model: claude-sonnet-5
|
||||
models:
|
||||
allow:
|
||||
- model: claude-sonnet-5
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertDoesNotThrow(cfg::validateModels);
|
||||
assertTrue(cfg.models().ids().contains("claude-sonnet-5"), "membership is unaffected");
|
||||
assertTrue(cfg.models().offIds().isEmpty(), "no enabled: field ⇒ nothing is off");
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 7 (the whole point of the ticket, mirroring criterion 3's shape): a profile naming a
|
||||
* model that is turned off ({@code enabled: false}) must still be VALID config —
|
||||
* {@code validateModels()} checks membership only, never the on/off state, so it must not throw.
|
||||
* Collapsing "off" into "removed from allow:" would make this test fail, which is exactly the
|
||||
* design trap the ticket calls out.
|
||||
*/
|
||||
@Test
|
||||
void anOffModelIsStillValidConfigForValidateModels(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: false
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertDoesNotThrow(cfg::validateModels,
|
||||
"an off model must stay a valid allow-list member — only the spawn gate reads enabled");
|
||||
assertTrue(cfg.models().ids().contains("deepseek-v4-flash"));
|
||||
assertTrue(cfg.models().offIds().contains("deepseek-v4-flash"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 8: one {@code enabled: false} entry names a model, not a profile — every profile
|
||||
* naming that model is off, on purpose (the live example: {@code deepseek-v4-flash} backs both
|
||||
* {@code local} and {@code local-direct}). Proven here at the {@code Models}/config-load level;
|
||||
* {@code CompositePeerLauncherTest} proves the spawn-time consequence for both profiles.
|
||||
*/
|
||||
@Test
|
||||
void turningOffOneModelIsIndependentOfHowManyProfilesNameIt(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
local-direct:
|
||||
baseUrl: http://local.gw:8001
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: false
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertDoesNotThrow(cfg::validateModels);
|
||||
// One allow-list entry, one off id — the fan-out to both profiles happens at the reader
|
||||
// (CompositePeerLauncher), not by duplicating the entry per profile.
|
||||
assertEquals(Set.of("deepseek-v4-flash"), cfg.models().offIds());
|
||||
assertEquals("deepseek-v4-flash", cfg.profiles().get("local").model());
|
||||
assertEquals("deepseek-v4-flash", cfg.profiles().get("local-direct").model());
|
||||
}
|
||||
|
||||
/** An {@code enabled: true} entry (explicit, not just absent) also stays on — not just null. */
|
||||
@Test
|
||||
void explicitlyEnabledTrueStaysOn(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx10.gw:8000
|
||||
model: claude-sonnet-5
|
||||
models:
|
||||
allow:
|
||||
- model: claude-sonnet-5
|
||||
enabled: true
|
||||
""");
|
||||
|
||||
FleetConfig cfg = FleetConfig.load(f);
|
||||
assertTrue(cfg.models().offIds().isEmpty());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -20,6 +20,7 @@ import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
||||
@@ -884,19 +885,22 @@ class CompletionResolverTest {
|
||||
@Test
|
||||
void coverageIsOffWhenNoProfileHasAPatternConfigured() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [terra])",
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra"), Set.of()));
|
||||
CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
Set.of("terra"), Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsFullWhenEveryProfileHasAPatternConfigured() {
|
||||
assertEquals("full (all profiles configured: [gx10, terra])",
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsPartialAndNamesWhichProfilesAreConfigured() {
|
||||
assertEquals("partial (configured: [terra]; not configured: [gx10])",
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra", "gx10"), Set.of("terra")));
|
||||
CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
Set.of("terra", "gx10"), Set.of("terra")));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -910,11 +914,42 @@ class CompletionResolverTest {
|
||||
*
|
||||
* <p>Every earlier test here passed the exhaustion case only, so none of them could see it. This
|
||||
* one pins that the message names the key the caller actually meant.
|
||||
*
|
||||
* <p>fleetd#415: the expected wording changed here too. {@code errorPattern} has a built-in
|
||||
* fallback ({@link CompletionResolver#BACKEND_ERROR}), so an empty {@code configuredProfiles}
|
||||
* for it is not "off" — see {@link #coverageDistinguishesOffFromBuiltInDefaultForTheSameEmptyInput}
|
||||
* for the test built specifically to pin that distinction.
|
||||
*/
|
||||
@Test
|
||||
void coverageNamesTheConfigKeyItsCallerMeansRatherThanAlwaysSayingExhaustedPattern() {
|
||||
assertEquals("off (no profile has an errorPattern configured; profiles: [gx10, terra])",
|
||||
CompletionResolver.coverage("errorPattern", Set.of("terra", "gx10"), Set.of()));
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])",
|
||||
CompletionResolver.coverage("errorPattern", CompletionResolver.UnsetMeaning.BUILT_IN_DEFAULT,
|
||||
Set.of("terra", "gx10"), Set.of()));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#415: {@code coverage()} measures pattern coverage (how many profiles set the key), but
|
||||
* for {@code errorPattern} the empty case is not the feature-off state — a profile with no
|
||||
* configured {@code errorPattern} still runs the classification against
|
||||
* {@link CompletionResolver#BACKEND_ERROR}. For {@code exhaustedPattern} there is no fallback,
|
||||
* so empty really is off. Same shape of input (empty {@code configuredProfiles}, one profile),
|
||||
* different {@link CompletionResolver.UnsetMeaning} — the wording must differ, or this method is
|
||||
* back to conflating pattern coverage with feature state for the one key where they disagree.
|
||||
*/
|
||||
@Test
|
||||
void coverageDistinguishesOffFromBuiltInDefaultForTheSameEmptyInput() {
|
||||
String exhaustedLine = CompletionResolver.coverage("exhaustedPattern",
|
||||
CompletionResolver.UnsetMeaning.OFF, Set.of("gx10", "terra"), Set.of());
|
||||
String errorLine = CompletionResolver.coverage("errorPattern",
|
||||
CompletionResolver.UnsetMeaning.BUILT_IN_DEFAULT, Set.of("gx10", "terra"), Set.of());
|
||||
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
exhaustedLine);
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])", errorLine);
|
||||
assertNotEquals(exhaustedLine, errorLine,
|
||||
"the same empty-coverage input must not read as the same feature state for both keys");
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: target-keyed backend-error pattern + typed sink ----------------------
|
||||
|
||||
@@ -201,14 +201,34 @@ class FleetMcpAuthzTest {
|
||||
*/
|
||||
@Test
|
||||
void pollingByTargetIsADrainAndPollingByTicketIsARead() {
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.pollAction("term_b"),
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.pollAction("term_b", null),
|
||||
"poll by target removes the replies — that is a drain, not an observation");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(null),
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(null, null),
|
||||
"poll by ticket changes nothing");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(" "),
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(" ", null),
|
||||
"a blank target is an absent target");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #421: a coordId branch is a THIRD operation behind fleet_poll's one name, and it must
|
||||
* map to {@link Authz.Action#COORD_READ} — never {@link Authz.Action#READ}, even though this
|
||||
* branch also consumes nothing. READ's grant is open to every authenticated role on the premise
|
||||
* that the roster carries no secrets; a lead-to-lead body is not the roster, so folding this
|
||||
* branch into READ would let any worker read every peer lead's mail in full. coordId also takes
|
||||
* priority over target when both happen to be present — it addresses a different inbox entirely.
|
||||
*/
|
||||
@Test
|
||||
void pollingByCoordIdIsACoordReadNeverAPlainRead() {
|
||||
assertEquals(Authz.Action.COORD_READ, FleetMcp.pollAction(null, "mac-opus"),
|
||||
"reading held peer mail must not be mapped to the everyone-readable READ action");
|
||||
assertEquals(Authz.Action.COORD_READ, FleetMcp.pollAction(" ", "mac-opus"),
|
||||
"a blank target must not fall through to READ/DRAIN when coordId is present");
|
||||
assertEquals(Authz.Action.READ, FleetMcp.pollAction(null, " "),
|
||||
"a blank coordId is an absent coordId, same as target/ticket");
|
||||
assertEquals(Authz.Action.COORD_READ, FleetMcp.pollAction("term_b", "mac-opus"),
|
||||
"coordId takes priority over target — this is a different inbox, not a drain");
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyRegisteredToolHasItsHandlerActionPinned() {
|
||||
Set<String> registered = toolsTheServerRegisters();
|
||||
@@ -230,6 +250,8 @@ class FleetMcpAuthzTest {
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_whoami", Map.of()));
|
||||
assertEquals(Authz.Action.READ, FleetMcp.toolAction("fleet_poll", Map.of("ticket", "task")));
|
||||
assertEquals(Authz.Action.DRAIN, FleetMcp.toolAction("fleet_poll", Map.of("target", "term_b")));
|
||||
assertEquals(Authz.Action.COORD_READ,
|
||||
FleetMcp.toolAction("fleet_poll", Map.of("coordId", "mac-opus")));
|
||||
}
|
||||
|
||||
private static Set<String> toolsTheServerRegisters() {
|
||||
@@ -249,11 +271,11 @@ class FleetMcpAuthzTest {
|
||||
void aWorkerMayNotDrainAnotherSessionsInboxByPolling() {
|
||||
FleetMcp m = mcp(true);
|
||||
|
||||
assertNotNull(m.denyFor(WORKER_A, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
assertNotNull(m.denyFor(WORKER_A, FleetMcp.pollAction("term_b", null), "term_b"),
|
||||
"a worker draining a peer's inbox would destroy replies queued for the primary");
|
||||
assertNotNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
assertNotNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction("term_b", null), "term_b"),
|
||||
"an architect has no lifecycle rights either — same gate as fleet_ack");
|
||||
assertNull(m.denyFor(PRIMARY, FleetMcp.pollAction("term_b"), "term_b"),
|
||||
assertNull(m.denyFor(PRIMARY, FleetMcp.pollAction("term_b", null), "term_b"),
|
||||
"collecting a held reply is the primary's job");
|
||||
}
|
||||
|
||||
@@ -265,11 +287,33 @@ class FleetMcpAuthzTest {
|
||||
void pollingAnOwnTicketStaysOpenToWorkersAndArchitects() {
|
||||
FleetMcp m = mcp(true);
|
||||
|
||||
assertNull(m.denyFor(WORKER_A, FleetMcp.pollAction(null), null));
|
||||
assertNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction(null), null),
|
||||
assertNull(m.denyFor(WORKER_A, FleetMcp.pollAction(null, null), null));
|
||||
assertNull(m.denyFor(ARCH_DESIGN, FleetMcp.pollAction(null, null), null),
|
||||
"an architect delegates with wait:false, so it must be able to poll its ticket");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #421 acceptance: the whole ticket, pinned at the policy-table layer. A worker must be
|
||||
* refused the coordId branch, the primary must be allowed, and (CB-548) an architect — which
|
||||
* holds READ today — must be refused too, because "not primary" means not-architect here.
|
||||
*/
|
||||
@Test
|
||||
void aWorkerAndAnArchitectMayNotReadHeldPeerMailOnlyThePrimaryMay() {
|
||||
FleetMcp m = mcp(true);
|
||||
Authz.Action coordRead = FleetMcp.pollAction(null, "mac-opus");
|
||||
|
||||
McpSchema.CallToolResult workerDenied = m.denyFor(WORKER_A, coordRead, null);
|
||||
assertNotNull(workerDenied, "a worker must not read held lead-to-lead mail");
|
||||
assertTrue(workerDenied.isError());
|
||||
|
||||
McpSchema.CallToolResult archDenied = m.denyFor(ARCH_DESIGN, coordRead, null);
|
||||
assertNotNull(archDenied, "an architect holds READ today, but not-primary must mean "
|
||||
+ "not-architect here too");
|
||||
assertTrue(archDenied.isError());
|
||||
|
||||
assertNull(m.denyFor(PRIMARY, coordRead, null), "reading its own held mail is the primary's job");
|
||||
}
|
||||
|
||||
// --- identity reconstruction from the transport context ------------------------------------
|
||||
|
||||
@Test
|
||||
|
||||
@@ -352,7 +352,7 @@ class FleetMcpTest {
|
||||
fail("a lead fleet_reply must not publish to the worker inbox");
|
||||
}
|
||||
@Override public List<InboxMessage> peek(String target) { return List.of(); }
|
||||
@Override public void ack(String target, String msgId) { }
|
||||
@Override public boolean ack(String target, String msgId) { return false; }
|
||||
};
|
||||
MessageService leadMessages = new MessageService(agents, new Injector(agents), new Rendezvous(),
|
||||
inboxThatRejectsPublishes);
|
||||
@@ -680,7 +680,11 @@ class FleetMcpTest {
|
||||
void listReportsHeldMessagesWithATruncatedPreviewNeverTheFullBody() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
String longContent = "x".repeat(200);
|
||||
// A homogeneous "x".repeat(200) body would not pin the exact 80-char boundary: a preview
|
||||
// widened by one (81 chars) still CONTAINS "x".repeat(80) + "…" as a substring one position
|
||||
// later, because every character is 'x'. Put a sentinel ("Y") exactly at index 80 — the
|
||||
// first character a widened cap would leak — so any preview past 80 chars is caught.
|
||||
String longContent = "x".repeat(80) + "Y" + "z".repeat(119);
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.hold(new LeadMessage("m1", "fleet01-lead", "mac-opus", longContent));
|
||||
|
||||
@@ -694,9 +698,113 @@ class FleetMcpTest {
|
||||
assertTrue(out.contains("\"msgId\":\"m1\""), out);
|
||||
assertTrue(out.contains("\"from\":\"fleet01-lead\""), out);
|
||||
assertFalse(out.contains(longContent), "fleet_list must never dump a held message's full body: " + out);
|
||||
assertFalse(out.contains("Y"),
|
||||
"the sentinel at index 80 must never appear — a preview past 80 chars leaked it: " + out);
|
||||
assertTrue(out.contains("x".repeat(80) + "…"), "expected an 80-char preview with an ellipsis: " + out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #421: {@code mailbox.pending} counts only broker-ready messages, so a blocked lead's
|
||||
* normal, healthy state is {@code "pending": 0} next to a non-empty {@code held[]} — which
|
||||
* invites the false reading that held mail is in-memory-only. {@code heldCount}/{@code
|
||||
* heldDurable} put an honest second number and fact beside {@code pending} instead of leaving it
|
||||
* as the only one.
|
||||
*/
|
||||
@Test
|
||||
void listReportsAnHonestHeldCountAndDurabilityNotJustPendingZero() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.withMailbox("mac-opus", LeadChannel.MailboxState.exists("mac-opus", 0, 1))
|
||||
.hold(new LeadMessage("m1", "fleet01-lead", "mac-opus", "one"))
|
||||
.hold(new LeadMessage("m2", "fleet01-lead", "mac-opus", "two"))
|
||||
.hold(new LeadMessage("m3", "fleet01-lead", "mac-opus", "three"));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, null,
|
||||
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), "",
|
||||
new FleetMcp.CoordinationSource(channel, List.of()));
|
||||
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"pending\":0"), out);
|
||||
assertTrue(out.contains("\"heldCount\":3"),
|
||||
"the honest count beside pending: 0 — three messages really are held: " + out);
|
||||
assertTrue(out.contains("\"heldDurable\":true"),
|
||||
"must state the durability fact, not leave pending as the only number next to held[]: " + out);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #440: {@code heldDurable} must be a derived fact, not a literal — so it can report
|
||||
* {@code false} when the channel behind it says held mail is not durable (a non-durable queue,
|
||||
* or a consumer running with {@code autoAck=true}). A test that only ever asserts {@code true}
|
||||
* repeats the defect this ticket fixes.
|
||||
*/
|
||||
@Test
|
||||
void listReportsHeldDurableFalseWhenTheChannelSaysMailIsNotDurable() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.withMailbox("mac-opus", LeadChannel.MailboxState.exists("mac-opus", 0, 1))
|
||||
.withHeldDurable(false)
|
||||
.hold(new LeadMessage("m1", "fleet01-lead", "mac-opus", "one"));
|
||||
|
||||
McpSchema.CallToolResult res = FleetMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, null,
|
||||
FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none(), Map.of(), "",
|
||||
new FleetMcp.CoordinationSource(channel, List.of()));
|
||||
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"heldDurable\":false"),
|
||||
"heldDurable must follow the channel, not a hardcoded true: " + out);
|
||||
}
|
||||
|
||||
// ── fleetd #421: a lead reads (never consumes) its own held peer mail ──────────────────────
|
||||
|
||||
@Test
|
||||
void pollWithCoordIdReturnsTheFullBodyWithoutAckingAndLeavesItHeld() {
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.hold(new LeadMessage("m1", "fleet01-lead", "mac-opus", "x".repeat(200)));
|
||||
|
||||
McpSchema.CallToolResult first = FleetMcp.poll(messages, channel, null, null, "mac-opus");
|
||||
assertNotEquals(Boolean.TRUE, first.isError(), textOf(first));
|
||||
String out1 = textOf(first);
|
||||
assertTrue(out1.contains("\"msgId\":\"m1\""), out1);
|
||||
assertTrue(out1.contains("\"from\":\"fleet01-lead\""), out1);
|
||||
assertTrue(out1.contains("\"content\":\"" + "x".repeat(200) + "\""),
|
||||
"the coordId route must return the FULL body, unlike fleet_list's preview: " + out1);
|
||||
|
||||
// Read again: identical bodies, and nothing was acked — peek() still holds it.
|
||||
McpSchema.CallToolResult second = FleetMcp.poll(messages, channel, null, null, "mac-opus");
|
||||
assertEquals(out1, textOf(second), "peek is non-destructive — reading twice must return the same bodies");
|
||||
assertTrue(channel.acked().isEmpty(), "a read must never ack — that is the whole point of the ticket");
|
||||
assertEquals(1, channel.peek().size(), "the message must still be held after being read");
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollWithCoordIdRefusesAPeersCoordIdInsteadOfReturningTheWrongMailOrNothing() {
|
||||
FakeLeadChannel channel = new FakeLeadChannel("mac-opus")
|
||||
.hold(new LeadMessage("m1", "fleet01-lead", "mac-opus", "secret coordination body"));
|
||||
|
||||
// The ORIGINAL fleetd #421 confusion: passing a PEER's id where a self-address was meant.
|
||||
McpSchema.CallToolResult res = FleetMcp.poll(messages, channel, null, null, "fleet01-lead");
|
||||
|
||||
assertEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertFalse(out.contains("secret coordination body"),
|
||||
"a refused read must never leak the body it refused to return: " + out);
|
||||
assertTrue(out.contains("mac-opus"), "the error must name this daemon's own coordId: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollWithCoordIdErrorsHonestlyWhenLeadCoordinationIsNotConfigured() {
|
||||
McpSchema.CallToolResult res = FleetMcp.poll(messages, null, null, null, "mac-opus");
|
||||
|
||||
assertEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).toLowerCase().contains("coordinat"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsEachDeclaredPeersLiveReachability() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
@@ -752,6 +860,9 @@ class FleetMcpTest {
|
||||
@Override
|
||||
public String selfCoordId() { return "mac-opus"; }
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() { return true; }
|
||||
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
started.countDown();
|
||||
@@ -1275,6 +1386,10 @@ class FleetMcpTest {
|
||||
|
||||
@Test
|
||||
void bridgeAckReturnsConfirmationForValidArgs() {
|
||||
// fleet_ack only reports success for a msgId actually queued in the target's inbox
|
||||
// (fleetd #437) — publish one via the inbox directly rather than asserting on a
|
||||
// fabricated id nothing ever queued.
|
||||
inbox.publish("term_a", "msg-1", "queued reply");
|
||||
McpSchema.CallToolResult res = FleetMcp.ack(messages, "term_a", "msg-1");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("msg-1"), "response should mention the msgId");
|
||||
@@ -1287,6 +1402,26 @@ class FleetMcpTest {
|
||||
assertTrue(FleetMcp.ack(messages, " ", "msg-1").isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckOfAnIdInNoInboxIsAnError() {
|
||||
// fleetd #437: fleet_ack used to say "acknowledged <msgId>" for a message it never
|
||||
// touched, because nothing in the chain reported hit vs. miss. "never-queued" is in no
|
||||
// inbox at all, so this must error rather than claim success.
|
||||
McpSchema.CallToolResult res = FleetMcp.ack(messages, "term_a", "never-queued");
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("never-queued"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckOfACoordIdTargetIsAnErrorNamingFleetPoll() {
|
||||
// A coord-id names a peer lead's held mailbox (LeadChannel/LeadMailbox), never a
|
||||
// worker's ReplyInbox — fleet_ack has no route to it and must say so, pointing at
|
||||
// fleet_poll{coordId} instead of reporting a false "acknowledged".
|
||||
McpSchema.CallToolResult res = FleetMcp.ack(messages, "coord-some-peer", "msg-1");
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("fleet_poll{coordId}"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckRemovesSpecificReply() {
|
||||
// Queue a reply and capture its msgId.
|
||||
@@ -1300,8 +1435,18 @@ class FleetMcpTest {
|
||||
var peeked = messages.drainReplies("term_a");
|
||||
assertEquals(1, peeked.size(), "one fresh reply in the inbox");
|
||||
|
||||
// ackReply works (no-op since published with a different UUID, but callable).
|
||||
assertDoesNotThrow(() -> messages.ackReply("term_a", msgId));
|
||||
// fleetd #437: msgId was already drained above (a fresh UUID each publish), so it is no
|
||||
// longer in the inbox — ackReply must now report that miss instead of pretending to ack.
|
||||
assertFalse(messages.ackReply("term_a", msgId));
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckRemovingARealQueuedReplyReportsSuccessAndRemovesIt() {
|
||||
// The worker path must not change behaviour: acking a reply that IS still in the inbox
|
||||
// still succeeds and still removes it (fleetd #437).
|
||||
inbox.publish("term_a", "real-1", "still queued");
|
||||
assertTrue(messages.ackReply("term_a", "real-1"), "ack of a real queued reply must report true");
|
||||
assertTrue(inbox.peek("term_a").isEmpty(), "the acked reply must be gone from the inbox");
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
@@ -0,0 +1,114 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #425: {@code fleet_profiles}' {@code "default"} field was captured once at boot
|
||||
* ({@code cfg.effectiveDefaultProfile()}, frozen into {@code CompositePeerLauncher.defaultProfile}
|
||||
* at construction) while an unqualified spawn resolves the same underlying key
|
||||
* ({@code fleet.developers}' first entry) live, on every call. Reordering {@code fleet.developers}
|
||||
* and reloading changed where a spawn landed without ever changing what {@code fleet_profiles}
|
||||
* reported — a lead following {@code CLAUDE.md}'s "check {@code fleet_profiles} once per session"
|
||||
* instruction was told a stale answer.
|
||||
*
|
||||
* <p>This test drives the exact caller {@code fleet_profiles} uses —
|
||||
* {@link FleetMcp#profilesView(PeerLauncher, FleetMcp.QuarantineSource, FleetMcp.OutageSource)} —
|
||||
* against a real, reloadable {@link ConfigRef}, so it fails if the reporting path is ever recoupled
|
||||
* to a frozen value instead of {@link CompositePeerLauncher#defaultProfile()}'s live answer.
|
||||
*/
|
||||
class FleetProfilesLiveDefaultTest {
|
||||
|
||||
/** A minimal fleetd.yaml whose dev pool is {@code profilesInOrder}, in that definition order. */
|
||||
private static String yamlWithDevPool(String... profilesInOrder) {
|
||||
StringBuilder devPool = new StringBuilder();
|
||||
for (int i = 0; i < profilesInOrder.length; i++) {
|
||||
devPool.append(" slot").append(i).append(":\n profile: ")
|
||||
.append(profilesInOrder[i]).append('\n');
|
||||
}
|
||||
return """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
opus:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: opus-coder
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet-coder
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
fleet:
|
||||
developers:
|
||||
""" + devPool;
|
||||
}
|
||||
|
||||
@Test
|
||||
void fleetProfilesDefaultTracksALiveDevPoolReorderAfterReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yamlWithDevPool("opus", "sonnet"));
|
||||
ConfigRef ref = new ConfigRef(f, FleetConfig.load(f));
|
||||
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"opus", new FleetConfig.Profile("opus", "http://gx00.gw:8000", "opus-coder", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, null, null),
|
||||
"sonnet", new FleetConfig.Profile("sonnet", "http://gx00.gw:8000", "sonnet-coder", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, null, null));
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "opus", _ -> null);
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
List.of(adapter), "opus", ref, _ -> 0, BackendQuarantine.none());
|
||||
|
||||
assertReportedDefaultMatchesAnUnqualifiedSpawn(workers, "opus");
|
||||
|
||||
Files.writeString(f, yamlWithDevPool("sonnet", "opus"));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), () -> "reload should apply cleanly: " + out.error());
|
||||
|
||||
assertReportedDefaultMatchesAnUnqualifiedSpawn(workers, "sonnet");
|
||||
}
|
||||
|
||||
/**
|
||||
* Asserts BOTH that {@code fleet_profiles}' {@code "default"} equals {@code expected}, AND that
|
||||
* it equals what a real unqualified {@code MemberRole#DEV} spawn actually gets placed on right
|
||||
* now — the two facts fleetd #425 found disagreeing.
|
||||
*/
|
||||
private static void assertReportedDefaultMatchesAnUnqualifiedSpawn(PeerLauncher workers, String expected) {
|
||||
Map<String, Object> view = FleetMcp.profilesView(
|
||||
workers, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
||||
assertEquals(expected, view.get("default"),
|
||||
"fleet_profiles' \"default\" must be the live dev-pool answer, not a boot-time snapshot");
|
||||
|
||||
String placed = workers.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile();
|
||||
assertEquals(expected, placed,
|
||||
"sanity: the profile an unqualified dev spawn actually lands on");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: {@code fleet_profiles}/{@code GET /profiles} — the reporting surface a
|
||||
* lead actually reads — must let it tell apart the three states {@link
|
||||
* dev.ltms.fleet.peer.PeerLauncher#disabledModels()} alone collapses into one empty set: no
|
||||
* {@code models:} block at all, a block armed with nothing currently off, and a block with N
|
||||
* models off. See {@link dev.ltms.fleet.peer.PeerLauncher.ModelGateState}'s javadoc for why a
|
||||
* bare {@code disabledModels()} read cannot make this distinction, and {@link
|
||||
* FleetMcp#profilesView} for where {@code modelGateArmed} is added alongside the existing {@code
|
||||
* modelsOff} key.
|
||||
*
|
||||
* <p>Every assertion here goes through {@link FleetMcp#profilesView}, never {@code
|
||||
* PeerLauncher.modelGateState()} directly — {@code CompositePeerLauncherTest} already proves the
|
||||
* accessor itself; this class proves the surface a lead reads (fleet_profiles / GET /profiles)
|
||||
* renders what that accessor reports.
|
||||
*/
|
||||
class FleetProfilesModelGateStateTest {
|
||||
|
||||
private static FleetConfig.Profile profile(String name, String model) {
|
||||
return new FleetConfig.Profile(name, "http://gx00.gw:8000", model, null, "FLEETD_WORKER_TOKEN",
|
||||
null, "tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
}
|
||||
|
||||
private static FleetMcp.QuarantineSource noQuarantine() {
|
||||
return new FleetMcp.QuarantineSource(_ -> null, BackendQuarantine.none(), _ -> false);
|
||||
}
|
||||
|
||||
private static HerdrPeerLauncher claudeAdapter(FakeHerdr h, Map<String, FleetConfig.Profile> profiles) {
|
||||
return new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "local", _ -> "tok");
|
||||
}
|
||||
|
||||
/**
|
||||
* State 1: no {@code models:} block at all — a plain {@code ClaudeCodeLauncher} (no {@code
|
||||
* models:} supplier exists for it to read) has nothing to gate against, matching the fleet01
|
||||
* host measured for this ticket: {@code grep -c '^models:' fleetd.yaml} returns 0 there.
|
||||
*/
|
||||
@Test
|
||||
void noModelsBlockReportsGateNotArmed() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", profile("local", "deepseek-v4-flash"));
|
||||
PeerLauncher workers = claudeAdapter(h, profiles);
|
||||
|
||||
Map<String, Object> view = FleetMcp.profilesView(workers, noQuarantine(), FleetMcp.OutageSource.none());
|
||||
|
||||
assertEquals(Boolean.FALSE, view.get("modelGateArmed"),
|
||||
"no models: block to read from — nothing is gated, and nothing can be");
|
||||
assertFalse(view.containsKey("modelsOff"), "nothing configured, so no off set to report either");
|
||||
}
|
||||
|
||||
/** State 2: a {@code models:} block is present, but nothing in it is currently turned off. */
|
||||
@Test
|
||||
void modelsBlockWithNothingOffReportsGateArmedAndZeroOff() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", profile("local", "deepseek-v4-flash"));
|
||||
FleetConfig.Models models = new FleetConfig.Models(
|
||||
List.of(new FleetConfig.Models.ModelEntry("deepseek-v4-flash", true)));
|
||||
PeerLauncher workers = new CompositePeerLauncher(List.of(claudeAdapter(h, profiles)), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(),
|
||||
new BackendOutagePolicy(System::nanoTime), () -> models);
|
||||
|
||||
Map<String, Object> view = FleetMcp.profilesView(workers, noQuarantine(), FleetMcp.OutageSource.none());
|
||||
|
||||
assertEquals(Boolean.TRUE, view.get("modelGateArmed"),
|
||||
"a models: block is present, so the gate is armed even though nothing is off yet");
|
||||
assertFalse(view.containsKey("modelsOff"),
|
||||
"nothing is off, so the key stays absent — an empty list here would be indistinguishable "
|
||||
+ "from today's modelsOff omission, exactly the ambiguity modelGateArmed exists to remove");
|
||||
}
|
||||
|
||||
/** State 3: a {@code models:} block is present with one model currently turned off. */
|
||||
@Test
|
||||
void modelsBlockWithModelsOffReportsGateArmedAndTheOffSet() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", profile("local", "deepseek-v4-flash"));
|
||||
FleetConfig.Models models = new FleetConfig.Models(
|
||||
List.of(new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
PeerLauncher workers = new CompositePeerLauncher(List.of(claudeAdapter(h, profiles)), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(),
|
||||
new BackendOutagePolicy(System::nanoTime), () -> models);
|
||||
|
||||
Map<String, Object> view = FleetMcp.profilesView(workers, noQuarantine(), FleetMcp.OutageSource.none());
|
||||
|
||||
assertEquals(Boolean.TRUE, view.get("modelGateArmed"));
|
||||
assertEquals(List.of("deepseek-v4-flash"), view.get("modelsOff"));
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package dev.ltms.fleet.member;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
@@ -22,8 +23,11 @@ import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
@@ -141,6 +145,13 @@ class CompositePeerLauncherTest {
|
||||
null, null, credentialId, null);
|
||||
}
|
||||
|
||||
/** Like {@link #stubWorker(String)}, but with an explicit {@code model:} for fleetd #422 tests. */
|
||||
private static FleetConfig.Profile stubWorkerModel(String profile, String model) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", model,
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* An <em>order-preserving</em> profile map. Never {@code Map.of} here: its iteration order is
|
||||
* salted per JVM run, and the weighted policy breaks an exact-weight tie on candidate order —
|
||||
@@ -525,6 +536,31 @@ class CompositePeerLauncherTest {
|
||||
assertEquals("claude", h.profile(), "the returned handle carries the resolved default profile");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #435: {@code FixedPlacementPolicy} — the default placement policy every config uses
|
||||
* unless {@code placement:} is set — never consulted {@code maxLoad}, so an unqualified spawn
|
||||
* (a blank profile, the normal delegation path) landed on a capped default anyway. Measured on
|
||||
* 7667727: a single dev profile at {@code maxLoad: 1} with 1 live, under {@code fixed()},
|
||||
* returned "SPAWNED on profile=a". This test goes through {@code CompositePeerLauncher.spawn}
|
||||
* with a blank profile, not the policy in isolation, so it proves the caller actually reaches
|
||||
* the fixed default's new cap check rather than only the {@code select} method.
|
||||
*/
|
||||
@Test
|
||||
void fixedPolicyGatesDefaultProfileAtMaxLoadOnUnqualifiedSpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(),
|
||||
"the default profile a is at maxLoad, so fixed placement must fall through to b");
|
||||
assertEquals(0, adapter.spawnCount("a"), "a is never spawned — it is already at cap");
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyGatesProfileAtMaxLoad() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -908,6 +944,89 @@ class CompositePeerLauncherTest {
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
// ── fleetd #425: defaultProfile()/defaultProfileFor() must track a live reload ─────────────
|
||||
|
||||
/** A minimal fleetd.yaml whose dev pool is {@code profilesInOrder}, in that definition order. */
|
||||
private static String yamlWithDevPool(String... profilesInOrder) {
|
||||
StringBuilder devPool = new StringBuilder();
|
||||
for (int i = 0; i < profilesInOrder.length; i++) {
|
||||
devPool.append(" slot").append(i).append(":\n profile: ")
|
||||
.append(profilesInOrder[i]).append('\n');
|
||||
}
|
||||
return """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
opus:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: opus-coder
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet-coder
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
fleet:
|
||||
developers:
|
||||
""" + devPool;
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 1 (fleetd #425): reorder {@code fleet.developers}, reload, and assert the reported
|
||||
* default ({@link CompositePeerLauncher#defaultProfile()} — what {@code fleet_profiles}' {@code
|
||||
* "default"} is built from, see {@code FleetMcp.profilesView}) matches what an unqualified
|
||||
* {@code MemberRole#DEV} spawn is actually placed on, both before and after the reorder. Asserts
|
||||
* {@code applied()} so the test proves the reload actually took, not that nothing changed.
|
||||
*/
|
||||
@Test
|
||||
void defaultProfileTracksALiveDevPoolReorderAfterReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yamlWithDevPool("opus", "sonnet"));
|
||||
ConfigRef ref = new ConfigRef(f, FleetConfig.load(f));
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, threeProfiles(), "opus", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "opus", ref, _ -> 0, BackendQuarantine.none());
|
||||
|
||||
assertEquals("opus", composite.defaultProfile(),
|
||||
"reported default starts at the dev pool's first entry");
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"an unqualified dev spawn must land on the same profile that was just reported");
|
||||
|
||||
Files.writeString(f, yamlWithDevPool("sonnet", "opus"));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), () -> "reload should apply cleanly: " + out.error());
|
||||
|
||||
assertEquals("sonnet", composite.defaultProfile(),
|
||||
"the reported default must follow the reorder with no daemon restart");
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"and it must still be exactly what an unqualified spawn actually gets");
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 2 — the mirror, and the load-bearing half (fleetd #425): with NOTHING configured (no
|
||||
* profiles at all, hence an empty pool for every role), the frozen {@code defaultProfile} field
|
||||
* is still what gets reported. A fix that always returns {@code poolFor(role).getFirst()} with no
|
||||
* empty-pool fallback throws or returns the wrong thing here even though criterion 1 above still
|
||||
* passes — this is the test that catches it.
|
||||
*/
|
||||
@Test
|
||||
void defaultProfileFallsBackToTheFrozenFieldWhenNothingIsConfiguredAtAll() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, Map.of(), "opus", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "opus", Map.of(), PlacementPolicies.fixed(), _ -> 0);
|
||||
|
||||
assertEquals("opus", composite.defaultProfile(),
|
||||
"with no profiles configured at all, the frozen field is the only answer available");
|
||||
assertEquals("opus", composite.defaultProfileFor(MemberRole.DEV));
|
||||
}
|
||||
|
||||
// ── CB-578 stage B: a BACKEND_EXHAUSTED classification quarantines the credential ──────────
|
||||
|
||||
@Test
|
||||
@@ -974,6 +1093,36 @@ class CompositePeerLauncherTest {
|
||||
assertEquals(0, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #425 rework, acceptance 1: {@link CompositePeerLauncher#routedProfileFor} must apply
|
||||
* the SAME quarantine filtering {@link CompositePeerLauncher#spawn} does, under {@code fixed()}
|
||||
* — the DEFAULT placement policy, deliberately not {@code weighted()} (which the regressed
|
||||
* round's own tests all used, and which never exercises {@code FixedPlacementPolicy}'s own
|
||||
* inline filter). This is the exact defect: the previous round's {@code defaultProfileFor}
|
||||
* blindly returns the pool's first entry ("sol", quarantined here) with no awareness of
|
||||
* quarantine at all, which is what turned a routine unqualified spawn into a hard throw once
|
||||
* {@code acquireWithWorktree} pre-resolved through it.
|
||||
*/
|
||||
@Test
|
||||
void routedProfileForSkipsAQuarantinedPoolFirstProfileUnderFixedPolicy() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
assertEquals("b", composite.routedProfileFor(MemberRole.DEV),
|
||||
"sol (the pool's first entry) is quarantined, so the routed answer must be b");
|
||||
assertEquals("sol", composite.defaultProfileFor(MemberRole.DEV),
|
||||
"sanity: defaultProfileFor stays blind to quarantine — that's the gap routedProfileFor closes");
|
||||
assertEquals(0, adapter.spawnCount("sol"), "routedProfileFor never spawns anything");
|
||||
assertEquals(0, adapter.spawnCount("b"), "routedProfileFor never spawns anything");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aQuarantineLiftsOnTheInjectedClockAndTheProfileBecomesSpawnableAgain() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
@@ -1189,4 +1338,415 @@ class CompositePeerLauncherTest {
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("claude", null, null)));
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("gemini", null, null)));
|
||||
}
|
||||
|
||||
// ── fleetd #422: the model on/off gate — a FOURTH, independent reason to refuse a spawn ──────
|
||||
// ── (an operator decision, never a backend-reported outage) — never merged with quarantine ────
|
||||
// ── or cool-off above ─────────────────────────────────────────────────────────────────────────
|
||||
|
||||
private static final BackendOutagePolicy NO_OUTAGE = new BackendOutagePolicy(() -> 0L);
|
||||
|
||||
/** Criterion 1: an explicit spawn onto a profile whose model is off is refused. */
|
||||
@Test
|
||||
void explicitSpawnOntoAnOffModelProfileIsRefused() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"sonnet", stubWorkerModel("sonnet", "claude-sonnet-5"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("local", null, null)));
|
||||
assertTrue(e.getMessage().contains("local"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("deepseek-v4-flash"),
|
||||
"message names the model: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("operator") && e.getMessage().contains("turned off"),
|
||||
"wording says the OPERATOR turned it off: " + e.getMessage());
|
||||
// Distinct from quarantine/cool-off wording (criterion 1's explicit requirement).
|
||||
assertFalse(e.getMessage().contains("quarantined"), "must not read like quarantine: " + e.getMessage());
|
||||
assertFalse(e.getMessage().contains("cooling off"), "must not read like cool-off: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("local"), "the off-model profile is never delegated to");
|
||||
}
|
||||
|
||||
/** A profile whose model is NOT off spawns normally even while another model is off. */
|
||||
@Test
|
||||
void explicitSpawnOntoAnEnabledModelProfileSucceedsWhileAnotherModelIsOff() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"sonnet", stubWorkerModel("sonnet", "claude-sonnet-5"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("sonnet", null, null)));
|
||||
assertEquals(1, adapter.spawnCount("sonnet"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 8: one {@code enabled: false} entry disables EVERY profile naming that model — the
|
||||
* live example, {@code deepseek-v4-flash} backing both {@code local} and {@code local-direct}.
|
||||
*/
|
||||
@Test
|
||||
void turningOffOneModelRefusesEveryProfileThatNamesIt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"local-direct", stubWorkerModel("local-direct", "deepseek-v4-flash"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("local", null, null)));
|
||||
assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("local-direct", null, null)),
|
||||
"local-direct shares local's model, so it must be refused too");
|
||||
assertEquals(0, adapter.spawnCount("local"));
|
||||
assertEquals(0, adapter.spawnCount("local-direct"));
|
||||
}
|
||||
|
||||
/** Criterion 3: an old-style entry with no {@code enabled} field never refuses a spawn. */
|
||||
@Test
|
||||
void anEntryWithNoEnabledFieldNeverRefusesASpawn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", stubWorkerModel("local", "deepseek-v4-flash"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
// The back-compat single-arg ModelEntry constructor — no enabled field in the shape at all.
|
||||
FleetConfig.Models models = new FleetConfig.Models(
|
||||
List.of(new FleetConfig.Models.ModelEntry("deepseek-v4-flash")));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("local", null, null)));
|
||||
assertEquals(1, adapter.spawnCount("local"));
|
||||
}
|
||||
|
||||
/** A model absent from {@code models.allow:} entirely (nothing to gate against) is never refused. */
|
||||
@Test
|
||||
void aModelNotConfiguredInTheAllowListIsNeverGated() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", stubWorkerModel("local", "unlisted-model"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("local", null, null)));
|
||||
}
|
||||
|
||||
/** Criterion 2: an unqualified spawn skips an off-model candidate and lands on another one. */
|
||||
@Test
|
||||
void placementSkipsAnOffModelProfileAndRoutesToAnotherOne() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"sonnet", stubWorkerModel("sonnet", "claude-sonnet-5"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("sonnet", h.profile(), "local's model is off, so an unqualified spawn must land on sonnet");
|
||||
assertEquals(0, adapter.spawnCount("local"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 2's second half: when EVERY candidate's model is off, the placement exception must
|
||||
* name that as the cause — not a generic "no candidates" message.
|
||||
*/
|
||||
@Test
|
||||
void automaticPlacementNamesModelOffWhenEveryCandidateIsOffModel() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"local-direct", stubWorkerModel("local-direct", "deepseek-v4-flash"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)),
|
||||
"both candidates share the off model — nothing is available");
|
||||
assertTrue(e.getMessage().contains("model") && e.getMessage().contains("turned off"),
|
||||
"message names model-off as the cause, not a generic no-candidates message: "
|
||||
+ e.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: {@code fixed} is the DEFAULT placement policy ({@code
|
||||
* PlacementPolicies.fromName} returns it for an absent/blank name), and it built its own inline
|
||||
* candidate filter instead of calling {@code PlacementPolicyUtil.available()} — so it never
|
||||
* checked {@code modelOff()}. Mirrors {@code placementSkipsAnOffModelProfileAndRoutesToAnotherOne}
|
||||
* above with only the policy swapped, to prove the gate now fires on the path most fleets use.
|
||||
*/
|
||||
@Test
|
||||
void fixedPlacementSkipsAnOffModelProfileToo() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"sonnet", stubWorkerModel("sonnet", "claude-sonnet-5"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("sonnet", h.profile(), "local's model is off, so an unqualified spawn must land on sonnet");
|
||||
assertEquals(0, adapter.spawnCount("local"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #425 rework, acceptance 2: same shape as {@link #fixedPlacementSkipsAnOffModelProfileToo}
|
||||
* above, but through {@link CompositePeerLauncher#routedProfileFor} rather than an actual
|
||||
* {@link CompositePeerLauncher#spawn} — the exact call {@code SessionManager.acquireWithWorktree}
|
||||
* makes to pre-resolve a profile for provisioning. This is the fleetd #429 case named in the
|
||||
* ticket: an operator turns a model off, and an unqualified worktree spawn must still route
|
||||
* around it instead of throwing "names model, which the operator has turned off" — the throw
|
||||
* {@link CompositePeerLauncher#enforceModelEnabled} raises only on the EXPLICIT-profile branch,
|
||||
* which is exactly the branch the regressed round accidentally routed every worktree spawn onto.
|
||||
*/
|
||||
@Test
|
||||
void routedProfileForSkipsAModelOffPoolFirstProfileUnderFixedPolicy() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"sonnet", stubWorkerModel("sonnet", "claude-sonnet-5"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
assertEquals("sonnet", composite.routedProfileFor(MemberRole.DEV),
|
||||
"local (the pool's first entry) names an off model, so the routed answer must be sonnet");
|
||||
assertEquals("local", composite.defaultProfileFor(MemberRole.DEV),
|
||||
"sanity: defaultProfileFor stays blind to model-off — that's the gap routedProfileFor closes");
|
||||
assertEquals(0, adapter.spawnCount("local"), "routedProfileFor never spawns anything");
|
||||
assertEquals(0, adapter.spawnCount("sonnet"), "routedProfileFor never spawns anything");
|
||||
}
|
||||
|
||||
/**
|
||||
* Criterion 4: turning a model off/on is HOT — no restart — proven through a REAL
|
||||
* {@code ConfigRef.reload()}, not a hand-rolled supplier swap. Also proves {@code models} is
|
||||
* correctly reclassified: the reload's {@link ConfigRef.Outcome#applied()} is {@code true} and
|
||||
* {@code "models"} never appears in {@link ConfigRef.Outcome#deferred()}.
|
||||
*/
|
||||
@Test
|
||||
void modelOnOffIsHotReloadedThroughARealConfigRef(@TempDir Path dir) throws Exception {
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
""");
|
||||
FleetConfig initial = FleetConfig.load(yaml);
|
||||
ConfigRef configRef = new ConfigRef(yaml, initial);
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr,
|
||||
Map.of("local", stubWorker("local")), "local", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "local", configRef, _ -> 0, BackendQuarantine.none(), NO_OUTAGE);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("local", null, null)),
|
||||
"the model starts enabled");
|
||||
assertEquals(1, adapter.spawnCount("local"));
|
||||
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: false
|
||||
""");
|
||||
ConfigRef.Outcome outcome = configRef.reload();
|
||||
assertTrue(outcome.applied(), "a models.allow on/off edit must apply live, never be refused");
|
||||
assertFalse(outcome.deferred().contains("models"),
|
||||
"models is hot-excluded now — it must never be reported as a deferred key");
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("local", null, null)),
|
||||
"the very next spawn must see the reload, with no restart");
|
||||
assertTrue(e.getMessage().contains("deepseek-v4-flash"));
|
||||
assertEquals(1, adapter.spawnCount("local"), "still just the one successful spawn from before");
|
||||
|
||||
// And back on, still hot, still no restart.
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: true
|
||||
""");
|
||||
ConfigRef.Outcome reEnabled = configRef.reload();
|
||||
assertTrue(reEnabled.applied());
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("local", null, null)));
|
||||
assertEquals(2, adapter.spawnCount("local"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up, acceptance criterion 2: {@link CompositePeerLauncher#modelGateState()}
|
||||
* is LIVE — no restart — proven through a REAL {@link ConfigRef#reload()}, exactly like {@link
|
||||
* #modelOnOffIsHotReloadedThroughARealConfigRef} above proves for the on/off gate itself. This
|
||||
* single reload sequence walks through all three states the ticket asks for: no {@code models:}
|
||||
* block, a block armed with nothing off, and a block with one model off — so a reload that flips
|
||||
* between any of the three is proven live, not just the on/off edit within an already-armed block.
|
||||
*/
|
||||
@Test
|
||||
void modelGateStateIsHotReloadedThroughARealConfigRef(@TempDir Path dir) throws Exception {
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
""");
|
||||
FleetConfig initial = FleetConfig.load(yaml);
|
||||
ConfigRef configRef = new ConfigRef(yaml, initial);
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr,
|
||||
Map.of("local", stubWorker("local")), "local", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "local", configRef, _ -> 0, BackendQuarantine.none(), NO_OUTAGE);
|
||||
|
||||
PeerLauncher.ModelGateState notConfigured = composite.modelGateState();
|
||||
assertFalse(notConfigured.configured(), "no models: block in the config at all");
|
||||
assertEquals(Set.of(), notConfigured.off());
|
||||
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: false
|
||||
""");
|
||||
assertTrue(configRef.reload().applied(), "adding a models: block must apply live, no restart");
|
||||
PeerLauncher.ModelGateState armedWithOneOff = composite.modelGateState();
|
||||
assertTrue(armedWithOneOff.configured(), "a models: block now exists — the gate is armed");
|
||||
assertEquals(Set.of("deepseek-v4-flash"), armedWithOneOff.off());
|
||||
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: true
|
||||
""");
|
||||
assertTrue(configRef.reload().applied(), "flipping the entry back on must apply live too");
|
||||
PeerLauncher.ModelGateState armedWithZeroOff = composite.modelGateState();
|
||||
assertTrue(armedWithZeroOff.configured(),
|
||||
"the block is still present — armed and reporting zero, not the same as no block at all");
|
||||
assertEquals(Set.of(), armedWithZeroOff.off());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up, acceptance criterion 3: the invariant is that an absent {@code
|
||||
* models:} block stays permitted and must never be fatal. Proved, not assumed — a config
|
||||
* without one loads, validates, reports the gate as not configured, AND still spawns normally
|
||||
* (no {@link PlacementException} from a gate that has nothing to check against), using the same
|
||||
* production-shaped {@code Supplier<FleetConfig>} wiring {@code Fleetd.main} actually uses.
|
||||
*/
|
||||
@Test
|
||||
void noModelsBlockConfigStillLoadsAndSpawnsNormally(@TempDir Path dir) throws Exception {
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
""");
|
||||
FleetConfig cfg = FleetConfig.load(yaml);
|
||||
assertDoesNotThrow(cfg::validateAll, "a config with no models: block must load and validate cleanly");
|
||||
ConfigRef configRef = new ConfigRef(yaml, cfg);
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr,
|
||||
Map.of("local", stubWorker("local")), "local", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "local", configRef, _ -> 0, BackendQuarantine.none(), NO_OUTAGE);
|
||||
|
||||
assertFalse(composite.modelGateState().configured());
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("local", null, null)),
|
||||
"no models: block means nothing to gate against — the spawn must go through");
|
||||
assertEquals(1, adapter.spawnCount("local"));
|
||||
}
|
||||
|
||||
/** {@code fleet_profiles}/{@code GET /profiles} must read the exact same live source the gate reads. */
|
||||
@Test
|
||||
void disabledModelsReportsWhatTheGateActuallyEnforces() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"local", stubWorkerModel("local", "deepseek-v4-flash"),
|
||||
"sonnet", stubWorkerModel("sonnet", "claude-sonnet-5"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "local", Set.of());
|
||||
FleetConfig.Models models = new FleetConfig.Models(List.of(
|
||||
new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false),
|
||||
new FleetConfig.Models.ModelEntry("claude-sonnet-5", true)));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(), NO_OUTAGE,
|
||||
() -> models);
|
||||
|
||||
assertEquals(Set.of("deepseek-v4-flash"), composite.disabledModels());
|
||||
}
|
||||
|
||||
/** A caller with no {@code models:} block (the map-based constructors) reports nothing off. */
|
||||
@Test
|
||||
void disabledModelsIsEmptyWithNoModelsConfigured() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(Set.of(), composite.disabledModels());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -16,6 +16,7 @@ import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
@@ -221,6 +222,31 @@ class AmqpReplyInboxContractTest {
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void ackReportsHitVsMissAgainstARealBroker() throws Exception {
|
||||
// fleetd #437: fleet_ack said "acknowledged <msgId>" for a message it never touched,
|
||||
// because ReplyInbox.ack() (void) could not tell a hit from a miss. Pin the fixed
|
||||
// boolean contract against a real broker — the adapter fleetd actually runs live.
|
||||
String target = "worker-ack-contract-" + System.nanoTime();
|
||||
try (AmqpReplyInbox inbox = AmqpReplyInbox.open(uri())) {
|
||||
inbox.own(target);
|
||||
|
||||
// Never held for this target at all: must report false, not throw.
|
||||
assertFalse(inbox.ack(target, "never-held"),
|
||||
"acking a msgId never held for an owned target must report false");
|
||||
|
||||
// A real message: first ack removes it and reports true...
|
||||
inbox.publish(target, "m1", "ack me");
|
||||
assertEquals(1, awaitPeek(inbox, target).size(), "the published reply should be held");
|
||||
assertTrue(inbox.ack(target, "m1"), "acking a held reply must report true");
|
||||
assertTrue(inbox.peek(target).isEmpty(), "an acked reply is dropped");
|
||||
|
||||
// ...and the second ack of the SAME msgId has nothing left to remove: false.
|
||||
assertFalse(inbox.ack(target, "m1"),
|
||||
"acking the same msgId twice must report false the second time");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void confirmedPublishDeliversNormally() throws Exception {
|
||||
String target = "worker-confirm-" + System.nanoTime();
|
||||
|
||||
@@ -28,11 +28,19 @@ public final class FakeLeadChannel implements LeadChannel {
|
||||
private volatile IllegalStateException publishFailure;
|
||||
/** Canned {@link #inspect} results by coord-id — absent for any coord-id not configured here. */
|
||||
private final Map<String, MailboxState> mailboxes = new ConcurrentHashMap<>();
|
||||
/** fleetd #440: matches {@link LeadMailbox}'s real default (durable queue + manual ack) unless overridden. */
|
||||
private volatile boolean heldDurable = true;
|
||||
|
||||
public FakeLeadChannel(String selfCoordId) {
|
||||
this.selfCoordId = selfCoordId;
|
||||
}
|
||||
|
||||
/** Make {@link #heldDurable()} report {@code durable} — the fleetd #440 seam for the false case. */
|
||||
public FakeLeadChannel withHeldDurable(boolean durable) {
|
||||
this.heldDurable = durable;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make {@link #inspect(String)} return {@code state} for {@code coordId} instead of "absent". */
|
||||
public FakeLeadChannel withMailbox(String coordId, MailboxState state) {
|
||||
mailboxes.put(coordId, state);
|
||||
@@ -80,6 +88,11 @@ public final class FakeLeadChannel implements LeadChannel {
|
||||
return mailboxes.getOrDefault(coordId, MailboxState.absent(coordId));
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return heldDurable;
|
||||
}
|
||||
|
||||
public List<LeadMessage> published() {
|
||||
return List.copyOf(published);
|
||||
}
|
||||
|
||||
@@ -41,20 +41,20 @@ class InMemoryReplyInboxTest {
|
||||
@Test
|
||||
void ackRemovesTheMessage() {
|
||||
inbox.publish("term_a", "m1", "hello");
|
||||
inbox.ack("term_a", "m1");
|
||||
assertTrue(inbox.ack("term_a", "m1"), "fleetd #437: ack of a real entry must report true");
|
||||
assertTrue(inbox.peek("term_a").isEmpty(), "after ack, the message is gone");
|
||||
}
|
||||
|
||||
@Test
|
||||
void ackForUnknownMsgIdIsNoOp() {
|
||||
inbox.publish("term_a", "m1", "hello");
|
||||
inbox.ack("term_a", "no-such-id"); // no-op
|
||||
assertFalse(inbox.ack("term_a", "no-such-id"), "fleetd #437: a miss must report false"); // no-op
|
||||
assertEquals(1, inbox.peek("term_a").size(), "the published message is still there");
|
||||
}
|
||||
|
||||
@Test
|
||||
void ackForUnknownTargetIsNoOp() {
|
||||
inbox.ack("no-such-target", "m1"); // no-op, should not throw
|
||||
assertFalse(inbox.ack("no-such-target", "m1"), "fleetd #437: a miss must report false"); // no-op, should not throw
|
||||
}
|
||||
|
||||
@Test
|
||||
@@ -167,7 +167,7 @@ class InMemoryReplyInboxTest {
|
||||
@Test
|
||||
void peekAndAckAreNoOpsForUnownedTarget() {
|
||||
assertTrue(inbox.peek("term_not_owned").isEmpty());
|
||||
inbox.ack("term_not_owned", "m1"); // no-op, should not throw
|
||||
assertFalse(inbox.ack("term_not_owned", "m1"), "fleetd #437: a miss must report false"); // no-op, should not throw
|
||||
}
|
||||
|
||||
@Test
|
||||
|
||||
@@ -207,6 +207,21 @@ class LeadMailboxTest {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #440: {@code heldDurable()} must be derived from what {@link LeadMailbox#own} actually
|
||||
* did against the real broker — a durable queue declare plus a manual-ack consumer — not a
|
||||
* hardcoded literal. This is the mutation-sensitive test: flip {@code own()}'s {@code autoAck}
|
||||
* local to {@code true} (or its {@code durableQueue} local to {@code false}) and this must fail.
|
||||
*/
|
||||
@Test
|
||||
void heldDurableReportsTrueBecauseTheQueueIsDurableAndTheConsumeIsManualAck() throws Exception {
|
||||
String self = coordId("lead-held-durable");
|
||||
try (LeadMailbox mailbox = LeadMailbox.open(uri(), self)) {
|
||||
assertTrue(mailbox.heldDurable(),
|
||||
"own() declares a durable queue and consumes with autoAck=false, so held mail is durable");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void inspectReportsAMissingMailboxAsAbsentRatherThanThrowing() throws Exception {
|
||||
String nobody = coordId("lead-inspect-nobody");
|
||||
|
||||
@@ -90,6 +90,63 @@ class PlacementPolicyTest {
|
||||
assertTrue(e.getMessage().contains("quarantined"), e.getMessage());
|
||||
}
|
||||
|
||||
// --- fleetd #422: FixedPlacementPolicy must consult modelOff too, at BOTH filter sites ------
|
||||
|
||||
/** The default-profile fast path (:44) must skip a default whose model is off. */
|
||||
@Test
|
||||
void fixedSkipsModelOffDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of(), Set.of(), Set.of(), Set.of("b"));
|
||||
assertEquals("a", policy.select(ctx).profile(),
|
||||
"the default 'b' names an off model, so fixed falls through to the first available candidate");
|
||||
}
|
||||
|
||||
/**
|
||||
* The fallback walk (:48-53) must skip an off-model candidate too — exercised independently of
|
||||
* the default-profile fast path by using no default at all, so this is the only filter that runs.
|
||||
*/
|
||||
@Test
|
||||
void fixedFallbackWalkSkipsModelOffCandidate() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null,
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of(), Set.of(), Set.of(), Set.of("a"));
|
||||
assertEquals("b", policy.select(ctx).profile(),
|
||||
"candidate 'a' names an off model, so the fallback walk skips it and picks 'b'");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedThrowsWhenDefaultAndEveryCandidateModelOff() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of(), Set.of(), Set.of(), Set.of("a", "b"));
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
assertTrue(e.getMessage().contains("turned off"),
|
||||
"message names model-off as the cause: " + e.getMessage());
|
||||
assertFalse(e.getMessage().contains("quarantined"), "must not read like quarantine: " + e.getMessage());
|
||||
assertFalse(e.getMessage().contains("cooling off"), "must not read like cool-off: " + e.getMessage());
|
||||
assertFalse(e.getMessage().contains("weight 0"), "must not read like weight-0: " + e.getMessage());
|
||||
}
|
||||
|
||||
/**
|
||||
* Quarantine still wins when a profile is both quarantined and model-off (mirrors {@code
|
||||
* fixedThrowsWhenDefaultAndEveryCandidateQuarantined}'s priority over cooling off).
|
||||
*/
|
||||
@Test
|
||||
void fixedReportsQuarantineNotModelOffWhenBothApply() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of(), Set.of("a", "b"), Set.of(), Set.of("a", "b"));
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
assertTrue(e.getMessage().contains("quarantined"), e.getMessage());
|
||||
assertFalse(e.getMessage().contains("turned off"),
|
||||
"quarantine takes priority over model-off in the message: " + e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void roundRobinCyclesThroughAvailableProfiles() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
@@ -404,6 +461,93 @@ class PlacementPolicyTest {
|
||||
assertTrue(e.getMessage().contains("weight 0"), e.getMessage());
|
||||
}
|
||||
|
||||
// --- fleetd #435: FixedPlacementPolicy must consult maxLoad too, at BOTH filter sites -------
|
||||
|
||||
/**
|
||||
* The default-profile fast path must skip a capped default. Measured on 7667727 before this
|
||||
* fix: a single dev profile at {@code maxLoad: 1} with 1 live, under {@code fixed()}, still
|
||||
* returned "SPAWNED on profile=a" — the cap was advisory for every unqualified spawn.
|
||||
*/
|
||||
@Test
|
||||
void fixedSkipsCappedDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a", 1.0f, null),
|
||||
PlacementCandidate.profile("b", 1.0f, 1)),
|
||||
name -> "b".equals(name) ? 1 : 0, Set.of(), Set.of(), Set.of());
|
||||
assertEquals("a", policy.select(ctx).profile(),
|
||||
"the default 'b' is at its maxLoad cap, so fixed falls through to the free candidate 'a'");
|
||||
}
|
||||
|
||||
/**
|
||||
* The fallback walk must skip a capped candidate too — exercised independently of the
|
||||
* default-profile fast path by using no default at all, so this is the only filter that runs.
|
||||
*/
|
||||
@Test
|
||||
void fixedFallbackWalkSkipsCappedCandidate() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null,
|
||||
List.of(PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, null)),
|
||||
name -> "a".equals(name) ? 1 : 0, Set.of(), Set.of(), Set.of());
|
||||
assertEquals("b", policy.select(ctx).profile(),
|
||||
"candidate 'a' is at its maxLoad cap, so the fallback walk skips it and picks 'b'");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedThrowsWhenDefaultAndEveryCandidateAtCap() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, 1)),
|
||||
_ -> 1, Set.of(), Set.of(), Set.of());
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), "message names the cap: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("1 cap"), "message names the cap value: " + e.getMessage());
|
||||
assertFalse(e.getMessage().contains("quarantined"), "must not read like quarantine: " + e.getMessage());
|
||||
assertFalse(e.getMessage().contains("turned off"), "must not read like model-off: " + e.getMessage());
|
||||
}
|
||||
|
||||
/** The mirror: an uncapped default is still chosen, so the new term cannot exclude everything. */
|
||||
@Test
|
||||
void fixedStillReturnsUncappedDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a", 1.0f, null),
|
||||
PlacementCandidate.profile("b", 1.0f, null)),
|
||||
noSessions(), Set.of(), Set.of(), Set.of());
|
||||
assertEquals("b", policy.select(ctx).profile(), "an uncapped default is returned unconditionally");
|
||||
}
|
||||
|
||||
/** CB-585: an explicit {@code maxLoad: 0} on the default caps it at zero live members. */
|
||||
@Test
|
||||
void fixedSkipsMaxLoadZeroDefaultEvenWithZeroLiveWorkers() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a", 1.0f, null),
|
||||
PlacementCandidate.profile("b", 1.0f, 0)),
|
||||
_ -> 0, Set.of(), Set.of(), Set.of());
|
||||
assertEquals("a", policy.select(ctx).profile(),
|
||||
"the default 'b' has maxLoad 0, so it is already at its cap with nobody live");
|
||||
}
|
||||
|
||||
/**
|
||||
* Quarantine still wins when a profile is both quarantined and at cap (mirrors {@code
|
||||
* fixedReportsQuarantineNotModelOffWhenBothApply}'s priority over model-off).
|
||||
*/
|
||||
@Test
|
||||
void fixedReportsQuarantineNotAtCapWhenBothApply() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext("b",
|
||||
List.of(PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, 1)),
|
||||
_ -> 1, Set.of(), Set.of("a", "b"), Set.of());
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
assertTrue(e.getMessage().contains("quarantined"), e.getMessage());
|
||||
assertFalse(e.getMessage().contains("maxLoad"),
|
||||
"quarantine takes priority over at-cap in the message: " + e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void unknownPolicyNameThrows() {
|
||||
assertThrows(IllegalArgumentException.class, () -> PlacementPolicies.fromName("random"));
|
||||
|
||||
@@ -6,12 +6,14 @@ import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.MemberLifecycle;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
@@ -20,9 +22,15 @@ import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
@@ -33,6 +41,7 @@ import java.util.concurrent.Future;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicInteger;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
@@ -1991,4 +2000,296 @@ class SessionManagerTest {
|
||||
sessions.rosterResolved();
|
||||
assertEquals(2, handle.callCount(), "once resolved, the id must not be looked up again");
|
||||
}
|
||||
|
||||
// ── fleetd #425 criterion 3: acquireWithWorktree must provision for the profile it actually
|
||||
// spawns, never a name resolved before a live pool change is accounted for ────────────────────
|
||||
|
||||
/** Two profiles with distinct {@code cwd}/{@code parityOverlay}, and a dev pool of {@code first,second}. */
|
||||
private static String worktreeReorderYaml(String first, String second) {
|
||||
return """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
a:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: coder-a
|
||||
b:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: coder-b
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
fleet:
|
||||
developers:
|
||||
slot0:
|
||||
profile: %s
|
||||
slot1:
|
||||
profile: %s
|
||||
""".formatted(first, second);
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireWithWorktreeProvisionsTheOverlayForTheProfileActuallySpawned(
|
||||
@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, worktreeReorderYaml("a", "b"));
|
||||
ConfigRef ref = new ConfigRef(f, FleetConfig.load(f));
|
||||
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"a", new FleetConfig.Profile("a", "http://gx00.gw:8000", "coder-a", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/a", List.of("a.mcp.json")),
|
||||
"b", new FleetConfig.Profile("b", "http://gx00.gw:8000", "coder-b", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/b", List.of("b.mcp.json")));
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "a", _ -> null);
|
||||
PeerLauncher launcher = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", ref, _ -> 0, BackendQuarantine.none());
|
||||
|
||||
// The pool changes AFTER the composite/launcher is built, and BEFORE the unqualified
|
||||
// worktree spawn — exactly the fleetd #425 scenario: the live pool's first entry is "b" by
|
||||
// the time acquireWithWorktree runs, even though nothing here was rebuilt.
|
||||
Files.writeString(f, worktreeReorderYaml("b", "a"));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), () -> "reload should apply cleanly: " + out.error());
|
||||
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(launcher, worktrees, () -> 0L);
|
||||
|
||||
MemberSession s = sessions.acquire(null, null, "/caller",
|
||||
null, new WorktreeRequest("fleetd-425", null));
|
||||
|
||||
assertEquals("b", s.profile(),
|
||||
"the live dev pool now starts at b, so the unqualified spawn must land there");
|
||||
FakeWorktrees.OverlayCall overlay = worktrees.lastOverlay();
|
||||
assertNotNull(overlay, "overlayParity must have been called");
|
||||
assertEquals(List.of("b.mcp.json"), overlay.requested(),
|
||||
"the worktree must be provisioned with profile b's overlay — the one actually "
|
||||
+ "spawned — never a's, the pool's stale first entry");
|
||||
}
|
||||
|
||||
/**
|
||||
* The deterministic, mutation-pinning half of criterion 3: {@code launcher.defaultProfile()}
|
||||
* only ever answers for {@link MemberRole#DEV} (see {@link CompositePeerLauncher#defaultProfile()}),
|
||||
* so resolving a worktree spawn's profile through it — instead of through {@link
|
||||
* PeerLauncher#defaultProfileFor(MemberRole)}, resolved against the CALLER's actual role — picks
|
||||
* the wrong pool's answer for any role other than DEV. No reload or race is needed to see it: an
|
||||
* ARCHITECT pool and a DEV pool that simply disagree, held constant, are enough.
|
||||
*/
|
||||
@Test
|
||||
void acquireWithWorktreeForANonDevRoleUsesThatRolesPoolNotTheDevPool() {
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of(
|
||||
"a", new FleetConfig.Profile("a", "http://gx00.gw:8000", "coder-a", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/a", List.of("a.mcp.json")),
|
||||
"b", new FleetConfig.Profile("b", "http://gx00.gw:8000", "coder-b", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/b", List.of("b.mcp.json")));
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "a", _ -> null);
|
||||
// developers -> a (first/only entry); architects -> b (first/only entry). The two pools
|
||||
// disagree on purpose, so a role-blind resolution (DEV's answer, "a") is visibly wrong for
|
||||
// an ARCHITECT spawn, which must land on "b".
|
||||
FleetConfig.Fleet fleet = new FleetConfig.Fleet(Map.of(),
|
||||
Map.of("s0", new FleetConfig.Slot("b")),
|
||||
Map.of("s0", new FleetConfig.Slot("a")),
|
||||
Map.of(), null);
|
||||
PeerLauncher launcher = new CompositePeerLauncher(List.of(adapter), "a", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, fleet);
|
||||
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(launcher, worktrees, () -> 0L);
|
||||
|
||||
MemberSession s = sessions.acquire(null, MemberRole.ARCHITECT, null, "/caller",
|
||||
null, new WorktreeRequest("fleetd-425b", null));
|
||||
|
||||
assertEquals("b", s.profile(),
|
||||
"an unqualified ARCHITECT worktree spawn must land on the architect pool's profile");
|
||||
FakeWorktrees.OverlayCall overlay = worktrees.lastOverlay();
|
||||
assertNotNull(overlay, "overlayParity must have been called");
|
||||
assertEquals(List.of("b.mcp.json"), overlay.requested(),
|
||||
"the worktree must be provisioned with profile b's overlay — the ARCHITECT pool's "
|
||||
+ "answer, the one actually spawned — never a's, the DEV pool's answer that "
|
||||
+ "launcher.defaultProfile() alone would have given");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #425 rework, acceptance 3: repoRoot, parityOverlay, AND the actual spawn must all name
|
||||
* the SAME routed profile, proven on the ROUTED path — a quarantine skips the pool's first entry
|
||||
* — not just the "pool reordered by a live reload" path the two tests above already cover.
|
||||
*
|
||||
* <p>This is the exact regression the rework fixes: the first round resolved
|
||||
* {@code acquireWithWorktree}'s profile through {@code launcher.defaultProfileFor(memberRole)},
|
||||
* which is blind to quarantine and just returns the pool's first entry ("a" here, quarantined).
|
||||
* That name went on to provision repoRoot/overlay for "a", and then the spawn itself — now an
|
||||
* EXPLICIT-profile spawn naming "a" — hit {@code CompositePeerLauncher.enforceNotQuarantined}
|
||||
* and threw, where the pre-fix code (a blank-profile spawn) would have routed around "a" onto
|
||||
* "b" without any trouble. {@code launcher.routedProfileFor(memberRole)} closes that gap by
|
||||
* running the SAME quarantine-aware selection {@code spawn} itself uses, so all three — repoRoot,
|
||||
* overlay, and the spawn — land on "b" together.
|
||||
*/
|
||||
@Test
|
||||
void acquireWithWorktreeRoutesAroundAQuarantinedPoolFirstProfile() {
|
||||
Map<String, FleetConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("a", new FleetConfig.Profile("a", "http://gx00.gw:8000", "coder-a", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/a", List.of("a.mcp.json"),
|
||||
null, null, null, null, null, null, null, null, "shared-cred", null));
|
||||
profiles.put("b", new FleetConfig.Profile("b", "http://gx00.gw:8000", "coder-b", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/b", List.of("b.mcp.json")));
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "a", _ -> null);
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-cred");
|
||||
PeerLauncher launcher = new CompositePeerLauncher(List.of(adapter), "a", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(launcher, worktrees, () -> 0L);
|
||||
|
||||
MemberSession s = sessions.acquire(null, null, "/caller",
|
||||
null, new WorktreeRequest("fleetd-425c", null));
|
||||
|
||||
assertEquals("b", s.profile(),
|
||||
"a is quarantined, so the unqualified worktree spawn must route to b");
|
||||
FakeWorktrees.OverlayCall overlay = worktrees.lastOverlay();
|
||||
assertNotNull(overlay, "overlayParity must have been called");
|
||||
assertEquals(List.of("b.mcp.json"), overlay.requested(),
|
||||
"parityOverlay must be provisioned for b — the profile actually spawned, never a's, "
|
||||
+ "the quarantined pool-first entry");
|
||||
FakeWorktrees.RepoRootCall repoRootCall = worktrees.repoRootCalls().getLast();
|
||||
assertTrue(repoRootCall.cwd().contains("/repo/b"),
|
||||
"repoRoot must be resolved through b's effectiveCwd, not a's: " + repoRootCall.cwd());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #425 rework, round 2: this is the exact probe that found round 1's maxLoad
|
||||
* regression. One dev profile ("a") is configured with {@code maxLoad: 1} and a liveCount
|
||||
* pinned at 1 — permanently at cap — under {@code PlacementPolicies.fixed()}, the default
|
||||
* policy, which deliberately never evaluates {@code maxLoad} during automatic selection (see
|
||||
* {@code CompositePeerLauncher}'s own javadoc on {@code place}/{@code FixedPlacementPolicy}).
|
||||
*
|
||||
* <p>Round 1 resolved {@code acquireWithWorktree}'s profile through
|
||||
* {@code launcher.routedProfileFor(memberRole)} and then fed that name back into
|
||||
* {@code launcher.spawn(SpawnRequest)} as an EXPLICIT profile. Naming a profile explicitly
|
||||
* takes {@code CompositePeerLauncher.spawn}'s THROWING branch, which calls
|
||||
* {@code enforceMaxLoad} — so the worktree path died with a {@code PlacementException} while
|
||||
* the exact same unqualified request, with no worktree, still spawned cleanly through the
|
||||
* routing branch that never checks {@code maxLoad} at all. One intent, two different answers,
|
||||
* depending only on whether a worktree was asked for — the #425 shape, moved to a different
|
||||
* filter instead of closed.
|
||||
*
|
||||
* <p>This test does not hardcode which of the two outcomes is correct — whether an unqualified
|
||||
* spawn SHOULD respect {@code maxLoad} is fleetd #435, a separate ticket. It only asserts that
|
||||
* the WITH-worktree and WITHOUT-worktree paths agree: both spawn on the same profile, or both
|
||||
* fail with the same exception type and message. That way this test stays correct however
|
||||
* #435 is eventually resolved, and only breaks if the two paths disagree again.
|
||||
*/
|
||||
@Test
|
||||
void unqualifiedAcquireAgreesWithAndWithoutAWorktreeWhenTheOnlyProfileIsAtMaxLoad() {
|
||||
Map<String, FleetConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("a", new FleetConfig.Profile("a", "http://gx00.gw:8000", "coder-a", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/a", List.of("a.mcp.json"),
|
||||
null, null, null, null, 1.0f, 1));
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "a", _ -> null);
|
||||
// liveCount pinned at 1 for "a", exactly matching maxLoad — "a" is permanently at cap,
|
||||
// regardless of how many times either branch below actually spawns.
|
||||
PeerLauncher launcher = new CompositePeerLauncher(List.of(adapter), "a", profiles,
|
||||
PlacementPolicies.fixed(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
Object without = attemptAcquire(() ->
|
||||
new SessionManager(launcher, new FakeWorktrees(), () -> 0L)
|
||||
.acquire(null, null, "/caller", null));
|
||||
Object with = attemptAcquire(() ->
|
||||
new SessionManager(launcher, new FakeWorktrees(), () -> 0L)
|
||||
.acquire(null, null, "/caller", null, new WorktreeRequest("fleetd-425-maxload", null)));
|
||||
|
||||
assertEquals(without, with, "an unqualified spawn on a profile at maxLoad must agree "
|
||||
+ "whether or not a worktree was requested — no condition may become newly fatal "
|
||||
+ "on the worktree path alone (fleetd #425 rework, round 2)");
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #425 rework, round 4: the exact regression a mutation test found that 186 green tests
|
||||
* missed — {@code acquireWithWorktree} dropping the {@link
|
||||
* dev.ltms.fleet.placement.PlacementDecision} it already resolved via {@code launcher.place},
|
||||
* and letting the unqualified spawn re-run placement a second time (a blank-profile {@code
|
||||
* launcher.spawn(spawnReq)}) instead of carrying that decision forward via {@code
|
||||
* launcher.spawn(spawnReq, decision)}. Every earlier test in this file uses {@code
|
||||
* PlacementPolicies.fixed()}, which returns the same answer on every {@code select()} call, so
|
||||
* dropping the decision is invisible under it — two {@code select()} calls simply agree by
|
||||
* accident. {@code PlacementPolicies.roundRobin()} is deterministic AND stateful: its {@code
|
||||
* select()} advances an internal index on every call, so two consecutive calls for the SAME
|
||||
* spawn (one from {@code place()} to provision the worktree, a second from a dropped-decision
|
||||
* blank-profile {@code spawn(spawnReq)}) land on DIFFERENT profiles from a two-profile pool —
|
||||
* index 0 ("a"), then index 1 ("b").
|
||||
*
|
||||
* <p>This test does not hardcode which profile wins — asserting one specific name would pass
|
||||
* for the wrong reason the moment the rotation order changes (round-4 brief invariant 3). It
|
||||
* asserts AGREEMENT instead: whichever profile the worktree's parity overlay was provisioned
|
||||
* for must be the SAME profile the member actually spawned on. Each profile's overlay list is
|
||||
* named after the profile itself ({@code "a.mcp.json"}/{@code "b.mcp.json"}), so comparing the
|
||||
* recorded overlay against {@code s.profile() + ".mcp.json"} checks agreement without ever
|
||||
* naming an expected winner.
|
||||
*/
|
||||
@Test
|
||||
void acquireWithWorktreeSpawnsOnTheSameProfileItProvisionedTheWorktreeForUnderARotatingPolicy() {
|
||||
Map<String, FleetConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("a", new FleetConfig.Profile("a", "http://gx00.gw:8000", "coder-a", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/a", List.of("a.mcp.json")));
|
||||
profiles.put("b", new FleetConfig.Profile("b", "http://gx00.gw:8000", "coder-b", null,
|
||||
"FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"worker: {profile} #{n}", null, "/repo/b", List.of("b.mcp.json")));
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "a", _ -> null);
|
||||
// roundRobin is deterministic AND stateful: the first select() call picks index 0 ("a"),
|
||||
// and the SAME policy instance's second select() call (reached only if the
|
||||
// PlacementDecision is dropped) picks index 1 ("b") — the two-call disagreement this test
|
||||
// needs to make a dropped decision observable, rather than merely probable.
|
||||
PeerLauncher launcher = new CompositePeerLauncher(List.of(adapter), "a", profiles,
|
||||
PlacementPolicies.roundRobin(), _ -> 0);
|
||||
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(launcher, worktrees, () -> 0L);
|
||||
|
||||
MemberSession s = sessions.acquire(null, null, "/caller",
|
||||
null, new WorktreeRequest("fleetd-425-round4", null));
|
||||
|
||||
FakeWorktrees.OverlayCall overlay = worktrees.lastOverlay();
|
||||
assertNotNull(overlay, "overlayParity must have been called");
|
||||
assertEquals(List.of(s.profile() + ".mcp.json"), overlay.requested(),
|
||||
"the worktree must be provisioned for the SAME profile the member actually spawned "
|
||||
+ "on — under a rotating policy, dropping the PlacementDecision makes the "
|
||||
+ "second, spawn-time select() call disagree with the first, place()-time "
|
||||
+ "call, so the member ends up on a profile whose worktree (repoRoot/parity "
|
||||
+ "overlay) was built for a DIFFERENT profile (fleetd #425 rework, round 4)");
|
||||
}
|
||||
|
||||
/**
|
||||
* Reduce one {@code acquire(...)} attempt to a value comparable across the with-worktree and
|
||||
* without-worktree paths: the spawned profile name on success, or the thrown exception's class
|
||||
* and message on failure. Comparing THIS — instead of asserting "both spawn" or "both throw" as
|
||||
* a hardcoded direction — is what keeps {@link
|
||||
* #unqualifiedAcquireAgreesWithAndWithoutAWorktreeWhenTheOnlyProfileIsAtMaxLoad} valid whichever
|
||||
* way fleetd #435 eventually resolves whether an unqualified spawn should respect maxLoad.
|
||||
*/
|
||||
private static Object attemptAcquire(Supplier<MemberSession> call) {
|
||||
try {
|
||||
return "spawned:" + call.get().profile();
|
||||
} catch (RuntimeException e) {
|
||||
return "threw:" + e.getClass().getName() + ":" + e.getMessage();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user