Compare commits
8 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 7fd914df1a | |||
| 766772763f | |||
| eab8185d7b | |||
| 7f672f0fb8 | |||
| ce74e164c6 | |||
| e60f892efd | |||
| ce05886831 | |||
| be123d0ac7 |
@@ -69,6 +69,7 @@ import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
@@ -231,6 +232,13 @@ public final class Fleetd {
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine,
|
||||
outagePolicy);
|
||||
// fleetd #422 follow-up: say which of the three model-gate states the daemon booted into —
|
||||
// no models: block at all, a block armed with nothing off, or a block with N off — the same
|
||||
// way exhaustedPatternCoverageLine/errorPatternCoverageLine report CB-578 stage A/fleetd
|
||||
// #201 Unit 5 coverage just below. Read from workers.modelGateState() (never a separate
|
||||
// config.get().models() here) so this line and fleet_profiles' modelGateArmed can never
|
||||
// disagree about what CompositePeerLauncher's spawn gate actually enforces.
|
||||
log.info("model gate (fleetd #422): {}", modelGateCoverageLine(workers.modelGateState()));
|
||||
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
@@ -352,7 +360,11 @@ public final class Fleetd {
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
// fleetd #424: MemberRegistry.live re-reads fleet.architects through `config` on every
|
||||
// reserve/requireSlotFor call, so a reload that removes or adds an architect slot governs
|
||||
// the next spawn with no restart — the frozen `new MemberRegistry(cfg.fleet())` this used
|
||||
// to be let a "revoked" slot keep granting new architect spawns forever.
|
||||
MemberRegistry members = MemberRegistry.live(() -> config.get().fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
@@ -380,8 +392,7 @@ public final class Fleetd {
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage("exhaustedPattern", cfg.profiles().keySet(),
|
||||
exhaustedPatternsByProfile.keySet()));
|
||||
exhaustedPatternCoverageLine(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
// fleetd #201 Unit 5: classify a completion-fallback scrape that matches a profile's
|
||||
// configured backend-error refusal (a credential outage, a provider 5xx) as a backend error
|
||||
// rather than handing it back as a real answer. Compiled once at startup, keyed by profile
|
||||
@@ -401,8 +412,7 @@ public final class Fleetd {
|
||||
BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,
|
||||
errorPatternsByProfile);
|
||||
log.info("backend-error classification (fleetd #201 Unit 5): {}",
|
||||
CompletionResolver.coverage("errorPattern", cfg.profiles().keySet(),
|
||||
errorPatternsByProfile.keySet()));
|
||||
errorPatternCoverageLine(cfg.profiles().keySet(), errorPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
@@ -807,6 +817,66 @@ public final class Fleetd {
|
||||
}, quarantine, profile -> startupExhaustedPatterns.containsKey(profile));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): package-private factory for the CB-578 stage A {@code
|
||||
* exhaustedPattern} startup coverage line, paired explicitly with {@link
|
||||
* CompletionResolver.UnsetMeaning#OFF} — {@code exhaustedPattern} has no fallback, so a
|
||||
* profile with none configured really does have the classification off.
|
||||
*
|
||||
* <p>Extracted out of {@code main} for the same reason {@link #capacitySource} and {@link
|
||||
* #worktreeBranchLookup} were: {@code coverage()}'s own tests ({@code CompletionResolverTest})
|
||||
* prove it words {@code OFF} and {@link CompletionResolver.UnsetMeaning#BUILT_IN_DEFAULT}
|
||||
* correctly when a test supplies the meaning itself — they cannot prove {@code main} pairs the
|
||||
* right meaning with the right key, which is the actual fleetd #415 defect. <b>Measured:</b>
|
||||
* swapping the {@code UnsetMeaning} arguments between this method and {@link
|
||||
* #errorPatternCoverageLine} — recreating #415's defect with the two keys exchanged — compiled
|
||||
* with 0 errors and left all 1506 existing tests green before {@code
|
||||
* FleetdPatternCoverageLineTest} was added to catch exactly that swap.
|
||||
*/
|
||||
static String exhaustedPatternCoverageLine(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
return CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
allProfiles, configuredProfiles);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): the {@code errorPattern} counterpart of {@link
|
||||
* #exhaustedPatternCoverageLine}, paired explicitly with {@link
|
||||
* CompletionResolver.UnsetMeaning#BUILT_IN_DEFAULT} — an unset {@code errorPattern} still runs
|
||||
* backend-error classification against {@code CompletionResolver}'s built-in {@code
|
||||
* BACKEND_ERROR} pattern, so the empty case is not "off". See {@link
|
||||
* #exhaustedPatternCoverageLine}'s javadoc for the measured swap mutation this pairing guards
|
||||
* against.
|
||||
*/
|
||||
static String errorPatternCoverageLine(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
return CompletionResolver.coverage("errorPattern", CompletionResolver.UnsetMeaning.BUILT_IN_DEFAULT,
|
||||
allProfiles, configuredProfiles);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: package-private factory for the startup line reporting which of the
|
||||
* three central {@code models.allow:} gate states the daemon booted into. Extracted the same
|
||||
* way {@link #exhaustedPatternCoverageLine}/{@link #errorPatternCoverageLine} are, so a
|
||||
* dedicated test can call it directly rather than parsing log output, and so {@code main}'s
|
||||
* only source for this line is {@link PeerLauncher#modelGateState()} — never a second,
|
||||
* independently-derived read of {@code cfg.models()} that could disagree with what {@code
|
||||
* CompositePeerLauncher}'s spawn gate actually enforces (the fleetd #404 lesson).
|
||||
*
|
||||
* <p>Unlike the two pattern-key lines above, there is no {@code UnsetMeaning} choice to make
|
||||
* here: {@link PeerLauncher.ModelGateState#configured()} already states, unambiguously, whether
|
||||
* an empty {@link PeerLauncher.ModelGateState#off()} means "no {@code models:} block to gate
|
||||
* with" or "a block armed and currently reporting zero off" — the exact two states a bare
|
||||
* {@code disabledModels()} read could not tell apart before this ticket.
|
||||
*/
|
||||
static String modelGateCoverageLine(PeerLauncher.ModelGateState state) {
|
||||
if (!state.configured()) {
|
||||
return "not configured (no models: block — nothing is gated, and nothing can be)";
|
||||
}
|
||||
Set<String> off = state.off();
|
||||
return off.isEmpty()
|
||||
? "armed (models: block present; 0 models currently turned off)"
|
||||
: "armed (" + off.size() + " model(s) turned off: " + new TreeSet<>(off) + ")";
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #416: production source for {@code fleet_list}'s per-profile capacity facts.
|
||||
*
|
||||
|
||||
@@ -11,6 +11,7 @@ import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
@@ -19,16 +20,43 @@ import java.util.Objects;
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/{@code reviewers}
|
||||
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
||||
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
||||
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
||||
* {@code CompositePeerLauncher}'s (see {@code ConfigRef}'s class doc) — so a config reload
|
||||
* that removes or adds an architect slot governs the <em>next</em> spawn with no restart.
|
||||
* Only {@link #MemberRegistry(FleetConfig.Fleet)} freezes the pool at construction, and that
|
||||
* constructor exists for tests and for the (rare) case of wiring a fixed, code-built config.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry, initially <em>empty</em>, and
|
||||
* <strong>never</strong> touched by a reload. Config declares no architect terminal, so at
|
||||
* startup every slot is idle and nothing resolves to an architect; a session only becomes one
|
||||
* when the spawn lifecycle {@linkplain #bind(String, String) binds} its terminal to a slot.
|
||||
* {@link CallerResolver} reads this through {@link #snapshot()} to turn a pane into an
|
||||
* {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The binding rule (fleetd #424): config governs what a bound slot still grants, as
|
||||
* well as what may be bound next.</strong> Removing a slot from config revokes it — that is the
|
||||
* ticket's entire point ("Revoking an architect slot does not revoke it"). Revoking it means an
|
||||
* architect already bound to that slot loses the ARCHITECT privilege on its very next request:
|
||||
* {@link #roleForSlot} and {@link #nameForSlot} read {@link #slots()} directly, with no cache, so
|
||||
* the moment a slot drops out of config, {@link CallerResolver#resolve} (which calls both on every
|
||||
* request from a bound pane, {@code CallerResolver.java:220}) can no longer confirm the pane's slot
|
||||
* is an architect slot, and the pane falls through to {@code Principal.worker(...)}. What does
|
||||
* <em>not</em> change is the {@code terminalToSlot} <em>occupancy</em> — the binding created by
|
||||
* {@link #bind} is untouched by a reload, on purpose: unbinding it here would double-book the slot
|
||||
* key (a second terminal could then bind to the "freed" key while the first is still the terminal
|
||||
* the operator actually meant to demote) and would silently break {@link #unbind}'s compare-safe
|
||||
* contract, which needs the original {@code terminal → slot} pair intact to remove it cleanly. So
|
||||
* the demoted session keeps occupying its slot — {@link #slotForTerminal} and {@link #snapshot()}
|
||||
* still name it — it just no longer resolves as an architect through that occupancy, and a fresh
|
||||
* spawn still cannot bind to the same key while it is occupied ({@link #reserve}/
|
||||
* {@link #requireSlotFor} refuse it anyway, since it is gone from {@link #slots()}). The demoted
|
||||
* session's own turn is unaffected: {@code fleet_reply}'s authorization
|
||||
* ({@code Authz.Action.REPLY}) is {@code caller.ownsSession(targetSession)} — identity by terminal,
|
||||
* not by role — so a demoted architect can still end its own turn normally.
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
@@ -55,14 +83,42 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
private final Supplier<FleetConfig.Fleet> fleet;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
/**
|
||||
* Freeze the pool at construction — for tests, and for the rare case of wiring a fixed,
|
||||
* code-built config. Production wiring should prefer {@link #live}, which re-reads
|
||||
* {@code fleet:} on every call.
|
||||
*/
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
this(() -> fleet);
|
||||
}
|
||||
|
||||
private MemberRegistry(Supplier<FleetConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
}
|
||||
|
||||
/**
|
||||
* Live variant (fleetd #424): {@code fleet} is read fresh on every {@link #slots()} call — pass
|
||||
* {@code () -> config.get().fleet()}, the same supplier shape {@code CompositePeerLauncher}
|
||||
* already uses for placement — so a reload that adds or removes an architect slot governs the
|
||||
* next spawn's {@link #reserve}/{@link #requireSlotFor} check with no restart. A separate,
|
||||
* private constructor rather than a same-arity public overload of
|
||||
* {@link #MemberRegistry(FleetConfig.Fleet)}: a {@code FleetConfig.Fleet} and a
|
||||
* {@code Supplier<FleetConfig.Fleet>} overload are ambiguous for a literal {@code null} — the
|
||||
* same reason {@code CallerResolver.withLeads} is a static factory rather than a fourth
|
||||
* constructor overload.
|
||||
*/
|
||||
public static MemberRegistry live(Supplier<FleetConfig.Fleet> fleet) {
|
||||
return new MemberRegistry(Objects.requireNonNull(fleet, "fleet"));
|
||||
}
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one map. Leaders are not members. */
|
||||
private static Map<String, Entry> flatten(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
@@ -74,18 +130,22 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
return Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
/**
|
||||
* The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot of
|
||||
* {@code fleet:} <em>as of this call</em> — see the class doc for which constructor makes that
|
||||
* live versus frozen.
|
||||
*/
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
return flatten(fleet.get());
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
/** The slots belonging to {@code role}, in definition order, as of this call. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
slots().forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
@@ -123,25 +183,35 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
/**
|
||||
* The role a qualified slot key belongs to, or {@code null} when the key is not currently
|
||||
* configured. Deliberately live, with no cache (fleetd #424, see the class doc's binding rule):
|
||||
* removing a slot from config must make {@link CallerResolver#resolve} stop granting the
|
||||
* ARCHITECT role for it on the very next request from a terminal that was bound to it, which is
|
||||
* the ticket's whole point — revoking a slot must actually revoke it, not just refuse the next
|
||||
* spawn.
|
||||
*/
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
/**
|
||||
* The unqualified configured name for a slot, or {@code null} if it is not currently configured.
|
||||
* Live for the same reason as {@link #roleForSlot} — see the class doc's binding rule.
|
||||
*/
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
return slots().containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -32,18 +32,26 @@ import java.util.function.Supplier;
|
||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||
* ({@code () -> config.get().fleet()}). <strong>But {@code fleet:} as a whole is NOT in this
|
||||
* class</strong>: {@code fleet.leaders} inside the same key is frozen, which is exactly what
|
||||
* makes {@code fleet:} split rather than hot — see below. {@code models:} (fleetd #422) joined
|
||||
* this class whole: {@link FleetConfig#validateModels()} re-runs fully against the fresh
|
||||
* config on every {@link #reload()} (via {@link FleetConfig#validateAll()}), refusing a bad
|
||||
* edit outright rather than caching a stale copy anywhere, and the on/off half added by
|
||||
* fleetd #422 is read live both by {@code CompositePeerLauncher}'s spawn gate
|
||||
* ({@code enforceModelEnabled} and its candidate filter) and by {@code fleet_profiles}/
|
||||
* {@code GET /profiles} (via {@code PeerLauncher.disabledModels()}). Nothing about
|
||||
* {@code models:} is baked into an object built at startup, so — unlike the deferred keys
|
||||
* below — there is no frozen half left to report; it moved here from deferred rather than
|
||||
* joining split.</li>
|
||||
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
||||
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
||||
* reads it live for placement (which profile an unqualified architect spawn may land on), and
|
||||
* {@code MemberRegistry} separately reads it live, through its own instance of the same
|
||||
* supplier shape, for identity — both which slot a spawn may bind to <em>and</em> what a slot
|
||||
* already bound still grants. Removing an architect slot from config therefore revokes the
|
||||
* {@link dev.ltms.fleet.auth.Role#ARCHITECT} role on the bound pane's very next request; only
|
||||
* the slot <em>occupancy</em> survives, so the demoted session still holds its slot key until
|
||||
* it unbinds. See {@code MemberRegistry}'s class doc for that binding rule.
|
||||
* <strong>But {@code fleet:} as a whole is NOT in this class</strong>: {@code fleet.leaders}
|
||||
* inside the same key is frozen, which is exactly what makes {@code fleet:} split rather than
|
||||
* hot — see below. {@code models:} (fleetd #422) joined this class whole: {@link
|
||||
* FleetConfig#validateModels()} re-runs fully against the fresh config on every {@link
|
||||
* #reload()} (via {@link FleetConfig#validateAll()}), refusing a bad edit outright rather than
|
||||
* caching a stale copy anywhere, and the on/off half added by fleetd #422 is read live both by
|
||||
* {@code CompositePeerLauncher}'s spawn gate ({@code enforceModelEnabled} and its candidate
|
||||
* filter) and by {@code fleet_profiles}/{@code GET /profiles} (via
|
||||
* {@code PeerLauncher.disabledModels()}). Nothing about {@code models:} is baked into an
|
||||
* object built at startup, so — unlike the deferred keys below — there is no frozen half left
|
||||
* to report; it moved here from deferred rather than joining split.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code idleSleepGuard:} ({@code Fleetd.java} reads it once, at startup, to decide whether
|
||||
@@ -525,26 +533,35 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||
+ "member's environment is read live on every spawn and already applied");
|
||||
}
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (architects, developers,
|
||||
// reviewers, charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, reviewers,
|
||||
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||
// with no restart note. Only fleet.leaders is frozen (Fleetd.java:281 reads
|
||||
// cfg.fleet().leaders() off the startup snapshot to build both the LeadTabScanner's
|
||||
// tab-label-to-name map, wired into CallerResolver.withLeadsAndMembers at Fleetd.java:620/624,
|
||||
// and — when herdr answered — LeadLauncher(...).ensureLeads() at Fleetd.java:315, which
|
||||
// auto-launches each lead up to its `instances` count; neither is rebuilt on reload). So this
|
||||
// compares fleet.leaders alone, not the whole Fleet record: comparing the whole record would
|
||||
// report "split" for a tabLabel-only or charters-only change that is actually fully hot,
|
||||
// which is the over-claim mirror of the under-claim bug this class exists to prevent.
|
||||
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
||||
// its consumers, not just the one this comment used to name: CompositePeerLauncher reads it
|
||||
// live for PLACEMENT through the () -> config.get().fleet() supplier named in the class doc's
|
||||
// Hot bullet, and MemberRegistry separately reads it live for IDENTITY (which slot a spawn
|
||||
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
||||
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
||||
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds. Only
|
||||
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
||||
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
||||
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
||||
// LeadLauncher(...).ensureLeads() at Fleetd.java:315, which auto-launches each lead up to its
|
||||
// `instances` count; neither is rebuilt on reload). So this compares fleet.leaders alone, not
|
||||
// the whole Fleet record: comparing the whole record would report "split" for a tabLabel-only
|
||||
// or architects-only change that is actually fully hot, which is the over-claim mirror of the
|
||||
// under-claim bug this class exists to prevent.
|
||||
if (!Objects.equals(leadersOf(old), leadersOf(fresh))) {
|
||||
changed.add("fleet: fleet.leaders (each lead's tab, workspace, cwd, profile and "
|
||||
+ "instances count) is read once at startup to build the LeadTabScanner's "
|
||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||
+ "lead; the rest of fleet: (architects, developers, reviewers, charters, "
|
||||
+ "tabLabel) is read live through the supplier on CompositePeerLauncher and "
|
||||
+ "already applied");
|
||||
+ "lead; the rest of fleet: (developers, reviewers, charters, tabLabel) is read "
|
||||
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
||||
+ "live through that same supplier for placement AND through a separate supplier "
|
||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
|
||||
@@ -698,17 +698,57 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* What an unset pattern key means for the classification it configures (fleetd#415).
|
||||
* {@code coverage()} cannot infer this from the key's name — the two keys it currently
|
||||
* describes disagree on it, and a string comparison on the name would just move the same bug
|
||||
* to a new spot — so every caller must state it explicitly.
|
||||
*
|
||||
* <p><strong>This alone does not prove a caller passes the right one for its key.</strong> A
|
||||
* test that calls {@code coverage()} directly and supplies the meaning itself only proves this
|
||||
* enum is worded correctly, never that {@code Fleetd}'s two call sites pair each key with its
|
||||
* true meaning — that pairing is #415's actual defect. Measured on review: swapping the two
|
||||
* {@code UnsetMeaning} arguments at those call sites (giving {@code exhaustedPattern} the
|
||||
* built-in-default wording and {@code errorPattern} the off wording — #415's exact defect with
|
||||
* the keys exchanged) compiled with 0 errors and left all 1506 existing tests green. See
|
||||
* {@code dev.ltms.fleet.Fleetd#exhaustedPatternCoverageLine}/{@code #errorPatternCoverageLine}
|
||||
* and {@code FleetdPatternCoverageLineTest}, which exists specifically to catch that swap.
|
||||
*/
|
||||
public enum UnsetMeaning {
|
||||
/** No fallback exists: a profile with no configured pattern truly has this classification off. */
|
||||
OFF,
|
||||
/** A built-in pattern applies when unset: the classification still runs for that profile. */
|
||||
BUILT_IN_DEFAULT
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for a fleetd#201/CB-578-style pattern-key classification, logged at startup
|
||||
* the way {@link dev.ltms.fleet.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* <p>fleetd#415: this method measures <em>pattern coverage</em> — how many profiles set the
|
||||
* key — which is not the same thing as <em>feature state</em> for a key with a fallback. For
|
||||
* {@code errorPattern}, an empty {@code configuredProfiles} still runs the classification
|
||||
* against {@code CompletionResolver}'s built-in compatibility pattern ({@link #BACKEND_ERROR}
|
||||
* at line ~84); for {@code exhaustedPattern} there is no fallback, so empty really does mean
|
||||
* off. {@code unsetMeaning} is the single, required source of that fact — see
|
||||
* {@link dev.ltms.fleet.config.FleetConfig#rejectMalformedProfilePatterns} lines ~2029-2032 for
|
||||
* where it is documented for config authors. It is a required parameter, not a defaulted
|
||||
* overload: a third pattern key added later must supply one to compile at all, rather than
|
||||
* silently inheriting whichever wording this method happened to default to.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry the pattern
|
||||
*/
|
||||
public static String coverage(String patternKey, Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
public static String coverage(String patternKey, UnsetMeaning unsetMeaning, Set<String> allProfiles,
|
||||
Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an " + patternKey + " configured; profiles: " + sorted(allProfiles) + ")";
|
||||
return switch (unsetMeaning) {
|
||||
case OFF -> "off (no profile has an " + patternKey + " configured; profiles: "
|
||||
+ sorted(allProfiles) + ")";
|
||||
case BUILT_IN_DEFAULT -> "built-in default for all profiles (no profile customises "
|
||||
+ patternKey + "; profiles: " + sorted(allProfiles) + ")";
|
||||
};
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
|
||||
@@ -1181,12 +1181,21 @@ public final class FleetMcp {
|
||||
result.put("coolingOff", coolingOff);
|
||||
}
|
||||
// fleetd #422: read the exact same accessor CompositePeerLauncher's spawn gate reads
|
||||
// (PeerLauncher.disabledModels(), which for the composite is models0().offIds()) — never a
|
||||
// (PeerLauncher.modelGateState(), which for the composite is models0() read live) — never a
|
||||
// separately-derived answer, so this status can never overstate or understate what the gate
|
||||
// actually enforces (the fleetd #404 lesson).
|
||||
Set<String> modelsOff = workers.disabledModels();
|
||||
if (!modelsOff.isEmpty()) {
|
||||
result.put("modelsOff", new ArrayList<>(modelsOff));
|
||||
//
|
||||
// fleetd #422 follow-up: "armed" and "off" come from the ONE modelGateState() call below,
|
||||
// never two independent reads of the gate — a reload landing between two separate reads
|
||||
// could otherwise make them disagree. modelGateArmed is reported unconditionally (never
|
||||
// omitted like quarantined/coolingOff above) precisely so a lead can tell "no models: block
|
||||
// at all" (false) apart from "a models: block with nothing currently off" (true, with
|
||||
// modelsOff simply absent below) — the two states PeerLauncher.disabledModels() alone
|
||||
// cannot distinguish, both reporting an empty set.
|
||||
PeerLauncher.ModelGateState modelGate = workers.modelGateState();
|
||||
result.put("modelGateArmed", modelGate.configured());
|
||||
if (!modelGate.off().isEmpty()) {
|
||||
result.put("modelsOff", new ArrayList<>(modelGate.off()));
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
@@ -644,14 +644,29 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422: read live off {@link #models0()} — the exact same accessor {@link
|
||||
* fleetd #422 follow-up: the single live read that answers both "is the models.allow: gate
|
||||
* armed" and "which models are off", off the exact same accessor ({@link #models0()}) {@link
|
||||
* #enforceModelEnabled} and {@link #modelOffProfiles} read — so {@code fleet_profiles}/{@code
|
||||
* GET /profiles} can never report a different answer than the gate enforces (the fleetd #404
|
||||
* lesson).
|
||||
* GET /profiles} (via {@link PeerLauncher#disabledModels()}, which now delegates here) can
|
||||
* never report a different answer than the gate enforces (the fleetd #404 lesson), and the
|
||||
* startup log line built from this can never disagree with either.
|
||||
*
|
||||
* <p>{@link #models0()} itself normalizes a {@code null} {@link #models} read to the shared
|
||||
* {@link #NO_MODELS_CONFIGURED} sentinel — deliberately the one object no config-supplied
|
||||
* {@code Models} instance can ever be identical to, since it is private to this class — so
|
||||
* comparing by reference here recovers exactly the fact {@code models0()}'s normalization
|
||||
* would otherwise erase: whether the live source was {@code null} (no {@code models:} block,
|
||||
* armed = false) or a real, config-supplied block (armed = true, even one whose {@code allow:}
|
||||
* is itself empty or absent — {@link FleetConfig.Models}'s "absent or empty allow: is off"
|
||||
* wording governs config-load validation, a distinct question from whether this gate is armed
|
||||
* for reporting).
|
||||
*/
|
||||
@Override
|
||||
public Set<String> disabledModels() {
|
||||
return models0().offIds();
|
||||
public PeerLauncher.ModelGateState modelGateState() {
|
||||
FleetConfig.Models m = models0();
|
||||
return m == NO_MODELS_CONFIGURED
|
||||
? PeerLauncher.ModelGateState.notConfigured()
|
||||
: PeerLauncher.ModelGateState.armed(m.offIds());
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -202,8 +202,56 @@ public interface PeerLauncher {
|
||||
* can drift from what the gate ({@code CompositePeerLauncher.enforceModelEnabled} and its
|
||||
* candidate filter) actually enforces. A default of {@code Set.of()} keeps every other {@link
|
||||
* PeerLauncher} implementer (the herdr adapters, and the two test-fake implementers) unchanged.
|
||||
*
|
||||
* <p>fleetd #422 follow-up: this alone cannot tell "no {@code models:} block at all" from "a
|
||||
* {@code models:} block where nothing is currently off" — both report an empty set here. Delegates
|
||||
* to {@link #modelGateState()} so the two facts always come from the one read {@link
|
||||
* #modelGateState()}'s implementer makes; do not override this method separately from that one.
|
||||
*/
|
||||
default Set<String> disabledModels() {
|
||||
return Set.of();
|
||||
return modelGateState().off();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the central {@code models.allow:} gate (fleetd #422) is armed at all, together with
|
||||
* which model ids are currently off — fleetd #422 follow-up. {@link #disabledModels()} alone
|
||||
* cannot distinguish two states that both report an empty set: a host with no {@code models:}
|
||||
* block (nothing is gated, and nothing can be) and a host WITH a {@code models:} block where
|
||||
* nothing is currently turned off (the gate is armed and reporting zero). This method exists so
|
||||
* a caller — the startup log, {@code fleet_profiles}/{@code GET /profiles} — can tell the two
|
||||
* apart, the same reason {@code CompletionResolver.UnsetMeaning} exists: an accessor that can
|
||||
* legitimately report "empty" must never let a caller guess why.
|
||||
*
|
||||
* <p>Default {@link ModelGateState#notConfigured()} — every launcher without a {@code models:}
|
||||
* block to read from (the herdr adapters, and the two test-fake implementers), matching {@link
|
||||
* #disabledModels()}'s own default of an empty set.
|
||||
*/
|
||||
default ModelGateState modelGateState() {
|
||||
return ModelGateState.notConfigured();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: the result of {@link #modelGateState()} — see that method's javadoc
|
||||
* for why "armed" and "off" must be reported together from one read rather than as two
|
||||
* separately-derived facts that a reload landing between them could make disagree.
|
||||
*
|
||||
* @param configured {@code true} when a {@code models:} block exists at all (armed), regardless
|
||||
* of whether anything in it is currently turned off; {@code false} when there
|
||||
* is no block to gate against
|
||||
* @param off the model ids currently turned off; always empty when {@code configured} is
|
||||
* {@code false}
|
||||
*/
|
||||
record ModelGateState(boolean configured, Set<String> off) {
|
||||
public ModelGateState {
|
||||
off = Set.copyOf(off);
|
||||
}
|
||||
|
||||
public static ModelGateState notConfigured() {
|
||||
return new ModelGateState(false, Set.of());
|
||||
}
|
||||
|
||||
public static ModelGateState armed(Set<String> off) {
|
||||
return new ModelGateState(true, off);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,62 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: {@code Fleetd.modelGateCoverageLine} is the startup-log counterpart of
|
||||
* {@code exhaustedPatternCoverageLine}/{@code errorPatternCoverageLine} — see {@code
|
||||
* FleetdPatternCoverageLineTest} for the identical shape this follows — except here there is no
|
||||
* {@code UnsetMeaning} choice for a caller to get backwards: {@link
|
||||
* PeerLauncher.ModelGateState#configured()} already states, unambiguously, whether an empty
|
||||
* {@link PeerLauncher.ModelGateState#off()} means "no {@code models:} block to gate with at all"
|
||||
* or "a block armed and currently reporting zero off". This class proves {@code
|
||||
* modelGateCoverageLine} words those two states — plus the third, N off — distinctly, so a
|
||||
* mutation that made it ignore {@code configured()} either way is caught here.
|
||||
*/
|
||||
class FleetdModelGateCoverageLineTest {
|
||||
|
||||
@Test
|
||||
@DisplayName("no models: block reports not configured, distinct from armed-with-zero")
|
||||
void noModelsBlockReportsNotConfigured() {
|
||||
String line = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.notConfigured());
|
||||
assertEquals("not configured (no models: block — nothing is gated, and nothing can be)", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a models: block armed with nothing off reports armed, distinct from not configured")
|
||||
void armedWithNothingOffReportsArmed() {
|
||||
String line = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of()));
|
||||
assertEquals("armed (models: block present; 0 models currently turned off)", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a models: block with N off names the off models")
|
||||
void armedWithModelsOffNamesThem() {
|
||||
String line = Fleetd.modelGateCoverageLine(
|
||||
PeerLauncher.ModelGateState.armed(Set.of("deepseek-v4-flash", "claude-opus-9000")));
|
||||
assertEquals("armed (2 model(s) turned off: [claude-opus-9000, deepseek-v4-flash])", line);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the three states produce pairwise-distinct wording for the same empty-looking input")
|
||||
void theThreeStatesProduceDistinctWording() {
|
||||
String notConfigured = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.notConfigured());
|
||||
String armedZero = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of()));
|
||||
String armedOne = Fleetd.modelGateCoverageLine(PeerLauncher.ModelGateState.armed(Set.of("x")));
|
||||
|
||||
// Pinned individually above; restated here so this test alone still catches a regression
|
||||
// even if one of the three tests above were ever deleted — the exact FleetdPatternCoverageLineTest
|
||||
// pattern, adapted from "two keys" to "three states of one gate".
|
||||
assertNotEquals(notConfigured, armedZero,
|
||||
"collapsing 'no models: block' into 'armed, zero off' is fleetd #422 follow-up's exact defect");
|
||||
assertNotEquals(armedZero, armedOne);
|
||||
assertNotEquals(notConfigured, armedOne);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,67 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
|
||||
/**
|
||||
* fleetd #415 (review follow-up): {@code CompletionResolverTest} proves {@code coverage()} words
|
||||
* {@code UnsetMeaning.OFF} and {@code UnsetMeaning.BUILT_IN_DEFAULT} correctly — but every one of
|
||||
* those tests supplies the meaning itself. That proves the enum's wording, never that {@code
|
||||
* Fleetd} pairs the right meaning with the right pattern key. That pairing is #415's actual
|
||||
* defect: {@code coverage()} had no way to know what unset meant for its key, so the fix moved
|
||||
* the fact to the caller — and nothing yet proved the caller states it correctly.
|
||||
*
|
||||
* <p><b>Measured or it didn't happen:</b> swapping the two {@code UnsetMeaning} arguments at
|
||||
* {@code Fleetd}'s two coverage call sites — giving {@code exhaustedPattern} the built-in-default
|
||||
* wording and {@code errorPattern} the off wording, #415's exact defect with the keys exchanged —
|
||||
* compiled with 0 errors and left all 1506 existing tests green. This class exists to turn that
|
||||
* swap red.
|
||||
*
|
||||
* <p>It calls {@link Fleetd#exhaustedPatternCoverageLine} and {@link Fleetd#errorPatternCoverageLine}
|
||||
* directly rather than reading {@code Fleetd.java} as source text (the shape {@code
|
||||
* FleetdCompletionResolverWiringTest} uses for a different wiring gap): those two methods are the
|
||||
* extracted call sites {@code main} actually invokes, following the same {@code static} factory +
|
||||
* dedicated-test pattern as {@link Fleetd#capacitySource} and {@link Fleetd#worktreeBranchLookup}.
|
||||
*/
|
||||
class FleetdPatternCoverageLineTest {
|
||||
|
||||
private static final Set<String> PROFILES = Set.of("terra", "gx10");
|
||||
|
||||
@Test
|
||||
@DisplayName("exhaustedPatternCoverageLine says off when no profile configures exhaustedPattern")
|
||||
void exhaustedPatternCoverageLineSaysOffWhenNoProfileConfiguresIt() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
Fleetd.exhaustedPatternCoverageLine(PROFILES, Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("errorPatternCoverageLine says built-in default when no profile configures errorPattern")
|
||||
void errorPatternCoverageLineSaysBuiltInDefaultWhenNoProfileConfiguresIt() {
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])",
|
||||
Fleetd.errorPatternCoverageLine(PROFILES, Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("the two keys produce different wording for the identical empty-coverage input")
|
||||
void theTwoKeysProduceDifferentWordingForTheSameEmptyInput() {
|
||||
String exhaustedLine = Fleetd.exhaustedPatternCoverageLine(PROFILES, Set.of());
|
||||
String errorLine = Fleetd.errorPatternCoverageLine(PROFILES, Set.of());
|
||||
|
||||
// Pinned individually above; restated here so this test alone still catches a swap even
|
||||
// if one of the two tests above were ever deleted.
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
exhaustedLine);
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])", errorLine);
|
||||
assertNotEquals(exhaustedLine, errorLine,
|
||||
"swapping which UnsetMeaning pairs with which pattern key at Fleetd's call sites "
|
||||
+ "must be caught here — that pairing, not coverage()'s own wording in isolation, "
|
||||
+ "is fleetd #415's actual defect");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,257 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* fleetd #424 — revoking (or granting) an architect slot must take effect on the next spawn with
|
||||
* no restart. A session already bound to a slot keeps its <em>binding</em> (the {@code
|
||||
* terminalToSlot} occupancy) even after that slot drops out of config, but NOT the ARCHITECT
|
||||
* <em>privilege</em> the slot used to grant — that is revoked on the bound session's very next
|
||||
* request. See {@link MemberRegistry}'s class doc for the exact rule: "config governs what a
|
||||
* bound slot still grants, as well as what may be bound next."
|
||||
*
|
||||
* <p>Every test here drives a REAL {@link ConfigRef#reload()} against a {@code @TempDir} file and
|
||||
* asserts {@link ConfigRef.Outcome#applied()}, rather than comparing two frozen
|
||||
* {@code MemberRegistry} instances in memory — the defect this ticket fixes is specifically that
|
||||
* {@link MemberRegistry} used to ignore a live reload, so a test that never reloads cannot tell the
|
||||
* fixed registry from the broken one. {@code requireSlotFor} and {@code reserve} are pinned in
|
||||
* separate tests, in both directions (removed and added), so a registry that simply refuses (or
|
||||
* simply allows) everything cannot pass by accident — see {@link MemberRegistryTest} for the
|
||||
* registry's other invariants (bind/unbind cardinality, thread-safety), which are unaffected by
|
||||
* this ticket and still exercised against the frozen constructor.
|
||||
*/
|
||||
class MemberRegistryLiveTest {
|
||||
|
||||
private static String yaml(String fleetBlock) {
|
||||
return """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
opus:
|
||||
baseUrl: http://gx00.gw:8001
|
||||
model: opus
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""" + fleetBlock;
|
||||
}
|
||||
|
||||
private static final String WITH_SONNET_SLOT = """
|
||||
fleet:
|
||||
architects:
|
||||
designer:
|
||||
profile: sonnet
|
||||
""";
|
||||
|
||||
/** No architect pool at all — developers is unrelated dead data for this registry (#424 out of scope). */
|
||||
private static final String WITHOUT_ARCHITECT_SLOTS = """
|
||||
fleet:
|
||||
developers:
|
||||
dev1:
|
||||
profile: sonnet
|
||||
""";
|
||||
|
||||
private static ConfigRef refFor(Path f) {
|
||||
return new ConfigRef(f, FleetConfig.load(f));
|
||||
}
|
||||
|
||||
// ── requireSlotFor is live (criteria 1, 2, 4) ──────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void requireSlotForRefusesAProfileWhoseSlotWasRemovedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertDoesNotThrow(() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"the slot is configured before the reload");
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"revoking the slot must refuse the NEXT spawn that names it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void requireSlotForAllowsAProfileWhoseSlotWasAddedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"no architect slot is configured yet");
|
||||
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertDoesNotThrow(() -> registry.requireSlotFor(MemberRole.ARCHITECT, "sonnet"),
|
||||
"a slot added by reload must be usable with no restart");
|
||||
}
|
||||
|
||||
// ── reserve is live too — tested separately from requireSlotFor (criteria 1, 2, 4) ────────
|
||||
|
||||
@Test
|
||||
void reserveRefusesAProfileWhoseSlotWasRemovedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation before = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertEquals("architect:designer", before.slot());
|
||||
registry.release(before); // free it back up so the reload-side reserve below starts clean
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.reserve(MemberRole.ARCHITECT, "sonnet"),
|
||||
"revoking the slot must refuse the NEXT reservation for it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reserveAllowsAProfileWhoseSlotWasAddedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.reserve(MemberRole.ARCHITECT, "sonnet"),
|
||||
"no architect slot is configured yet");
|
||||
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
MemberLifecycle.SlotReservation after = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertEquals("architect:designer", after.slot(),
|
||||
"a slot added by reload must be reservable with no restart");
|
||||
}
|
||||
|
||||
// ── a bound architect is demoted, but the binding itself is not touched (fleetd #424) ───────
|
||||
// The lead's corrected ruling: the PRIVILEGE a slot grants is revoked on the bound session's
|
||||
// very next request, but the terminalToSlot BINDING itself is untouched by a reload — dropping
|
||||
// it would double-book the slot key and break unbind's compare-safe contract. See the class
|
||||
// doc's binding rule.
|
||||
|
||||
/** A caller identity resolving the one canned pane (terminal {@code term_a}) in {@link FakeHerdr}. */
|
||||
private static ConnectionIdentity boundPaneIdentity() {
|
||||
return new ConnectionIdentity(new PaneLocator(new FakeHerdr()), _ -> FakeHerdr.WORKER_PID);
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectAlreadyBoundToASlotIsDemotedByReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation reservation = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertTrue(registry.bind(reservation, "term_a"));
|
||||
assertEquals("architect:designer", registry.slotForTerminal("term_a"));
|
||||
|
||||
// Drive the real caller path, not the roleForSlot seam directly: CallerResolver.resolve is
|
||||
// what a live request actually goes through (CallerResolver.java:220), and a resolver that
|
||||
// ignored roleForSlot entirely would still pass a test that only checked the seam.
|
||||
CallerResolver resolver = CallerResolver.withLeadsAndMembers(
|
||||
boundPaneIdentity(), false, null, Map::of, registry);
|
||||
|
||||
Principal before = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.ARCHITECT, before.role(), "sanity check: the harness binds term_a as an architect");
|
||||
assertEquals("designer", before.name());
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
Principal after = resolver.resolve("127.0.0.1", 42, null);
|
||||
assertEquals(Role.WORKER, after.role(),
|
||||
"removing the slot from config must demote the bound session to worker on its "
|
||||
+ "NEXT request — this is the ticket's whole point");
|
||||
assertEquals("term_a", after.terminal(), "same pane, same terminal — only the role changed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theOriginalBindingStillOccupiesTheRemovedSlotSoASecondTerminalCannotClaimIt(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation reservation = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertTrue(registry.bind(reservation, "term_a"));
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome removed = ref.reload();
|
||||
assertTrue(removed.applied(), "the reload must actually take effect: " + removed.summary());
|
||||
|
||||
// The binding survives the removal untouched.
|
||||
assertEquals("architect:designer", registry.slotForTerminal("term_a"),
|
||||
"a live binding must never be retroactively unbound by a config edit");
|
||||
assertEquals(Map.of("term_a", "architect:designer"), registry.snapshot());
|
||||
|
||||
// Bring the slot back into config. If the binding had been silently dropped by the removal
|
||||
// (rather than merely losing the privilege it grants), a second terminal could now claim
|
||||
// the "freed" key — the exact double-booking the class doc's binding rule rules out.
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef.Outcome restored = ref.reload();
|
||||
assertTrue(restored.applied(), "the reload must actually take effect: " + restored.summary());
|
||||
|
||||
assertFalse(registry.bind("architect:designer", "term_b"),
|
||||
"the slot is still occupied by term_a — a second terminal must not bind to it");
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> registry.reserve(MemberRole.ARCHITECT, "sonnet"),
|
||||
"the slot is still occupied by term_a — a fresh reservation must not find it free");
|
||||
assertEquals("architect:designer", registry.slotForTerminal("term_a"),
|
||||
"the original binding is unchanged throughout");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unbindStillSucceedsForTheOriginalTerminalAfterItsSlotIsRemoved(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml(WITH_SONNET_SLOT));
|
||||
ConfigRef ref = refFor(f);
|
||||
MemberRegistry registry = MemberRegistry.live(() -> ref.get().fleet());
|
||||
|
||||
MemberLifecycle.SlotReservation reservation = registry.reserve(MemberRole.ARCHITECT, "sonnet");
|
||||
assertTrue(registry.bind(reservation, "term_a"));
|
||||
|
||||
Files.writeString(f, yaml(WITHOUT_ARCHITECT_SLOTS));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertTrue(out.applied(), "the reload must actually take effect: " + out.summary());
|
||||
|
||||
assertTrue(registry.unbind("architect:designer", "term_a"),
|
||||
"unbind must still work for a slot that config has since removed, or a session "
|
||||
+ "that outlives its slot's removal could never release it");
|
||||
assertNull(registry.slotForTerminal("term_a"));
|
||||
assertEquals(Map.of(), registry.snapshot());
|
||||
}
|
||||
}
|
||||
@@ -20,6 +20,7 @@ import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
||||
@@ -884,19 +885,22 @@ class CompletionResolverTest {
|
||||
@Test
|
||||
void coverageIsOffWhenNoProfileHasAPatternConfigured() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [terra])",
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra"), Set.of()));
|
||||
CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
Set.of("terra"), Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsFullWhenEveryProfileHasAPatternConfigured() {
|
||||
assertEquals("full (all profiles configured: [gx10, terra])",
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsPartialAndNamesWhichProfilesAreConfigured() {
|
||||
assertEquals("partial (configured: [terra]; not configured: [gx10])",
|
||||
CompletionResolver.coverage("exhaustedPattern", Set.of("terra", "gx10"), Set.of("terra")));
|
||||
CompletionResolver.coverage("exhaustedPattern", CompletionResolver.UnsetMeaning.OFF,
|
||||
Set.of("terra", "gx10"), Set.of("terra")));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -910,11 +914,42 @@ class CompletionResolverTest {
|
||||
*
|
||||
* <p>Every earlier test here passed the exhaustion case only, so none of them could see it. This
|
||||
* one pins that the message names the key the caller actually meant.
|
||||
*
|
||||
* <p>fleetd#415: the expected wording changed here too. {@code errorPattern} has a built-in
|
||||
* fallback ({@link CompletionResolver#BACKEND_ERROR}), so an empty {@code configuredProfiles}
|
||||
* for it is not "off" — see {@link #coverageDistinguishesOffFromBuiltInDefaultForTheSameEmptyInput}
|
||||
* for the test built specifically to pin that distinction.
|
||||
*/
|
||||
@Test
|
||||
void coverageNamesTheConfigKeyItsCallerMeansRatherThanAlwaysSayingExhaustedPattern() {
|
||||
assertEquals("off (no profile has an errorPattern configured; profiles: [gx10, terra])",
|
||||
CompletionResolver.coverage("errorPattern", Set.of("terra", "gx10"), Set.of()));
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])",
|
||||
CompletionResolver.coverage("errorPattern", CompletionResolver.UnsetMeaning.BUILT_IN_DEFAULT,
|
||||
Set.of("terra", "gx10"), Set.of()));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#415: {@code coverage()} measures pattern coverage (how many profiles set the key), but
|
||||
* for {@code errorPattern} the empty case is not the feature-off state — a profile with no
|
||||
* configured {@code errorPattern} still runs the classification against
|
||||
* {@link CompletionResolver#BACKEND_ERROR}. For {@code exhaustedPattern} there is no fallback,
|
||||
* so empty really is off. Same shape of input (empty {@code configuredProfiles}, one profile),
|
||||
* different {@link CompletionResolver.UnsetMeaning} — the wording must differ, or this method is
|
||||
* back to conflating pattern coverage with feature state for the one key where they disagree.
|
||||
*/
|
||||
@Test
|
||||
void coverageDistinguishesOffFromBuiltInDefaultForTheSameEmptyInput() {
|
||||
String exhaustedLine = CompletionResolver.coverage("exhaustedPattern",
|
||||
CompletionResolver.UnsetMeaning.OFF, Set.of("gx10", "terra"), Set.of());
|
||||
String errorLine = CompletionResolver.coverage("errorPattern",
|
||||
CompletionResolver.UnsetMeaning.BUILT_IN_DEFAULT, Set.of("gx10", "terra"), Set.of());
|
||||
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [gx10, terra])",
|
||||
exhaustedLine);
|
||||
assertEquals("built-in default for all profiles (no profile customises errorPattern; "
|
||||
+ "profiles: [gx10, terra])", errorLine);
|
||||
assertNotEquals(exhaustedLine, errorLine,
|
||||
"the same empty-coverage input must not read as the same feature state for both keys");
|
||||
}
|
||||
|
||||
// --- fleetd#201 Unit 1: target-keyed backend-error pattern + typed sink ----------------------
|
||||
|
||||
@@ -0,0 +1,109 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: {@code fleet_profiles}/{@code GET /profiles} — the reporting surface a
|
||||
* lead actually reads — must let it tell apart the three states {@link
|
||||
* dev.ltms.fleet.peer.PeerLauncher#disabledModels()} alone collapses into one empty set: no
|
||||
* {@code models:} block at all, a block armed with nothing currently off, and a block with N
|
||||
* models off. See {@link dev.ltms.fleet.peer.PeerLauncher.ModelGateState}'s javadoc for why a
|
||||
* bare {@code disabledModels()} read cannot make this distinction, and {@link
|
||||
* FleetMcp#profilesView} for where {@code modelGateArmed} is added alongside the existing {@code
|
||||
* modelsOff} key.
|
||||
*
|
||||
* <p>Every assertion here goes through {@link FleetMcp#profilesView}, never {@code
|
||||
* PeerLauncher.modelGateState()} directly — {@code CompositePeerLauncherTest} already proves the
|
||||
* accessor itself; this class proves the surface a lead reads (fleet_profiles / GET /profiles)
|
||||
* renders what that accessor reports.
|
||||
*/
|
||||
class FleetProfilesModelGateStateTest {
|
||||
|
||||
private static FleetConfig.Profile profile(String name, String model) {
|
||||
return new FleetConfig.Profile(name, "http://gx00.gw:8000", model, null, "FLEETD_WORKER_TOKEN",
|
||||
null, "tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
}
|
||||
|
||||
private static FleetMcp.QuarantineSource noQuarantine() {
|
||||
return new FleetMcp.QuarantineSource(_ -> null, BackendQuarantine.none(), _ -> false);
|
||||
}
|
||||
|
||||
private static HerdrPeerLauncher claudeAdapter(FakeHerdr h, Map<String, FleetConfig.Profile> profiles) {
|
||||
return new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "local", _ -> "tok");
|
||||
}
|
||||
|
||||
/**
|
||||
* State 1: no {@code models:} block at all — a plain {@code ClaudeCodeLauncher} (no {@code
|
||||
* models:} supplier exists for it to read) has nothing to gate against, matching the fleet01
|
||||
* host measured for this ticket: {@code grep -c '^models:' fleetd.yaml} returns 0 there.
|
||||
*/
|
||||
@Test
|
||||
void noModelsBlockReportsGateNotArmed() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", profile("local", "deepseek-v4-flash"));
|
||||
PeerLauncher workers = claudeAdapter(h, profiles);
|
||||
|
||||
Map<String, Object> view = FleetMcp.profilesView(workers, noQuarantine(), FleetMcp.OutageSource.none());
|
||||
|
||||
assertEquals(Boolean.FALSE, view.get("modelGateArmed"),
|
||||
"no models: block to read from — nothing is gated, and nothing can be");
|
||||
assertFalse(view.containsKey("modelsOff"), "nothing configured, so no off set to report either");
|
||||
}
|
||||
|
||||
/** State 2: a {@code models:} block is present, but nothing in it is currently turned off. */
|
||||
@Test
|
||||
void modelsBlockWithNothingOffReportsGateArmedAndZeroOff() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", profile("local", "deepseek-v4-flash"));
|
||||
FleetConfig.Models models = new FleetConfig.Models(
|
||||
List.of(new FleetConfig.Models.ModelEntry("deepseek-v4-flash", true)));
|
||||
PeerLauncher workers = new CompositePeerLauncher(List.of(claudeAdapter(h, profiles)), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(),
|
||||
new BackendOutagePolicy(System::nanoTime), () -> models);
|
||||
|
||||
Map<String, Object> view = FleetMcp.profilesView(workers, noQuarantine(), FleetMcp.OutageSource.none());
|
||||
|
||||
assertEquals(Boolean.TRUE, view.get("modelGateArmed"),
|
||||
"a models: block is present, so the gate is armed even though nothing is off yet");
|
||||
assertFalse(view.containsKey("modelsOff"),
|
||||
"nothing is off, so the key stays absent — an empty list here would be indistinguishable "
|
||||
+ "from today's modelsOff omission, exactly the ambiguity modelGateArmed exists to remove");
|
||||
}
|
||||
|
||||
/** State 3: a {@code models:} block is present with one model currently turned off. */
|
||||
@Test
|
||||
void modelsBlockWithModelsOffReportsGateArmedAndTheOffSet() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("local", profile("local", "deepseek-v4-flash"));
|
||||
FleetConfig.Models models = new FleetConfig.Models(
|
||||
List.of(new FleetConfig.Models.ModelEntry("deepseek-v4-flash", false)));
|
||||
PeerLauncher workers = new CompositePeerLauncher(List.of(claudeAdapter(h, profiles)), "local", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, BackendQuarantine.none(),
|
||||
new BackendOutagePolicy(System::nanoTime), () -> models);
|
||||
|
||||
Map<String, Object> view = FleetMcp.profilesView(workers, noQuarantine(), FleetMcp.OutageSource.none());
|
||||
|
||||
assertEquals(Boolean.TRUE, view.get("modelGateArmed"));
|
||||
assertEquals(List.of("deepseek-v4-flash"), view.get("modelsOff"));
|
||||
}
|
||||
}
|
||||
@@ -1453,6 +1453,108 @@ class CompositePeerLauncherTest {
|
||||
assertEquals(2, adapter.spawnCount("local"));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up, acceptance criterion 2: {@link CompositePeerLauncher#modelGateState()}
|
||||
* is LIVE — no restart — proven through a REAL {@link ConfigRef#reload()}, exactly like {@link
|
||||
* #modelOnOffIsHotReloadedThroughARealConfigRef} above proves for the on/off gate itself. This
|
||||
* single reload sequence walks through all three states the ticket asks for: no {@code models:}
|
||||
* block, a block armed with nothing off, and a block with one model off — so a reload that flips
|
||||
* between any of the three is proven live, not just the on/off edit within an already-armed block.
|
||||
*/
|
||||
@Test
|
||||
void modelGateStateIsHotReloadedThroughARealConfigRef(@TempDir Path dir) throws Exception {
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
""");
|
||||
FleetConfig initial = FleetConfig.load(yaml);
|
||||
ConfigRef configRef = new ConfigRef(yaml, initial);
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr,
|
||||
Map.of("local", stubWorker("local")), "local", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "local", configRef, _ -> 0, BackendQuarantine.none(), NO_OUTAGE);
|
||||
|
||||
PeerLauncher.ModelGateState notConfigured = composite.modelGateState();
|
||||
assertFalse(notConfigured.configured(), "no models: block in the config at all");
|
||||
assertEquals(Set.of(), notConfigured.off());
|
||||
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: false
|
||||
""");
|
||||
assertTrue(configRef.reload().applied(), "adding a models: block must apply live, no restart");
|
||||
PeerLauncher.ModelGateState armedWithOneOff = composite.modelGateState();
|
||||
assertTrue(armedWithOneOff.configured(), "a models: block now exists — the gate is armed");
|
||||
assertEquals(Set.of("deepseek-v4-flash"), armedWithOneOff.off());
|
||||
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
models:
|
||||
allow:
|
||||
- model: deepseek-v4-flash
|
||||
enabled: true
|
||||
""");
|
||||
assertTrue(configRef.reload().applied(), "flipping the entry back on must apply live too");
|
||||
PeerLauncher.ModelGateState armedWithZeroOff = composite.modelGateState();
|
||||
assertTrue(armedWithZeroOff.configured(),
|
||||
"the block is still present — armed and reporting zero, not the same as no block at all");
|
||||
assertEquals(Set.of(), armedWithZeroOff.off());
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up, acceptance criterion 3: the invariant is that an absent {@code
|
||||
* models:} block stays permitted and must never be fatal. Proved, not assumed — a config
|
||||
* without one loads, validates, reports the gate as not configured, AND still spawns normally
|
||||
* (no {@link PlacementException} from a gate that has nothing to check against), using the same
|
||||
* production-shaped {@code Supplier<FleetConfig>} wiring {@code Fleetd.main} actually uses.
|
||||
*/
|
||||
@Test
|
||||
void noModelsBlockConfigStillLoadsAndSpawnsNormally(@TempDir Path dir) throws Exception {
|
||||
Path yaml = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(yaml, """
|
||||
bind:
|
||||
port: 8080
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://local.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
""");
|
||||
FleetConfig cfg = FleetConfig.load(yaml);
|
||||
assertDoesNotThrow(cfg::validateAll, "a config with no models: block must load and validate cleanly");
|
||||
ConfigRef configRef = new ConfigRef(yaml, cfg);
|
||||
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr,
|
||||
Map.of("local", stubWorker("local")), "local", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "local", configRef, _ -> 0, BackendQuarantine.none(), NO_OUTAGE);
|
||||
|
||||
assertFalse(composite.modelGateState().configured());
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("local", null, null)),
|
||||
"no models: block means nothing to gate against — the spawn must go through");
|
||||
assertEquals(1, adapter.spawnCount("local"));
|
||||
}
|
||||
|
||||
/** {@code fleet_profiles}/{@code GET /profiles} must read the exact same live source the gate reads. */
|
||||
@Test
|
||||
void disabledModelsReportsWhatTheGateActuallyEnforces() {
|
||||
|
||||
Reference in New Issue
Block a user