CB-553: enforce maxLoad on explicit-profile spawns (no cap bypass)

This commit is contained in:
Dai Ha
2026-08-13 21:12:32 +02:00
parent 293a305748
commit 3a7ef0adbd
2 changed files with 90 additions and 1 deletions
@@ -9,6 +9,7 @@ import dev.ltms.bridged.peer.PeerUnreachableException;
import dev.ltms.bridged.peer.SpawnRequest;
import dev.ltms.bridged.placement.PlacementCandidate;
import dev.ltms.bridged.placement.PlacementContext;
import dev.ltms.bridged.placement.PlacementException;
import dev.ltms.bridged.placement.PlacementPolicies;
import dev.ltms.bridged.placement.PlacementPolicy;
import org.slf4j.Logger;
@@ -139,8 +140,12 @@ public final class CompositePeerLauncher implements PeerLauncher {
public PeerHandle spawn(SpawnRequest req) {
String requestedProfile = req.profileName();
if (requestedProfile != null && !requestedProfile.isBlank()) {
// An explicit profile bypasses the policy entirely.
// An explicit profile bypasses the placement policy, but not the capacity cap: maxLoad
// is documented as an unconditional limit on this profile (BridgedConfig.Worker), and
// the charter makes explicit-profile spawns the normal path — so skipping the check
// here would leave the cap dead config in real operation.
HerdrPeerLauncher d = route(requestedProfile);
enforceMaxLoad(requestedProfile);
PeerHandle handle = d.spawn(req);
spawnedBy.put(handle.id(), d);
return handle;
@@ -190,6 +195,42 @@ public final class CompositePeerLauncher implements PeerLauncher {
+ " candidate(s): " + String.join(", ", unreachable));
}
/**
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
*
* <p>maxLoad is a documented, unconditional capacity limit (see {@code BridgedConfig.Worker#maxLoad}),
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
* ({@link dev.ltms.bridged.placement.PlacementPolicyUtil}) would leave the cap dead config on every
* call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
*
* <p>Deliberately no fallback to another profile: the caller named {@code profile} for a cost/model
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
*
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
* the readiness gate and is a far worse trade.
*
* @param profile the profile the caller explicitly named
* @throws PlacementException when the profile is at capacity
*/
private void enforceMaxLoad(String profile) {
// Absent config, or a config whose maxLoad normalized to null (non-positive ⇒ unlimited at
// load), means no cap — never cap what wasn't configured.
BridgedConfig.Worker cfg = profileConfigs.get(profile);
Integer cap = (cfg == null) ? null : cfg.maxLoad();
if (cap == null) {
return;
}
int live = liveCount.apply(profile);
if (live >= cap) {
throw new PlacementException("worker profile '" + profile + "' is at maxLoad: " + live
+ " live >= " + cap + " cap; refusing spawn — no fallback to another profile");
}
}
/** Build the candidate list from the configured profiles, in definition order. */
private List<PlacementCandidate> candidates() {
List<PlacementCandidate> out = new ArrayList<>();