Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 83cac07f6e | |||
| f472e0f782 |
@@ -9,7 +9,6 @@ import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import dev.ltms.bridged.placement.PlacementCandidate;
|
||||
import dev.ltms.bridged.placement.PlacementContext;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.placement.PlacementPolicies;
|
||||
import dev.ltms.bridged.placement.PlacementPolicy;
|
||||
import org.slf4j.Logger;
|
||||
@@ -140,12 +139,8 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String requestedProfile = req.profileName();
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
// An explicit profile bypasses the placement policy, but not the capacity cap: maxLoad
|
||||
// is documented as an unconditional limit on this profile (BridgedConfig.Worker), and
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
// An explicit profile bypasses the policy entirely.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
@@ -195,42 +190,6 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
|
||||
*
|
||||
* <p>maxLoad is a documented, unconditional capacity limit (see {@code BridgedConfig.Worker#maxLoad}),
|
||||
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
|
||||
* ({@link dev.ltms.bridged.placement.PlacementPolicyUtil}) would leave the cap dead config on every
|
||||
* call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
|
||||
*
|
||||
* <p>Deliberately no fallback to another profile: the caller named {@code profile} for a cost/model
|
||||
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
|
||||
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
|
||||
*
|
||||
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
|
||||
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
|
||||
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
|
||||
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
|
||||
* the readiness gate and is a far worse trade.
|
||||
*
|
||||
* @param profile the profile the caller explicitly named
|
||||
* @throws PlacementException when the profile is at capacity
|
||||
*/
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (non-positive ⇒ unlimited at
|
||||
// load), means no cap — never cap what wasn't configured.
|
||||
BridgedConfig.Worker cfg = profileConfigs.get(profile);
|
||||
Integer cap = (cfg == null) ? null : cfg.maxLoad();
|
||||
if (cap == null) {
|
||||
return;
|
||||
}
|
||||
int live = liveCount.apply(profile);
|
||||
if (live >= cap) {
|
||||
throw new PlacementException("worker profile '" + profile + "' is at maxLoad: " + live
|
||||
+ " live >= " + cap + " cap; refusing spawn — no fallback to another profile");
|
||||
}
|
||||
}
|
||||
|
||||
/** Build the candidate list from the configured profiles, in definition order. */
|
||||
private List<PlacementCandidate> candidates() {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
|
||||
@@ -378,54 +378,6 @@ class CompositePeerLauncherTest {
|
||||
assertEquals(1, adapter.spawnCount("b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnAtMaxLoadThrowsPlacementExceptionNamingProfileLiveAndCap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Worker> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 2 : 0);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("a", null, null)));
|
||||
assertTrue(e.getMessage().contains("'a'"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 live"), "message names the live count: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 cap"), "message names the cap: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("a"), "at cap, the spawn is refused before any delegation");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnUnderMaxLoadStillSucceeds() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Worker> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile under its cap accepts an explicit spawn");
|
||||
assertEquals(1, adapter.spawnCount("a"), "the under-cap spawn is delegated");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnWithNullMaxLoadIsNeverCapped() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Worker> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, null),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
// A deliberately absurd live count: an unset maxLoad means unlimited, so it must never refuse.
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> 1000);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile with no maxLoad is never capped, however many live workers");
|
||||
}
|
||||
|
||||
@Test
|
||||
void emptyCandidateSetThrowsClearException() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
# CB-500 — Multi-Tier Coordination (Stage 6)
|
||||
|
||||
**Status:** design note (proposal — ticket split deferred)
|
||||
**Status:** design note. Developments A/B remain proposals; Development C (§6 and Figures 7–8) is
|
||||
**SUPERSEDED** by the advisory-architect design in Gitea issue #16 and the `architects:` configuration
|
||||
block (CB-548).
|
||||
**Depends on:** CB-401/402 (Peer Launcher SPI + composite router — placement-neutral spawn),
|
||||
CB-308 (per-agent broker channels + global id + federated roster — the addressing substrate),
|
||||
CB-307 (durable inbox + push loop), CB-301/303 (session FSM + context-cap/idle-ttl), CB-304
|
||||
@@ -16,8 +18,9 @@ workers) into a **multi-tier** one, along three axes the lead has asked for:
|
||||
1. **Sandboxed workers** — each worker runs in a **separated, peer-owned sandbox** carrying its own
|
||||
toolchain (Claude routed via `ANTHROPIC_BASE_URL`, a headless IDE, git, MCP, dev-tools), with
|
||||
**per-role** sandboxes (a backend-agent image, a frontend-agent image).
|
||||
2. **Main-agent pairs** — the "main" tier becomes a **pair** (on-subscription Opus + one cloud
|
||||
module) collaborating, instead of a lone primary.
|
||||
2. **Main-agent pairs** — **SUPERSEDED.** The considered model made the "main" tier a pair
|
||||
(on-subscription Opus + one cloud module). The actual fleet is one human-driven lead plus two
|
||||
short-lived advisory architects on different model families.
|
||||
3. **An orchestrator tier** — a supervisor **above** the mains that owns their **session identity**
|
||||
(naming, resume) and **curates context**, so every main→worker delegation carries the *exact*
|
||||
slice of context it needs and nothing else.
|
||||
@@ -63,6 +66,10 @@ Four concrete bake-ins assume a single tier:
|
||||
|
||||
## 3. Target multi-tier architecture
|
||||
|
||||
> **SUPERSEDED fleet sketch.** Figure 2 records the former two-main model. The actual fleet is one
|
||||
> lead, two independent advisory architects, and N workers; architects are sideways peers, not leads
|
||||
> and not a tier above the lead. See Gitea issue #16 and the `architects:` block.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
human["human"]
|
||||
@@ -93,10 +100,9 @@ flowchart TB
|
||||
chan --- roster
|
||||
```
|
||||
|
||||
*Figure 2 — three tiers. Tier 0 owns the mains' session identity + context scope; Tier 1 is a
|
||||
collaborating pair, each an MCP client with its own pull inbox; Tier 2 is peer-owned sandboxes the
|
||||
bus launches into. The middle is CB-308's per-agent-channel + federated-roster substrate, now
|
||||
carrying tier-to-tier traffic, not just host-to-host.*
|
||||
*Figure 2 — **SUPERSEDED historical fleet sketch.** It proposed a collaborating pair of managed mains.
|
||||
The actual fleet keeps one human-driven lead and uses two independent, short-lived advisory architects
|
||||
on different model families, so agreement is evidence rather than correlated echo.*
|
||||
|
||||
The recursion is the key idea: **`orchestrator : mains :: main : workers`** — the same
|
||||
spawn/name/resume/scope verbs at two levels.
|
||||
@@ -166,6 +172,11 @@ gateway, because herdr keystroke-injection needs a locally-owned PTY.**
|
||||
|
||||
## 5. Development B — Main-agent pairs
|
||||
|
||||
> **SUPERSEDED — do not implement this model.** The two-main fleet was replaced by one human-driven
|
||||
> lead and two independent advisory architects. They are deliberately different model families (Claude
|
||||
> Sonnet 5 and GPT-5.6 through opencode), receive the same brief, and work independently so agreement
|
||||
> is evidence rather than correlated echo. See Gitea issue #16 and `architects:`.
|
||||
|
||||
Both mains are MCP **clients**, so **neither can be called into** — each needs a **pull-based
|
||||
per-agent inbox**, which is precisely CB-308 item #1 (per-agent AMQP channels). The primary machinery
|
||||
that is singular today (single-slot `PrimaryRegistry`, a push-loop aimed at one terminal, "these
|
||||
@@ -221,6 +232,19 @@ push-loop fan-out; relax "orchestration tools only the primary calls" to "any re
|
||||
|
||||
## 6. Development C — Orchestrator tier
|
||||
|
||||
> **SUPERSEDED — do not implement this model.** The operator rejected a supervisor above the lead.
|
||||
> The human continues to drive the pre-existing lead directly; bridged neither spawns nor resumes that
|
||||
> lead. What replaced this proposal is **one lead, two short-lived advisory architects, and N workers**:
|
||||
> the lead engages architects sideways for a strong-model assessment, then discards them. Architect
|
||||
> slots are declared in `architects:` (see Gitea issue #16), rather than making leads managed sessions.
|
||||
> The two architects deliberately use different model families — Claude Sonnet 5 and GPT-5.6 through
|
||||
> opencode — and receive the same brief independently. Agreement is evidence, not correlated echo
|
||||
> from one provider or one conversation.
|
||||
|
||||
> **Historical alternative retained.** The text and figures below record the considered model and why it
|
||||
> was rejected: it re-rooted the human-facing session above the lead, violating the still-true premise
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by bridged.
|
||||
|
||||
The orchestrator is **`SessionManager` recursed one tier up**: today it spawns/names/reaps *worker*
|
||||
sessions; the orchestrator does the same for *main* sessions, and adds **context scoping**.
|
||||
|
||||
@@ -249,11 +273,10 @@ flowchart TB
|
||||
m2 -->|"scoped delegation"| w
|
||||
```
|
||||
|
||||
*Figure 7 — the recursion. Tiers 1 and 2 run the identical spawn/name/resume machinery; the
|
||||
orchestrator merely operates it one level higher. **Re-rooting caveat:** today the primary IS the
|
||||
human's live session; here the human drives the orchestrator, and the mains become managed,
|
||||
resumable sessions. That moves the human-facing top up a tier — an intentional re-root, not an
|
||||
add-on.*
|
||||
*Figure 7 — **SUPERSEDED historical alternative.** The recursion re-rooted the human-facing session:
|
||||
the human drove an orchestrator and the mains became managed, resumable sessions. The operator rejected
|
||||
that re-root. The replacement keeps the human-driven, pre-existing lead and engages architects sideways
|
||||
as short-lived advisory peers; see Gitea issue #16 and `architects:`.*
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
@@ -270,10 +293,9 @@ sequenceDiagram
|
||||
O->>O: fold into orchestrator context, pick next main/turn
|
||||
```
|
||||
|
||||
*Figure 8 — context focus. The orchestrator holds the global context and hands each main only the
|
||||
slice a given delegation needs, so the main→worker conversation stays on-point. Context *scoping* is
|
||||
coordination (the bus already owns session/turn lifecycle) — it stays inside the identity boundary
|
||||
(§7), unlike toolchain ownership which does not.*
|
||||
*Figure 8 — **SUPERSEDED historical alternative.** This proposed an orchestrator holding global context
|
||||
and slicing it for managed mains. The replacement has the human-driven lead send the same advisory brief
|
||||
issue #16 and `architects:`.*
|
||||
|
||||
**Deltas:** a second, higher `SessionManager` instance whose "peers" are mains; the orchestrator
|
||||
becomes the top MCP client; context-slice selection (new) layered on CB-303's `context_cap` +
|
||||
@@ -315,7 +337,7 @@ flowchart LR
|
||||
cb402["CB-401/402<br/>Peer Launcher SPI + composite<br/>(DONE / in-flight)"]
|
||||
A["A · SandboxLauncher<br/>(placement-neutral, independent)"]
|
||||
cb308["CB-308 substrate<br/>per-agent channels + global id<br/>+ federated roster"]
|
||||
B["B · main-agent pair<br/>(multi-slot PrimaryRegistry)"]
|
||||
B["B · main-agent pair (SUPERSEDED)<br/>(multi-slot PrimaryRegistry)"]
|
||||
C["C · orchestrator tier<br/>(SessionManager recursed up)"]
|
||||
cb402 --> A
|
||||
cb402 --> cb308
|
||||
@@ -333,7 +355,7 @@ flowchart LR
|
||||
2. **A · SandboxLauncher** — independent; a second proof of the SPI (placement-neutral). Ships anytime.
|
||||
3. **CB-308 substrate** — per-agent channels + global id + federated roster (the multi-host work,
|
||||
promoted from host-to-host to tier-to-tier).
|
||||
4. **B · main-agent pair** — multi-slot `PrimaryRegistry` + per-main inbox routing, on the substrate.
|
||||
4. **B · main-agent pair** — **SUPERSEDED** by lead + two advisory architects.
|
||||
5. **C · orchestrator tier** — the capstone; the recursive session manager + context scoping.
|
||||
|
||||
## 9. Open questions (to resolve at ticket-split)
|
||||
@@ -341,8 +363,8 @@ flowchart LR
|
||||
- **Sandbox mechanism:** container (`docker exec`) vs devcontainer — how the role→image mapping is
|
||||
expressed on the profile. *(Topology **resolved** in §11: distributed = gateway-per-host × local
|
||||
sandboxes; the remaining choice is only the local launch mechanism, not the shape.)*
|
||||
- **Pair semantics:** are the two mains fully symmetric peers, or is one a co-primary that may also
|
||||
delegate? Affects how `PrimaryRegistry` and the "orchestration tools" identity relax.
|
||||
- **Pair semantics:** **SUPERSEDED.** The two-main question is replaced by the architect role's
|
||||
least-privilege boundary: advisory architects can send/reply/ask/read but cannot spawn/stop/drain.
|
||||
- **Orchestrator drivenness:** the mains become programmatically spawned/resumed — does the human
|
||||
still ever type directly into a main, or only into the orchestrator? (The re-root caveat, Fig 7.)
|
||||
- **Context-slice selection:** who decides the slice — orchestrator heuristics, explicit tool args,
|
||||
|
||||
+1
-1
Submodule wiki updated: 8c63db5da6...e5424f4665
Reference in New Issue
Block a user