Compare commits
281 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 847e8bd3fa | |||
| a55079afbd | |||
| e18ad4723b | |||
| 6de8ac8972 | |||
| a052975420 | |||
| 6d82ca95a4 | |||
| 8067ee4ec4 | |||
| 3743789e8d | |||
| 388aba7632 | |||
| ad587eafa3 | |||
| 08ce9aef11 | |||
| 4ac688b6d9 | |||
| a814d1ef00 | |||
| ea98856130 | |||
| a49e96835a | |||
| 63c19dcba7 | |||
| 2823349c8e | |||
| 6fc301d62c | |||
| bc99d64786 | |||
| 615af4ed0a | |||
| 045d229728 | |||
| d1fd5700f5 | |||
| 2d55b0b9a5 | |||
| d89ae94a2e | |||
| e6193c4098 | |||
| b66f0677ed | |||
| 23ada1981e | |||
| a22480c117 | |||
| 24f404f989 | |||
| a237fbff9d | |||
| 31d5516991 | |||
| 17af61e8dd | |||
| 6af87b6ad6 | |||
| 5ba05d0bdb | |||
| fc655e78c2 | |||
| 25726a5ae7 | |||
| 11c3ff67b6 | |||
| d867c87100 | |||
| b0c4cedfab | |||
| 85417d5215 | |||
| 4accc746bd | |||
| 21c4c8cbef | |||
| 430f5b0dae | |||
| 42731833d0 | |||
| 65acf066ad | |||
| ee8f570fd7 | |||
| fa3f910d44 | |||
| 46ac6e4e38 | |||
| 3bfa82839b | |||
| 65ccf2e4ad | |||
| 7a3b27f76f | |||
| 82e7be564c | |||
| c5e24197bf | |||
| bcb402b688 | |||
| 7c4170ff6d | |||
| a6095743f0 | |||
| 66e776d178 | |||
| 26f64cba45 | |||
| 51047848f1 | |||
| d8c0b657e8 | |||
| 312c0584ce | |||
| da2625acfb | |||
| 97d9cebc59 | |||
| 3ce76a5d69 | |||
| 2e138a199b | |||
| 450a5ed9c5 | |||
| edabccd885 | |||
| 1dbe3a03fc | |||
| 6058b8472b | |||
| f29968c334 | |||
| 7b98cca967 | |||
| 2757bc7185 | |||
| 7655f1b51a | |||
| a5efb7c676 | |||
| 5a8cf4cb4d | |||
| a3fc7e4df8 | |||
| d811b30df3 | |||
| 3f4ac2b24e | |||
| 4644359128 | |||
| 95a8dbcea9 | |||
| 83b50753fe | |||
| 6f968d59a4 | |||
| d432df8e5c | |||
| 4b822731e6 | |||
| e38eac1a33 | |||
| 8101290933 | |||
| 6e7fc12f89 | |||
| 50df14a50f | |||
| ccd882f2fc | |||
| ecc590f344 | |||
| 3b7cdf9365 | |||
| 9b50dd69d8 | |||
| 08f1a79800 | |||
| 51bdec22e7 | |||
| d43f670285 | |||
| c135583402 | |||
| 8d2893b67e | |||
| 555715ced9 | |||
| 446a9d395c | |||
| 8311f2db7f | |||
| f4570ff274 | |||
| 3aea4e1ec6 | |||
| 516366b796 | |||
| d82d0157cb | |||
| 1deb0c90d4 | |||
| 41a3114d03 | |||
| 76e4b577a0 | |||
| 03473a286b | |||
| 5cabd09705 | |||
| cc47672b7c | |||
| 37edd9134b | |||
| 80f167b1f7 | |||
| bd6547fca3 | |||
| 7e97f5bff5 | |||
| 7d41ccccee | |||
| bf616e192a | |||
| f8bd5d0c51 | |||
| ac40de1d30 | |||
| 7930a31b94 | |||
| 850fb12807 | |||
| b14b66ab03 | |||
| 15ff6bcde5 | |||
| 7822772905 | |||
| 0efe1567c0 | |||
| 837fed7690 | |||
| aa4ee64a34 | |||
| bb750cdba3 | |||
| d56c77b368 | |||
| 83e2ff06cf | |||
| f5deaafd06 | |||
| fe46311266 | |||
| 08968bb1b7 | |||
| 27bbd11f06 | |||
| 48d7841fbf | |||
| cc1df11f69 | |||
| fdfd4ac491 | |||
| d5dd5639ae | |||
| 7a120b3256 | |||
| 863d477966 | |||
| cec48832be | |||
| a36b7ccd7c | |||
| 32bf324a1e | |||
| 16de9df000 | |||
| 65c9deb4d1 | |||
| 28ae27b8e1 | |||
| 81a0cf4710 | |||
| 8a837a2830 | |||
| 0a2b3a4a56 | |||
| 613ece92dc | |||
| 8d4206c2b5 | |||
| 03286a589b | |||
| 88b9503c3b | |||
| c553d795d8 | |||
| 3ba6d6784c | |||
| 78ca24dc3f | |||
| 4fa6553db5 | |||
| 2124e043ce | |||
| e4f3620acb | |||
| 032a59a34d | |||
| e689090024 | |||
| 1cc34888fd | |||
| 0331ecd5d3 | |||
| 831a918c30 | |||
| 3db5277ae8 | |||
| 6939e0cbbc | |||
| f0095bf8b2 | |||
| 5206679efd | |||
| ac044e7573 | |||
| 6ebad2a91f | |||
| 3d10ed385c | |||
| 6d0c94dbdb | |||
| e01563a550 | |||
| 30e3225a3c | |||
| ef186a1516 | |||
| 5d5b3bdc76 | |||
| 2f8c98dac9 | |||
| 2db7189067 | |||
| 94476ac109 | |||
| 5fe02b7c98 | |||
| a1052f4fd1 | |||
| 8c9904a7c4 | |||
| 23af5dfdff | |||
| 3a10f6ad17 | |||
| a3842c873d | |||
| c29c3f063d | |||
| 758d62a396 | |||
| 6725642274 | |||
| a1f4dc365a | |||
| 165b62ee20 | |||
| 72d3481de3 | |||
| 29ccb747ad | |||
| 4749e27453 | |||
| e501d39988 | |||
| 6b6cf25862 | |||
| 180de840eb | |||
| 8fd2d7e5e7 | |||
| 129dd4a838 | |||
| e09cac6f1f | |||
| c00a86b32c | |||
| d3ae0350a2 | |||
| 2c2196a1f1 | |||
| 35ade14630 | |||
| 541df87272 | |||
| 337b6ccd6e | |||
| 42f46dfe9a | |||
| 9088d2b2c5 | |||
| 500bfa2c33 | |||
| 9ca9c43dfa | |||
| b525b0f08f | |||
| 1966c69994 | |||
| 976eff8ad1 | |||
| 0af902ec43 | |||
| 9118ce2537 | |||
| 2f48e08f1f | |||
| fec284e7cb | |||
| 8e2e4c5e73 | |||
| 0edc6615fc | |||
| b745e159de | |||
| aac29d604c | |||
| c884802b13 | |||
| 5f5573a24e | |||
| 74b0087ebb | |||
| 927e0151d4 | |||
| 5275922d1d | |||
| e186c7945a | |||
| 33a6e77f0e | |||
| 4e47489d53 | |||
| c76b2f149e | |||
| 826fffe05b | |||
| 75f57cdba7 | |||
| f556af5d4e | |||
| d5f33f0c6e | |||
| 16e17b32ad | |||
| 988494e18b | |||
| c4549a5e20 | |||
| 46fa4f38d5 | |||
| 20e0e68ad7 | |||
| 17468a234a | |||
| 8a53d5bfc6 | |||
| 695da7418e | |||
| abd26c796b | |||
| 24559d81ac | |||
| 6c1c2c3994 | |||
| 01fab15713 | |||
| 0cd00e71c3 | |||
| bf0ff2adbf | |||
| ed4bbc1c56 | |||
| c456402cc5 | |||
| 4f0bf667b1 | |||
| 61af9aa574 | |||
| 3b59b34e76 | |||
| 65b38997f7 | |||
| 7510f7649c | |||
| 619792a81c | |||
| cea1183f75 | |||
| e2af4c5ae4 | |||
| dfd5f82894 | |||
| e81944cef6 | |||
| 7bdd39ab9a | |||
| f4b38f040e | |||
| 799668e129 | |||
| 890190263e | |||
| f8522edacd | |||
| b958747855 | |||
| 078bde2c02 | |||
| 0b10ea987b | |||
| 1553d38182 | |||
| d04b075996 | |||
| 644927636d | |||
| 95e45007aa | |||
| 7a583c4045 | |||
| 65bce058bb | |||
| 48b437083b | |||
| fa0612859b | |||
| 4ffbcd0b7d | |||
| a0cd053fd9 | |||
| f9a5e066b5 | |||
| e9bc192160 | |||
| e0a57988ad | |||
| b414a74c26 | |||
| 2c3796d598 |
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the bridged MCP gateway.",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
@@ -8,7 +8,7 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the bridged MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"description": "Make a project bridge-ready: mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
name: architect
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
name: dev
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,274 @@
|
||||
---
|
||||
name: fleets-status
|
||||
description: Report the status of every fleet that shares one LavinMQ instance. Use for local daemon health, broker-wide fleet presence, and cross-host lead coordination checks.
|
||||
---
|
||||
|
||||
# Status of every fleet on the shared LavinMQ instance
|
||||
|
||||
**The headline: always report what is missing.** This skill starts with the local fleet, then adds
|
||||
broker-wide facts when its read-only credential exists. A missing fleet must appear as `unknown` or
|
||||
`not reachable`, with the reason and the fix. Never leave it out.
|
||||
|
||||
The known topology has one LavinMQ instance on `10.10.20.13` (`fleet01`). AMQP uses port `5672`,
|
||||
and the management API uses port `15672`. The Mac fleet owns vhost `/mac`. The fleet01 fleet owns
|
||||
vhost `/fleet01`.
|
||||
|
||||
## 1. Protect credentials before any probe
|
||||
|
||||
**Hard rule — never print `LAVINMQ_URI`.** It is an AMQP URI with its password inline. It only
|
||||
resolves in a login shell because `${SHARED_ENV}/tools/secrets.sh` supplies it. A non-login shell
|
||||
can make every broker probe look empty.
|
||||
|
||||
- Never run `echo "$LAVINMQ_URI"`.
|
||||
- Never put `${LAVINMQ_URI:-something}` in output. That form expands to the secret value when set.
|
||||
- Parse the user, host, and password into shell or Python variables. Use them without printing them.
|
||||
- Prefer `resolves` or `does not resolve` over any part of the value.
|
||||
- Every command that can read `LAVINMQ_URI` must send all output through this redaction before it
|
||||
reaches the report:
|
||||
|
||||
```bash
|
||||
sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
```
|
||||
|
||||
**The `g` flag is not optional.** Without it `sed` replaces only the first match on each line, so a
|
||||
line carrying two URIs leaks the second one. `scripts/redeploy-fleetd.sh --check` prints lines like
|
||||
that. Checked on 2026-08-27: without `g`, `amqp://u1:p1@h1/mac and http://u2:p2@h2:15672/api`
|
||||
redacts the first pair and prints `u2:p2` in the clear.
|
||||
|
||||
Keep `pipefail` on when applying that filter. Otherwise the filter can hide a failed probe. Apply
|
||||
the same no-print rule to the management password below, even though it is not in an AMQP URI.
|
||||
|
||||
## 2. Tier 1 — this fleet (always run)
|
||||
|
||||
Start here even when the broker tier is blocked. Work from the local fleetd checkout.
|
||||
|
||||
First run the read-only deployment check. It already checks the daemon process, deployed jar versus
|
||||
the checkout `HEAD`, launchd state, and whether each configured token resolves in a login shell.
|
||||
Do not copy those checks into new shell code. The script reads `LAVINMQ_URI`, so redact all output:
|
||||
|
||||
```bash
|
||||
set -o pipefail
|
||||
scripts/redeploy-fleetd.sh --check 2>&1 \
|
||||
| sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
git rev-parse HEAD
|
||||
```
|
||||
|
||||
Treat jar drift as a top-level warning. A merge is not a deployment. State the running jar result
|
||||
as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into a match.
|
||||
|
||||
Report the process identifier (PID) and uptime too:
|
||||
|
||||
```bash
|
||||
PIDS="$(pgrep -f 'target/fleetd.jar' || true)"
|
||||
if [ -z "$PIDS" ]; then
|
||||
printf '%s\n' 'fleetd: not running'
|
||||
else
|
||||
for PID in $PIDS; do
|
||||
ps -p "$PID" -o pid=,etime=,lstart=,command=
|
||||
done
|
||||
fi
|
||||
```
|
||||
|
||||
Read the full health response. Keep the HTTP status because `503` means fleetd is running but herdr
|
||||
is not reachable. Report both `herdr.version` and `herdr.protocol` when present:
|
||||
|
||||
```bash
|
||||
curl -sS --max-time 5 -w '\nHTTP %{http_code}\n' http://127.0.0.1:8765/healthz
|
||||
```
|
||||
|
||||
Call `fleet_whoami`, then call `fleet_list`. Preserve its sections in the report:
|
||||
|
||||
- `leads`, including which row is this lead;
|
||||
- `members`, including state, role, profile, branch, and worktree when present;
|
||||
- every per-profile `capacity` row, including `maxLoad`, `live`, `free`, and quarantine facts;
|
||||
- the exact `healthCoverage` value.
|
||||
|
||||
Do not describe an empty `members` list as an empty fleet. It says only that no members are spawned.
|
||||
Also do not hide a profile with `free: 0`; say whether load or credential quarantine caused it.
|
||||
|
||||
Show only WARN, ERROR, and SEVERE lines after the last `fleetd listening` line. This anchor stops an
|
||||
old incident from looking current:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
path = Path("fleetd/fleetd.out")
|
||||
if not path.exists():
|
||||
print("cannot check current WARN/ERROR: fleetd/fleetd.out does not exist")
|
||||
else:
|
||||
lines = path.read_text(errors="replace").splitlines()
|
||||
starts = [i for i, line in enumerate(lines) if "fleetd listening" in line]
|
||||
if not starts:
|
||||
print("cannot anchor WARN/ERROR: no 'fleetd listening' line exists")
|
||||
else:
|
||||
current = lines[starts[-1]:]
|
||||
alerts = [line for line in current if re.search(r"\b(?:WARN|ERROR|SEVERE)\b", line)]
|
||||
print(f"current WARN/ERROR/SEVERE count: {len(alerts)}")
|
||||
for line in alerts[-50:]:
|
||||
print(line)
|
||||
PY
|
||||
```
|
||||
|
||||
**What this tier cannot see:** it proves facts only about the Mac daemon at `127.0.0.1:8765`.
|
||||
It cannot show the fleet01 daemon, broker queue depth, or broker consumers. The fleet01 REST service
|
||||
at `10.10.20.13:8765` is not reachable from the Mac. Say this in the report rather than omitting
|
||||
fleet01.
|
||||
|
||||
**But fleet01 IS reachable over SSH — checked 2026-08-28.** An older version of this line said SSH
|
||||
was denied. That is true only for the user `dai.ha`. The host alias `fleet01` maps to user `ltms`,
|
||||
and `ssh fleet01` works with key auth:
|
||||
|
||||
```bash
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=6 fleet01 'echo $(id -un)@$(hostname)'
|
||||
```
|
||||
|
||||
So fleet01's daemon PID, uptime, jar and `/healthz` **can** be reported — over SSH, not over REST.
|
||||
Do that rather than writing `not reachable`. `ltms` also has passwordless sudo there.
|
||||
|
||||
## 3. Tier 2 — the shared broker (run when management access exists)
|
||||
|
||||
**This tier is blocked today.** The AMQP user in `LAVINMQ_URI` can connect on port `5672`, but gets
|
||||
HTTP `401` from the management API on port `15672`. An AMQP connection does not grant monitoring
|
||||
access.
|
||||
|
||||
The operator must create a separate, read-only LavinMQ management user with the `monitoring` tag.
|
||||
It needs access to inspect both `/mac` and `/fleet01`. Store its values as
|
||||
`LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD` in
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Do not reuse or print the AMQP URI. Full multi-fleet status stays
|
||||
blocked until this user exists.
|
||||
|
||||
When both variables resolve, run this from a login shell. It calls `GET /api/overview`,
|
||||
`GET /api/vhosts`, `GET /api/queues`, and `GET /api/connections`. It prints selected status fields,
|
||||
but never the user, password, Authorization header, or AMQP URI:
|
||||
|
||||
```bash
|
||||
zsh -lc 'python3 - "$@"' -- <<'PY'
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
base = "http://10.10.20.13:15672"
|
||||
user = os.environ.get("LAVINMQ_MANAGEMENT_USER", "")
|
||||
password = os.environ.get("LAVINMQ_MANAGEMENT_PASSWORD", "")
|
||||
if not user or not password:
|
||||
print("broker tier: BLOCKED — management credential does not resolve in a login shell")
|
||||
sys.exit(0)
|
||||
|
||||
token = base64.b64encode(f"{user}:{password}".encode()).decode()
|
||||
|
||||
def get(path):
|
||||
request = urllib.request.Request(
|
||||
base + path,
|
||||
headers={"Authorization": "Basic " + token, "Accept": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=5) as response:
|
||||
return json.load(response)
|
||||
|
||||
try:
|
||||
overview = get("/api/overview")
|
||||
vhosts = get("/api/vhosts")
|
||||
queues = get("/api/queues")
|
||||
connections = get("/api/connections")
|
||||
except urllib.error.HTTPError as error:
|
||||
print(f"broker tier: BLOCKED — management API returned HTTP {error.code}")
|
||||
sys.exit(0)
|
||||
except Exception as error:
|
||||
print(f"broker tier: BLOCKED — management API is not reachable: {type(error).__name__}")
|
||||
sys.exit(0)
|
||||
|
||||
fleet_names = {"/mac": "Mac fleet", "/fleet01": "fleet01 fleet"}
|
||||
print(json.dumps({
|
||||
"overview": {
|
||||
"lavinmq_version": overview.get("lavinmq_version"),
|
||||
"rabbitmq_version": overview.get("rabbitmq_version"),
|
||||
"queue_totals": overview.get("queue_totals", {}),
|
||||
"object_totals": overview.get("object_totals", {}),
|
||||
},
|
||||
"fleets": [
|
||||
{
|
||||
"fleet": fleet_names.get(vhost.get("name"), "UNKNOWN FLEET"),
|
||||
"vhost": vhost.get("name"),
|
||||
"queues": [
|
||||
{
|
||||
"name": queue.get("name"),
|
||||
"messages": queue.get("messages", 0),
|
||||
"messages_ready": queue.get("messages_ready", 0),
|
||||
"messages_unacknowledged": queue.get("messages_unacknowledged", 0),
|
||||
"consumers": queue.get("consumers", 0),
|
||||
}
|
||||
for queue in queues if queue.get("vhost") == vhost.get("name")
|
||||
],
|
||||
"connections": [
|
||||
{
|
||||
"name": connection.get("name"),
|
||||
"peer_host": connection.get("peer_host"),
|
||||
"state": connection.get("state"),
|
||||
}
|
||||
for connection in connections if connection.get("vhost") == vhost.get("name")
|
||||
],
|
||||
}
|
||||
for vhost in vhosts
|
||||
],
|
||||
}, indent=2, sort_keys=True))
|
||||
PY
|
||||
```
|
||||
|
||||
Map `/mac` to the Mac fleet and `/fleet01` to the fleet01 fleet. Keep any other vhost in the
|
||||
report as `UNKNOWN FLEET`; do not drop it. For each vhost, total the ready, unacknowledged, and all
|
||||
messages. Report every queue's consumer count and each live connection.
|
||||
|
||||
**A vhost with queues but zero consumers means that fleet's daemon is down while its durable state
|
||||
survives. Call this out as a top-level warning.** This is the main reason to use the management API
|
||||
instead of calling each remote daemon.
|
||||
|
||||
**What this tier cannot see:** without the new `monitoring` credential it cannot enumerate any
|
||||
vhost, queue, depth, consumer, or connection. With the credential it still cannot report fleet01's
|
||||
daemon PID, uptime, jar revision, `/healthz`, herdr version, or member capacity. Those need reachable
|
||||
fleet01 REST or SSH access, which the Mac does not have today.
|
||||
|
||||
## 4. Tier 3 — cross-fleet lead coordination
|
||||
|
||||
Use the queue data from Tier 2. Select queues whose names match `lead.<coordId>.inbox`. Report each
|
||||
queue's vhost, depth, consumer count, and the `coordId` between the prefix and suffix.
|
||||
|
||||
- A lead inbox with a consumer shows that a lead mailbox is live on that vhost.
|
||||
- A durable lead inbox with zero consumers shows saved coordination state, but no live receiver.
|
||||
- No lead inbox is not proof that coordination is disabled. The daemon may be down before declaring
|
||||
its queue, or this account may not be allowed to see the vhost.
|
||||
|
||||
This Mac fleet currently sets both `broker.uriEnv` and `coordinator.uriEnv` to the same variable,
|
||||
`LAVINMQ_URI`. Therefore its coordinator connects to `/mac`. Cross-host `fleet_send{coordId}` routes
|
||||
only when both leads share the same coordinator vhost. If the fleet01 lead uses `/fleet01` for its
|
||||
coordinator, the leads cannot see each other and the send will not route.
|
||||
|
||||
**Open question:** the fleet01 coordinator vhost has not been checked. Surface this question in
|
||||
every report until a live `lead.<coordId>.inbox` consumer or fleet01's config proves the answer. Do
|
||||
not claim that fleet01 uses `/fleet01` just because its member queues do.
|
||||
|
||||
Also compare these broker facts with the `leads` rows from local `fleet_list`. A missing remote lead
|
||||
is `not visible from this coordinator`, not `down`, unless the broker consumer facts prove it.
|
||||
|
||||
**What this tier cannot see:** without Tier 2 management access it cannot list lead inboxes or their
|
||||
consumers. Even with that access, a stopped fleet01 daemon leaves only durable queue history. That
|
||||
history cannot prove which coordinator URI its current config would use after restart.
|
||||
|
||||
## 5. Report all fleets
|
||||
|
||||
Use one row per known or discovered fleet. Include blocked rows.
|
||||
|
||||
| Fleet | Daemon | Deployment | Herdr | Members/capacity | Queues/consumers | Lead coordination | Cannot check |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| Mac (`/mac`) | PID + uptime | jar vs `HEAD` | health + version + protocol | `fleet_list` + `healthCoverage` | facts or blocked reason | inbox facts or open question | exact missing facts |
|
||||
| fleet01 (`/fleet01`) | reachable/down/unknown | value or `not reachable` | value or `not reachable` | value or `not reachable` | facts or blocked reason | inbox facts plus coordinator-vhost question | exact missing facts and fix |
|
||||
|
||||
Add rows for unknown vhosts. End with three short sections: `Current warnings`, `Checks that were
|
||||
blocked`, and `Operator action`. Until the management user exists, `Operator action` must say:
|
||||
|
||||
> Create a read-only LavinMQ management user with the `monitoring` tag and access to `/mac` and
|
||||
> `/fleet01`. Put its user and password in `${SHARED_ENV}/tools/secrets.sh` as
|
||||
> `LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD`.
|
||||
@@ -1,11 +1,11 @@
|
||||
---
|
||||
name: implementer
|
||||
description: Implementer-role procedure for a bridged worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over bridged.
|
||||
description: Implementer-role procedure for a fleetd worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over fleetd.
|
||||
---
|
||||
|
||||
# Implementer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge, never commit `.mcp.json` or `wiki/`) is in **`CLAUDE.md` → Bridge communication →
|
||||
Worker** and already applies. This skill is only the *implement-and-hand-off procedure*.
|
||||
|
||||
@@ -49,7 +49,7 @@ test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-t
|
||||
your worktree*:
|
||||
|
||||
```bash
|
||||
cd "$(git rev-parse --show-toplevel)/bridged" && mvn clean install
|
||||
cd "$(git rev-parse --show-toplevel)/fleetd" && mvn clean install
|
||||
echo "exit=$?"
|
||||
```
|
||||
|
||||
@@ -85,7 +85,7 @@ host (`GITEA_HOST`) into your env for exactly this — the token can create a PR
|
||||
merge**.
|
||||
|
||||
```bash
|
||||
API="${GITEA_HOST%/}/api/v1/repos/lms/claude-bridge/pulls"
|
||||
API="${GITEA_HOST%/}/api/v1/repos/fleet/fleetd/pulls"
|
||||
BRANCH="$(git branch --show-current)"
|
||||
curl -sS -X POST "$API" \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
@@ -103,7 +103,7 @@ fix it if the cause is yours (e.g. branch not pushed yet), and report the failur
|
||||
inventing a URL. If `GITEA_TOKEN` is unset your profile was not granted PR-create: push the branch
|
||||
and report its name so the lead opens the PR.
|
||||
|
||||
## 6. Hand off — what goes in `bridge_reply`
|
||||
## 6. Hand off — what goes in `fleet_reply`
|
||||
|
||||
The reply is the entire handoff; the lead cannot see your terminal.
|
||||
|
||||
@@ -130,7 +130,7 @@ sequenceDiagram
|
||||
I->>G: git push -u origin HEAD
|
||||
I->>G: POST /pulls (GITEA_TOKEN) — open PR to main
|
||||
G-->>I: html_url
|
||||
I->>L: bridge_reply(PR url, branch, files, tests)
|
||||
I->>L: fleet_reply(PR url, branch, files, tests)
|
||||
Note over L,G: the lead reviews the PR and merges on green — you never merge
|
||||
```
|
||||
|
||||
|
||||
@@ -53,7 +53,7 @@ Map each entry from `.mcp.json`:
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"instructions": ["CLAUDE.md"],
|
||||
"mcp": {
|
||||
"bridged": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"fleetd": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"context7": { "type": "remote", "url": "https://example.dev/mcp", "enabled": true,
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" } },
|
||||
"gitea": { "type": "local", "command": ["gitea-mcp", "-t", "stdio"], "enabled": true,
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Reviewer-role procedure for a bridged worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over bridged.
|
||||
description: Reviewer-role procedure for a fleetd worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over fleetd.
|
||||
---
|
||||
|
||||
# Reviewer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies. This
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
@@ -24,13 +24,13 @@ wrong.
|
||||
covers what it was given.
|
||||
- Do **not** edit files or run the build. You review; the owner acts.
|
||||
|
||||
## 3. Reach for `bridge_ask` only for a genuine fork
|
||||
## 3. Reach for `fleet_ask` only for a genuine fork
|
||||
|
||||
Ambiguous requirement, a missing acceptance criterion, "intended or a bug?", or two defensible
|
||||
fixes with different consequences — those are the lead's call, and guessing produces a
|
||||
confident-but-wrong finding. Anything you could settle by reading more code is yours to settle.
|
||||
|
||||
## 4. The finding — what goes in `bridge_reply`
|
||||
## 4. The finding — what goes in `fleet_reply`
|
||||
|
||||
Report the **single most important** real issue in the scope, in these four lines, under
|
||||
~90 words:
|
||||
|
||||
@@ -14,7 +14,7 @@ jobs:
|
||||
# does not depend on the wiki repo being reachable.
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The runner image ships an older default-jdk; bridged sets maven.compiler.release=25, so
|
||||
# The runner image ships an older default-jdk; fleetd sets maven.compiler.release=25, so
|
||||
# provision the JDK explicitly rather than apt-installing whatever "default" means today.
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
@@ -32,7 +32,7 @@ jobs:
|
||||
mvn -version
|
||||
|
||||
- name: Build and test
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
# This IS the mock-socket surface CB-503 asks for: the pom's `default-excludes` profile
|
||||
# already sets excludedGroups=contract, so the @Tag("contract") tests — which need a live
|
||||
# herdr socket and a RabbitMQ container — are excluded without any flag here. Everything
|
||||
@@ -46,7 +46,7 @@ jobs:
|
||||
# the log instead, where they are actually readable.
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||
# straight to it — no Docker, no skipped tests. This separation (build job hermetic and
|
||||
# Docker-free; contract job broker-provided) is deliberate — see the default-excludes/contract
|
||||
# profiles in bridged/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# profiles in fleetd/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# exactly as in the build job above.
|
||||
contract:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -69,8 +69,6 @@ jobs:
|
||||
env:
|
||||
RABBITMQ_DEFAULT_USER: guest
|
||||
RABBITMQ_DEFAULT_PASS: guest
|
||||
ports:
|
||||
- 5672:5672
|
||||
env:
|
||||
# Service containers are reachable from the job by their network alias on their internal port.
|
||||
AMQP_URI: amqp://guest:guest@rabbitmq:5672
|
||||
@@ -93,12 +91,12 @@ jobs:
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
- name: Contract tests
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
|
||||
+12
-4
@@ -6,8 +6,16 @@
|
||||
# Settings backups inherit the env block — and secrets with it.
|
||||
.claude/settings.local.json.bak*
|
||||
|
||||
# Daemon runtime artefacts. bridged appends its log wherever it is launched from, so both the
|
||||
# repo root and bridged/ collect one; neither belongs in git.
|
||||
bridged.out
|
||||
bridged/bridged.out
|
||||
# The default profile parityOverlay copies these primary→worktree, so they appear in EVERY worker
|
||||
# worktree. Two reasons they must be ignored. They hold environment values, which is reason enough.
|
||||
# And since CB-576 a release preserves any worktree that `git status --porcelain` calls dirty —
|
||||
# untracked files included, deliberately. An untracked overlay file would therefore make every
|
||||
# COMPLETED release preserve its worktree, and worktrees would pile up with no error to notice.
|
||||
.env
|
||||
.envrc
|
||||
|
||||
# Daemon runtime artefacts. fleetd appends its log wherever it is launched from, so both the
|
||||
# repo root and fleetd/ collect one; neither belongs in git.
|
||||
fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
[submodule "wiki"]
|
||||
path = wiki
|
||||
url = ssh://git@git.ltms.dev:2224/lms/claude-bridge.wiki.git
|
||||
url = ssh://git@git.ltms.dev:2224/fleet/fleetd.wiki.git
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -4,48 +4,52 @@
|
||||
|
||||
> **Canonical block.** Everything down to §Layering is the portable bridge charter, copied verbatim
|
||||
> into every project that mounts the bridge MCP. Keep it byte-identical with the template in the
|
||||
> wiki ([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
|
||||
If no `bridge_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
`bridged` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **worker**) mount the *same* MCP server and talk only
|
||||
through its `bridge_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
`fleetd` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **member**) mount the *same* MCP server and talk only
|
||||
through its `fleet_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
|
||||
### Which role am I? — settle this before acting
|
||||
|
||||
**Both roles read this file.** A worker runs in a git worktree of this same repo, so it inherits
|
||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||
|
||||
**Call `bridge_whoami`.** It returns `{"role":"primary"}` or `{"role":"worker","sessionId":…,
|
||||
"profile":…,"worktree":…,"branch":…}`, resolved by the daemon from your connection — unforgeable,
|
||||
and the same resolution its authorization gate uses. Don't infer what you can ask.
|
||||
**Call `fleet_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
||||
was bound to. Don't infer what you can ask.
|
||||
|
||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||
fires: the reply charter in your system prompt (*"You are an off-subscription worker in the
|
||||
claude-bridge fleet"*) ⇒ **worker**; bridge tools prefixed `mcp__bridge__*` ⇒ **worker** (the
|
||||
launcher fixes that mount name; a primary's mount is named by whoever wrote its `.mcp.json`, so it
|
||||
varies); `ANTHROPIC_BASE_URL` set ⇒ **worker** (Claude-model workers run on a clean env, so its
|
||||
*absence* proves nothing). **Still unsure ⇒ act as a worker.** The two mistakes are not symmetric: a
|
||||
primary acting as a worker is refused by the authorization gate — loud and self-correcting — while a
|
||||
worker acting as the primary ends its turn with no `bridge_reply`, and the sender silently receives
|
||||
nothing. Fail toward the recoverable error.
|
||||
fires: the reply charter in your system prompt (*"You are a spawned member in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; fleet tools prefixed `mcp__fleet__*` ⇒ **spawned
|
||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||
`.mcp.json`, so it varies — and a member spawned before CB-632 still says `mcp__bridge__*`); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
||||
only `fleet_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
### Invariants — both roles, no exceptions
|
||||
|
||||
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
||||
stays on subscription; only the bridge puts a worker off it, at spawn. Mounting the bridge must
|
||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||
never move a session across that boundary.
|
||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||
side cannot see your screen. An answer that isn't in a `bridge_*` call is silently discarded.
|
||||
side cannot see your screen. An answer that isn't in a `fleet_*` call is silently discarded.
|
||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||
cannot act as another session. Spawn/stop/send/drain are lead-only; reply/ask are
|
||||
only-as-itself — any peer may answer for its own pane, and for no other. A call outside your
|
||||
role is refused, not queued.
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
||||
outside your role is refused, not queued.
|
||||
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`/`blocked`.
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
|
||||
@@ -53,10 +57,10 @@ nothing. Fail toward the recoverable error.
|
||||
|
||||
**Delegate by default — that is the job.** With the bridge mounted you are an orchestrator on a
|
||||
metered subscription, and workers are cheap, parallel, and disposable. The default answer to "who
|
||||
does this?" is **a worker**, not you. Reach for `bridge_send` before you reach for `Edit`. The steps
|
||||
does this?" is **a worker**, not you. Reach for `fleet_send` before you reach for `Edit`. The steps
|
||||
below are the procedure — run them in order, every task, not only the big ones.
|
||||
|
||||
0. **Know your role** — `bridge_whoami`, once per session, before anything else.
|
||||
0. **Know your role** — `fleet_whoami`, once per session, before anything else.
|
||||
1. **Split.** Write the unit list. Every unit carries: scope · the files or PR in question ·
|
||||
acceptance criteria · exactly what to report back. A unit with no acceptance criteria is not
|
||||
ready to delegate — refine it or keep it.
|
||||
@@ -66,20 +70,22 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
delegate. The keep-list is closed: the conversation with the user, decomposition and planning,
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `bridge_spawn{profile, worktree:true, ticket}`, one per
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
4. **Then send them all** — `bridge_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
brief instead. The brief is self-contained — the worker sees your message and the repo, nothing
|
||||
of your context, your plan, or your screen.
|
||||
5. **Collect** — `bridge_poll{ticket}` → `bridge_ack{ticket, msgId}`. Answer a worker's `bridge_ask`
|
||||
with `bridge_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`bridge_status`, never by reading its terminal.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker mounts only the bridge MCP and
|
||||
cannot run your other tooling, and a piped command (`… | tail`) hides failures behind a zero
|
||||
exit — never promote a worker's "clean" to a fact.
|
||||
5. **Collect** — `fleet_poll{ticket}` → `fleet_ack{target, msgId}`. Answer a worker's `fleet_ask`
|
||||
with `fleet_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||
@@ -87,11 +93,11 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
read it yourself.
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `bridge_stop{paneId}`.
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
prefer `wait:false` + `bridge_poll` for anything non-trivial: a blocking `bridge_send` is capped by
|
||||
prefer `wait:false` + `fleet_poll` for anything non-trivial: a blocking `fleet_send` is capped by
|
||||
*your own* MCP client call timeout (~60s), well below the task's real runtime.
|
||||
|
||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||
@@ -99,28 +105,29 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
|
||||
| Intent | Tool |
|
||||
|---|---|
|
||||
| Confirm your own role | `bridge_whoami` |
|
||||
| See backends available | `bridge_profiles` |
|
||||
| Start a worker | `bridge_spawn{profile?, cwd?, worktree?, ticket?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `bridge_list` → `leads` (your peers) + `workers` · one peer's state: `bridge_status{sessionId}` |
|
||||
| Delegate (blocking) | `bridge_send{sessionId, content}` |
|
||||
| Delegate (long task) | `bridge_send{sessionId, content, wait:false}` → ticket → `bridge_poll{ticket}` |
|
||||
| Answer a worker's `bridge_ask` | `bridge_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** | `bridge_send{sessionId: <their terminal>, content}` — `bridge_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `bridge_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `bridge_poll{target}` · then `bridge_ack{target, msgId}` |
|
||||
| Tear down | `bridge_stop{paneId}` |
|
||||
| Confirm your own role | `fleet_whoami` |
|
||||
| See backends available | `fleet_profiles` |
|
||||
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `fleet_status{sessionId}` |
|
||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
`bridge_list` returns `leads` alongside `workers`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own workers, and its own judgment. An empty
|
||||
`workers` array means no workers are spawned; it says nothing about peers.
|
||||
`fleet_list` returns `leads` alongside `members`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own members, and its own judgment. An empty
|
||||
`members` array means no members are spawned; it says nothing about peers.
|
||||
|
||||
**A lead never assigns a task to another lead.** Work goes to workers — only ever downward, never
|
||||
sideways. Sending a peer a brief with acceptance criteria is a category error: a brief is a worker's
|
||||
**A lead never assigns a task to another lead.** Work goes to members — only ever downward, never
|
||||
sideways. Sending a peer a brief with acceptance criteria is a category error: a brief is a member's
|
||||
artefact, and a peer is not yours to task. If a unit needs doing and it falls in your area, spawn a
|
||||
worker and delegate it yourself; if it falls in the peer's area, say so and let the peer assign it.
|
||||
member and delegate it yourself; if it falls in the peer's area, say so and let the peer assign it.
|
||||
The traffic between leads is coordination and nothing else:
|
||||
|
||||
1. **Divide the map, not the work.** Agree who owns which area, then each of you assigns inside your
|
||||
@@ -134,22 +141,30 @@ The traffic between leads is coordination and nothing else:
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `bridge_reply`, and push back on
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
|
||||
### Worker — the turn contract
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
1. **Load the playbook skill the lead named** before doing anything else.
|
||||
2. **Do the assigned scope only.** Note anything you spot outside it in one line; don't go hunt it.
|
||||
3. **`bridge_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||
Don't ask what you could decide yourself.
|
||||
4. **End the turn with exactly one `bridge_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `bridge_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
5. **Report honestly.** State only what you actually ran and its real output, including failures.
|
||||
You mount **only** the bridge MCP — the primary's other servers (IDE, forge, docs) are not yours,
|
||||
so never claim the result of a check you had no way to run.
|
||||
4. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||
lead with its end cut off.
|
||||
5. **Report honestly.** State only what you actually ran and its real output, including failures,
|
||||
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
||||
find there holds a deliberately blocked credential and fails every call, by design.
|
||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
@@ -157,29 +172,92 @@ you.
|
||||
|
||||
| Layer | Scope | Reaches |
|
||||
|---|---|---|
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `bridge_reply`* | every worker, at launch, every peer kind |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every Claude worker — tracked in git, so worktrees inherit it |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a worker told to load one |
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `fleet_reply`* | every spawned member, at launch, every peer kind — never a lead |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every member that reads the repo — tracked in git, so worktrees inherit it |
|
||||
| role agent definition files | role contract and per-job procedure | a member whose launcher binds its role to the matching file in its worktree |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a member told to load one |
|
||||
| the bridge's own docs | design detail, flows, error model | on demand |
|
||||
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. Peers that don't read
|
||||
`CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they* must obey belongs in the
|
||||
charter, not here.
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. A member without a
|
||||
repo checkout still gets the launcher's reply charter, which is why that one rule stays there.
|
||||
Peers that don't read `CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they*
|
||||
must obey belongs in the charter, not here.
|
||||
|
||||
## Project addendum — claude-bridge (not part of the canonical block)
|
||||
|
||||
- **This repo is the bridge.** The daemon is `bridged`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/BridgeMcp` (tools), `auth/Authz` (the role table),
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace).
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace) and
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `bridge_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` §6, kept out of this file because it loads into every
|
||||
session's context.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
@@ -192,15 +270,15 @@ Before you call any work done, check the row that matches what you touched:
|
||||
|
||||
| You changed… | Re-read and update… |
|
||||
|---|---|
|
||||
| a `bridge_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| a `fleet_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| `Authz` / the role table | invariant 3, and the primary-only vs worker-only claims |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `bridge_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__bridge__*`), and the layering table's top row |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `fleet_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__fleet__*`), and the layering table's top row |
|
||||
| the injector / status gating | invariant 4 |
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `bridged.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
|
||||
That last row is not bookkeeping. Chapters 1–10 answer *how is this built* and *why this way*;
|
||||
none of them has a home for *what can it do and how do I turn it on*, so for twenty tickets a
|
||||
@@ -211,7 +289,7 @@ is a Roadmap line. A change that touches none of the three earns no entry, and t
|
||||
outcome rather than an omission.
|
||||
|
||||
Then **propagate**: the block in this file and the template in the wiki
|
||||
([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
block*) must stay byte-identical, and other projects carrying the block need the same edit. Verify
|
||||
rather than trust:
|
||||
|
||||
@@ -234,10 +312,10 @@ PY
|
||||
|
||||
Two IDE MCP servers are connected: **intellij-index** (semantic code intelligence) and
|
||||
**jetbrains** (file problems, reformat, debugger). IntelliJ has multiple projects open; our
|
||||
module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
module is **`fleetd`**. Always pass these to IDE MCP tools:
|
||||
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/bridged`
|
||||
- IDE paths are relative to `bridged/` (e.g. `src/main/java/dev/ltms/bridged/...`)
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/fleetd`
|
||||
- IDE paths are relative to `fleetd/` (e.g. `src/main/java/dev/ltms/fleet/...`)
|
||||
|
||||
### After editing any file — mandatory
|
||||
|
||||
@@ -251,7 +329,7 @@ module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
whole-project gate before declaring work done or committing.
|
||||
|
||||
**Whenever dependencies change (or a `pom.xml` edit), validate CVEs with
|
||||
`jetbrains get_file_problems{filePath: "bridged/pom.xml"}`** — its Mend.io check reflects the
|
||||
`jetbrains get_file_problems{filePath: "fleetd/pom.xml"}`** — its Mend.io check reflects the
|
||||
dependencies on disk. (Note: `ide_diagnostics` / intellij-index does NOT re-resolve dependencies
|
||||
after a pom edit without a full Maven reimport, so it reports stale CVE results — don't trust it
|
||||
for this.) Treat a CVE warning like any other: bump to a patched version and confirm
|
||||
|
||||
@@ -9,32 +9,33 @@ Sibling of [`crush-bridge`](https://git.ltms.dev/systems/vms) (which drives a he
|
||||
process*, so it inherits `CLAUDE.md`, hooks, skills, and MCP — just pointed at a
|
||||
cheaper/local model.
|
||||
|
||||
## Leading approach — herdr-centric message server (`bridged`)
|
||||
## Leading approach — herdr-centric message server (`fleetd`)
|
||||
|
||||
A small always-on message server, **`bridged`**, controls
|
||||
A small always-on message server, **`fleetd`**, controls
|
||||
[herdr](https://herdr.dev) (an agent multiplexer) over its Unix-socket API and exposes a
|
||||
clean 2-way messaging API as an **MCP server that both the primary and the workers mount** —
|
||||
one unified Claude setup and the **sole communication gateway** (REST/SSE stays for non-Claude
|
||||
clients; any broker is `bridged`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `bridged` owns
|
||||
clients; any broker is `fleetd`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `fleetd` owns
|
||||
policy (subscription boundary, session lifecycle, status-gated delivery) and the client
|
||||
contract. The worker `claude` launches with `ANTHROPIC_BASE_URL=https://ollama.ltms.dev` + a
|
||||
bearer token; the primary Opus stays env-clean and calls `bridged`'s MCP tools.
|
||||
contract. A Claude member launches with `ANTHROPIC_BASE_URL` pointed at the gateway,
|
||||
`https://llm.ltms.dev/anthropic`, plus a bearer token; the lead stays env-clean and calls
|
||||
`fleetd`'s MCP tools. See the wiki's **[13 User Guide](wiki/13-User-Guide.md)** to run it.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["bridged — standalone daemon (not a claude process)"]
|
||||
subgraph BD["fleetd — standalone daemon (not a claude process)"]
|
||||
SRV["SERVER face<br/>MCP · REST/SSE · policy"]
|
||||
CLI["CLIENT face<br/>status-gated injector · herdr socket"]
|
||||
SRV --> CLI
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
M["ollama.ltms.dev<br/>(worker model)"]
|
||||
M["llm.ltms.dev<br/>(the one gateway)"]
|
||||
|
||||
OPUS -->|"MCP bridge_send (blocks)"| SRV
|
||||
W -.->|"MCP bridge_reply"| SRV
|
||||
OPUS -->|"MCP fleet_send (blocks)"| SRV
|
||||
W -.->|"MCP fleet_reply"| SRV
|
||||
CLI -->|"Unix socket<br/>send_text · events.subscribe"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
W -->|"inference"| M
|
||||
@@ -46,24 +47,26 @@ flowchart LR
|
||||
```
|
||||
|
||||
- **Subscription boundary:** the *primary* never sets `ANTHROPIC_BASE_URL` (stays on
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `bridged` itself is a
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `fleetd` itself is a
|
||||
plain daemon (no Anthropic quota), so it may poll/subscribe freely.
|
||||
- **One gateway (unified MCP setup):** `bridged` is the **sole communication path** for every
|
||||
- **One gateway (unified MCP setup):** `fleetd` is the **sole communication path** for every
|
||||
Claude session. Primary and workers each mount it as an MCP server (one `claude mcp add`
|
||||
line, same on both) and talk over MCP tools — `bridge_send` / `bridge_reply` /
|
||||
`bridge_status` (with `bridge_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `bridged`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`bridge_send`);
|
||||
`bridged` holds it open until the worker calls `bridge_reply` or its turn hits
|
||||
line, same on both) and talk over MCP tools — `fleet_send` / `fleet_reply` /
|
||||
`fleet_status` (with `fleet_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `fleetd`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
**Tool naming:** the tools are `fleet_*` (renamed from `bridge_*` in CB-622). The old
|
||||
`bridge_*` names were removed in CB-634 — only `fleet_*` answers now.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`fleet_send`);
|
||||
`fleetd` holds it open until the worker calls `fleet_reply` or its turn hits
|
||||
`agent_status=done`, then returns the reply as the tool result. No cross-turn busy-poll, so
|
||||
no quota burn. SSE is an optional side-channel for humans/dashboards watching status.
|
||||
- **Worker → primary** rides `bridged`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `bridged` **injects the primary's idle pane** when it's
|
||||
- **Worker → primary** rides `fleetd`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `fleetd` **injects the primary's idle pane** when it's
|
||||
ready), so *no keystroke-into-primary and no broker are involved, even single-host*. The one
|
||||
exception: a split-host primary that isn't a herdr pane wakes via its own `Stop`-hook, which
|
||||
polls **`bridged`** (never a broker). See the wiki for the two topologies.
|
||||
polls **`fleetd`** (never a broker). See the wiki for the two topologies.
|
||||
- **Different model per process** sidesteps Claude Code's lack of per-subagent provider
|
||||
routing — the worker isn't a subagent, it's its own configured process.
|
||||
- **AgentAPI** ([`coder/agentapi`](https://github.com/coder/agentapi)) is retained only as a
|
||||
@@ -72,11 +75,11 @@ flowchart LR
|
||||
|
||||
## Docs
|
||||
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/lms/claude-bridge/wiki)**,
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/fleet/fleetd/wiki)**,
|
||||
vendored here as a submodule under [`wiki/`](./wiki):
|
||||
|
||||
```bash
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/lms/claude-bridge.git
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/fleet/fleetd.git
|
||||
# or, after a plain clone:
|
||||
git submodule update --init
|
||||
```
|
||||
@@ -86,7 +89,7 @@ Gitea wiki.
|
||||
|
||||
## Status
|
||||
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`bridged`** message server is built and in
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`fleetd`** message server is built and in
|
||||
real use: an Opus primary delegates tasks to off-subscription workers that reply through the
|
||||
bridge (code reviews delegated this way have produced committed bug fixes). Selected as the
|
||||
primary approach 2026-07-11, superseding the AgentAPI plan (2026-07-08); AgentAPI retained as a
|
||||
@@ -97,9 +100,9 @@ tests run separately via `mvn test -Pcontract`):
|
||||
|
||||
- **Core gateway** — herdr socket client (contract-tested vs live 0.7.0); guard-checked worker
|
||||
spawn with `ANTHROPIC_BASE_URL` injected only into the worker's env; status-gated injector;
|
||||
blocking `bridge_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `bridge_send` / `bridge_reply` / `bridge_status` (messaging) and `bridge_spawn`
|
||||
/ `bridge_list` / `bridge_stop` / `bridge_profiles` / `bridge_poll` (fleet). Caller identity is
|
||||
blocking `fleet_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `fleet_send` / `fleet_reply` / `fleet_status` (messaging) and `fleet_spawn`
|
||||
/ `fleet_list` / `fleet_stop` / `fleet_profiles` / `fleet_poll` (fleet). Caller identity is
|
||||
connection-based (loopback peer PID → herdr pane), so the same mount serves primary and workers.
|
||||
- **Delivery reliability** — completion fallback (a confirmed `working→idle` turn resolves a
|
||||
send); async fire-and-poll (beats the caller's MCP call timeout for long tasks); and failure
|
||||
@@ -107,7 +110,7 @@ tests run separately via `mvn test -Pcontract`):
|
||||
- **Fleet** — multiple worker profiles, each with an independent base_url guard check; workers
|
||||
inherit the primary's working directory (never `$HOME`); a readiness gate holds delivery until
|
||||
a worker's Claude has connected the bridge MCP (no paste lost into its boot window).
|
||||
- **Blocked-worker path** — `bridge_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
- **Blocked-worker path** — `fleet_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
ask the primary and resumes the *same* turn with the answer (CB-205).
|
||||
- **Session lifecycle** — session manager with spawn/reuse/recycle, `idle_ttl` reaper, `context_cap`,
|
||||
and graceful drain on shutdown (CB-301/CB-303); per-worker git worktrees on their own branch with
|
||||
|
||||
@@ -1,14 +0,0 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from bridged.example.yaml)
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -1,391 +0,0 @@
|
||||
# bridged configuration (example). Copy to bridged.yaml and adjust.
|
||||
#
|
||||
# bridged is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — bridged REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
|
||||
# API authentication (CB-501). Governs how a caller that is NOT an on-host worker pane proves it
|
||||
# is the primary. Worker identity never depends on this: a loopback peer PID that maps to a herdr
|
||||
# pane is unforgeable and is always honoured, so turning auth on cannot lock the fleet out.
|
||||
#
|
||||
# mode: loopback-trust → DEFAULT, and the historical behaviour: any loopback caller that is not
|
||||
# a worker is the primary, no credential needed. Sound ONLY because the
|
||||
# OS refuses remote connections to a loopback socket.
|
||||
# mode: token → such a caller must send `Authorization: Bearer <token>`; without it it
|
||||
# is anonymous and authorized for nothing. REQUIRED for a non-loopback
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# BRIDGED_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
# location / { proxy_pass http://127.0.0.1:8765; proxy_set_header Authorization $http_authorization; }
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: BRIDGED_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from bridge_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
# pushReminders: 5 # max nudges before giving up (default 5)
|
||||
# pushBackoffMs: 15000 # delay between nudges (default 15000)
|
||||
|
||||
# CB-530: MORE THAN ONE LEAD. `primary:` above is singular by construction — every other pane
|
||||
# resolves as a worker — which is right for one lead driving a fleet and wrong the moment two leads
|
||||
# (say a Claude lead and an opencode lead) work as peers: the second is silently demoted and refused
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# terminal → the ONLY field identity depends on; get it from that session's bridge_whoami
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by bridge_whoami
|
||||
#
|
||||
# A lead is never spawned — it pre-exists, which is exactly why it must be named rather than created.
|
||||
# `bridge_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
# destination for its nudges. If both name the same terminal, the `fleet.leaders:` entry wins.
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing bridged places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open bridge_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
# Opt-in on purpose — it SPENDS the operator's subscription on its own initiative (each nudge starts
|
||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||
# block = feature off, exactly as before.
|
||||
#
|
||||
# Three knobs, each with a default that errs on the side of not burning context:
|
||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||
# leadHeartbeat:
|
||||
# idleAfterSeconds: 300
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
# it is for; that is the member's role, and roles live under `fleet:` below. Which profile an
|
||||
# unqualified spawn lands on comes from that role's pool, not from a global default.
|
||||
#
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → bridged mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
# cwd on an MCP spawn, else the daemon's cwd — never $HOME. See
|
||||
# docs/Worker-Startup-and-Trust.md.
|
||||
# configDir → CLAUDE_CONFIG_DIR for the worker, so it inherits that profile's
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.claude/settings.local.json, .env, .envrc].
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
# handed a worker the primary's IDE servers, which are bound to the primary's
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# bridged neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
# Opt-in by design — omit and the worker gets no PR-create grant (push over
|
||||
# SSH is unaffected). The token value itself is never stored in this file.
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. bridged hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# bridged now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.bridged.plist and deploy/bridged.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
# entry cannot repoint a worker past the SubscriptionGuard — which is checked
|
||||
# against `baseUrl` alone.
|
||||
# Put `defaultMode: "auto"` in each ccs profile so the worker runs autonomously.
|
||||
profiles:
|
||||
gx10: # ccs profile name (NOT a hostname)
|
||||
kind: claude-code # which adapter spawns this profile (default; may omit)
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: BRIDGED_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
weight: 0.5 # relative selection weight for placement: weighted
|
||||
maxLoad: 2 # max live workers on this profile (omit for unlimited)
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".claude/settings.local.json", ".env", ".envrc"] # never add .mcp.json — see above
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
# Pin an auto-compact window BELOW the served model's context ceiling. The global
|
||||
# ~/.claude/settings.json value is shared by every ccs instance and the primary, so the
|
||||
# per-profile override belongs here. Equal to the ceiling means auto-compact never fires
|
||||
# before the server rejects the prompt, which kills a worker mid-turn (CB-523).
|
||||
env:
|
||||
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "280000"
|
||||
# CB-402: a second coding-agent kind, proving the PeerLauncher SPI is provider-neutral.
|
||||
# opencode is provider-agnostic and uses NONE of Claude's private seams: no ANTHROPIC_BASE_URL /
|
||||
# SubscriptionGuard (so it needs no `guard` host entry), no --mcp-config / --append-system-prompt.
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → bridge_send →
|
||||
# structured bridge_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
#
|
||||
# CB-508: point an opencode profile at your OWN OpenAI-compatible endpoint (local vLLM, llama.cpp,
|
||||
# LM Studio, TGI…) instead of opencode's gateway. Setting `baseUrl` on a `kind: opencode` profile
|
||||
# makes the bridge emit a custom `provider` block into the generated opencode.json — opencode has
|
||||
# no ANTHROPIC_BASE_URL seam, so this is how the endpoint is pinned.
|
||||
# baseUrl → a bare host:port gets `/v1` appended (where these servers mount the API); a URL that
|
||||
# already has a path is used verbatim, so a custom mount point still works.
|
||||
# model → MUST be "<provider>/<model>". The provider half names the generated block; the model
|
||||
# half must match an id the server reports at /v1/models. One field drives both the
|
||||
# declaration and the `-m` flag, so they cannot drift apart. A bare model name with a
|
||||
# baseUrl set is rejected at spawn rather than silently using the default gateway.
|
||||
# tokenEnv → optional; its value becomes the provider apiKey. Most local servers ignore the key,
|
||||
# so a placeholder is used when unset (the AI SDK still requires a non-empty one).
|
||||
# NOTE: no `guard` entry is needed even with a baseUrl set. The SubscriptionGuard exists to stop a
|
||||
# worker borrowing the primary's Anthropic subscription, and an opencode process has no Anthropic
|
||||
# credential path at all.
|
||||
# opencode-local:
|
||||
# kind: opencode
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
# How an unqualified spawn chooses a profile: fixed (default, reproduces pre-CB-518 behaviour),
|
||||
# round-robin, or weighted. Omitting this key is a strict no-op for existing configs.
|
||||
placement: weighted
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded bridged keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. bridged checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
# Not every key can move under a running daemon, and the difference is about what already exists
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool and
|
||||
# `tabLabel`), `placement:`, and an existing profile's weight / maxLoad. Those are
|
||||
# hot because the placement policy reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# DEFERRED → accepted into the new config, but the wiring built at startup keeps the old value
|
||||
# until you restart: `lifecycle:`, `leadHeartbeat:`, `guard:`, `worktreeRoot:`,
|
||||
# `spawnReadyTimeoutMs` / `spawnReadyPollMs`, ADDING or REMOVING a profile (a new
|
||||
# backend needs its own launcher, and launchers are built once), AND an existing
|
||||
# profile's launch settings — model, baseUrl, argv, env, configDir, mcpUrl, tabLabel.
|
||||
# The launcher takes a copy of `profiles:` at startup and resolves every spawn out of
|
||||
# that copy, so those never reach a launch until you restart. The reload logs them by
|
||||
# name rather than pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
# bound, the broker connection is open, and the auth mode decides who may reach the
|
||||
# port that is already listening.
|
||||
#
|
||||
# A changed COLD key refuses the WHOLE reload — not the hot half applied and the cold half warned
|
||||
# about. A half-applied reload would leave the daemon matching no file on disk, which is the worst
|
||||
# thing a reload can do to an operator debugging one. A file that fails to parse or fails a startup
|
||||
# validator is refused the same way, and the running config stays live.
|
||||
# configReload:
|
||||
# enabled: true
|
||||
# intervalSeconds: 10
|
||||
|
||||
# THE FLEET (CB-557) — who the daemon may run, and under which role. This one block replaced four
|
||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||
#
|
||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
||||
# file, the playbook skill and the authz row.
|
||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||
# which is why the two cannot be one field.
|
||||
#
|
||||
# The ROLE IS THE CONTAINING KEY, not a `role:` field. That is not only tidier: a misspelled role
|
||||
# used to parse into a member with no contract at all, while a misspelled pool name here simply
|
||||
# declares nothing.
|
||||
#
|
||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
||||
# with being listed here; the entry key just names the entry.
|
||||
fleet:
|
||||
# Optional. Template for a member tab's label; {role}, {profile}, {model} and {n} are substituted.
|
||||
# {n} counts per role+profile, so `dev: sonnet #2` really is the second sonnet dev. Because {role}
|
||||
# comes from a closed enum, a generated label can never begin with a lead's tabPrefix.
|
||||
# tabLabel: "{role}: {profile} #{n}"
|
||||
|
||||
# Panes that orchestrate rather than are orchestrated. A lead may now be CREATED as well as
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Give it only a `terminal:` and it is recognise-only, as before.
|
||||
#
|
||||
# `tabPrefix` is the naming convention that finds a lead without pasting a terminal id: label the
|
||||
# tab `lead: <name>` when you open it and the pane is recognised on the next rescan. Reopen the
|
||||
# tab later and the id changes; the label does not.
|
||||
#
|
||||
# A lead the daemon launches is labelled BY the daemon, using the same convention, so it is found
|
||||
# by the same scan. A lead counts as live only when herdr also reports a running agent in that
|
||||
# tab — a label left behind by a session that died does not block the relaunch.
|
||||
#
|
||||
# An auto-launched lead is NOT a member: it gets no worker reply charter, is never registered with
|
||||
# the session lifecycle (the idle reaper would kill your orchestrator), and stays on the
|
||||
# subscription — ANTHROPIC_BASE_URL/AUTH_TOKEN are stripped from its env whatever the profile says.
|
||||
# leaders:
|
||||
# opus-5.0:
|
||||
# profile: opus # omit to never create this lead, only recognise it
|
||||
# instances: 1 # desired live count; only the shortfall is launched. 0 = off
|
||||
# terminal: term_0123456789abcd # optional hand-pin; usually found by tabPrefix instead.
|
||||
# # A running agent on this terminal also counts as live, so a
|
||||
# # lead you opened by hand is not relaunched under you.
|
||||
# tabPrefix: "lead:" # `lead: opus-5.0` ⇒ a lead named opus-5.0 (case-insensitive)
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: bridged's own)
|
||||
# kind: claude # descriptive; reported by bridge_whoami
|
||||
# gpt-sol-5.6:
|
||||
# terminal: term_fedcba9876543
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
# architect-2:
|
||||
# profile: sol # a different vendor on purpose — two architects that share a
|
||||
# # model share its blind spots
|
||||
developers:
|
||||
gx10:
|
||||
profile: gx10
|
||||
# reviewers:
|
||||
# gx10:
|
||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Every profile above must have its host listed here.
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- gx01.gw
|
||||
|
||||
# Spawn-readiness gate (CB-306). The launcher blocks until the worker's herdr status is
|
||||
# injectable (IDLE/BLOCKED/DONE) or the timeout elapses. 0 disables the gate.
|
||||
# NOTE: keys are camelCase — config is bound by plain Jackson with no naming strategy and
|
||||
# unknown keys are ignored, so a snake_case key would be silently dropped (default kept).
|
||||
# spawnReadyTimeoutMs: 20000
|
||||
# spawnReadyPollMs: 300
|
||||
|
||||
# Worktree provisioning root (CB-301-ext). Where per-worker git worktrees are checked out so
|
||||
# each worker owns an isolated branch instead of sharing the primary's tree. Omit to default
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
# contextCap → force-release a session after this many delegated turns
|
||||
# drainTimeoutSeconds → seconds to wait for BUSY sessions on shutdown before forced teardown
|
||||
# lifecycle:
|
||||
# idleTtlSeconds: 300
|
||||
# contextCap: 10
|
||||
# drainTimeoutSeconds: 5
|
||||
|
||||
# Durable reply delivery (CB-307 Stage 2). OMIT this block entirely to keep the default
|
||||
# in-memory, soft-state reply inbox (late worker replies are held only until a daemon bounce).
|
||||
# Set a broker uri to swap in the AMQP-backed inbox: worker replies with no open send are held
|
||||
# on a durable per-target queue (agent.<target>.inbox) and survive a restart — the broker
|
||||
# redelivers anything the primary had not yet drained. Production default is LavinMQ; a stock
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# broker:
|
||||
# uri: amqp://guest:guest@127.0.0.1:5672
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open bridge_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
# terminal → pin the primary's herdr terminal id. Omit to learn it from the connection on
|
||||
# the first orchestration-side MCP call (the normal case). An off-host or
|
||||
# non-herdr primary leaves this unresolved → the loop is a no-op and delivery
|
||||
# degrades to pull; the reply is still never lost.
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just bridged-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off bridge_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
# primary:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
@@ -1,488 +0,0 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.config.ConfigRef;
|
||||
import dev.ltms.bridged.config.ConfigWatcher;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.LeadTabScanner;
|
||||
import dev.ltms.bridged.lead.LeadLauncher;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.CompletionResolver;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.inject.StatusPoller;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.auth.MemberRegistry;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.mcp.BridgeMcp;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.bridged.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.bridged.msg.AmqpReplyInbox;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.ReplyInbox;
|
||||
import dev.ltms.bridged.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.bridged.msg.ReplyPushLoop;
|
||||
import dev.ltms.bridged.rest.BridgedApp;
|
||||
import dev.ltms.bridged.session.GitWorktrees;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.session.SessionReaper;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.member.CompositePeerLauncher;
|
||||
import dev.ltms.bridged.member.HerdrPeerLauncher;
|
||||
import dev.ltms.bridged.member.OpenCodeLauncher;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* {@code bridged} entry point. Wires the real herdr socket client to the REST app and
|
||||
* starts listening. Before anything else it asserts its own environment is clean —
|
||||
* {@code bridged} is not a Claude process and must never carry a base_url.
|
||||
*/
|
||||
public final class Bridged {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Bridged.class);
|
||||
|
||||
/** How often the injector samples a busy worker's status while it has queued work. */
|
||||
private static final long INJECT_POLL_MILLIS = 250;
|
||||
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = Path.of(args.length > 0 ? args[0] : "bridged.yaml");
|
||||
BridgedConfig cfg = BridgedConfig.load(configPath);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched bridged must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, BridgedConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, BridgedConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet().tabLabel()));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(agents, spaces,
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet().tabLabel()));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName));
|
||||
// CB-504: under supervision (launchd/systemd) bridged can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the static registry, discover leads by the tab labels the operator
|
||||
// writes. CB-557 moved the settings onto the lead they describe, so scanning is on whenever
|
||||
// a `fleet.leaders:` entry exists — with no leads configured the supplier is a constant and
|
||||
// never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
// One scanner, so one prefix and one interval. Distinct per-lead prefixes would need a
|
||||
// scanner each; until a config actually wants that, take the first entry's settings and
|
||||
// say so, rather than silently honouring one lead's prefix and dropping another's.
|
||||
var scan = leaders.values().iterator().next();
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(Objects::nonNull)
|
||||
.collect(Collectors.toSet());
|
||||
leads = new LeadTabScanner(herdr, scan.tabPrefix(), memberSpaces, leadTerminals,
|
||||
TimeUnit.SECONDS.toNanos(scan.scanIntervalSeconds()), System::nanoTime);
|
||||
log.info("lead scan: tabs labelled '{}…' host a lead (rescan every {}s, member spaces {} "
|
||||
+ "excluded)",
|
||||
scan.tabPrefix(), scan.scanIntervalSeconds(), memberSpaces);
|
||||
long distinctPrefixes = leaders.values().stream()
|
||||
.map(BridgedConfig.Leader::tabPrefix).distinct().count();
|
||||
if (distinctPrefixes > 1) {
|
||||
log.warn("fleet.leaders declares {} different tabPrefix values; only '{}' is scanned "
|
||||
+ "for. Give every lead the same tabPrefix, or leads under the others "
|
||||
+ "will not be discovered.",
|
||||
distinctPrefixes, scan.tabPrefix());
|
||||
}
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(agents, spaces, cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
completion.onDelivered(target);
|
||||
sessions.onDelivered(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Injector injector = new Injector(agents, turnListener, deliverableTo(presence, leads),
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, INJECT_POLL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block (with a uri) selects the AMQP-backed durable adapter;
|
||||
// absent, bridged stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox;
|
||||
if (cfg.broker() != null && cfg.broker().isConfigured()) {
|
||||
replyInbox = AmqpReplyInbox.open(cfg.broker().uri());
|
||||
log.info("reply inbox: AMQP broker (durable) at {}", cfg.broker().uri());
|
||||
} else {
|
||||
replyInbox = new InMemoryReplyInbox();
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
}
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open bridge_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = BridgedMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, agents, replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, agents, replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(terminal -> {
|
||||
messages.abandon(terminal, "the worker session was released before it replied");
|
||||
replyInbox.release(terminal);
|
||||
primaryRegistry.forgetDelegation(terminal); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): bridge_send/bridge_reply/bridge_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting bridged");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads,
|
||||
members::snapshot);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads,
|
||||
members::snapshot);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
BridgeMcp mcp = new BridgeMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics);
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
herdr.close();
|
||||
}));
|
||||
|
||||
Javalin app = new BridgedApp(herdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("bridged listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a worker whose
|
||||
* agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> worker's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code BridgeMcp} marks presence
|
||||
* only for a worker, deliberately, since that map doubles as the worker roster's availability
|
||||
* signal and a lead counted there would show up as an available worker. So without the second
|
||||
* disjunct a lead is permanently un-deliverable: every lead→lead send sat on the gate for
|
||||
* {@code READINESS_GRACE_POLLS} (~60s) and then failed having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Bridged() {
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,59 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
* identity (CB-105). Given the PID that opened an MCP connection, {@link #terminalForPid} finds
|
||||
* the agent pane whose process tree contains it, so {@code bridged} can tell <em>which worker</em>
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane owns it (e.g. the caller is the primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(paneId, pid)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private boolean paneOwnsPid(String paneId, long pid) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -1,265 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* The CB-106 completion fallback: bridges the {@link Injector}'s turn-completion signal to the
|
||||
* {@link Rendezvous} so a blocking {@code bridge_send} resolves even when the worker finishes its
|
||||
* task without ever calling {@code bridge_reply} — the common case for a real delegated coding task.
|
||||
*
|
||||
* <p>On a confirmed {@code working → idle} boundary it scrapes the worker's recent transcript and
|
||||
* resolves the awaiting send with that tail (a {@link Rendezvous.Kind#COMPLETION} resolution, so the
|
||||
* caller can tell a scrape from a structured reply). It scrapes only when a send is actually waiting
|
||||
* — a fleet worker's own turns, or a send that already timed out, cost no herdr traffic. An explicit
|
||||
* {@code bridge_reply} that raced in first wins; {@link Rendezvous#resolveCompletion} is then a no-op.
|
||||
*
|
||||
* <p>It also handles the CB-109 stall signal ({@link #onTurnFailed}): a worker that ran a turn then
|
||||
* wedged in an {@code unknown} state resolves the send as a failure (with the error screen as
|
||||
* context) rather than leaving it to time out.
|
||||
*
|
||||
* <p>The scrape is cleaned to the last {@code ⏺} assistant block (stripping TUI chrome) and guarded
|
||||
* against misattribution (CB-115): the pane content is baselined on delivery ({@link #onDelivered}),
|
||||
* and a completion whose scrape is unchanged from that baseline — the previous turn's wind-down
|
||||
* sampled as this turn's boundary on a rapid back-to-back send — is suppressed rather than resolving
|
||||
* the send with a stale answer.
|
||||
*
|
||||
* <p><strong>Waiter-specific resolution (CB-116).</strong> On delivery we also capture the exact
|
||||
* {@link Rendezvous} waiter this turn belongs to, and the completion/failure fallbacks resolve
|
||||
* <em>that</em> waiter — never "whatever send is waiting now". A completion fallback runs on a virtual
|
||||
* thread and can land after the worker's {@code bridge_reply} already resolved the turn and the
|
||||
* <em>next</em> send opened its own waiter on the same session; resolving the current waiter would
|
||||
* then deliver turn N's stale scrape as turn N+1's answer. Targeting the captured waiter makes a late
|
||||
* completion a harmless no-op (its waiter is already done) instead of a cross-turn stale reply.
|
||||
*
|
||||
* <p>Wired as the {@link Injector}'s {@link TurnListener}; the handlers hand off to a virtual thread
|
||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||
*/
|
||||
public final class CompletionResolver implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source for the completion scrape. {@code recent} returns the tail of
|
||||
* the transcript (the worker's last output), which is what a delegator wants when the worker
|
||||
* didn't structure a reply.
|
||||
*/
|
||||
static final String SCRAPE_SOURCE = "recent";
|
||||
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
* delivering send opened, plus the assistant block present when it was delivered.
|
||||
*
|
||||
* <p>The {@code waiter} is what makes a late fallback safe (CB-116): we resolve it, not "whoever
|
||||
* is waiting now", so a completion that fires after the next send has opened its own waiter is a
|
||||
* no-op rather than a cross-turn stale reply. The {@code baseline} is the CB-115 staleness
|
||||
* reference: a completion scrape equal to it means the worker produced no new output (the previous
|
||||
* turn's wind-down sampled as this boundary), so it is suppressed. Overwritten on each delivery;
|
||||
* cleared when the turn resolves. Package-private so tests can capture and replay a specific turn.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
}
|
||||
String baseline;
|
||||
try {
|
||||
// Clip to the same cap resolve() applies to the tail (line ~134): the CB-115 misattribution
|
||||
// guard compares baseline.equals(tail), so both sides must be the same capped representation.
|
||||
// An unclipped baseline vs a clipped tail would never match for a >MAX_SCRAPE_CHARS block,
|
||||
// defeating the guard and letting a stale completion resolve the send.
|
||||
baseline = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
InFlight inFlight(String target) {
|
||||
return inFlight.get(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
// Read the in-flight turn on the poller thread — before any next-turn delivery can overwrite
|
||||
// it — then off-load the scrape (a herdr round-trip we must not block polling on) to a vthread.
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("completion-" + target).start(() -> resolve(target, turn));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the completed turn before adapter housekeeping can erase its rendered output. This is
|
||||
* intentionally synchronous and used only when a post-turn context reset is enabled; the normal
|
||||
* path remains off-loaded so polling is not blocked by a scrape.
|
||||
*/
|
||||
public void resolveBeforePostAction(String target) {
|
||||
resolve(target, inFlight.get(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
void resolve(String target, InFlight turn) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = turn == null ? null : turn.waiter();
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
// Nobody is blocked on THIS turn (it had no send, or its bridge_reply already won). Skip
|
||||
// the scrape; resolving the current waiter here would be the CB-116 cross-turn stale reply.
|
||||
inFlight.remove(target, turn);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
tail = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
// The worker finished but we couldn't read its screen — still resolve the send so the
|
||||
// caller unblocks; an empty tail beats hanging until the caller's timeout.
|
||||
log.warn("completion scrape for {} failed; resolving with an empty tail: {}",
|
||||
target, e.getMessage());
|
||||
tail = "";
|
||||
scrapeFailed = true;
|
||||
}
|
||||
// Misattribution guard (CB-115): if the scrape is byte-identical to the pane content at
|
||||
// delivery, this turn produced no new output — the boundary belongs to the previous turn's
|
||||
// wind-down (common on rapid back-to-back sends). Suppress rather than resolve the send with
|
||||
// a stale answer; the real bridge_reply (or a later genuine completion) resolves it instead.
|
||||
// A scrape that failed to read is exempt — an empty tail there is "couldn't see", not "no change".
|
||||
String baseline = turn.baseline();
|
||||
if (!scrapeFailed && baseline != null && baseline.equals(tail)) {
|
||||
log.debug("suppressing misattributed completion for {} (no output change since delivery)",
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
}
|
||||
if (rendezvous.resolveCompletion(waiter, tail)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.debug("resolved send to {} via turn-completion fallback ({} chars scraped)",
|
||||
target, tail.length());
|
||||
}
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
CompletableFuture<Rendezvous.Resolution> waiter =
|
||||
turn != null ? turn.waiter() : rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason;
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.debug("failed send to {} via turn-stall fallback", target);
|
||||
}
|
||||
}
|
||||
|
||||
private static String clip(String s) {
|
||||
if (s == null) return "";
|
||||
String trimmed = s.strip();
|
||||
return trimmed.length() <= MAX_SCRAPE_CHARS
|
||||
? trimmed
|
||||
: trimmed.substring(trimmed.length() - MAX_SCRAPE_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the last assistant message from a raw Claude Code pane scrape (CB-115). Claude Code
|
||||
* prefixes each assistant turn with {@code ⏺}; the delegator wants that answer, not the TUI
|
||||
* chrome around it. Take everything from the final {@code ⏺} onward and stop at the <em>first</em>
|
||||
* hard interface boundary below it — the spinner/status line, input box, {@code ❯} prompt (which
|
||||
* may echo the <em>next</em> turn's text), footer, or tips/warnings. Stopping at the first
|
||||
* boundary (rather than trimming only trailing chrome) is what keeps a following turn's echoed
|
||||
* prompt out of this reply. Blank lines are not boundaries, so a multi-paragraph answer survives;
|
||||
* trailing blanks are trimmed at the end. With no {@code ⏺} marker (an unusual render) the whole
|
||||
* text is scanned the same way, so we never lose the reply.
|
||||
*
|
||||
* <p>Package-private and pure so it is unit-testable without herdr.
|
||||
*/
|
||||
static String lastAssistantBlock(String raw) {
|
||||
if (raw == null || raw.isBlank()) return "";
|
||||
int marker = raw.lastIndexOf('⏺');
|
||||
String block = marker >= 0 ? raw.substring(marker + 1) : raw;
|
||||
StringBuilder out = new StringBuilder();
|
||||
int kept = 0;
|
||||
for (String line : block.split("\n", -1)) {
|
||||
if (isBoundary(line)) break; // first TUI boundary ends the assistant message
|
||||
if (kept++ > 0) out.append('\n');
|
||||
out.append(line);
|
||||
}
|
||||
return out.toString().strip();
|
||||
}
|
||||
|
||||
/**
|
||||
* A hard TUI boundary line that marks the end of an assistant message and the start of interface
|
||||
* chrome (input box, prompt, spinner, footer, tips/warnings). Blank lines are <em>not</em>
|
||||
* boundaries — an answer may contain them — so they are kept and trimmed only if trailing.
|
||||
*/
|
||||
private static boolean isBoundary(String line) {
|
||||
String t = line.strip();
|
||||
if (t.isEmpty()) return false;
|
||||
// A horizontal rule / all box-drawing separators (e.g. "──────").
|
||||
if (t.chars().allMatch(c -> c == '─' || c == '—' || c == '━' || c == '═' || c == '-')) {
|
||||
return true;
|
||||
}
|
||||
String lower = t.toLowerCase();
|
||||
return t.startsWith("╭") || t.startsWith("│") || t.startsWith("╰") || t.startsWith("┌")
|
||||
|| t.startsWith("└") || t.startsWith("❯") || t.startsWith("⏵")
|
||||
|| t.startsWith("⎿") || t.startsWith("⚠")
|
||||
// Status/spinner lines Claude Code renders below a settled or in-flight turn,
|
||||
// e.g. "✻ Baked for 21s", "✶ Forming…".
|
||||
|| t.startsWith("✻") || t.startsWith("✳") || t.startsWith("✽") || t.startsWith("·")
|
||||
|| t.startsWith("●") || t.startsWith("◐") || t.startsWith("✢") || t.startsWith("✶")
|
||||
|| lower.contains("auto mode") || lower.contains("for shortcuts")
|
||||
|| lower.contains("esc to interrupt") || lower.contains("bypass permissions");
|
||||
}
|
||||
}
|
||||
@@ -1,350 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>Claude Code</strong> — the safe path from a
|
||||
* delegation request to a running off-subscription Claude.
|
||||
*
|
||||
* <p>Everything transport-related (tab/pane placement, the CB-306 spawn-readiness gate, unique
|
||||
* naming, CB-117 orphan reap, teardown, listing, cwd resolution) lives in the base. This class
|
||||
* supplies only the two Claude-specific seams:
|
||||
* <ul>
|
||||
* <li>the {@code claude} name prefix (so reap matches {@code claude-*} panes, never another
|
||||
* adapter's), and</li>
|
||||
* <li>{@link #buildLaunch}, which encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em> touching
|
||||
* herdr, and mount the bridge MCP + reply charter as inline launch flags. A worker's base_url
|
||||
* lives in the env map handed to herdr and nowhere else; {@code bridged}'s own environment is
|
||||
* never mutated, and nothing is written to the worker's profile.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "claude";
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
* Standing instruction appended to the worker's system prompt so it returns its result via
|
||||
* {@code bridge_reply}. Injected as a launch flag, so nothing is written to the worker's
|
||||
* profile — it is guidance, and a worker that never replies is caught by the send's timeout.
|
||||
*/
|
||||
static final String REPLY_CHARTER =
|
||||
"You are an off-subscription worker in the claude-bridge fleet. Every message you "
|
||||
+ "receive arrives through the bridge, and the ONLY channel back to the sender is the "
|
||||
+ "bridge_reply MCP tool. Text you write in your terminal is NOT sent anywhere — the "
|
||||
+ "sender cannot see your screen, so an in-terminal answer is silently discarded. "
|
||||
+ "Therefore you MUST end EVERY turn by calling bridge_reply with `content` set to your "
|
||||
+ "complete response. This holds for every message without exception — tasks, questions, "
|
||||
+ "clarifications, acknowledgements, and ordinary back-and-forth conversation. Call "
|
||||
+ "bridge_reply exactly once, as the final action of your turn, with your full answer in "
|
||||
+ "`content`; never wait for confirmation first. If you end a turn without calling "
|
||||
+ "bridge_reply, the sender receives nothing and the exchange stalls.";
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so
|
||||
* existing deployments and tests keep the legacy non-blocking spawn semantics.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with spawn-ready gate enabled. The gate polls {@code agents.status()}
|
||||
* until the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
tabLabelTemplate);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply
|
||||
* fakes for the clock ({@code nowMillis}) and poll-loop wait ({@code sleeper}). The
|
||||
* {@code sleeper} is never called when the gate is disabled ({@code spawnReadyTimeoutMs == 0}).
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param guard subscription-boundary guard (checked before spawning)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (e.g. {@code () -> Thread.sleep(pollMs)}); it
|
||||
* already encodes the poll interval, so the 8th positional argument
|
||||
* (poll ms) is accepted for API symmetry but otherwise unused here
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param tabLabelTemplate {@code fleet.tabLabel}; {@code null}/blank ⇒
|
||||
* {@link BridgedConfig.Fleet#DEFAULT_TAB_LABEL}
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, tabLabelTemplate);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>A legacy spawn with no session identity is a fresh, launcher-derived session — delegate to
|
||||
* the session-aware form with no name and no resume id.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg) {
|
||||
return buildLaunch(cfg, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: assert the profile's base_url is on
|
||||
* the allowlist <em>before</em> any herdr call, then build the worker env with
|
||||
* {@code ANTHROPIC_*}, the parity-neutral git-forge grant, and the bridge MCP + reply charter
|
||||
* mounted as inline launch flags. When the request carries session identity (CB-547a) it is
|
||||
* applied here — see {@link #applySessionIdentity}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, String sessionName, String resumeSessionId) {
|
||||
// CB-539: a profile may deliberately opt into the subscription (subscription: true) when no
|
||||
// off-subscription endpoint exists for it — e.g. `sonnet` on `ccs`. That profile gets no
|
||||
// ANTHROPIC_BASE_URL/AUTH_TOKEN (there is nothing to point them at) and the guard's base_url
|
||||
// requirement is skipped FOR IT ONLY. Every other profile keeps the hard boundary below.
|
||||
boolean onSubscription = cfg.isSubscription();
|
||||
String baseUrl = cfg.baseUrl();
|
||||
|
||||
if (onSubscription) {
|
||||
// NO SILENT CONTRADICTION: subscription:true + a baseUrl state opposite intents; refuse
|
||||
// loudly rather than pick a winner.
|
||||
if (baseUrl != null && !baseUrl.isBlank()) {
|
||||
throw new IllegalStateException("profile '" + cfg.profile()
|
||||
+ "' sets both subscription: true and a baseUrl ('" + baseUrl + "') — the two "
|
||||
+ "are contradictory: a subscription profile must not point at an endpoint. "
|
||||
+ "Drop baseUrl, or drop subscription: true.");
|
||||
}
|
||||
// Visible without anyone going looking for it: this worker bills the subscription.
|
||||
log.warn("spawning profile '{}' on the Claude subscription (subscription: true) — this "
|
||||
+ "worker WILL bill the operator's subscription", cfg.profile());
|
||||
} else {
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
}
|
||||
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
if (onSubscription) {
|
||||
// CB-542 belt-and-braces: on the subscription path no guard vets these two keys, and the
|
||||
// profile's env: is layered in by baseEnv — so strip any that rode in there. Config load
|
||||
// already rejects this (loudly, naming the profile); this makes the boundary hold even
|
||||
// for a profile built in code that never passed through that validation.
|
||||
workerEnv.remove("ANTHROPIC_BASE_URL");
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so bridged chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
// handle is known BEFORE the agent has written anything; a resume spawn adopts its prior
|
||||
// id via -r and passes no --session-id (the two conflict). Both are injected before the
|
||||
// model flag so --model keeps outranking the operator's own argv.
|
||||
// mutableArgv: argvWithBridge may hand back the profile's own (immutable) List.of when it
|
||||
// has no MCP — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithBridge(cfg));
|
||||
String agentSessionId = applySessionIdentity(argv, sessionName, resumeSessionId);
|
||||
return new Launch(workerEnv, argvWithModel(argv, cfg), agentSessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
}
|
||||
if (resuming) {
|
||||
argv.add("-r");
|
||||
argv.add(resumeSessionId);
|
||||
return resumeSessionId;
|
||||
}
|
||||
String minted = UUID.randomUUID().toString();
|
||||
argv.add("--session-id");
|
||||
argv.add(minted);
|
||||
return minted;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus — when {@code worker.mcpUrl} is set — inline {@code --mcp-config} for
|
||||
* the bridge server and {@code --append-system-prompt} for the {@link #REPLY_CHARTER}. Neither
|
||||
* touches the profile's config; both are pure command-line flags. This inline-flag mount is
|
||||
* Claude Code specific — other adapters mount MCP and instructions their own way.
|
||||
*/
|
||||
private List<String> argvWithBridge(BridgedConfig.Profile cfg) {
|
||||
if (!cfg.hasMcp()) {
|
||||
return cfg.argv();
|
||||
}
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(REPLY_CHARTER);
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin the model on the command line as well as in {@code ANTHROPIC_MODEL} (CB-533).
|
||||
*
|
||||
* <p>The env var alone is not a reliable pin for this adapter, because the argv is usually a
|
||||
* launcher rather than {@code claude} itself — {@code ["ccs", "<profile>"]} — and {@code ccs}
|
||||
* exports its profile's own model family ({@code ANTHROPIC_MODEL}, {@code DEFAULT_OPUS/SONNET/
|
||||
* HAIKU}, {@code CLAUDE_CODE_SUBAGENT_MODEL}) over whatever it inherited. A worker profile that
|
||||
* set {@code model:} therefore got silently overruled by its own launcher. Claude Code's
|
||||
* {@code --model} flag outranks the environment, and {@code ccs <profile> [claude-args...]}
|
||||
* passes trailing arguments through, so the flag survives the wrapper.
|
||||
*
|
||||
* <p>Appended last so it also outranks anything in the operator's own {@code argv}. Profiles
|
||||
* that deliberately leave {@code model:} unset (letting {@code ccs} own model selection, as
|
||||
* {@code gx10} does) are untouched — this adds nothing when there is nothing to add. This is
|
||||
* the {@code kind: claude} counterpart of the opencode adapter's {@code -m provider/model}.
|
||||
*/
|
||||
private static List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() == null || cfg.model().isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withModel = mutableArgv(argv);
|
||||
withModel.add("--model");
|
||||
withModel.add(cfg.model());
|
||||
return withModel;
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.CONTEXT_RESET, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_NAME, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
String target = agentTarget(id);
|
||||
if (target == null) {
|
||||
return false;
|
||||
}
|
||||
// This deliberately bypasses Injector: /clear is housekeeping, not a delegated turn.
|
||||
agents().send(target, "/clear");
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (Claude prefix), kept for direct unit testing -------------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a Claude Code bridge worker started by a <em>different</em> process
|
||||
* than {@code currentNonce}. A thin {@code claude}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -1,738 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ConcurrentMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Abstract base for {@link PeerLauncher} adapters that materialize a peer as a <em>herdr</em>
|
||||
* agent (a CLI coding agent running in a herdr tab/pane). It owns everything that is the same
|
||||
* regardless of <em>which</em> coding agent runs: tab/pane placement, the CB-306 spawn-readiness
|
||||
* gate, unique naming, CB-117 orphan reap, teardown, {@link #list() listing}, and cwd resolution.
|
||||
*
|
||||
* <p>Two seams are peer-specific and supplied by the concrete adapter:
|
||||
* <ul>
|
||||
* <li>{@code namePrefix} (constructor arg) — the label prefix ({@code claude}, {@code opencode})
|
||||
* that drives both unique naming and the orphan-reap pattern, so each adapter reaps only its
|
||||
* own kind of pane and never another's.</li>
|
||||
* <li>{@link #buildLaunch(BridgedConfig.Profile)} — the peer-specific env map + argv, including any
|
||||
* subscription/guard check, MCP mount, and instruction injection. The base never sees how the
|
||||
* peer is configured; it only places and starts the returned {@link Launch}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Placement: in the default {@code tab} policy a peer lands in its own tab inside a dedicated
|
||||
* worker space (found-or-created once, then shared), so peers never split or clutter the user's
|
||||
* real work spaces. Teardown removes the peer's pane <em>and</em> its now-empty tab, tolerating an
|
||||
* already-gone peer so a repeated DELETE is harmless.
|
||||
*/
|
||||
public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
/**
|
||||
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
||||
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
||||
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
||||
*/
|
||||
private static final int SHELL_READY_RETRIES = 20;
|
||||
|
||||
private final String namePrefix; // label prefix: naming + reap scheme
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final Map<String, BridgedConfig.Profile> profiles; // profile name → spawn settings
|
||||
private final String defaultProfile; // profile a no-arg spawn uses (nullable)
|
||||
|
||||
/** Host env lookup (injectable for tests); adapters read it in {@link #buildLaunch}. */
|
||||
protected final Function<String, String> env;
|
||||
|
||||
private final AtomicLong nameSeq = new AtomicLong(); // per-peer counter (herdr agent names only)
|
||||
|
||||
/**
|
||||
* The {@code fleet.tabLabel} template; a {@code null} supplier or a {@code null}/blank value ⇒
|
||||
* {@link BridgedConfig.Fleet#DEFAULT_TAB_LABEL}. A profile's own {@code tabLabel} still
|
||||
* overrides it.
|
||||
*
|
||||
* <p>CB-559: a supplier rather than a String, so a config reload renames the <em>next</em> tab
|
||||
* without a restart. Existing tabs keep the label they were given — bridged does not rewrite a
|
||||
* label it already wrote.
|
||||
*/
|
||||
private final Supplier<String> tabLabelTemplate;
|
||||
|
||||
/**
|
||||
* Tab numbers, counted per {@code role/profile} pair (CB-557).
|
||||
*
|
||||
* <p>Deliberately not {@link #nameSeq}. That counter is shared by every profile this launcher
|
||||
* serves, because its job is to make herdr <em>agent names</em> unique. Reusing it for the tab
|
||||
* label made the numbers global, so sibling tabs read {@code #4}, {@code #9}, {@code #17} — gaps
|
||||
* that look like a member died. Counting per role+profile makes {@code dev: sonnet #2} mean the
|
||||
* second sonnet dev, which is what a reader assumes it means.
|
||||
*
|
||||
* <p>Resets when the daemon restarts, and that is fine: the label is a human-facing hint, not an
|
||||
* identity. Identity is {@link PeerHandle#id()}.
|
||||
*/
|
||||
private final ConcurrentMap<String, AtomicLong> labelSeq = new ConcurrentHashMap<>();
|
||||
|
||||
private final long spawnReadyTimeoutMs; // 0 = disable gate (legacy non-blocking spawn)
|
||||
private final LongSupplier nowMillis; // monotonic clock (injectable for tests)
|
||||
private final Runnable sleeper; // sleep/wait hook (injectable for tests; never real-sleep in unit tests)
|
||||
|
||||
// Per-process token mixed into each peer name so a fresh process (nameSeq back at 0) cannot
|
||||
// collide with same-profile peers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
|
||||
// CB-519: PeerHandle.id() is a host-unique opaque UUID, decoupled from the herdr pane id. The
|
||||
// routing/registry key is the UUID; the herdr pane id is a launcher-private placement/teardown
|
||||
// coordinate. This map bridges the two so stop(id) can resolve a host-unique key back to the
|
||||
// exact pane it must tear down. The pane id is launcher-private (never the routing key) — see
|
||||
// PeerHandle.id().
|
||||
private final ConcurrentMap<String, String> paneByAgentId = new ConcurrentHashMap<>();
|
||||
private final AtomicBoolean resetUnsupportedLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* @param namePrefix label prefix for this peer kind (drives naming and reap)
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured peer profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (never called when the gate is disabled); the poll
|
||||
* interval is baked into this hook, so the base needs no poll field
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(namePrefix, agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus the {@code fleet.tabLabel} template (CB-557).
|
||||
*
|
||||
* @param tabLabelTemplate fleet-wide tab-label template, read per spawn (CB-559); {@code null},
|
||||
* or a supplier yielding {@code null}/blank ⇒
|
||||
* {@link BridgedConfig.Fleet#DEFAULT_TAB_LABEL}. A separate constructor
|
||||
* rather than a new parameter on the one above, so every existing call
|
||||
* site keeps the default without an edit.
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
this.tabLabelTemplate = tabLabelTemplate;
|
||||
this.namePrefix = namePrefix;
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.profiles = Map.copyOf(profiles);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.env = env;
|
||||
this.spawnReadyTimeoutMs = spawnReadyTimeoutMs;
|
||||
this.nowMillis = nowMillis;
|
||||
this.sleeper = sleeper;
|
||||
}
|
||||
|
||||
// --- adapter seams -------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Build the peer-specific launch for {@code cfg}: the environment map and argv handed to herdr.
|
||||
* Any subscription/guard check, MCP mount, and instruction injection happen here. The env map
|
||||
* and argv are adapter-private; the base only places and starts what is returned.
|
||||
*/
|
||||
protected abstract Launch buildLaunch(BridgedConfig.Profile cfg);
|
||||
|
||||
/**
|
||||
* Session-aware variant of {@link #buildLaunch(BridgedConfig.Profile)} (CB-547a). Default
|
||||
* discards the session identity and delegates to the profile-only form, so an adapter that
|
||||
* carries no durable peer session (opencode, say) inherits byte-identical behaviour and needs
|
||||
* no change. An adapter that does (Claude Code) overrides this to mint/resume the id and to
|
||||
* surface it on the returned {@link Launch#agentSessionId()}.
|
||||
*
|
||||
* @param cfg the resolved profile to spawn
|
||||
* @param sessionName the bridge's logical session name, or null/blank for launcher-derived
|
||||
* @param resumeSessionId the peer's own prior session id to resume, or null/blank for fresh
|
||||
*/
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, String sessionName, String resumeSessionId) {
|
||||
return buildLaunch(cfg);
|
||||
}
|
||||
|
||||
/** Direct transport access for peer-specific, non-turn control operations. */
|
||||
protected final AgentControl agents() {
|
||||
return agents;
|
||||
}
|
||||
|
||||
/** Resolve the public peer id to the launcher's private herdr target. */
|
||||
protected final String agentTarget(String id) {
|
||||
return paneByAgentId.get(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
if (resetUnsupportedLogged.compareAndSet(false, true)) {
|
||||
log.warn("context reset is unsupported for peer kind {}; clearAfterTurn is a no-op",
|
||||
namePrefix);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* A peer-specific launch: the herdr {@code env} map and {@code argv}, plus — for an adapter
|
||||
* that carries durable session identity (CB-547a) — the peer's OWN session id
|
||||
* ({@link PeerHandle#agentSessionId()}), known before the peer has written anything. Null for
|
||||
* a launch that carries no identity.
|
||||
*/
|
||||
protected record Launch(Map<String, String> env, List<String> argv, String agentSessionId) {
|
||||
|
||||
/** A launch without a discoverable agent session id (an adapter that carries none). */
|
||||
Launch(Map<String, String> env, List<String> argv) {
|
||||
this(env, argv, null);
|
||||
}
|
||||
}
|
||||
|
||||
// --- profile surface -----------------------------------------------------------------------
|
||||
|
||||
/** The configured peer profile names (what {@code spawn(profile)} accepts). */
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return profiles.keySet();
|
||||
}
|
||||
|
||||
/** The parity-overlay file list for {@code profileName} (default list when unset). */
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
return cfg == null ? List.of() : cfg.parityOverlay();
|
||||
}
|
||||
|
||||
/** The profile a no-argument spawn uses, or {@code null} if none is configured. */
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/** The configured profiles, for adapter capability decisions (e.g. any git-token grant). */
|
||||
protected Collection<BridgedConfig.Profile> profileConfigs() {
|
||||
return profiles.values();
|
||||
}
|
||||
|
||||
/** Resolve {@code profileName} (null/blank → default) to its config, or throw with the options. */
|
||||
protected BridgedConfig.Profile requireProfile(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
throw new IllegalArgumentException("no default worker profile is configured — "
|
||||
+ "pass a profile; configured: " + profiles.keySet());
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
if (cfg == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile '" + name
|
||||
+ "' — configured: " + profiles.keySet());
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
// --- spawn ---------------------------------------------------------------------------------
|
||||
|
||||
/** A started peer plus the launch's agent-session id (the resume handle, or null). */
|
||||
private record Spawned(Agent agent, String agentSessionId) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer. {@code profileName} null/blank → the default profile. The working directory
|
||||
* (CB-112) is resolved by {@link #resolveCwd}: an explicit {@code requestedCwd}, else the
|
||||
* profile's configured {@code cwd}, else {@code callerCwd} (the primary's cwd, when the spawn
|
||||
* came from the primary over MCP), else the daemon's cwd — never assumed to be {@code $HOME}.
|
||||
* The adapter's {@link #buildLaunch} runs before any herdr call.
|
||||
*/
|
||||
protected Agent spawnInternal(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, null, null).agent();
|
||||
}
|
||||
|
||||
/** Pre-CB-557 shape: no explicit role, so the tab is labelled as a {@code dev}. */
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId,
|
||||
MemberRole.DEV);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer with session identity (CB-547a). {@code sessionName} and {@code resumeSessionId}
|
||||
* are threaded from the {@link SpawnRequest} into {@link #buildLaunch(BridgedConfig.Profile,
|
||||
* String, String)}, and the launch's resolved agent-session id is returned alongside the agent
|
||||
* so the caller can put it on the {@link PeerHandle}.
|
||||
*/
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId, MemberRole role) {
|
||||
BridgedConfig.Profile cfg = requireProfile(profileName);
|
||||
Launch launch = buildLaunch(cfg, sessionName, resumeSessionId);
|
||||
String cwd = resolveCwd(requestedCwd, cfg, callerCwd);
|
||||
Agent agent = cfg.tabPlacement()
|
||||
? spawnInTab(cfg, launch.env(), launch.argv(), cwd, role)
|
||||
: spawnAsPane(cfg, launch.env(), launch.argv(), cwd);
|
||||
return new Spawned(agent, launch.agentSessionId());
|
||||
}
|
||||
|
||||
/**
|
||||
* The next tab number for {@code role} on {@code profile}, starting at 1.
|
||||
*
|
||||
* <p>Starts at 1 rather than 0 because the number is read by a person: {@code "dev: sonnet #1"}
|
||||
* is the first one, and {@code #0} invites the question of where {@code #1} went.
|
||||
*/
|
||||
private long nextLabelSeq(MemberRole role, String profile) {
|
||||
String key = (role == null ? "" : role.wireName()) + "/" + profile;
|
||||
return labelSeq.computeIfAbsent(key, _ -> new AtomicLong()).incrementAndGet();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to {@link #spawnInternal} and wraps the resulting herdr {@link Agent} in a
|
||||
* {@link WorkerHandle} whose {@link PeerHandle#id()} is a fresh <em>host-unique</em> opaque
|
||||
* UUID (CB-519), deliberately decoupled from the herdr pane id: the id is the registry/routing
|
||||
* key and must never collide across daemon processes on the same host, while the herdr pane id
|
||||
* stays a launcher-private placement/teardown coordinate, remembered here so {@link #stop}
|
||||
* can resolve the host-unique key back to its pane. When {@code spawnReadyTimeoutMs > 0},
|
||||
* blocks until the peer's herdr status is injectable or the timeout elapses; on timeout the
|
||||
* pane is closed (no orphan) and a {@link PeerUnreachableException} is thrown.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
Spawned spawned = spawnInternal(req.profileName(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
Agent agent = spawned.agent();
|
||||
String paneId = agent.paneId();
|
||||
if (spawnReadyTimeoutMs > 0) {
|
||||
waitUntilInjectableOrThrow(paneId);
|
||||
}
|
||||
// CB-519: the handle id is a host-unique UUID; the herdr pane it maps to stays internal.
|
||||
String id = UUID.randomUUID().toString();
|
||||
paneByAgentId.put(id, paneId);
|
||||
return new WorkerHandle(id, agent.terminalId(), requireProfile(req.profileName()).profile(),
|
||||
req.sessionName(), spawned.agentSessionId());
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return effectiveCwd(req.profileName(), req.requestedCwd(), req.callerCwd());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-301: the effective working directory a spawn for {@code profileName} would use, without
|
||||
* actually spawning.
|
||||
*/
|
||||
private String effectiveCwd(String profileName, String requestedCwd, String callerCwd) {
|
||||
return resolveCwd(requestedCwd, requireProfile(profileName), callerCwd);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-112 cwd resolution: spawn arg → profile config → the primary's cwd → the daemon's cwd.
|
||||
* Never returns {@code null}/blank: {@code "."} (the daemon's own working directory) is the
|
||||
* guaranteed last resort so a pathological environment with an unset {@code user.dir} still
|
||||
* honours the "never assume {@code $HOME}" contract rather than letting herdr default the pane.
|
||||
*/
|
||||
private static String resolveCwd(String requestedCwd, BridgedConfig.Profile cfg, String callerCwd) {
|
||||
return firstNonBlank(requestedCwd, cfg.cwd(), callerCwd, System.getProperty("user.dir"), ".");
|
||||
}
|
||||
|
||||
private static String firstNonBlank(String... values) {
|
||||
for (String v : values) {
|
||||
if (v != null && !v.isBlank()) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Dedicated worker space → own tab (carrying cwd+env) → start the peer into the seed pane. */
|
||||
private Agent spawnInTab(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd, MemberRole role) {
|
||||
Workspace space = spaces.ensureWorkspace(cfg.workspace());
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId(), cwd, workerEnv);
|
||||
log.info("spawning {} profile={} space={} tab={} cwd={}",
|
||||
namePrefix, cfg.profile(), space.workspaceId(), tab.tab().tabId(), cwd);
|
||||
|
||||
Started started;
|
||||
try {
|
||||
if (tab.rootPaneId() == null) {
|
||||
// Protocol 19 starts the agent INTO the seed pane — without one there is nowhere
|
||||
// to start, and a partial tab would be left behind.
|
||||
throw new IllegalStateException("tab " + tab.tab().tabId()
|
||||
+ " had no seed pane in the create response — cannot start a peer in it");
|
||||
}
|
||||
started = startUniquelyNamed(cfg, argv, tab.rootPaneId());
|
||||
} catch (RuntimeException e) {
|
||||
// The peer never started — don't leave the tab we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned tab {} after spawn error: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
|
||||
// The peer is LIVE now, in the seed pane itself (no shell pane to drop — protocol 19).
|
||||
// Labelling is cosmetic: it must not fail the spawn or orphan the running peer — on error
|
||||
// we log and still return it so the caller gets its paneId and can tear it down.
|
||||
tidy("label tab " + tab.tab().tabId(),
|
||||
() -> spaces.renameTab(tab.tab().tabId(),
|
||||
cfg.renderTabLabel(
|
||||
tabLabelTemplate == null ? null : tabLabelTemplate.get(),
|
||||
role, nextLabelSeq(role, cfg.profile()))));
|
||||
log.info("{} started pane={} tab={} terminal={}",
|
||||
namePrefix, started.agent().paneId(), started.agent().tabId(), started.agent().terminalId());
|
||||
return started.agent();
|
||||
}
|
||||
|
||||
/** Run a best-effort post-start cleanup step, logging (not throwing) on failure. */
|
||||
private void tidy(String what, Runnable step) {
|
||||
try {
|
||||
step.run();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("post-start step failed ({}) — peer is running regardless: {}", what, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Legacy placement: split the currently-focused tab; the peer still starts in {@code cwd}. */
|
||||
private Agent spawnAsPane(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd) {
|
||||
log.info("spawning {} (pane placement) profile={} cwd={} argv={}",
|
||||
namePrefix, cfg.profile(), cwd, argv);
|
||||
String paneId = spaces.splitPane(cwd, workerEnv);
|
||||
if (paneId == null) {
|
||||
throw new IllegalStateException("pane.split returned no pane — cannot start a peer");
|
||||
}
|
||||
Agent peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
log.info("{} started pane={} terminal={}", namePrefix, peer.paneId(), peer.terminalId());
|
||||
return peer;
|
||||
}
|
||||
|
||||
/** A started peer together with the sequence its unique name/label used. */
|
||||
private record Started(Agent agent, long seq) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the peer under a unique herdr agent name. herdr requires each running agent's
|
||||
* {@code name} to be distinct (a 2nd identical {@code name} fails {@code agent_name_taken}) —
|
||||
* the exact case that makes multiple peers useful. The name is
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>}: {@code seq} distinguishes peers within this process,
|
||||
* and the per-process {@code nonce} keeps a fresh process (whose {@code seq} restarts at 0) from
|
||||
* colliding with same-profile peers that outlived a restart. The retry is a belt-and-braces
|
||||
* backstop for the astronomically unlikely nonce+seq clash; the name is a label only — herdr
|
||||
* detects kind and status from terminal output, not from it.
|
||||
*/
|
||||
private Started startUniquelyNamed(BridgedConfig.Profile cfg, List<String> argv, String paneId) {
|
||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(startAwaitingShellPrompt(name, args, paneId), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("peer name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
||||
HerdrException busy = null;
|
||||
for (int attempt = 0; attempt < SHELL_READY_RETRIES; attempt++) {
|
||||
try {
|
||||
return agents.start(name, namePrefix, args, paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_pane_busy".equals(e.code())) throw e;
|
||||
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
||||
busy = e;
|
||||
sleeper.run();
|
||||
}
|
||||
}
|
||||
throw busy;
|
||||
}
|
||||
|
||||
// --- discovery + reap ----------------------------------------------------------------------
|
||||
|
||||
/** All herdr-tracked agents — discovery for "what peers exist". */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
return agents.list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Reap peer panes left behind by an earlier daemon process (CB-117). herdr keeps a peer's pane
|
||||
* alive across a daemon restart <em>by design</em>, and that pane's id is held only by its
|
||||
* spawner — so a peer whose owning process exited before issuing the matching teardown leaks
|
||||
* with nothing tracking it. On boot we scan herdr for agents whose name matches our
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>} scheme with a nonce <em>other</em> than this
|
||||
* process's {@link #nameNonce}, and tear each one down (its pane and, via {@link #stop}, its
|
||||
* now-empty dedicated tab). A current-nonce peer is ours and live, so it is left running; a
|
||||
* user's own session carries no such name and is never touched. A peer from a <em>different</em>
|
||||
* adapter (different prefix) is likewise never touched. Best-effort: a failed listing, or a
|
||||
* failure to stop any one peer, is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
List<Agent> all;
|
||||
try {
|
||||
all = agents.list();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("orphan-peer reap skipped — agent.list failed: {}", e.getMessage());
|
||||
return 0;
|
||||
}
|
||||
int reaped = 0;
|
||||
for (Agent a : all) {
|
||||
if (!isForeignWorker(namePrefix, a.name(), nameNonce)) continue;
|
||||
try {
|
||||
stop(a.paneId());
|
||||
reaped++;
|
||||
log.info("reaped orphan {} {} (pane={} tab={}) left by a prior daemon",
|
||||
namePrefix, a.name(), a.paneId(), a.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("could not reap orphan {} {} (pane={}): {}",
|
||||
namePrefix, a.name(), a.paneId(), e.getMessage());
|
||||
}
|
||||
}
|
||||
if (reaped > 0) {
|
||||
log.info("orphan-peer reap complete — {} stale {} peer(s) removed at startup", reaped, namePrefix);
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The {@code <prefix>-<profile>-<nonce>-<seq>} name pattern; group 1 captures the 6-hex nonce. */
|
||||
static Pattern workerNamePattern(String prefix) {
|
||||
return Pattern.compile(prefix + "-.*-([0-9a-f]{6})-\\d+");
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a peer of kind {@code prefix} started by a <em>different</em> process
|
||||
* than {@code currentNonce} — the reap predicate (CB-117). True only for the prefix's naming
|
||||
* scheme with a foreign nonce: a non-peer name, a different adapter's name, or our own live
|
||||
* nonce is excluded. Pure and package-private so the decision is unit-testable without herdr.
|
||||
*/
|
||||
static boolean isForeignWorker(String prefix, String name, String currentNonce) {
|
||||
String nonce = workerNonce(prefix, name);
|
||||
return nonce != null && !nonce.equals(currentNonce);
|
||||
}
|
||||
|
||||
/** The 6-hex nonce embedded in a {@code prefix} peer name, or {@code null} if not one. */
|
||||
static String workerNonce(String prefix, String name) {
|
||||
if (name == null) return null;
|
||||
Matcher m = workerNamePattern(prefix).matcher(name);
|
||||
return m.matches() ? m.group(1) : null;
|
||||
}
|
||||
|
||||
/** This process's peer-name nonce (a label component only; exposed for reaper tests). */
|
||||
String nameNonce() {
|
||||
return nameNonce;
|
||||
}
|
||||
|
||||
// --- teardown ------------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Tear a peer down: close the pane, and close its tab <em>only</em> when the peer is that tab's
|
||||
* sole occupant. The single-pane check is what makes this safe regardless of how the peer was
|
||||
* placed (or a placement-config change across a restart): a pane-placement peer sitting in one
|
||||
* of the user's shared tabs has siblings, so its tab is never closed — we only ever remove a
|
||||
* tab we created to hold one peer.
|
||||
*
|
||||
* <p>{@code idOrPane} is the {@link PeerHandle#id()} of a peer this launcher spawned (CB-519's
|
||||
* host-unique opaque UUID), resolved through {@link #paneByAgentId} to the pane it must tear
|
||||
* down. An argument that is not one of our ids is treated as a raw herdr pane id — the
|
||||
* {@link #reapOrphanWorkers() orphan-reap} and spawn-gate-timeout paths, plus any caller that
|
||||
* passes a pane directly, keep working without an owning id.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed peer) is treated as success; any other failure propagates so a
|
||||
* genuinely failed teardown is not reported as done.
|
||||
*/
|
||||
@Override
|
||||
public void stop(String idOrPane) {
|
||||
// Teardown knows only the pane, not which profile spawned it. Attempt tab cleanup when any
|
||||
// profile uses tab placement (so the bridge may have created a dedicated peer tab); the
|
||||
// single-occupant check below is what actually protects the user's shared tabs.
|
||||
String paneId = paneByAgentId.remove(idOrPane);
|
||||
if (paneId == null) {
|
||||
paneId = idOrPane; // raw-pane fallback (reap, gate timeout, pane-addressed callers)
|
||||
}
|
||||
WorkspaceControl.PaneLocation loc = usesTabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) throw e;
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated peer tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether any configured profile places peers in their own tab (so tabs may need cleanup). */
|
||||
private boolean usesTabPlacement() {
|
||||
return profiles.values().stream().anyMatch(BridgedConfig.Profile::tabPlacement);
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
// --- spawn-readiness gate (CB-306) ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until the pane reports an injectable state or the configured
|
||||
* timeout elapses. On timeout, close the pane (self-reap) and throw.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long deadline = nowMillis.getAsLong() + spawnReadyTimeoutMs;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (agents.status(paneId).injectable()) {
|
||||
log.debug("peer pane={} reached injectable state", paneId);
|
||||
return;
|
||||
}
|
||||
sleeper.run();
|
||||
}
|
||||
log.warn("peer pane={} did not become injectable within {}ms — closing", paneId, spawnReadyTimeoutMs);
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " did not reach injectable state within "
|
||||
+ spawnReadyTimeoutMs + "ms");
|
||||
}
|
||||
|
||||
/**
|
||||
* A concrete {@link PeerHandle} wrapping herdr agent coordinates, the profile that spawned it,
|
||||
* and the session identity the launch resolved (CB-547a): the bridge's logical name and the
|
||||
* peer's own session id, both null when the spawn carried no identity.
|
||||
*/
|
||||
private record WorkerHandle(String id, String terminalId, String profile,
|
||||
String sessionName, String agentSessionId) implements PeerHandle {
|
||||
}
|
||||
|
||||
// --- shared helpers ------------------------------------------------------------------------
|
||||
|
||||
/** Put {@code k → v} only when {@code v} is present (non-null, non-blank). */
|
||||
protected static void putIfPresent(Map<String, String> m, String k, String v) {
|
||||
if (v != null && !v.isBlank()) {
|
||||
m.put(k, v);
|
||||
}
|
||||
}
|
||||
|
||||
/** Host env lookup that tolerates an unconfigured (null/blank) var name — returns null then. */
|
||||
protected String resolveEnv(String name) {
|
||||
return (name == null || name.isBlank()) ? null : env.apply(name);
|
||||
}
|
||||
|
||||
/**
|
||||
* The parity-neutral git-forge token grant (CB-302): when {@code cfg} opts in via
|
||||
* {@code gitTokenEnv} and the token resolves, inject {@code GITEA_TOKEN} plus its paired
|
||||
* {@code GITEA_HOST}. Push over SSH is unaffected; the only incremental grant is PR-create.
|
||||
* Peer-neutral, so every herdr adapter reuses it unchanged.
|
||||
*/
|
||||
protected void applyGitToken(Map<String, String> workerEnv, BridgedConfig.Profile cfg) {
|
||||
if (!cfg.hasGitToken()) {
|
||||
return;
|
||||
}
|
||||
String gitToken = resolveEnv(cfg.gitTokenEnv());
|
||||
if (gitToken != null) {
|
||||
workerEnv.put("GITEA_TOKEN", gitToken);
|
||||
putIfPresent(workerEnv, "GITEA_HOST", resolveEnv(cfg.gitHostEnv()));
|
||||
}
|
||||
}
|
||||
|
||||
/** A fresh mutable env map — the conventional starting point for {@link #buildLaunch}. */
|
||||
/**
|
||||
* Seed a worker's environment (CB-511): the daemon's own {@code PATH}, then the profile's
|
||||
* {@code env:} entries.
|
||||
*
|
||||
* <p>Why this exists: bridged passes herdr an explicit env map, and herdr merges it into
|
||||
* <em>its own</em> process environment. So before this, a worker inherited whatever PATH the
|
||||
* herdr server happened to be started with — on this host, one from weeks earlier with no JDK
|
||||
* and no Maven, which left workers unable to run the build they were being asked to run. The
|
||||
* worker's toolchain must follow from configuration, not from how a long-lived daemon was
|
||||
* launched.
|
||||
*
|
||||
* <p>Adapter-specific variables are layered on top of this by {@code buildLaunch} and therefore
|
||||
* win. That ordering is deliberate and load-bearing: it stops a profile's {@code env:} from
|
||||
* overriding {@code ANTHROPIC_BASE_URL} and slipping past {@link
|
||||
* dev.ltms.bridged.guard.SubscriptionGuard}, which is checked against the profile's
|
||||
* {@code baseUrl} and nothing else.
|
||||
*/
|
||||
protected Map<String, String> baseEnv(BridgedConfig.Profile cfg) {
|
||||
Map<String, String> workerEnv = new LinkedHashMap<>();
|
||||
String path = env.apply("PATH");
|
||||
if (path != null && !path.isBlank()) {
|
||||
workerEnv.put("PATH", path);
|
||||
}
|
||||
if (cfg != null && cfg.env() != null) {
|
||||
workerEnv.putAll(cfg.env());
|
||||
}
|
||||
return workerEnv;
|
||||
}
|
||||
|
||||
/** Defensive copy of {@code argv} plus room to append launch flags. */
|
||||
protected static List<String> mutableArgv(List<String> argv) {
|
||||
return new ArrayList<>(argv);
|
||||
}
|
||||
|
||||
/**
|
||||
* Uninterruptible sleep — the production {@link #sleeper}. Tests supply their own no-op /
|
||||
* fast-faking sleeper so they never real-sleep.
|
||||
*/
|
||||
protected static void sleepUninterruptibly(long ms) {
|
||||
try {
|
||||
Thread.sleep(ms);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
// preserve the interrupt flag but continue — poll loops should not be aborted by an
|
||||
// interrupt that was not meant for them.
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,122 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a bridged worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every bridged worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A bridged worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
*/
|
||||
String sessionIdForDirectory(String directory) {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
} catch (IOException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,242 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import com.rabbitmq.client.AMQP;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.Connection;
|
||||
import com.rabbitmq.client.ConnectionFactory;
|
||||
import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
* port {@link InMemoryReplyInbox} implements as soft state.
|
||||
*
|
||||
* <p><strong>Mapping — consume-and-hold with deferred manual ack.</strong> Each target has a durable
|
||||
* queue {@code agent.<target>.inbox}. The gateway that owns the target starts a manual-ack consumer
|
||||
* ({@link #own}) that pulls persistent messages off that queue into an in-memory <em>held</em> map
|
||||
* (keyed by {@code msgId}) but does <em>not</em> ack them. {@link #peek} returns that snapshot;
|
||||
* {@link #ack} acks the broker delivery-tag and drops the entry. Because messages stay unacked until
|
||||
* the owning gateway actually drains them, a crash (or a {@code java -jar} bounce) before caller-ack
|
||||
* leaves them on the broker — it redelivers on reconnect. That is the durability the in-memory
|
||||
* adapter cannot give, with the port contract preserved.
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> {@link #own} declares the queue and starts the consumer;
|
||||
* {@link #release} cancels it. {@link #publish} sends to the queue but does <em>not</em> imply ownership
|
||||
* and does not attach a consumer. This split is required by CB-308 federation, where one gateway may
|
||||
* publish to an agent owned by another gateway; in that case the publisher must not compete for
|
||||
* deliveries.
|
||||
*
|
||||
* <p><strong>Dedup.</strong> The consumer keys the held map by {@code msgId}; a redelivered duplicate
|
||||
* (at-least-once, or a producer double-publish) is acked-and-dropped on arrival, so it never
|
||||
* double-queues.
|
||||
*
|
||||
* <p><strong>Visibility.</strong> Unlike the in-memory adapter, publish → broker → consumer is
|
||||
* asynchronous, so a {@link #peek} immediately after {@link #publish} may not yet see the message
|
||||
* (broker delivery latency). Callers that need the reply drained poll (as the primary already does);
|
||||
* the contract test waits for visibility. This is inherent to broker-backed delivery, not a defect.
|
||||
*
|
||||
* <p>The default deploy targets LavinMQ; a stock RabbitMQ speaks the same AMQP 0-9-1 (URI-only swap),
|
||||
* so the {@code @Tag("contract")} integration test runs against a RabbitMQ container.
|
||||
*/
|
||||
public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(AmqpReplyInbox.class);
|
||||
|
||||
private static final String QUEUE_PREFIX = "agent.";
|
||||
private static final String QUEUE_SUFFIX = ".inbox";
|
||||
|
||||
private final Connection connection;
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
/** A message pulled off the broker but not yet acked: its delivery-tag plus the port payload. */
|
||||
private record Held(long deliveryTag, InboxMessage message) {}
|
||||
|
||||
/** Connect to {@code uri} (e.g. {@code amqp://guest:guest@127.0.0.1:5672/}) and open the inbox. */
|
||||
public static AmqpReplyInbox open(String uri) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new AmqpReplyInbox(factory.newConnection("bridged-reply-inbox"));
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection) {
|
||||
this.connection = connection;
|
||||
try {
|
||||
this.channel = connection.createChannel();
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot open AMQP channel", e);
|
||||
}
|
||||
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
|
||||
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue).
|
||||
if (connection instanceof Recoverable recoverable) {
|
||||
recoverable.addRecoveryListener(new RecoveryListener() {
|
||||
@Override
|
||||
public void handleRecovery(Recoverable recoverable) {
|
||||
held.clear();
|
||||
log.info("AMQP connection recovered; cleared held replies for fresh redelivery");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void handleRecoveryStarted(Recoverable recoverable) {
|
||||
// no-op: we act once recovery completes
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void own(String target) {
|
||||
synchronized (channelLock) {
|
||||
if (consumerTags.containsKey(target)) {
|
||||
return; // already owning this target
|
||||
}
|
||||
String queue = queueName(target);
|
||||
try {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
AMQP.BasicProperties props = new AMQP.BasicProperties.Builder()
|
||||
.messageId(msgId)
|
||||
.deliveryMode(2) // persistent — survives a broker restart
|
||||
.contentType("text/plain")
|
||||
.build();
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
channel.basicPublish("", queueName(target), props, content.getBytes(StandardCharsets.UTF_8));
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot publish reply to " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
return perTarget.values().stream().map(Held::message).toList();
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return;
|
||||
}
|
||||
Held h;
|
||||
synchronized (perTarget) {
|
||||
h = perTarget.remove(msgId);
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(h.deliveryTag(), false);
|
||||
}
|
||||
} catch (IOException e) {
|
||||
// Ack didn't reach the broker: restore the entry so a later ack (or a redelivery after
|
||||
// reconnect) can retry. Keeps the at-least-once contract — a reply is never silently lost.
|
||||
synchronized (perTarget) {
|
||||
perTarget.putIfAbsent(msgId, h);
|
||||
}
|
||||
throw new IllegalStateException("cannot ack reply " + msgId + " on " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback(String target) {
|
||||
return (_, delivery) -> {
|
||||
String msgId = delivery.getProperties().getMessageId();
|
||||
long tag = delivery.getEnvelope().getDeliveryTag();
|
||||
if (msgId == null || msgId.isBlank()) {
|
||||
msgId = Long.toHexString(tag); // synthesize an id so dedup still has a key
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
duplicate = true;
|
||||
} else {
|
||||
perTarget.put(msgId, new Held(tag, new InboxMessage(msgId, target, content)));
|
||||
duplicate = false;
|
||||
}
|
||||
}
|
||||
if (duplicate) {
|
||||
// Redelivered duplicate: ack the new tag and drop it so the broker stops resending.
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(tag, false);
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
private static String queueName(String target) {
|
||||
return QUEUE_PREFIX + target + QUEUE_SUFFIX;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
try {
|
||||
channel.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP channel close: {}", e.toString());
|
||||
}
|
||||
try {
|
||||
connection.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP connection close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,569 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionException;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.locks.ReentrantLock;
|
||||
|
||||
/**
|
||||
* The blocking delegation feature (CB-104): deliver {@code content} into a worker and block until
|
||||
* the worker returns a <em>structured reply</em> via {@code bridge_reply} (the {@link Rendezvous}),
|
||||
* then hand that reply back. Delivery is the {@link Injector}'s job (the background poller sends it
|
||||
* when the worker is injectable); this service never drives the injector or scrapes the terminal —
|
||||
* completion is the worker's explicit reply, not a guess about {@code agent_status}.
|
||||
*
|
||||
* <p>Sends are serialized per session so exactly one reply can be outstanding per worker, which is
|
||||
* what lets a reply map unambiguously to its send (no cross-talk between concurrent callers).
|
||||
*
|
||||
* <p>If the worker never replies within the timeout, the caller gets a typed "still working" /
|
||||
* "queued" outcome — the message may still be mid-flight. A finished-but-unreplied turn is caught
|
||||
* by the CB-106 completion fallback (see {@link Rendezvous#resolveCompletion}).
|
||||
*
|
||||
* <p><strong>Async fire-and-poll (CB-107).</strong> A caller's MCP client caps a blocking call at
|
||||
* ~60s, but a real delegated task runs for minutes. {@link #sendAsync} therefore runs the same
|
||||
* blocking {@link #send} on a background virtual thread and hands back a <em>ticket</em> the caller
|
||||
* polls with {@link #poll}. The blocking and async paths share one code path (and the same per-target
|
||||
* serialization), so async inherits the reply + completion resolution behaviour for free.
|
||||
*/
|
||||
public final class MessageService {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MessageService.class);
|
||||
|
||||
/**
|
||||
* The window a fire-and-poll send waits for resolution — generous, since no caller is blocked on
|
||||
* it; a real delegated task resolves (reply or completion) well within this, and only a genuinely
|
||||
* hung worker rides it out.
|
||||
*/
|
||||
private static final long ASYNC_TIMEOUT_MS = 30 * 60 * 1_000L;
|
||||
|
||||
/** How long a finished (terminal) ticket is retained for polling before it is pruned. */
|
||||
private static final long TICKET_TTL_NANOS = 10 * 60 * 1_000_000_000L;
|
||||
|
||||
/** Outcome of a blocking send. */
|
||||
public enum Outcome {
|
||||
/** The worker called {@code bridge_reply}; {@code text} holds the structured answer. */
|
||||
REPLIED,
|
||||
/**
|
||||
* The worker's delegated turn finished without a {@code bridge_reply} (CB-106 fallback);
|
||||
* {@code text} is the scraped transcript tail rather than a structured answer.
|
||||
*/
|
||||
COMPLETED_UNREPLIED,
|
||||
/**
|
||||
* The worker ran the turn then wedged in an unrecoverable state (CB-109); {@code text} is the
|
||||
* failure context (e.g. the error screen). Terminal, but not a successful completion.
|
||||
*/
|
||||
WORKER_FAILED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||
* {@link #answer(String, String, long)} and the turn resumes.
|
||||
*/
|
||||
QUESTION,
|
||||
/** Timed out after the message was delivered — the worker is still working. */
|
||||
TIMED_OUT_WORKING,
|
||||
/** Timed out before delivery — the message is still queued for the worker. */
|
||||
TIMED_OUT_QUEUED,
|
||||
/** Another send to this session was in flight for the whole window. */
|
||||
BUSY,
|
||||
/**
|
||||
* An answer ({@link #answer(String, String, long)}) referenced a {@code turnId} that is no
|
||||
* longer open — the worker's {@code bridge_ask} already timed out or was answered.
|
||||
*/
|
||||
STALE_TURN
|
||||
}
|
||||
|
||||
/**
|
||||
* @param outcome how the send ended (or paused)
|
||||
* @param text the worker's answer when {@link #completed()} (a structured {@code bridge_reply}
|
||||
* for {@link Outcome#REPLIED}, a scraped transcript tail for
|
||||
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
||||
* else {@code null}
|
||||
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
||||
* {@link #answer(String, String, long)}), else {@code null}
|
||||
*/
|
||||
public record Reply(Outcome outcome, String text, String turnId) {
|
||||
/** A reply with no correlation id (the common terminal outcomes). */
|
||||
public Reply(Outcome outcome, String text) {
|
||||
this(outcome, text, null);
|
||||
}
|
||||
|
||||
/** Whether the worker's turn actually finished with an answer (replied or scraped). */
|
||||
public boolean completed() {
|
||||
return outcome == Outcome.REPLIED || outcome == Outcome.COMPLETED_UNREPLIED;
|
||||
}
|
||||
}
|
||||
|
||||
/** How a worker's {@code bridge_ask} (CB-205) resolved. */
|
||||
public enum AskOutcome {
|
||||
/** The primary answered; {@link AskResult#answer} carries it. */
|
||||
ANSWERED,
|
||||
/** No delegation was open to surface the question to — the worker has no one to ask. */
|
||||
NO_WAITER,
|
||||
/** The primary did not answer within the window. */
|
||||
TIMED_OUT
|
||||
}
|
||||
|
||||
/** The outcome of a worker's {@code bridge_ask}: how it resolved and (if answered) the answer. */
|
||||
public record AskResult(AskOutcome outcome, String answer) {
|
||||
}
|
||||
|
||||
/** Lifecycle phase of an async delegation ticket. */
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
PENDING,
|
||||
/** The worker's turn finished; {@link TaskView#reply} holds the answer. */
|
||||
DONE,
|
||||
/** The delegation could not complete (timed out, worker gone, or busy). */
|
||||
FAILED
|
||||
}
|
||||
|
||||
/**
|
||||
* A poll snapshot of an async delegation.
|
||||
*
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, else {@code null}
|
||||
* @param replySource {@code "reply"} (structured {@code bridge_reply}) or {@code "transcript"}
|
||||
* (completion scrape) when {@link Phase#DONE}, else {@code null}
|
||||
* @param detail a human note (live worker status while pending, or the failure reason)
|
||||
*/
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail) {
|
||||
}
|
||||
|
||||
/** An in-flight or finished async delegation, keyed by its ticket. */
|
||||
private record Task(String target, CompletableFuture<Reply> future, long createdNanos) {
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Injector injector;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ReplyInbox inbox;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
|
||||
/**
|
||||
* Create with an explicit {@link ReplyInbox} and optional {@link ReplyPushLoop}.
|
||||
*
|
||||
* @param pushLoop nullable — when non-null, the push loop is notified on the no-waiter reply
|
||||
* branch ({@link #reply}) so it can nudge the primary to drain the inbox
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with a metric registry (CB-502). Instrumenting here rather than at the REST and MCP
|
||||
* edges means both surfaces are counted by one piece of code and cannot drift.
|
||||
*
|
||||
* @param metrics nullable — when null, nothing is recorded
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop, Metrics metrics) {
|
||||
this.agents = agents;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
this.pushLoop = pushLoop;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Create with an explicit {@link ReplyInbox} and no push loop. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox) {
|
||||
this(agents, injector, rendezvous, inbox, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor that uses a default {@link InMemoryReplyInbox}. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous) {
|
||||
this(agents, injector, rendezvous, new InMemoryReplyInbox());
|
||||
}
|
||||
|
||||
/** Current lifecycle status of a worker (the {@code GET /sessions/{id}/status} surface). */
|
||||
public AgentStatus status(String target) {
|
||||
return agents.status(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code bridge_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
* result is <em>not</em> a failure — the reply is held for later drain.
|
||||
*
|
||||
* <p><strong>Do NOT use this for mid-turn questions.</strong> {@code bridge_ask} /
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* @return always {@code true} — the reply either resolved a live send or was queued
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(BridgedMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content);
|
||||
// A rising inbox share is the signal CB-307 exists to make visible: the worker finished but
|
||||
// nobody was waiting, so delivery now depends on the push loop and a drain.
|
||||
count(BridgedMetrics.REPLIES, "path", "inbox");
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
}
|
||||
|
||||
/** Record a counter sample when a registry is wired; a no-op in unit tests. */
|
||||
private void count(String name, String... labels) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(name, labels);
|
||||
}
|
||||
}
|
||||
|
||||
/** Count a send's terminal outcome and pass the reply through unchanged. */
|
||||
private Reply recorded(Reply r) {
|
||||
String label = sendOutcomeLabel(r.outcome());
|
||||
if (label != null) {
|
||||
count(BridgedMetrics.SENDS, "outcome", label);
|
||||
}
|
||||
return r;
|
||||
}
|
||||
|
||||
/** Map a terminal send outcome to its metric label, or {@code null} for non-terminal ones. */
|
||||
private static String sendOutcomeLabel(Outcome o) {
|
||||
return switch (o) {
|
||||
case REPLIED -> "replied";
|
||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Abandon any send still waiting on {@code target} because its session has gone away (CB-516).
|
||||
*
|
||||
* <p>Without this, tearing a worker down left its rendezvous waiter open: a blocking
|
||||
* {@code bridge_send} kept blocking, and an async one kept reporting {@code PENDING} until
|
||||
* {@link #ASYNC_TIMEOUT_MS} — thirty minutes — even though the worker provably no longer
|
||||
* existed and the delegation could never complete. Worse, {@code poll} already had the evidence
|
||||
* (it calls {@code liveStatus} to build its detail string and gets back {@code "unknown"}) and
|
||||
* reported {@code PENDING} anyway.
|
||||
*
|
||||
* <p>Resolving the waiter as a failure — rather than letting it time out — also means the
|
||||
* outcome is counted, so a torn-down delegation stops being invisible to {@code /metrics}.
|
||||
*
|
||||
* @return true if a live waiter was failed
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
return false; // nobody is blocked on this worker — nothing to abandon
|
||||
}
|
||||
boolean failed = rendezvous.resolveFailure(waiter, reason);
|
||||
if (failed) {
|
||||
log.debug("abandoned send to {}: {}", target, reason);
|
||||
}
|
||||
return failed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain (peek + ack) all pending inbox replies for {@code target}. At-least-once: returns the
|
||||
* messages and acknowledges them; an in-flight failure between returning and the caller
|
||||
* processing them re-surfaces them on a subsequent drain (the ack is local).
|
||||
*
|
||||
* @return the drained messages, newest last (FIFO); empty list if none
|
||||
*/
|
||||
public List<ReplyInbox.InboxMessage> drainReplies(String target) {
|
||||
var messages = inbox.peek(target);
|
||||
for (var msg : messages) {
|
||||
inbox.ack(target, msg.msgId());
|
||||
}
|
||||
return messages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis) {
|
||||
return send(target, content, timeoutMillis, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
||||
*
|
||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||
* queued, so a throwing hook fails the send cleanly (the waiter it already opened is closed and
|
||||
* nothing is left queued). It is <em>not</em> invoked when the send is {@link Outcome#BUSY}
|
||||
* (lock never taken). A caller uses this to record that <em>it</em> now owns the delegation's
|
||||
* reply routing (CB-548: {@code PrimaryRegistry} delegator ownership) — recording only on
|
||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
// Open the waiter BEFORE queueing delivery (CB-548). A fast reply — the worker already
|
||||
// injectable the instant we enqueue — otherwise arrives before the waiter is registered
|
||||
// and orphans into the inbox while this send blocks to the timeout (the enqueue-before-
|
||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
try {
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
return recorded(new Reply(
|
||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
}
|
||||
} finally {
|
||||
rendezvous.close(target, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker's mid-turn question (CB-205 reverse rendezvous): surface {@code question} to the
|
||||
* primary by resolving its open blocking {@code bridge_send}, then block this (worker) call until
|
||||
* the primary answers via {@link #answer} or {@code timeoutMillis} elapses. Identity is the
|
||||
* worker's own session — it does not address the primary.
|
||||
*
|
||||
* <p>Returns {@link AskOutcome#NO_WAITER} when no delegation is open to surface the question to
|
||||
* (nothing to answer it), {@link AskOutcome#ANSWERED} with the primary's answer, or
|
||||
* {@link AskOutcome#TIMED_OUT} if the primary stayed silent. The worker resumes its turn either
|
||||
* way — an answered ask hands back the answer; an unanswered one leaves it to proceed alone.
|
||||
*/
|
||||
public AskResult ask(String workerSession, String question, long timeoutMillis) {
|
||||
Rendezvous.AskTicket ticket = rendezvous.openAsk(workerSession);
|
||||
// Only the freshly-opening caller surfaces the question; a coalesced duplicate simply blocks on
|
||||
// the shared answer future that the fresh owner is already responsible for.
|
||||
if (ticket.fresh()) {
|
||||
// Register the reverse waiter first, then surface the question — so the answer, which can
|
||||
// arrive the instant the primary reacts, always finds an open waiter to resolve.
|
||||
if (!rendezvous.resolveQuestion(workerSession, question, ticket.turnId())) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("bridge_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the primary's answer for " + workerSession, e);
|
||||
} finally {
|
||||
// Only the fresh owner tears down the shared turn; a duplicate must leave it open.
|
||||
if (ticket.fresh()) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The primary's answer to a worker's {@code bridge_ask} (CB-205): resolve the worker's blocked
|
||||
* question identified by {@code turnId}, then — like a fresh {@link #send} — block for the worker's
|
||||
* eventual {@code bridge_reply} as it finishes the resumed turn. The worker session is derived from
|
||||
* {@code turnId}, never a caller argument.
|
||||
*
|
||||
* <p>Unlike {@link #send} this does not re-inject through the {@link Injector}: the worker is
|
||||
* mid-turn (already picked up), so the answer flows back through its own open {@code bridge_ask}
|
||||
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
||||
* is unblocked so a reply that lands the instant it resumes is not lost.
|
||||
*/
|
||||
public Reply answer(String turnId, String content, long timeoutMillis) {
|
||||
String workerSession = rendezvous.askSession(turnId);
|
||||
if (workerSession == null) {
|
||||
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
||||
}
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null);
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
// turn (it never re-entered the injector), so a silent worker rides out the window.
|
||||
return new Reply(Outcome.TIMED_OUT_WORKING, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fire-and-poll variant of {@link #send}: deliver {@code content} to {@code target} on a
|
||||
* background virtual thread and return immediately with a ticket to {@link #poll}. This is how a
|
||||
* long task is delegated without tripping the caller's MCP client call timeout.
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content) {
|
||||
return sendAsync(target, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(String, String)}, with the accepted-delivery hook of
|
||||
* {@link #send(String, String, long, Runnable)} — the running {@code send} invokes {@code onAccepted}
|
||||
* the moment it becomes the accepted target turn, so async flooding records delegator ownership
|
||||
* exactly as the blocking path does (CB-548).
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
CompletableFuture<Reply> future = CompletableFuture.supplyAsync(
|
||||
() -> send(target, content, ASYNC_TIMEOUT_MS, onAccepted), asyncExecutor);
|
||||
tasks.put(ticket, new Task(target, future, System.nanoTime()));
|
||||
pruneTerminalTickets();
|
||||
log.debug("async send {} -> {}", ticket, target);
|
||||
return ticket;
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket;
|
||||
* otherwise a {@link Phase#PENDING} view (with the live worker status as detail), a
|
||||
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
||||
*/
|
||||
public TaskView poll(String ticket) {
|
||||
Task task = tasks.get(ticket);
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future();
|
||||
if (!f.isDone()) {
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target()));
|
||||
}
|
||||
Reply r;
|
||||
try {
|
||||
r = f.getNow(null);
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
Throwable cause = (e instanceof CompletionException ce && ce.getCause() != null) ? ce.getCause() : e;
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage());
|
||||
}
|
||||
if (r.completed()) {
|
||||
String source = r.outcome() == Outcome.REPLIED ? "reply" : "transcript";
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null);
|
||||
}
|
||||
// A wedged worker (CB-109) carries the error context as its reason; the timeout/busy
|
||||
// outcomes carry none, so fall back to the outcome name.
|
||||
String detail = r.outcome() == Outcome.WORKER_FAILED && r.text() != null
|
||||
? r.text()
|
||||
: "no reply — " + r.outcome().name().toLowerCase();
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
return agents.status(target).name().toLowerCase();
|
||||
} catch (RuntimeException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/** Drop finished tickets older than the TTL so the registry cannot grow without bound. */
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = System.nanoTime() - TICKET_TTL_NANOS;
|
||||
tasks.values().removeIf(t -> t.future().isDone() && t.createdNanos() < cutoff);
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
public void close() {
|
||||
asyncExecutor.shutdown();
|
||||
}
|
||||
|
||||
/** Map a rendezvous {@link Rendezvous.Kind} onto its send {@link Outcome} (shared by send/answer). */
|
||||
private static Outcome outcomeOf(Rendezvous.Kind kind) {
|
||||
return switch (kind) {
|
||||
case REPLY -> Outcome.REPLIED;
|
||||
case COMPLETION -> Outcome.COMPLETED_UNREPLIED;
|
||||
case FAILED -> Outcome.WORKER_FAILED;
|
||||
case QUESTION -> Outcome.QUESTION;
|
||||
};
|
||||
}
|
||||
|
||||
private static boolean tryLock(ReentrantLock lock, long millis) {
|
||||
try {
|
||||
return lock.tryLock(Math.max(0, millis), TimeUnit.MILLISECONDS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the session send lock", e);
|
||||
}
|
||||
}
|
||||
|
||||
private static long remainingMillis(long deadlineNanos) {
|
||||
return (deadlineNanos - System.nanoTime()) / 1_000_000L;
|
||||
}
|
||||
}
|
||||
@@ -1,199 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Mechanism (b) of CB-307: a dedicated, status-gated push loop that nudges the primary's own
|
||||
* herdr pane when a worker reply lands with no live {@code bridge_send} to resolve it.
|
||||
*
|
||||
* <p>The loop is triggered by {@link #onReplyQueued(String)} (called from
|
||||
* {@link MessageService#reply} after the durable inbox publish). It checks four conditions
|
||||
* at each tick via {@link #decide(String, int)}, then either injects a drain nudge,
|
||||
* waits for the primary to become injectable, or stops reminding.
|
||||
*
|
||||
* <p>Bounded: at most {@link #maxReminders} nudges per target, with a configurable backoff
|
||||
* between them. The reply is never lost — the durable inbox is the backstop.
|
||||
*/
|
||||
public final class ReplyPushLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ReplyPushLoop.class);
|
||||
static final String NUDGE_FORMAT = "Worker %s returned a reply — run bridge_poll(target=%s) to collect it";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final ReplyInbox inbox;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final int maxReminders;
|
||||
private final long backoffMs;
|
||||
private final Metrics metrics; // CB-512: nullable — no registry in unit tests
|
||||
|
||||
/** Track targets that have an active schedule. */
|
||||
private final ConcurrentHashMap<String, Boolean> activeTargets = new ConcurrentHashMap<>();
|
||||
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs) {
|
||||
this(primaryRegistry, agents, inbox, scheduler, maxReminders, backoffMs, null);
|
||||
}
|
||||
|
||||
/** As above, with a metric registry (CB-512) so push outcomes are counted. */
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs, Metrics metrics) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.inbox = inbox;
|
||||
this.scheduler = scheduler;
|
||||
this.maxReminders = maxReminders;
|
||||
this.backoffMs = backoffMs;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.PUSH_NUDGES, "outcome", outcome);
|
||||
}
|
||||
}
|
||||
|
||||
// --- decision logic (package-private for unit-testing) -------------------------------------
|
||||
|
||||
/** The action the loop should take for a target at the given reminder count. */
|
||||
enum Action { INJECT, WAIT_BUSY, STOP }
|
||||
|
||||
/**
|
||||
* Pure decision function: examine the current state and return what the loop should do.
|
||||
*
|
||||
* @param target the worker session (target terminal id)
|
||||
* @param reminderCount how many nudges have been sent so far for this target
|
||||
* @return the action the caller should take
|
||||
*/
|
||||
Action decide(String target, int reminderCount) {
|
||||
// CB-532: the destination is per-delegation — the lead that sent this worker its work, not
|
||||
// "the primary". With two leads orchestrating one fleet the singular question has no right
|
||||
// answer, and answering it anyway interrupted whichever lead happened to call bridge_send
|
||||
// first with results it never asked for.
|
||||
var nudgeTarget = primaryRegistry.nudgeTargetFor(target);
|
||||
if (nudgeTarget.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}, stopping reminder", target);
|
||||
return Action.STOP;
|
||||
}
|
||||
if (inbox.peek(target).isEmpty()) {
|
||||
log.debug("push: inbox empty for {}, stopping reminder", target);
|
||||
return Action.STOP;
|
||||
}
|
||||
if (reminderCount >= maxReminders) {
|
||||
log.debug("push: reminder cap ({}) reached for {}, stopping", maxReminders, target);
|
||||
countNudge("exhausted");
|
||||
return Action.STOP;
|
||||
}
|
||||
String leadTerminal = nudgeTarget.get();
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(leadTerminal);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("push: status check failed for lead {}, will retry", leadTerminal, e);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
if (status.injectable()) {
|
||||
return Action.INJECT;
|
||||
}
|
||||
log.debug("push: lead {} is {} (not injectable), waiting", leadTerminal, status);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
|
||||
// --- public entrypoint ---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Called when a reply is queued for {@code target}. Idempotent per target: a second call while
|
||||
* a schedule is active is a no-op. The schedule nudges the primary, then schedules a follow-up
|
||||
* check (reminder on backoff, or re-check on WAIT_BUSY), until the inbox is empty or the cap
|
||||
* is reached.
|
||||
*/
|
||||
public void onReplyQueued(String target) {
|
||||
if (activeTargets.putIfAbsent(target, Boolean.TRUE) != null) {
|
||||
log.debug("push: already active for {}, ignoring duplicate trigger", target);
|
||||
return; // already scheduled
|
||||
}
|
||||
log.debug("push: starting reminder loop for {}", target);
|
||||
scheduleNext(target, 0);
|
||||
}
|
||||
|
||||
/** Execute one loop tick — called on the scheduler thread. */
|
||||
private void tick(String target, int reminderCount) {
|
||||
var action = decide(target, reminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(target, reminderCount);
|
||||
scheduleNext(target, reminderCount + 1);
|
||||
}
|
||||
// Re-check after the configured backoff; the primary may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(target, reminderCount);
|
||||
case STOP -> {
|
||||
activeTargets.remove(target);
|
||||
log.debug("push: reminder loop ended for {}", target);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Send the nudge and log the event. */
|
||||
private void injectNudge(String target, int reminderCount) {
|
||||
// Re-read rather than threading it down from decide(): the delegating lead can change
|
||||
// between the decision and the injection, and the nudge should follow the current one.
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: lead for {} disappeared before the nudge could be sent", target);
|
||||
return;
|
||||
}
|
||||
String leadTerminal = lead.get();
|
||||
String nudge = NUDGE_FORMAT.formatted(target, target);
|
||||
try {
|
||||
agents.send(leadTerminal, nudge);
|
||||
log.debug("push: nudge {}/{} sent to lead {} for target {}",
|
||||
reminderCount + 1, maxReminders, leadTerminal, target);
|
||||
countNudge("delivered");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} for target {} (reminder {}/{}): {}",
|
||||
leadTerminal, target, reminderCount + 1, maxReminders, e.toString());
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
private void scheduleNext(String target, int nextReminderCount) {
|
||||
scheduler.schedule(() -> tick(target, nextReminderCount), backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Whether any reminder loop is currently active for some target (CB-551). The idle-lead heartbeat
|
||||
* uses this to stand aside: while the push loop is actively nudging the lead, a concurrent
|
||||
* heartbeat injection would start a second competing turn in the same pane — racing loops multiply
|
||||
* turns and context burn. "Active" means a schedule exists in {@link #activeTargets}; the set is
|
||||
* bounded by what has been triggered, not by any persistent state.
|
||||
*/
|
||||
public boolean isActive() {
|
||||
return !activeTargets.isEmpty();
|
||||
}
|
||||
|
||||
/** Shut down the scheduler. Outstanding reminders are cancelled. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
activeTargets.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
public void close() {
|
||||
stop();
|
||||
}
|
||||
}
|
||||
@@ -1,94 +0,0 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* SPI for materializing a connected peer — the only way the bridge core creates or tears down
|
||||
* a peer process. Every launcher is a first-party, in-tree adapter selected by (future) profile
|
||||
* config; today's single adapter is the {@code ClaudeCodeLauncher} / Claude Code over herdr.
|
||||
*
|
||||
* <p>The core delegates spawn and teardown to this interface without knowing how the peer is set
|
||||
* up. Environment variables, CLI flags, subscription guards, transport (herdr tab/pane) layout,
|
||||
* and naming conventions are all adapter-private — the core sees only the returned
|
||||
* {@link PeerHandle} whose {@code id()} is the registry/routing key.
|
||||
*
|
||||
* <p>The interface is a superset of what {@code SessionManager} and {@code Bridged.main} call
|
||||
* on the concrete launcher today.
|
||||
*/
|
||||
public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The set of {@link Capability capabilities} this launcher declares. A peer whose profile
|
||||
* opts into a git-forge token should include {@link Capability#SELF_PR}; the base set for
|
||||
* the Claude Code herdr adapter is always {@code MID_TURN_ASK, WORKTREE, ORPHAN_REAP}.
|
||||
*/
|
||||
Set<Capability> capabilities();
|
||||
|
||||
/**
|
||||
* {@code profileName}/requestedCwd null/blank → default resolution. Returns after the peer
|
||||
* process is live (env + argv + placement complete). Never returns {@code null}.
|
||||
*
|
||||
* @param req the spawn parameters (profile, requested cwd, caller cwd)
|
||||
* @return a handle whose {@link PeerHandle#id()} is the registry/routing key
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
PeerHandle spawn(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The configured worker profile names — the set of names {@code spawn(profileName)} accepts.
|
||||
*/
|
||||
Set<String> profiles();
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
*
|
||||
* @return the resolved absolute path, never null/blank
|
||||
*/
|
||||
String effectiveCwd(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The parity-overlay file list for {@code profileName} (default list when unset). Used by
|
||||
* worktree provisioning to copy config files into the isolated checkout before spawning.
|
||||
*/
|
||||
List<String> parityOverlay(String profileName);
|
||||
|
||||
/**
|
||||
* The set of all agents this launcher currently tracks, transport-specific. Each element
|
||||
* exposes at minimum a pane-like {@code id()} matching this launcher's {@link PeerHandle}
|
||||
* scheme, plus transport-level status. Callers merge this set with the session registry to
|
||||
* build a live roster view.
|
||||
*/
|
||||
List<?> list();
|
||||
|
||||
/**
|
||||
* Reap orphaned peers left behind by a prior daemon process. Only peers whose naming scheme
|
||||
* matches this launcher's and whose nonce differs from the current process are eligible.
|
||||
* Best-effort: a failure to list or to stop any one peer is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
int reapOrphanWorkers();
|
||||
|
||||
/**
|
||||
* Tear a peer down by its registry/routing key ({@link PeerHandle#id()}). Tolerates an
|
||||
* already-gone peer. Also cleans up launcher-private resources (e.g. empty dedicated tabs)
|
||||
* when safe to do so.
|
||||
*/
|
||||
void stop(String id);
|
||||
|
||||
/**
|
||||
* Discard the context of the peer identified by {@code id}. Implementations must bypass normal
|
||||
* bridge delivery/turn accounting. Unsupported peer kinds return {@code false} without sending
|
||||
* a guessed command.
|
||||
*
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
}
|
||||
@@ -1,22 +0,0 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank()) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
PlacementCandidate first = ctx.candidates().getFirst();
|
||||
return new PlacementCandidate(first.profile(), null, first.weight(), first.maxLoad());
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
}
|
||||
@@ -1,277 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStreamReader;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* Production {@link Worktrees} implementation that shells {@code git} via {@link ProcessBuilder}.
|
||||
* Non-zero exits become {@link WorktreeException}. Worktree directories live under a configurable
|
||||
* root (default: a sibling {@code .bridged-worktrees} of the repo root) so they are never nested
|
||||
* inside the primary working tree.
|
||||
*/
|
||||
public final class GitWorktrees implements Worktrees {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(GitWorktrees.class);
|
||||
|
||||
/** Project-level MCP config. Present in the repo, so every worktree would otherwise inherit the
|
||||
* primary's IDE server mounts (a CB-523 worker edited the primary checkout; see the isolation
|
||||
* javadoc). Neutralized unconditionally. */
|
||||
private static final String MCP_CONFIG = ".mcp.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .mcp.json}: a valid, explicitly empty server map. */
|
||||
private static final String NEUTRAL_MCP_CONFIG = "{\n \"mcpServers\": {}\n}\n";
|
||||
|
||||
/** OpenCode's repo-level config. Tracked here, so it lands in every worktree, and it mounts the
|
||||
* primary's gitea and context7 servers with the primary's credentials. Neutralized so the worker
|
||||
* gets only the config its launcher writes via {@code OPENCODE_CONFIG}.
|
||||
*
|
||||
* <p>The reason has changed shape and is now stronger. It used to be a crash: the file carried
|
||||
* {@code {file:.secrets/...}} references to gitignored files that never reached a worktree, and
|
||||
* opencode refuses to start on a dangling reference (CB-543). Those credentials now live in one
|
||||
* shell-level store and the file reads them as {@code {env:...}}, so in a worktree the reference
|
||||
* resolves instead of failing. That is worse, not better: a member would silently inherit the
|
||||
* primary's admin-scoped {@code GITEA_ACCESS_TOKEN}. A loud crash became a quiet privilege leak,
|
||||
* so this entry protects a boundary now rather than papering over a startup error. */
|
||||
private static final String OPENCODE_CONFIG = "opencode.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code opencode.json}: a valid, empty JSON object. */
|
||||
private static final String NEUTRAL_OPENCODE_CONFIG = "{}\n";
|
||||
|
||||
/** Autoenv's repo-level config. Not tracked today, but re-landing it must stay safe: autoenv
|
||||
* authorizes by path, so a fresh worktree path is always unauthorized and its interactive prompt
|
||||
* would block every spawn — neutralize it so it can never be committed. */
|
||||
private static final String AUTOENV_CONFIG = ".autoenv";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .autoenv}: a valid, empty env file. */
|
||||
private static final String NEUTRAL_AUTOENV_CONFIG = "";
|
||||
|
||||
/**
|
||||
* A tracked project config that is hostile in a provisioned worktree, and what to replace it
|
||||
* with. {@link #file} is the repo-relative path; {@link #stub} is a neutral but VALID payload for
|
||||
* that file's format — a malformed stub would only trade one crash for another;
|
||||
* {@link #createIfAbsent} keeps {@code .mcp.json}'s long-standing behaviour of writing its stub
|
||||
* even when the repo carries no such file, whereas the others are only touched when present.
|
||||
*/
|
||||
private record WorktreeHostileConfig(String file, String stub, boolean createIfAbsent) {}
|
||||
|
||||
/** The worktree-hostile configs neutralized in every provisioned worktree, in order. */
|
||||
private static final List<WorktreeHostileConfig> WORKTREE_HOSTILE_CONFIGS = List.of(
|
||||
new WorktreeHostileConfig(MCP_CONFIG, NEUTRAL_MCP_CONFIG, true),
|
||||
new WorktreeHostileConfig(OPENCODE_CONFIG, NEUTRAL_OPENCODE_CONFIG, false),
|
||||
new WorktreeHostileConfig(AUTOENV_CONFIG, NEUTRAL_AUTOENV_CONFIG, false)
|
||||
);
|
||||
|
||||
private final String configuredRoot;
|
||||
private final SecureRandom random = new SecureRandom();
|
||||
private final AtomicLong seq = new AtomicLong();
|
||||
|
||||
/** Default constructor: worktree root is derived per-repo as {@code <repoRoot>/../.bridged-worktrees}. */
|
||||
public GitWorktrees() {
|
||||
this(null);
|
||||
}
|
||||
|
||||
/** @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of the repo root. */
|
||||
public GitWorktrees(String configuredRoot) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
String base = (baseRef == null || baseRef.isBlank()) ? "HEAD" : baseRef;
|
||||
String nonce = nonce();
|
||||
Path root = resolveRoot(repoRoot);
|
||||
Path path = root.resolve(nonce);
|
||||
try {
|
||||
Files.createDirectories(root);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
isolateToolSurface(wt);
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Neutralize the worktree's worktree-hostile project configs so a worker inherits only the tools
|
||||
* and environment its launcher mounts (the bridge via {@code --mcp-config}, the opencode config
|
||||
* via {@code OPENCODE_CONFIG}) — never the primary's.
|
||||
*
|
||||
* <p>This is unconditional, and it is not the same job as the parity overlay. The repo's own
|
||||
* committed {@code .mcp.json} declares the primary's IDE servers, so a fresh checkout mounts them
|
||||
* whether or not the overlay copies anything; a worker that inherits them navigates and edits
|
||||
* through tools bound to the <em>primary's</em> IntelliJ project, which silently hands it absolute
|
||||
* paths outside its own worktree. That is not hypothetical: a CB-523 worker made all 59 of its
|
||||
* edits in the primary checkout while compiling its worktree, so every build it ran was of code
|
||||
* that did not contain its changes. {@code opencode.json} is the same trap one tool over — tracked,
|
||||
* so it lands in every worktree, and it mounts gitea and context7 with the primary's own
|
||||
* credentials, which a member must never hold. {@code .autoenv} extends the principle to a
|
||||
* config that is not tracked today: autoenv authorizes by path, so a fresh worktree path is always
|
||||
* unauthorized and its interactive prompt would block every spawn, so re-landing one must be safe.
|
||||
*
|
||||
* <p>Where a config exists it is replaced by a valid neutral stub (an explicitly empty
|
||||
* map/object, or an empty env file — never a deletion, which would still let a later
|
||||
* {@code git checkout} restore the hostile copy). The {@code --skip-worktree} bit keeps the
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
}
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot neutralize " + cfg.file() + " in the worktree: "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing to remove", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.info("removing worktree {}", worktreePath);
|
||||
exec("git", "-C", repoRoot, "worktree", "remove", "--force", worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlay == null || overlay.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Path srcRoot = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
Path dstRoot = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (String rel : overlay) {
|
||||
Path src = srcRoot.resolve(rel).normalize();
|
||||
if (!Files.exists(src)) {
|
||||
log.debug("parity overlay source missing — skipping {}", rel);
|
||||
continue;
|
||||
}
|
||||
Path dst = dstRoot.resolve(rel).normalize();
|
||||
try {
|
||||
Files.createDirectories(dst.getParent());
|
||||
Files.copy(src, dst, StandardCopyOption.REPLACE_EXISTING, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
log.debug("copied parity overlay {}", rel);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy overlay " + rel + ": " + e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(dstRoot, rel)) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", rel);
|
||||
log.debug("marked overlay --skip-worktree {}", rel);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
String out = exec("git", "-C", cwd, "rev-parse", "--show-toplevel");
|
||||
return Path.of(out.trim()).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/** Resolve the directory that will hold per-session worktree checkouts. */
|
||||
private Path resolveRoot(String repoRoot) {
|
||||
if (configuredRoot != null && !configuredRoot.isBlank()) {
|
||||
return Path.of(configuredRoot).toAbsolutePath().normalize();
|
||||
}
|
||||
Path repo = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
return repo.resolveSibling(".bridged-worktrees");
|
||||
}
|
||||
|
||||
private String nonce() {
|
||||
return String.format("%06x", random.nextInt(1 << 24)) + "-" + seq.incrementAndGet();
|
||||
}
|
||||
|
||||
private boolean isTracked(Path worktreeRoot, String rel) {
|
||||
return exitCode("git", "-C", worktreeRoot.toString(), "ls-files", "--error-unmatch", rel) == 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a command and return its stdout. Non-zero exit → {@link WorktreeException} with both
|
||||
* stdout and stderr (merged by redirectErrorStream).
|
||||
*/
|
||||
private String exec(String... command) {
|
||||
String out;
|
||||
int code;
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try (BufferedReader r = new BufferedReader(new InputStreamReader(p.getInputStream(), StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(Collectors.joining("\n"));
|
||||
} catch (IOException e) {
|
||||
p.destroyForcibly();
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command) + "\n" + out);
|
||||
}
|
||||
code = p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
if (code != 0) {
|
||||
throw new WorktreeException("exit " + code + " for: " + String.join(" ", command)
|
||||
+ (out.isBlank() ? "" : "\n" + out));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
private int exitCode(String... command) {
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command));
|
||||
}
|
||||
return p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,70 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Periodic virtual-thread reaper that tears down {@code READY}/{@code DONE} sessions which have
|
||||
* exceeded their idle TTL. Modeled on {@link dev.ltms.bridged.inject.StatusPoller}: a single
|
||||
* virtual-thread loop, idempotent start/stop, and no {@code ScheduledExecutorService}.
|
||||
*/
|
||||
public final class SessionReaper {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
||||
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
||||
|
||||
private final SessionManager sessions;
|
||||
private final long idleTtlNanos;
|
||||
private final long intervalMillis;
|
||||
private volatile boolean running;
|
||||
private Thread thread;
|
||||
|
||||
/** Construct a reaper with the default 5-second polling interval. */
|
||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds) {
|
||||
this(sessions, idleTtlSeconds, DEFAULT_INTERVAL_MILLIS);
|
||||
}
|
||||
|
||||
/** Construct a reaper with an explicit polling interval (useful for tests). */
|
||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis) {
|
||||
this.sessions = sessions;
|
||||
this.idleTtlNanos = TimeUnit.SECONDS.toNanos(idleTtlSeconds);
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
/** Start the reaper loop on a virtual thread. Idempotent. */
|
||||
public synchronized void start() {
|
||||
if (running) return;
|
||||
running = true;
|
||||
thread = Thread.ofVirtual().name("session-reaper").start(this::loop);
|
||||
log.info("session reaper started (idle ttl {}s, interval {}ms)",
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos), intervalMillis);
|
||||
}
|
||||
|
||||
private void loop() {
|
||||
while (running) {
|
||||
try {
|
||||
sessions.reapIdle(idleTtlNanos);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("session reaper iteration failed; continuing", e);
|
||||
}
|
||||
sleep();
|
||||
}
|
||||
}
|
||||
|
||||
private void sleep() {
|
||||
try {
|
||||
Thread.sleep(intervalMillis);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
running = false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Stop the reaper loop. Idempotent. */
|
||||
public synchronized void stop() {
|
||||
running = false;
|
||||
if (thread != null) thread.interrupt();
|
||||
}
|
||||
}
|
||||
@@ -1,18 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Seam between {@link SessionManager} and git worktree operations. Tests use a recording fake. */
|
||||
public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree add <path> -b <branch> <baseRef|HEAD>. Returns the worktree path. */
|
||||
String add(String repoRoot, String branch, String baseRef);
|
||||
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/** Copy each existing overlay path repoRoot→worktree; mark tracked ones --skip-worktree. */
|
||||
void overlayParity(String repoRoot, String worktreePath, List<String> overlay);
|
||||
|
||||
/** git -C <cwd> rev-parse --show-toplevel — the repo root that owns cwd. */
|
||||
String repoRoot(String cwd);
|
||||
}
|
||||
@@ -1,27 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** Unit tests for PID → pane resolution (the herdr half of connection-based MCP identity). */
|
||||
class PaneLocatorTest {
|
||||
|
||||
private final PaneLocator loc = new PaneLocator(new FakeHerdr());
|
||||
|
||||
@Test
|
||||
void resolvesTerminalForAForegroundPid() {
|
||||
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForAPidInNoPane() {
|
||||
assertNull(loc.terminalForPid(999_999));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForNonPositivePid() {
|
||||
assertNull(loc.terminalForPid(0));
|
||||
assertNull(loc.terminalForPid(-1));
|
||||
}
|
||||
}
|
||||
@@ -1,304 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
||||
class CompletionResolverTest {
|
||||
|
||||
@Test
|
||||
void skipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
resolver.resolve("term_a", null); // no in-flight turn captured for this target
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a turn nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failSkipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
resolver.fail("term_a", null); // no in-flight turn, and no registered waiter to fall back to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a wedge nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void captureBaselineSkipsTheReadWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ X\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
resolver.captureBaseline("term_a"); // no send to attribute a later completion to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"with no waiting send there is no turn to baseline — skip the scrape");
|
||||
}
|
||||
|
||||
// --- CB-115 clean scrape: extract the last assistant block ----------------
|
||||
|
||||
@Test
|
||||
void extractsTheLastAssistantBlockStrippingChrome() {
|
||||
String raw = """
|
||||
⏺ Reading the file…
|
||||
|
||||
⏺ Done. The bug was an off-by-one in the loop bound.
|
||||
|
||||
╭──────────────────────────────────────╮
|
||||
│ > │
|
||||
╰──────────────────────────────────────╯
|
||||
⏵⏵ auto mode on · ? for shortcuts
|
||||
""";
|
||||
assertEquals("Done. The bug was an off-by-one in the loop bound.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void keepsMultiLineAssistantContent() {
|
||||
String raw = "⏺ Line one.\nLine two.\n❯ ";
|
||||
assertEquals("Line one.\nLine two.", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void fallsBackToRawTextWhenThereIsNoMarker() {
|
||||
String raw = "plain worker output with no glyph";
|
||||
assertEquals("plain worker output with no glyph", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void blankScrapeYieldsEmpty() {
|
||||
assertTrue(CompletionResolver.lastAssistantBlock("").isEmpty());
|
||||
assertTrue(CompletionResolver.lastAssistantBlock(null).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void stripsSpinnerAndRuleChrome() {
|
||||
String raw = """
|
||||
⏺ Channel check confirmed — your message got through.
|
||||
|
||||
✻ Brewed for 11s
|
||||
|
||||
─────────────────────────────────────
|
||||
""";
|
||||
assertEquals("Channel check confirmed — your message got through.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void cutsANextTurnPromptEchoAndTrailingTipsFromTheBlock() {
|
||||
// The exact turn-2 leak: the scrape captured the settled answer, then a "✻ Cooked" spinner,
|
||||
// then the NEXT turn's echoed prompt, then a "✶ Forming…" spinner and trailing tips/warnings
|
||||
// whose lines (⎿, ⚠) are not themselves chrome-terminated. Stopping at the first boundary
|
||||
// (the ✻ spinner) is what keeps every one of those interface lines out of the reply.
|
||||
String raw = """
|
||||
⏺ Channel confirmed — the bridge reply delivered successfully.
|
||||
|
||||
✻ Cooked for 9s
|
||||
|
||||
❯ Thanks. Now a small task: what is 17 * 23? Show just the number.
|
||||
|
||||
|
||||
|
||||
✶ Forming…
|
||||
⎿ Tip: Name your conversations with /rename
|
||||
⚠ claude.ai connectors are disabled because ANTHROPIC_API_KEY is set
|
||||
""";
|
||||
assertEquals("Channel confirmed — the bridge reply delivered successfully.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
// --- CB-115 misattribution guard: suppress a stale (unchanged) completion -------
|
||||
|
||||
@Test
|
||||
void suppressesACompletionWhoseScrapeIsUnchangedFromDelivery() {
|
||||
// Rapid back-to-back turn: the pane still shows the PREVIOUS turn's answer when this turn's
|
||||
// (misattributed) completion boundary fires. The scrape == the delivery baseline, so the
|
||||
// send must NOT be resolved with the stale answer.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ 391\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
// The turn as captured at delivery: its waiter, and the previous turn's answer still on screen.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn); // scrape still "391" == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(), "a completion with no output change must not resolve the send");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesACompletionWhoseScrapeChangedSinceDelivery() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ No, 391 = 17 × 23.\n❯ "); // the worker's real answer
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
// Delivery baseline was the previous turn's "391"; the scrape now differs → resolve.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(), "a completion with new output must resolve the send");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("No, 391 = 17 × 23.", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesSynchronouslyBeforePostTurnContextClearing() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ previous answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.captureBaseline("term_a");
|
||||
herdr.readText("⏺ answer that /clear would erase\n❯ ");
|
||||
|
||||
resolver.resolveBeforePostAction("term_a");
|
||||
|
||||
assertTrue(waiter.isDone(), "the answer is captured before the adapter sends /clear");
|
||||
assertEquals("answer that /clear would erase", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void suppressesAnUnchangedCompletionEvenWhenTheBlockExceedsTheScrapeCap() {
|
||||
// The fan-out issue-hunt finding: captureBaseline once stored the RAW (unclipped) assistant
|
||||
// block while resolve compares against a clip()'d tail. For a block longer than MAX_SCRAPE_CHARS
|
||||
// the two capped representations differ even when the pane never changed, so the CB-115
|
||||
// byte-identical guard failed to fire and a stale completion could resolve the send. Both sides
|
||||
// must clip identically; here an unchanged >cap block on rapid back-to-back turns stays suppressed.
|
||||
String longBlock = "⏺ " + "x".repeat(CompletionResolver.MAX_SCRAPE_CHARS + 500) + "\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(longBlock);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
resolver.captureBaseline("term_a"); // baseline is the clipped >cap block
|
||||
var turn = resolver.inFlight("term_a");
|
||||
assertEquals(CompletionResolver.MAX_SCRAPE_CHARS, turn.baseline().length(),
|
||||
"the delivery baseline is clipped to the same cap resolve() applies to the tail");
|
||||
|
||||
resolver.resolve("term_a", turn); // scrape unchanged → clipped tail == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(),
|
||||
"an unchanged >cap block must still be recognised as stale and suppressed");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenThereIsNoBaseline() {
|
||||
// No delivery baseline (e.g. the pre-turn read failed) ⇒ never suppress; the completion resolves.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ hello\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "with no baseline a completion resolves as before");
|
||||
assertEquals("hello", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenTheScrapeItselfFailsEvenWithABaselinePresent() {
|
||||
// The most important branch of the CB-115 guard: a failed read means the resolver could not
|
||||
// SEE the screen — "couldn't see", not "no change". It must still resolve the send (an empty
|
||||
// tail beats hanging until the caller's timeout), even though a baseline was captured. The
|
||||
// baseline here is "" (an empty pane at delivery), so without the !scrapeFailed clause the
|
||||
// byte-identical guard would wrongly match the empty tail and suppress.
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false); // agent.read throws HerdrException
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, ""); // empty pane baselined at delivery
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(),
|
||||
"a failed scrape must still resolve the send, not hang until the caller's timeout");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("", waiter.getNow(null).text(), "the tail is empty because the screen was unreadable");
|
||||
}
|
||||
|
||||
// --- CB-115/CB-116 fail guard: an already-done or absent waiter is left alone ---------
|
||||
|
||||
@Test
|
||||
void failLeavesAnAlreadyResolvedWaiterUntouchedAndSkipsTheScrape() {
|
||||
// The send was already resolved (e.g. by the worker's explicit reply) before fail fired.
|
||||
// fail must not overwrite that value, and must not even scrape the worker — nobody needs it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied"));
|
||||
|
||||
resolver.fail("term_a", turn);
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"fail must not scrape a waiter that is already done");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"fail must not overwrite the existing resolution");
|
||||
assertEquals("already replied", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void failFallsBackToTheRegisteredWaiterWhenThereIsNoInFlightTurn() {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fail falls back to the waiter currently registered on the Rendezvous and fails it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // send registered, but no captureBaseline ever ran
|
||||
resolver.fail("term_a", null); // no in-flight turn → fall back to the registered waiter
|
||||
|
||||
assertTrue(waiter.isDone(), "fail falls back to the registered waiter when no turn is in flight");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals("stuck on an error screen", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
// --- CB-116 waiter identity: a late completion never crosses into the next turn ---------
|
||||
|
||||
@Test
|
||||
void aLateCompletionForOneTurnNeverResolvesTheNextTurnsWaiter() {
|
||||
// The cross-turn stale reply the conversation test surfaced: turn N's completion fallback
|
||||
// fires AFTER turn N was resolved by an explicit bridge_reply and turn N+1 has opened its own
|
||||
// waiter on the same session. Resolving "whatever is waiting now" would hand turn N's stale
|
||||
// scrape to turn N+1; targeting turn N's captured waiter makes the late completion a no-op.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ turn N answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiterN = rendezvous.open("term_a"); // turn N's send
|
||||
// The turn as the injector captured it at delivery (waiter + pre-turn baseline).
|
||||
var turnN = new CompletionResolver.InFlight(waiterN, "an earlier answer");
|
||||
|
||||
// Turn N is resolved by the worker's explicit reply, and its send deregisters the waiter.
|
||||
assertTrue(rendezvous.resolve("term_a", "N replied"));
|
||||
rendezvous.close("term_a", waiterN); // the sender's finally, before the next turn opens
|
||||
|
||||
// Turn N+1's send opens its own waiter on the same session (CB-548: open fails if the
|
||||
// previous waiter is still registered, so a clean turn deregisters it first as above).
|
||||
var waiterN1 = rendezvous.open("term_a");
|
||||
|
||||
resolver.resolve("term_a", turnN); // turn N's completion fallback finally fires
|
||||
|
||||
assertFalse(waiterN1.isDone(), "turn N's late completion must not resolve turn N+1's waiter");
|
||||
assertEquals(Rendezvous.Kind.REPLY, waiterN.getNow(null).kind(),
|
||||
"turn N stays resolved by its own reply");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "turn N+1 is still awaiting its own resolution");
|
||||
}
|
||||
}
|
||||
@@ -1,590 +0,0 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.session.FakeWorktrees;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.WorktreeRequest;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Parity tests for the MCP tool adapters — they must produce the same outcomes as the REST routes,
|
||||
* since both drive the same {@link MessageService}/{@link Rendezvous}. The MCP wire protocol itself
|
||||
* is the SDK's concern; here we test the thin adapter logic directly.
|
||||
*/
|
||||
class BridgeMcpTest {
|
||||
|
||||
private static final String T = "term_a";
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private final Rendezvous rendezvous = new Rendezvous();
|
||||
private final InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
private final MessageService messages = new MessageService(agents, new Injector(agents), rendezvous, inbox);
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
// CB-520: the inbox only peeks/acks targets it owns.
|
||||
inbox.own(T);
|
||||
}
|
||||
|
||||
private static String textOf(McpSchema.CallToolResult r) {
|
||||
return ((McpSchema.TextContent) r.content().getFirst()).text();
|
||||
}
|
||||
|
||||
private static ClaudeCodeLauncher workerService(FakeHerdr h, String baseUrl, Set<String> allow) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", baseUrl, "coder", null, "BRIDGED_WORKER_TOKEN", null,
|
||||
"tab", "bridged-workers", "worker: {profile} #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(allow), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> "tok");
|
||||
}
|
||||
|
||||
private static SessionManager sessionManager(FakeHerdr h, String baseUrl, Set<String> allow) {
|
||||
return new SessionManager(workerService(h, baseUrl, allow));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendThenReplyRoundTrips() throws Exception {
|
||||
// bridge_send blocks; bridge_reply resolves it with the worker's structured answer.
|
||||
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.send(messages, "term_a", "review this", 4000L));
|
||||
|
||||
// Wait until the send has opened its waiter so the reply resolves it (CB-307: reply now
|
||||
// queues in the inbox if no waiter is open, which would break the round-trip).
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "send should have opened its waiter");
|
||||
|
||||
McpSchema.CallToolResult reply = BridgeMcp.reply(messages, "term_a", "LGTM");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
|
||||
McpSchema.CallToolResult res = send.get(6, TimeUnit.SECONDS);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertEquals("LGTM", textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void asyncSendReturnsATicketThenPollReportsTheReply() throws Exception {
|
||||
// wait:false parity — a ticket is issued, resolved by a reply, and surfaced by bridge_poll.
|
||||
McpSchema.CallToolResult accepted = BridgeMcp.sendAsync(messages, "term_a", "do it");
|
||||
assertNotEquals(Boolean.TRUE, accepted.isError());
|
||||
String out = textOf(accepted);
|
||||
assertTrue(out.contains("ticket="), out);
|
||||
String ticket = out.substring(out.indexOf("ticket=") + "ticket=".length()).trim();
|
||||
|
||||
// Wait until the send has opened its waiter before replying (CB-307: reply never errors,
|
||||
// so the old retry-on-error pattern no longer works — it would queue instead of resolve).
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "send should have opened its waiter");
|
||||
|
||||
McpSchema.CallToolResult reply = BridgeMcp.reply(messages, "term_a", "async LGTM");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
|
||||
// Poll until the async send completes and reports the reply.
|
||||
McpSchema.CallToolResult polled = BridgeMcp.poll(messages, ticket, null);
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (!textOf(polled).contains("async LGTM") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(10);
|
||||
polled = BridgeMcp.poll(messages, ticket, null);
|
||||
}
|
||||
assertEquals("async LGTM", textOf(polled));
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollUnknownTicketIsAnError() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.poll(messages, "task-999", null);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("unknown ticket"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendTimesOutWithAWorkingNote() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.send(messages, "term_a", "hi", 120L);
|
||||
assertNotEquals(Boolean.TRUE, res.isError(), "a timeout is informational, not a tool error");
|
||||
assertTrue(textOf(res).contains("no reply"), "got: " + textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendRejectsMissingArgs() {
|
||||
assertTrue(BridgeMcp.send(messages, null, "hi", null).isError());
|
||||
assertTrue(BridgeMcp.send(messages, "term_a", " ", null).isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyWithNoPendingSendIsQueuedNotError() {
|
||||
// CB-307: a reply with no open send is now queued in the inbox, not an error.
|
||||
McpSchema.CallToolResult res = BridgeMcp.reply(messages, "term_a", "orphan");
|
||||
assertNotEquals(Boolean.TRUE, res.isError(), "a queued reply is not an error");
|
||||
assertEquals("delivered", textOf(res));
|
||||
|
||||
// The reply is drainable by target.
|
||||
var drained = messages.drainReplies("term_a");
|
||||
assertEquals(1, drained.size());
|
||||
assertEquals("orphan", drained.getFirst().content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgePollWithTargetDrainsReplies() {
|
||||
// A reply with no open send queues it in the inbox.
|
||||
BridgeMcp.reply(messages, "term_a", "queued-msg");
|
||||
|
||||
// bridge_poll with target drains the inbox.
|
||||
McpSchema.CallToolResult res = BridgeMcp.poll(messages, null, "term_a");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String text = textOf(res);
|
||||
assertTrue(text.contains("queued-msg"), "the drained reply should appear in the result");
|
||||
|
||||
// Second drain returns empty.
|
||||
McpSchema.CallToolResult empty = BridgeMcp.poll(messages, null, "term_a");
|
||||
assertEquals("[]", textOf(empty));
|
||||
}
|
||||
|
||||
@Test
|
||||
void askThenAnswerRoundTrips() throws Exception {
|
||||
// The primary delegates and blocks; wait until its waiter is open before the worker asks.
|
||||
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.send(messages, "term_a", "do X", 5000L));
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send must be waiting for the ask to surface to");
|
||||
|
||||
// The worker asks mid-turn; the call blocks for the primary's answer.
|
||||
CompletableFuture<McpSchema.CallToolResult> ask = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.ask(messages, "term_a", "which config?", 5000L));
|
||||
|
||||
// The primary's send unblocks with the question and a turnId to answer on.
|
||||
McpSchema.CallToolResult q = send.get(6, TimeUnit.SECONDS);
|
||||
assertNotEquals(Boolean.TRUE, q.isError());
|
||||
String qt = textOf(q);
|
||||
assertTrue(qt.contains("[question]"), qt);
|
||||
String afterMarker = qt.substring(qt.indexOf("turnId=\"") + "turnId=\"".length());
|
||||
String turnId = afterMarker.substring(0, afterMarker.indexOf('"'));
|
||||
|
||||
// The primary answers via bridge_send(turnId); this blocks again for the worker's reply.
|
||||
CompletableFuture<McpSchema.CallToolResult> answer = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.answer(messages, turnId, "config.yaml", 5000L));
|
||||
|
||||
// The worker's ask returns the answer — it resumes the same turn.
|
||||
assertEquals("config.yaml", textOf(ask.get(6, TimeUnit.SECONDS)));
|
||||
|
||||
// The resumed worker replies, resolving the answering send (wait for the reopened waiter).
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the answer should have reopened a waiter");
|
||||
McpSchema.CallToolResult reply = BridgeMcp.reply(messages, "term_a", "done");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
assertEquals("done", textOf(answer.get(6, TimeUnit.SECONDS)));
|
||||
}
|
||||
|
||||
@Test
|
||||
void askFromANonWorkerConnectionIsAnError() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.ask(messages, null, "which config?", 500L);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("workers only"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void answerToAStaleTurnIsAnError() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.answer(messages, "term_a#999", "too late", 500L);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("no longer open"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsTheNewWorkersSessionAndPane() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sm = sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(sm, null);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_new_1\""), out);
|
||||
// CB-519: the "paneId" wire field now carries the host-unique opaque id, not the herdr pane.
|
||||
MemberSession s = sm.roster().getFirst();
|
||||
assertTrue(out.contains("\"paneId\":\"" + s.paneId() + "\""), out);
|
||||
assertNotEquals("w9:pRoot_1", s.paneId(), "the id is decoupled from the herdr pane coordinate");
|
||||
assertTrue(out.contains("\"status\":\"spawning\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnOffAllowlistProfileWithoutTouchingHerdr() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res =
|
||||
BridgeMcp.spawn(sessionManager(h, "https://api.anthropic.com", Set.of("gx00.gw")), null);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("subscription boundary"));
|
||||
assertFalse(h.called("agent.start"), "the guard must block before any spawn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnUnknownProfileAsAnError() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res =
|
||||
BridgeMcp.spawn(sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), "nope");
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("unknown worker profile"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnPassesTheRequestedCwdToTheWorker() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, null, "/req/dir", null, null, null);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
// Protocol 19: the requested cwd roots the worker's pane at creation (tab.create).
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, Object> create = (Map<String, Object>) h.lastCall("tab.create").params();
|
||||
assertEquals("/req/dir", create.get("cwd"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesListsConfiguredProfilesAndDefault() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.profiles(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("ltms-local"), out);
|
||||
assertTrue(out.contains("\"default\":\"ltms-local\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsTrackedWorkers() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary",
|
||||
new WorktreeRequest("cb-304", null));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), "");
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"sessionId\":\"" + s.terminalId() + "\""), out);
|
||||
assertTrue(out.contains("\"paneId\":\"" + s.paneId() + "\""), out);
|
||||
assertTrue(out.contains("\"profile\":\"ltms-local\""), out);
|
||||
assertTrue(out.contains("\"state\":\"spawning\""), out);
|
||||
assertTrue(out.contains("\"worktree\":\"" + s.worktree() + "\""), out);
|
||||
assertTrue(out.contains("\"branch\":\"" + s.branch() + "\""), out);
|
||||
assertTrue(out.contains("\"owner\":\"term_primary\""), out);
|
||||
assertTrue(out.contains("\"liveStatus\":\"unknown\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsLeadsAndFlagsTheCallersOwnRow() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions,
|
||||
Map.of("term_me", "opus-5.0", "term_peer", "gpt-sol-5.6"), "term_me");
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"name\":\"opus-5.0\""), out);
|
||||
assertTrue(out.contains("\"name\":\"gpt-sol-5.6\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_peer\""), out);
|
||||
// The caller's own row is flagged, and only the caller's — a peer must be distinguishable
|
||||
// from self without a second bridge_whoami call.
|
||||
assertEquals(1, out.split("\"self\":true", -1).length - 1, out);
|
||||
assertTrue(out.indexOf("term_me") < out.indexOf("\"self\":true"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsBothHalvesEvenWhenEmpty() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), "");
|
||||
|
||||
// An absent "leads" key is what made an empty member roster read as "no peers" (CB-535).
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"leads\":[]"), out);
|
||||
assertTrue(out.contains("\"members\":[]"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsALeadHerdrCannotSeeAsUnknown() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions,
|
||||
Map.of("term_ghost", "gone-away"), "term_me");
|
||||
|
||||
// Reported, not hidden: an unreachable peer is exactly what a would-be sender needs to see.
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"name\":\"gone-away\""), out);
|
||||
assertTrue(out.contains("\"status\":\"unknown\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopTearsDownAWorkerByPane() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.stop(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), "w9:pW");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertEquals("stopped w9:pW", textOf(res));
|
||||
assertTrue(h.called("pane.close"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopRequiresAPaneId() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
assertTrue(BridgeMcp.stop(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), " ").isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckReturnsConfirmationForValidArgs() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.ack(messages, "term_a", "msg-1");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("msg-1"), "response should mention the msgId");
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckRejectsMissingArgs() {
|
||||
assertTrue(BridgeMcp.ack(messages, null, "msg-1").isError());
|
||||
assertTrue(BridgeMcp.ack(messages, "term_a", null).isError());
|
||||
assertTrue(BridgeMcp.ack(messages, " ", "msg-1").isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckRemovesSpecificReply() {
|
||||
// Queue a reply and capture its msgId.
|
||||
BridgeMcp.reply(messages, "term_a", "orphan");
|
||||
var before = messages.drainReplies("term_a");
|
||||
assertEquals(1, before.size(), "one reply in the inbox");
|
||||
String msgId = before.getFirst().msgId();
|
||||
|
||||
// Publish the same reply again and ack it via bridge_ack surface.
|
||||
BridgeMcp.reply(messages, "term_a", "orphan-again");
|
||||
var peeked = messages.drainReplies("term_a");
|
||||
assertEquals(1, peeked.size(), "one fresh reply in the inbox");
|
||||
|
||||
// ackReply works (no-op since published with a different UUID, but callable).
|
||||
assertDoesNotThrow(() -> messages.ackReply("term_a", msgId));
|
||||
}
|
||||
|
||||
@Test
|
||||
void statusReportsLiveAgentStatus() {
|
||||
FakeHerdr blocked = new FakeHerdr().agentStatus("blocked");
|
||||
AgentControl blockedAgents = new AgentControl(blocked);
|
||||
McpSchema.CallToolResult res = BridgeMcp.status(
|
||||
new MessageService(blockedAgents, new Injector(blockedAgents), rendezvous), "term_a");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertEquals("blocked", textOf(res));
|
||||
}
|
||||
|
||||
// --- bridge_whoami: the caller's own identity, so an agent never has to guess its role -------
|
||||
|
||||
@Test
|
||||
void whoamiReportsThePrimaryAsPrimaryAndNothingElse() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(
|
||||
Principal.primary(100), sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"primary\""), out);
|
||||
// The primary owns no session — leaking a sessionId here would invite it to reply as one.
|
||||
assertFalse(out.contains("sessionId"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void whoamiReportsAWorkerWithItsRegisteredSession() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary",
|
||||
new WorktreeRequest("cb-517", null));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(Principal.worker(s.terminalId(), 200), sessions);
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"worker\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"" + s.terminalId() + "\""), out);
|
||||
assertTrue(out.contains("\"profile\":\"ltms-local\""), out);
|
||||
assertTrue(out.contains("\"worktree\":\"" + s.worktree() + "\""), out);
|
||||
assertTrue(out.contains("\"branch\":\"" + s.branch() + "\""), out);
|
||||
assertTrue(out.contains("\"owner\":\"term_primary\""), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker the registry has no record of — it outlived a daemon restart — must still learn the
|
||||
* load-bearing fact. Degrading to "I don't know who you are" would put it back to guessing,
|
||||
* which is the failure this tool exists to remove.
|
||||
*/
|
||||
@Test
|
||||
void whoamiStillReportsWorkerRoleWhenTheSessionIsUnregistered() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(Principal.worker("term_orphan", 200),
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"worker\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_orphan\""), out);
|
||||
assertFalse(out.contains("profile"), out); // nothing invented for a session we don't track
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548: an architect reports its role and which gateway-local slot its pane is bound to —
|
||||
* the same shape as a lead, under the architect key, so it can tell a peer where to reach it.
|
||||
*/
|
||||
@Test
|
||||
void whoamiReportsAnArchitectWithItsSlotAndPane() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(
|
||||
Principal.architect("lead-designer", "term_design", 400),
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"architect\""), out);
|
||||
assertTrue(out.contains("\"architect\":\"lead-designer\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_design\""), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548: an architect SEND delegates as its own pane (recording the per-target delegation) but
|
||||
* must NEVER become the legacy singleton "primary" fallback — the per-target map does not cure
|
||||
* the singleton, so an architect left there would draw no-delegation inbox nudges meant for a
|
||||
* primary. Only PRIMARY callers (the unnamed primary and named leads alike) may claim it, and
|
||||
* the decision keys on the resolved role, not name/kind sniffing.
|
||||
*/
|
||||
@Test
|
||||
void architectSendDoesNotClaimThePrimarySingletonButALeadSendStillCan() {
|
||||
// Architect SEND: does not change the legacy primary fallback.
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(reg, "term_design", Principal.architect("design", "term_design", 400));
|
||||
assertTrue(reg.primaryTerminal().isEmpty(),
|
||||
"an architect must never become the legacy primary fallback");
|
||||
|
||||
// Lead SEND (a named PRIMARY) still claims it — preserved from CB-530/CB-532.
|
||||
PrimaryRegistry leadReg = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(leadReg, "term_lead_opus", Principal.leader("opus", "term_lead_opus", 100));
|
||||
assertEquals("term_lead_opus", leadReg.primaryTerminal().orElseThrow(),
|
||||
"a named lead is a primary and may claim the fallback");
|
||||
|
||||
// Unnamed primary likewise.
|
||||
PrimaryRegistry primaryReg = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(primaryReg, "term_p", Principal.primary(50));
|
||||
assertEquals("term_p", primaryReg.primaryTerminal().orElseThrow(),
|
||||
"an unnamed primary may claim the fallback");
|
||||
|
||||
// A null caller (legacy/no-auth path) records nothing.
|
||||
PrimaryRegistry legacy = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(legacy, "term_x", null);
|
||||
assertTrue(legacy.primaryTerminal().isEmpty(), "no caller means nothing is recorded");
|
||||
}
|
||||
|
||||
// --- member taxonomy (CB-557) ---------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void bridgeListReportsMembersNotWorkersAndCarriesEachRole() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
sessions.acquire("ltms-local", MemberRole.REVIEWER, null, "/caller/proj", "term_primary", null);
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), "");
|
||||
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"members\":"), "the roster half is named members: " + out);
|
||||
assertFalse(out.contains("\"workers\":"), "the old key must be gone: " + out);
|
||||
assertTrue(out.contains("\"role\":\"reviewer\""), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* Role and profile are separate axes, so the roster has to report both. Two members on one
|
||||
* backend may still be allowed to do entirely different things.
|
||||
*/
|
||||
@Test
|
||||
void aRosterRowCarriesBothItsRoleAndItsProfile() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
sessions.acquire("ltms-local", MemberRole.DEV, null, "/caller/proj", "term_primary", null);
|
||||
sessions.acquire("ltms-local", MemberRole.REVIEWER, null, "/caller/proj", "term_primary", null);
|
||||
|
||||
String out = textOf(BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), ""));
|
||||
|
||||
assertTrue(out.contains("\"role\":\"dev\""), out);
|
||||
assertTrue(out.contains("\"role\":\"reviewer\""), out);
|
||||
assertEquals(2, out.split("\"profile\":\"ltms-local\"", -1).length - 1,
|
||||
"both members share one profile — that is the point: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnDefaultsTheRoleToDev() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, null, null, null, null, null);
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("\"role\":\"dev\""), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnAcceptsAnExplicitRole() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, "architect",
|
||||
null, null, null, null);
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("\"role\":\"architect\""), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnUnknownRoleAndNamesTheValidOnes() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, "worker",
|
||||
null, null, null, null);
|
||||
|
||||
assertEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("architect, dev, reviewer"), textOf(res));
|
||||
}
|
||||
}
|
||||
@@ -1,881 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.GuardException;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** The step-4 launch-flag injection: the bridge MCP + reply charter are appended to the argv. */
|
||||
class ClaudeCodeLauncherTest {
|
||||
|
||||
private ClaudeCodeLauncher service(FakeHerdr herdr, List<String> argv, String mcpUrl) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
argv, "tab", "bridged-workers", "worker: {profile} #{n}", mcpUrl, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
}
|
||||
|
||||
/** The {@code args} of the last agent.start — protocol 19: everything after the executable. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private List<String> spawnedArgs(FakeHerdr herdr) {
|
||||
return (List<String>) ((Map<String, Object>) herdr.lastCall("agent.start").params()).get("args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void appendsBridgeMcpAndReplyCharterWhenMcpUrlSet() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("claude"), "http://127.0.0.1:8765/mcp").spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertTrue(args.contains("--mcp-config"));
|
||||
assertTrue(args.stream().anyMatch(a -> a.contains("\"bridge\"") && a.contains("http://127.0.0.1:8765/mcp")),
|
||||
"inline bridge MCP config present");
|
||||
assertTrue(args.contains("--append-system-prompt"));
|
||||
assertTrue(args.stream().anyMatch(a -> a.contains("bridge_reply")), "reply charter present");
|
||||
}
|
||||
|
||||
@Test
|
||||
void startRetriesWhileTheSeedShellBoots() {
|
||||
// tab.create returns before the seed shell reaches its prompt; herdr refuses agent.start
|
||||
// into a not-ready pane with agent_pane_busy. The launcher must wait it out, not fail.
|
||||
FakeHerdr herdr = new FakeHerdr().agentPaneBusyTimes(2);
|
||||
long[] clock = {0};
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
Map.of("ltms-local", new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null)),
|
||||
"ltms-local", _ -> null,
|
||||
0, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn succeeds once the shell is ready");
|
||||
assertEquals(3, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"two busy rejections, then the successful start");
|
||||
}
|
||||
|
||||
@Test
|
||||
void startResolvesTheExecutableFromKindAndDropsArgvZero() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("claude"), null).spawn();
|
||||
|
||||
Map<?, ?> start = (Map<?, ?>) herdr.lastCall("agent.start").params();
|
||||
assertEquals("claude", start.get("kind"), "herdr launches the canonical executable by kind");
|
||||
// CB-533: the shared fixture pins model "coder", so the model flag is the whole args list.
|
||||
// What this test guards is that argv[0] is NOT repeated — herdr supplies it from `kind`.
|
||||
assertEquals(List.of("--model", "coder"), start.get("args"),
|
||||
"the configured executable is not repeated in args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noBridgeFlagsWhenMcpUrlAbsent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("claude", "--verbose"), null).spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("--mcp-config"), "no bridge mount without mcpUrl");
|
||||
assertFalse(args.contains("--append-system-prompt"), "no reply charter without mcpUrl");
|
||||
// CB-533: the model flag is independent of the MCP mount — pinning the model is not part of
|
||||
// "mount the bridge", so an unmounted worker still runs the model its profile names.
|
||||
assertEquals(List.of("--verbose", "--model", "coder"), args,
|
||||
"the operator's own args are preserved, in order, ahead of the model flag");
|
||||
}
|
||||
|
||||
private ClaudeCodeLauncher multiProfile(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile gx10 = new BridgedConfig.Profile("gx10", "http://gx10.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
BridgedConfig.Profile ollama = new BridgedConfig.Profile("ollama", "http://ollama.ltms.dev", null,
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx10.gw", "ollama.ltms.dev")),
|
||||
Map.of("gx10", gx10, "ollama", ollama), "gx10", _ -> "tok");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnPicksTheNamedProfilesBaseUrl() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
multiProfile(herdr).spawn("ollama");
|
||||
|
||||
assertEquals("http://ollama.ltms.dev", startEnv(herdr).get("ANTHROPIC_BASE_URL"),
|
||||
"the named profile's base_url");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnUnknownProfile() {
|
||||
try (FakeHerdr herdr = new FakeHerdr()) {
|
||||
assertThrows(IllegalArgumentException.class, () -> multiProfile(herdr).spawn("nope"));
|
||||
}
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static String startCwd(FakeHerdr herdr) {
|
||||
// Protocol 19: the worker's cwd is set at pane creation (tab.create), where the seed
|
||||
// shell — which the agent starts into — is rooted.
|
||||
Object v = ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("cwd");
|
||||
return v == null ? null : v.toString();
|
||||
}
|
||||
|
||||
@Test
|
||||
void requestedCwdRootsTheWorker() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("ccs", "ltms-local"), null).spawn("ltms-local", "/work/proj", "/caller/home");
|
||||
assertEquals("/work/proj", startCwd(herdr), "an explicit spawn cwd wins over everything");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profileConfigCwdBeatsTheCallerCwd() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile("ltms-local", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"w #{n}", null, "/pinned/dir", null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("ltms-local", cfg), "ltms-local", _ -> null);
|
||||
svc.spawn("ltms-local", null, "/caller/home");
|
||||
assertEquals("/pinned/dir", startCwd(herdr), "a profile-pinned cwd overrides the caller's");
|
||||
}
|
||||
|
||||
@Test
|
||||
void inheritsTheCallerCwdWhenNothingElseIsSet() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("ccs", "ltms-local"), null).spawn("ltms-local", null, "/primary/project");
|
||||
assertEquals("/primary/project", startCwd(herdr), "no explicit/config cwd → inherit the primary's");
|
||||
}
|
||||
|
||||
// --- CB-302 git-forge token injection (worker checkpoint grant) ------------
|
||||
|
||||
/** Protocol 19: the worker's env is injected at pane creation (tab.create), not agent.start. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, String> startEnv(FakeHerdr herdr) {
|
||||
return (Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||
}
|
||||
|
||||
@Test
|
||||
void injectsForgeTokenAndHostWhenProfileGrantsIt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"impl", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "impl"), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
"GITEA_ACCESS_TOKEN", null); // parityOverlay null; gitHostEnv null → defaults to GITEA_HOST
|
||||
Function<String, String> host = name -> switch (name) {
|
||||
case "GITEA_ACCESS_TOKEN" -> "gt-secret";
|
||||
case "GITEA_HOST" -> "git.ltms.dev";
|
||||
default -> null;
|
||||
};
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("impl", cfg), "impl", host).spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertEquals("gt-secret", env.get("GITEA_TOKEN"), "the forge token is injected for a granting profile");
|
||||
assertEquals("git.ltms.dev", env.get("GITEA_HOST"), "the paired forge host rides along with the token");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noForgeTokenWhenProfileDoesNotGrantIt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// gitTokenEnv unset (12-arg ctor); the env would resolve a token if asked, proving the gate
|
||||
// is the profile config, not a missing env var.
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile("ltms-local", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("ccs"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("ltms-local", cfg), "ltms-local",
|
||||
_ -> "would-be-secret").spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("GITEA_TOKEN"), "no forge token when the profile does not opt in");
|
||||
assertNull(env.get("GITEA_HOST"), "no forge host without a granted token");
|
||||
}
|
||||
|
||||
// --- CB-117 orphan reap: the pure predicate --------------------------------
|
||||
|
||||
@Test
|
||||
void isForeignWorkerMatchesOurSchemeWithANonSelfNonce() {
|
||||
assertTrue(ClaudeCodeLauncher.isForeignWorker("claude-ollama-be09c2-2", "aaaaaa"),
|
||||
"a bridge worker name with a different nonce is a prior daemon's orphan");
|
||||
assertTrue(ClaudeCodeLauncher.isForeignWorker("claude-gx10-4127af-11", "aaaaaa"),
|
||||
"profile and multi-digit seq are still parsed; foreign nonce ⇒ reap");
|
||||
}
|
||||
|
||||
@Test
|
||||
void isForeignWorkerSparesOurOwnLiveWorkersAndNonWorkers() {
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("claude-ollama-abcdef-3", "abcdef"),
|
||||
"a worker with THIS process's nonce is ours and live — never reap it");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker(null, "abcdef"), "an unnamed agent is not a worker");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("claude", "abcdef"), "a bare kind name is not a worker");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("my-repl", "abcdef"), "a user's own label is not a worker");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("claude-ollama-XYZ123-2", "abcdef"),
|
||||
"a non-hex nonce does not match our scheme");
|
||||
}
|
||||
|
||||
// --- CB-117 orphan reap: the wiring through stop() -------------------------
|
||||
|
||||
private static long paneCloseCount(FakeHerdr herdr, String paneId) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("pane.close"))
|
||||
.filter(c -> paneId.equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapsAForeignOrphanButSparesOurOwnWorkerAndUserSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
herdr.withAgent("claude-ollama-be09c2-2", "term_orphan", "wQ:pF", "wQ:t8") // prior daemon's leak
|
||||
.withAgent("claude-gx10-" + svc.nameNonce() + "-1", "term_mine", "wQ:pMine", "wQ:tMine"); // ours, live
|
||||
// (the fake's default unnamed term_a stands in for a user's own Claude session)
|
||||
|
||||
int reaped = svc.reapOrphanWorkers();
|
||||
|
||||
assertEquals(1, reaped, "exactly the one foreign-nonce orphan is reaped");
|
||||
assertEquals(1, paneCloseCount(herdr, "wQ:pF"), "the orphan's pane is closed");
|
||||
assertEquals(0, paneCloseCount(herdr, "wQ:pMine"), "our own live worker's pane is left running");
|
||||
assertEquals(0, paneCloseCount(herdr, "w2:p7"), "a user's own session is never touched");
|
||||
assertTrue(herdr.called("tab.close"), "the orphan's now-empty dedicated tab is closed too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapCountsAnAlreadyGoneOrphanAsReaped() {
|
||||
FakeHerdr herdr = new FakeHerdr().paneCloseFailsWith("pane_not_found");
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
herdr.withAgent("claude-ollama-0d856d-3", "term_gone", "wQ:pS", "wQ:tD");
|
||||
|
||||
assertEquals(1, svc.reapOrphanWorkers(),
|
||||
"a pane that vanished between list and close is a successful reap, not a failure");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIsSkippedWhenHerdrCannotBeListed() {
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false); // agent.list throws
|
||||
assertEquals(0, multiProfile(herdr).reapOrphanWorkers(), "a listing failure reaps nothing and does not throw");
|
||||
}
|
||||
|
||||
// --- PeerHandle indirection ----------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void spawnReturnsPeerHandleWithHostUniqueOpaqueId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn must return a non-null handle");
|
||||
// CB-519: id() is a host-unique opaque UUID, decoupled from the herdr pane coordinate.
|
||||
assertNotEquals("w9:pRoot_1", handle.id(),
|
||||
"handle.id() must NOT be the herdr pane id");
|
||||
assertDoesNotThrow(() -> UUID.fromString(handle.id()),
|
||||
"handle.id() must be a UUID: " + handle.id());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsPeerHandleWithCorrectTerminalId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, "/caller"));
|
||||
|
||||
assertEquals("term_new_1", handle.terminalId(), "handle.terminalId() must equal the agent's terminalId");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesIncludeMidTurnAskWorktreeOrphanReap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
Set<Capability> caps = svc.capabilities();
|
||||
|
||||
assertTrue(caps.contains(Capability.MID_TURN_ASK), "every Claude Code peer supports mid-turn ask");
|
||||
assertTrue(caps.contains(Capability.WORKTREE), "every CLI peer supports worktree cwd");
|
||||
assertTrue(caps.contains(Capability.ORPHAN_REAP), "every herdr launcher supports orphan reap");
|
||||
assertTrue(caps.contains(Capability.CONTEXT_RESET), "Claude Code supports /clear");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearContextUsesTheClaudeCommandThroughTheOwningHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("claude"), null);
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertTrue(svc.clearContext(handle.id()));
|
||||
|
||||
Map<?, ?> prompt = (Map<?, ?>) herdr.lastCall("agent.prompt").params();
|
||||
assertEquals("/clear", prompt.get("text"));
|
||||
assertEquals("w9:pRoot_1", prompt.get("target"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesIncludeSelfPrWhenProfileHasGitToken() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"impl", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "impl"), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
"GITEA_ACCESS_TOKEN", null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("impl", cfg), "impl",
|
||||
_ -> "tok");
|
||||
|
||||
assertTrue(svc.capabilities().contains(Capability.SELF_PR),
|
||||
"a profile with a git token grants SELF_PR");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesExcludeSelfPrWhenNoGitToken() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
assertFalse(svc.capabilities().contains(Capability.SELF_PR),
|
||||
"no git token profile → no SELF_PR capability");
|
||||
}
|
||||
|
||||
@Test
|
||||
void effectiveCwdViaSpawnRequestMatchesExistingResolution() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
String cwd = svc.effectiveCwd(new SpawnRequest("ltms-local", "/work/proj", "/caller/home"));
|
||||
|
||||
assertEquals("/work/proj", cwd, "effectiveCwd via SpawnRequest must match the three-arg resolution");
|
||||
}
|
||||
|
||||
// --- CB-547a: durable session identity (mint / resume / no-identity legacy) -----------------
|
||||
|
||||
@Test
|
||||
void freshSpawnMintsASessionIdAndPassesTheName() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, null, "my-session", null));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int flag = args.indexOf("--session-id");
|
||||
assertTrue(flag >= 0 && flag + 1 < args.size(), "--session-id present: " + args);
|
||||
String minted = args.get(flag + 1);
|
||||
assertDoesNotThrow(() -> UUID.fromString(minted), "--session-id is a valid UUID: " + minted);
|
||||
assertEquals("my-session", args.get(args.indexOf("-n") + 1), "the logical name rides as -n");
|
||||
assertEquals(minted, handle.agentSessionId(),
|
||||
"the resume handle is the minted id, known before the agent has written anything");
|
||||
assertEquals("my-session", handle.sessionName(), "the handle carries the logical name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSpawnPassesDashRAndNeverASessionId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, null, "my-session", "cb-resume-1"));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("--session-id"), "--session-id must NOT be passed on a resume (conflicts with -r)");
|
||||
assertEquals("cb-resume-1", args.get(args.indexOf("-r") + 1), "-r carries the prior session id");
|
||||
assertEquals("cb-resume-1", handle.agentSessionId(), "a resume adopts the prior id as its own");
|
||||
assertEquals("my-session", handle.sessionName(), "the logical name survives a resume");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noIdentitySpawnKeepsTheLegacyArgvAndCarriesNoSessionHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, null));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("--session-id"), "no identity → no --session-id");
|
||||
assertFalse(args.contains("-n"), "no identity → no -n");
|
||||
assertFalse(args.contains("-r"), "no identity → no -r");
|
||||
assertNull(handle.agentSessionId(), "no identity → no resume handle");
|
||||
assertNull(handle.sessionName(), "no identity → no logical name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesIncludeSessionNameAndSessionResume() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
Set<Capability> caps = svc.capabilities();
|
||||
assertTrue(caps.contains(Capability.SESSION_NAME), "Claude Code surfaces the bridge's logical name (-n)");
|
||||
assertTrue(caps.contains(Capability.SESSION_RESUME), "Claude Code can relaunch onto a prior conversation (-r)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesViaPeerLauncherMatchesExistingApi() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
|
||||
assertEquals(Set.of("gx10", "ollama"), svc.profiles(), "profiles() via PeerLauncher must match");
|
||||
}
|
||||
|
||||
@Test
|
||||
void defaultProfileViaPeerLauncherMatches() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
|
||||
assertEquals("gx10", svc.defaultProfile(), "defaultProfile() via PeerLauncher must match");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopViaPeerLauncherTearsDownByHandleId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
svc.stop(handle.id());
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "stop via handle.id() must close the pane");
|
||||
}
|
||||
|
||||
// --- CB-519: host-unique id, decoupled from the pane coordinate ------------------------------
|
||||
|
||||
@Test
|
||||
void twoSpawnsOnTheSamePaneNeverCollideOnHostUniqueId() {
|
||||
// Two spawns may be placed on the same herdr pane coordinate (e.g. a pane that was reused
|
||||
// or re-reported after a restart); the host-unique id must not collide even then.
|
||||
FakeHerdr herdr = new FakeHerdr().pinNextStarts(2, "term_shared", "w9:pShared");
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle a = svc.spawn(new SpawnRequest(null, null, null));
|
||||
PeerHandle b = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotEquals(a.id(), b.id(),
|
||||
"two spawns on the same pane coordinate get distinct host-unique ids");
|
||||
assertNotEquals("w9:pShared", a.id(), "id is not the pane coordinate");
|
||||
assertNotEquals("w9:pShared", b.id(), "id is not the pane coordinate");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopResolvesTheHostUniqueIdToThePaneThatSpawnedIt() {
|
||||
// CB-519: id() != paneId, so stop(id) must tear down the exact pane the id names — and no
|
||||
// other live peer's pane.
|
||||
FakeHerdr herdr = new FakeHerdr(); // deterministic panes w9:pRoot_1, w9:pRoot_2 per spawn
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle a = svc.spawn(new SpawnRequest(null, null, null));
|
||||
PeerHandle b = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
svc.stop(b.id());
|
||||
|
||||
assertEquals(1, paneCloseCount(herdr, "w9:pRoot_2"), "stop(b.id()) closes only b's pane");
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"), "a's pane is untouched");
|
||||
}
|
||||
|
||||
// --- CB-306 spawn-readiness gate -----------------------------------------------------------
|
||||
|
||||
private static Map<String, BridgedConfig.Profile> workerConfigMap(String profile, String mcpUrl) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
profile, "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", profile), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", mcpUrl, null, null);
|
||||
return Map.of(cfg.profile(), cfg);
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnWaitsUntilInjectableThenReturnsHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // first status call sees UNKNOWN
|
||||
long[] clock = {0};
|
||||
boolean[] firstSleep = {true};
|
||||
// The sleeper: advance the fake clock, and on the first call flip the
|
||||
// agent status to IDLE so the next poll succeeds.
|
||||
Runnable sleeper = () -> {
|
||||
clock[0] += 300;
|
||||
if (firstSleep[0]) {
|
||||
herdr.agentStatus("idle");
|
||||
firstSleep[0] = false;
|
||||
}
|
||||
};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
5000, () -> clock[0], sleeper);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn returns a handle when worker becomes injectable");
|
||||
assertNotEquals("w9:pRoot_1", handle.id(),
|
||||
"handle id is a host-unique opaque id, not the started pane");
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"no pane.close when worker becomes injectable before timeout");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnThrowsPeerUnreachableWhenNeverInjectableAndReapsPane() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // always UNKNOWN
|
||||
long[] clock = {0};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
1000, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(
|
||||
PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertTrue(ex.getMessage().contains("w9:pRoot_1"),
|
||||
"exception message references the paneId: " + ex.getMessage());
|
||||
assertTrue(ex.getMessage().contains("1000"),
|
||||
"exception message references the timeout: " + ex.getMessage());
|
||||
assertTrue(clock[0] >= 1000, "fake clock advanced past the timeout: " + clock[0]);
|
||||
assertEquals(1, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"pane was closed on timeout (no orphan left behind)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsImmediatelyWhenGateIsDisabled() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// The default 6-arg constructor has spawnReadyTimeoutMs=0 (gate disabled).
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn returns a handle when the gate is disabled");
|
||||
assertFalse(herdr.called("agent.get"),
|
||||
"agent.get is never called when the gate is disabled (no polling)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnGateRespectsZeroTimeoutEvenWithFullConstructor() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
long[] clock = {0};
|
||||
|
||||
// Explicit zero timeout with the full testability constructor — should
|
||||
// skip polling entirely, just like the legacy default path.
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
0, () -> clock[0], () -> clock[0] += 1);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn still succeeds with zero timeout");
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"no orphan pane close from the gate path");
|
||||
assertDoesNotThrow(() -> UUID.fromString(handle.id()));
|
||||
}
|
||||
|
||||
// --- CB-511: worker environment seeding -----------------------------------------------------
|
||||
|
||||
@Test
|
||||
void workerInheritsTheDaemonPath() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
k -> "PATH".equals(k) ? "/opt/tools/bin:/usr/bin" : null).spawn();
|
||||
|
||||
assertEquals("/opt/tools/bin:/usr/bin", startEnv(herdr).get("PATH"),
|
||||
"a worker with no PATH cannot run the build it is asked to run");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profileEnvIsInjectedIntoTheWorker() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null, null, null,
|
||||
null, Map.of("JAVA_HOME", "/opt/jdk", "PATH", "/profile/bin"), null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
k -> "PATH".equals(k) ? "/daemon/bin" : null).spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertEquals("/opt/jdk", env.get("JAVA_HOME"), "profile env: is passed through");
|
||||
assertEquals("/profile/bin", env.get("PATH"), "an explicit profile PATH overrides the daemon's");
|
||||
}
|
||||
|
||||
/**
|
||||
* The security-relevant ordering. {@code SubscriptionGuard} is checked against the profile's
|
||||
* {@code baseUrl} only, so if a profile's {@code env:} could overwrite ANTHROPIC_BASE_URL a
|
||||
* worker could be pointed at an unguarded host while the guard passed on a benign one.
|
||||
*/
|
||||
@Test
|
||||
void profileEnvCannotOverrideGuardCheckedAnthropicVars() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null, null, null,
|
||||
null, Map.of("ANTHROPIC_BASE_URL", "http://evil.example.com"), null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null).spawn();
|
||||
|
||||
assertEquals("http://gx00.gw:8000", startEnv(herdr).get("ANTHROPIC_BASE_URL"),
|
||||
"the guard-checked baseUrl must win over any env: entry, or the boundary is bypassable");
|
||||
}
|
||||
|
||||
// ── CB-533: the model is pinned on the command line, not only in the environment ────────────
|
||||
|
||||
/** A launcher for a profile identical but for its {@code model:} — the only variable here. */
|
||||
private ClaudeCodeLauncher serviceWithModel(FakeHerdr herdr, String model) {
|
||||
BridgedConfig.Profile cfg = profileWithModel(model);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null);
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile profileWithModel(String model) {
|
||||
return new BridgedConfig.Profile("sonnet", "http://gx00.gw:8000", model, null,
|
||||
"BRIDGED_WORKER_TOKEN", List.of("ccs", "sonnet"), "tab", "bridged-workers",
|
||||
"w #{n}", "http://127.0.0.1:8765/mcp", null, null);
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConfiguredModelIsPassedAsAModelFlagAsWellAsTheEnvVar() {
|
||||
// ANTHROPIC_MODEL alone loses to `ccs`, which exports its own model family over whatever it
|
||||
// inherited — so a profile that set model: was silently overruled by its own launcher.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithModel(herdr, "claude-sonnet-5").spawn("sonnet", null, null);
|
||||
|
||||
assertEquals("claude-sonnet-5", startEnv(herdr).get("ANTHROPIC_MODEL"));
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int flag = args.indexOf("--model");
|
||||
assertTrue(flag >= 0, "the flag is what survives a wrapper argv like [ccs, sonnet]");
|
||||
assertEquals("claude-sonnet-5", args.get(flag + 1));
|
||||
}
|
||||
|
||||
@Test
|
||||
void theModelFlagComesLastSoItOutranksTheOperatorsOwnArgv() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithModel(herdr, "claude-sonnet-5").spawn("sonnet", null, null);
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertEquals(args.size() - 2, args.indexOf("--model"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithNoModelGetsNoModelFlag() {
|
||||
// gx10 deliberately leaves model: unset so ccs owns selection; adding a flag would make
|
||||
// this file a second source of truth for exactly the thing it declines to decide.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithModel(herdr, null).spawn("sonnet", null, null);
|
||||
|
||||
assertFalse(spawnedArgs(herdr).contains("--model"));
|
||||
assertNull(startEnv(herdr).get("ANTHROPIC_MODEL"));
|
||||
}
|
||||
|
||||
// --- CB-539: subscription-profile opt-in ----------------------------------------------------
|
||||
|
||||
/** A claude-code profile on the subscription: no baseUrl (by design), no off-sub endpoint. */
|
||||
private static BridgedConfig.Profile subscriptionCfg(String profile, String baseUrl) {
|
||||
return new BridgedConfig.Profile(
|
||||
profile, baseUrl, "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", profile), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
null, null, null, Map.of(), null, null, true);
|
||||
}
|
||||
|
||||
@Test
|
||||
void defaultRefusalIsPreservedForClaudeProfileWithNoBaseUrl() {
|
||||
// Requirement 1: absent subscription:true ⇒ byte-identical refusal to today. A claude-code
|
||||
// profile with no baseUrl and no subscription must still be refused (it would bill the sub).
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", null, "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
|
||||
GuardException ex = assertThrows(GuardException.class, () -> svc.spawn("ltms-local", null, null));
|
||||
assertTrue(ex.getMessage().contains("no ANTHROPIC_BASE_URL"),
|
||||
"the refusal names the missing baseUrl: " + ex.getMessage());
|
||||
assertEquals(0, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"nothing was spawned before the refusal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void subscriptionProfileSpawnsWithoutInjectedAnthropicVars() {
|
||||
// Requirement on subscription:true: no baseUrl is required (or injected), and neither
|
||||
// ANTHROPIC_BASE_URL nor ANTHROPIC_AUTH_TOKEN is injected even though the token env would
|
||||
// resolve one if asked.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = subscriptionCfg("sonnet", null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "would-be-token").spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"), "no baseUrl injected for a subscription profile");
|
||||
assertNull(env.get("ANTHROPIC_AUTH_TOKEN"), "no auth token injected for a subscription profile");
|
||||
assertEquals("sonnet", env.get("ANTHROPIC_MODEL"),
|
||||
"the model alias is still injected; only the subscription-boundary vars are dropped");
|
||||
}
|
||||
|
||||
@Test
|
||||
void subscriptionPlusBaseUrlIsRefused() {
|
||||
// Requirement 2: subscription:true + a baseUrl state opposite intents — refuse at spawn,
|
||||
// naming the profile, rather than silently picking a winner.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = subscriptionCfg("sonnet", "http://gx00.gw:8000");
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class,
|
||||
() -> svc.spawn("sonnet", null, null));
|
||||
assertTrue(ex.getMessage().contains("sonnet"), "refusal names the profile: " + ex.getMessage());
|
||||
assertTrue(ex.getMessage().contains("subscription"), "refusal explains the contradiction: " + ex.getMessage());
|
||||
assertEquals(0, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"nothing was spawned before the contradiction was refused");
|
||||
}
|
||||
|
||||
@Test
|
||||
void nonSubscriptionProfilesAreStillAllowlistChecked() {
|
||||
// Requirement 3: the guard keeps its teeth for every other profile — a base_url whose host is
|
||||
// not on the allowlist is still refused, whether or not any subscription profile exists.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile rogue = new BridgedConfig.Profile(
|
||||
"rogue", "http://evil.example.com:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "rogue"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(rogue.profile(), rogue), rogue.profile(), _ -> null);
|
||||
|
||||
GuardException ex = assertThrows(GuardException.class, () -> svc.spawn("rogue", null, null));
|
||||
assertTrue(ex.getMessage().contains("not on the"), "refusal cites the allowlist: " + ex.getMessage());
|
||||
assertEquals(0, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"nothing was spawned before the allowlist refusal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSubscriptionProfileHasAnEnvSuppliedAnthropicBindingStripped() {
|
||||
// CB-542: even a subscription profile whose env: carries ANTHROPIC_BASE_URL (or AUTH_TOKEN)
|
||||
// must not hand them to the worker — on the subscription path no guard would vet them. Config
|
||||
// load refuses this loudly; this launcher-side strip is the belt-and-braces that makes the
|
||||
// invariant hold for a profile built in code that never passed through that validation.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"sonnet", null, "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "sonnet"), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
null, null, null,
|
||||
Map.of("ANTHROPIC_BASE_URL", "http://evil.example.com",
|
||||
"ANTHROPIC_AUTH_TOKEN", "sk-ant-bad", "JAVA_HOME", "/opt/jdk"),
|
||||
null, null, true);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "would-be-token").spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"),
|
||||
"the unguarded endpoint must not survive into the worker");
|
||||
assertNull(env.get("ANTHROPIC_AUTH_TOKEN"),
|
||||
"the unguarded token must not survive into the worker");
|
||||
assertEquals("/opt/jdk", env.get("JAVA_HOME"),
|
||||
"only the Anthropic binding keys are stripped; the rest of env: still applies");
|
||||
}
|
||||
|
||||
// ── CB-557: role-aware tab labels ─────────────────────────────────────────────────────────
|
||||
|
||||
/** The {@code label} of every {@code tab.rename}, in call order. */
|
||||
private List<String> tabLabels(FakeHerdr herdr) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("tab.rename"))
|
||||
.map(c -> (String) ((Map<?, ?>) c.params()).get("label"))
|
||||
.toList();
|
||||
}
|
||||
|
||||
/** A profile with no {@code tabLabel:} of its own — the fleet template decides. */
|
||||
private ClaudeCodeLauncher labelService(FakeHerdr herdr, Supplier<String> fleetTemplate) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", null, null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null, 0, 0L, fleetTemplate);
|
||||
}
|
||||
|
||||
/**
|
||||
* The knob must reach the rename call. It was inert once — {@code HerdrPeerLauncher} accepted a
|
||||
* template while {@code Bridged} passed none, and the label stayed right only because the
|
||||
* fallback happened to match. Pin the wiring, not the coincidence.
|
||||
*/
|
||||
@Test
|
||||
void theFleetTemplateNamesTheRoleTheMemberWasSpawnedFor() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = labelService(herdr, () -> "{role}: {profile} #{n}");
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.REVIEWER));
|
||||
|
||||
assertEquals(List.of("reviewer: sonnet #1"), tabLabels(herdr));
|
||||
}
|
||||
|
||||
/** The counter is per role+profile, so a dev and a reviewer on one profile both start at #1. */
|
||||
@Test
|
||||
void theCounterRunsPerRoleAndProfileNotPerFleet() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = labelService(herdr, () -> "{role}: {profile} #{n}");
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.REVIEWER));
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertEquals(List.of("dev: sonnet #1", "reviewer: sonnet #1", "dev: sonnet #2"),
|
||||
tabLabels(herdr));
|
||||
}
|
||||
|
||||
/** No fleet template configured ⇒ the built-in default, still role-first. */
|
||||
@Test
|
||||
void aBlankFleetTemplateFallsBackToTheRoleFirstDefault() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
labelService(herdr, () -> null).spawn(
|
||||
new SpawnRequest("sonnet", null, null, null, null, MemberRole.ARCHITECT));
|
||||
|
||||
assertEquals(List.of("architect: sonnet #1"), tabLabels(herdr));
|
||||
assertEquals("{role}: {profile} #{n}", BridgedConfig.Fleet.DEFAULT_TAB_LABEL);
|
||||
}
|
||||
|
||||
/** A profile that wants its own label still outranks the fleet template. */
|
||||
@Test
|
||||
void aProfileTabLabelOverridesTheFleetTemplate() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "pinned {profile}", null, null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null, 0, 0L, () -> "{role}: {profile} #{n}")
|
||||
.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.REVIEWER));
|
||||
|
||||
assertEquals(List.of("pinned sonnet"), tabLabels(herdr));
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-559: the template is read per spawn, not captured at construction. This is what makes
|
||||
* {@code fleet.tabLabel} a hot key — a launcher built at boot must see an edit made an hour later
|
||||
* without being rebuilt.
|
||||
*/
|
||||
@Test
|
||||
void theTemplateIsReadOnEverySpawnSoAnEditTakesEffect() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicReference<String> template = new AtomicReference<>("{role}: {profile} #{n}");
|
||||
ClaudeCodeLauncher svc = labelService(herdr, template::get);
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
template.set("[{profile}] {role} {n}");
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertEquals(List.of("dev: sonnet #1", "[sonnet] dev 2"), tabLabels(herdr));
|
||||
}
|
||||
}
|
||||
@@ -1,555 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The composite router: profile → owning adapter for spawn/cwd/parity, pane id → owner for stop,
|
||||
* and fleet-wide union/dedup for list/reap/caps/profiles. Exercised through two real adapters —
|
||||
* claude-code + opencode — over one FakeHerdr, so each call is observed reaching the right adapter
|
||||
* (the started herdr agent name carries that adapter's {@code claude-}/{@code opencode-} prefix).
|
||||
*/
|
||||
class CompositePeerLauncherTest {
|
||||
|
||||
private ClaudeCodeLauncher claudeAdapter(FakeHerdr herdr) {
|
||||
// 12-arg back-compat Worker ctor → kind defaults to claude-code.
|
||||
BridgedConfig.Profile claude = new BridgedConfig.Profile("claude", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("claude", claude), "claude", _ -> null);
|
||||
}
|
||||
|
||||
private OpenCodeLauncher opencodeAdapter(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile gemini = new BridgedConfig.Profile("gemini", null, "google/gemini-2.5-pro",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("opencode"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null, "GITEA_ACCESS_TOKEN", null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", gemini), "gemini", _ -> "tok");
|
||||
}
|
||||
|
||||
private CompositePeerLauncher composite(FakeHerdr herdr) {
|
||||
return new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(herdr), opencodeAdapter(herdr)), "claude");
|
||||
}
|
||||
|
||||
/**
|
||||
* A minimal concrete HerdrPeerLauncher for policy tests. It either returns a fake handle for the
|
||||
* requested profile or throws, depending on {@code failProfiles}. buildLaunch is a stub; only
|
||||
* spawn/stop/list/caps/reap are exercised by the composite.
|
||||
*/
|
||||
private static final class StubLauncher extends HerdrPeerLauncher {
|
||||
private final Set<String> failProfiles;
|
||||
private final Map<String, Integer> spawnCounts = new HashMap<>();
|
||||
|
||||
StubLauncher(String prefix, FakeHerdr herdr,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Set<String> failProfiles) {
|
||||
super(prefix, new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
profiles, defaultProfile, _ -> null, 0L, System::currentTimeMillis, () -> { });
|
||||
this.failProfiles = Set.copyOf(failProfiles);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg) {
|
||||
return new Launch(Map.of(), List.of());
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String p = (req.profileName() == null || req.profileName().isBlank())
|
||||
? defaultProfile() : req.profileName();
|
||||
spawnCounts.merge(p, 1, Integer::sum);
|
||||
if (failProfiles.contains(p)) {
|
||||
throw new PeerUnreachableException(p + " is down");
|
||||
}
|
||||
return new PeerHandle() {
|
||||
@Override public String id() { return "pane-" + p; }
|
||||
@Override public String terminalId() { return "term-" + p; }
|
||||
@Override public String profile() { return p; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) { }
|
||||
|
||||
@Override
|
||||
public List<Agent> list() { return List.of(); }
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() { return EnumSet.noneOf(Capability.class); }
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() { return 0; }
|
||||
|
||||
int spawnCount(String profile) {
|
||||
return spawnCounts.getOrDefault(profile, 0);
|
||||
}
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile stubWorker(String profile) {
|
||||
return new BridgedConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null);
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile stubWorker(String profile, float weight, Integer maxLoad) {
|
||||
return new BridgedConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null,
|
||||
weight, maxLoad);
|
||||
}
|
||||
|
||||
/**
|
||||
* An <em>order-preserving</em> profile map. Never {@code Map.of} here: its iteration order is
|
||||
* salted per JVM run, and the weighted policy breaks an exact-weight tie on candidate order —
|
||||
* so a {@code Map.of} would make "which profile is tried first" a coin flip per run and any
|
||||
* assertion about the first attempt intermittently false.
|
||||
*/
|
||||
private static Map<String, BridgedConfig.Profile> ordered(String first, BridgedConfig.Profile a,
|
||||
String second, BridgedConfig.Profile b) {
|
||||
Map<String, BridgedConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put(first, a);
|
||||
m.put(second, b);
|
||||
return m;
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static String startedName(FakeHerdr herdr) {
|
||||
return (String) ((Map<String, Object>) herdr.lastCall("agent.start").params()).get("name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRoutesEachProfileToItsOwningAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
|
||||
composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("opencode-"),
|
||||
"the gemini profile is spawned by the opencode adapter: " + startedName(herdr));
|
||||
|
||||
composite.spawn(new SpawnRequest("claude", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"the claude profile is spawned by the claude-code adapter: " + startedName(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullProfileResolvesTheDefaultAndRoutesToItsOwner() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
composite(herdr).spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"a no-profile spawn resolves the default (claude) and routes to its adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unknownProfileIsRejected() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> composite.spawn(new SpawnRequest("nope", null, null)),
|
||||
"a profile no adapter declares is an error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesAndDefaultAreExposedAcrossAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(Set.of("claude", "gemini"), composite.profiles(),
|
||||
"profiles are the union of every adapter's profiles");
|
||||
assertEquals("claude", composite.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesAreTheUnionOfEveryAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher claude = claudeAdapter(herdr);
|
||||
OpenCodeLauncher opencode = opencodeAdapter(herdr);
|
||||
PeerLauncher composite = new CompositePeerLauncher(List.of(claude, opencode), "claude");
|
||||
|
||||
assertTrue(composite.capabilities().containsAll(claude.capabilities()),
|
||||
"the fleet offers every claude-code capability");
|
||||
assertTrue(composite.capabilities().containsAll(opencode.capabilities()),
|
||||
"the fleet offers every opencode capability (incl. SELF_PR from its git-token profile)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void listIsDeduplicatedByPaneIdAcrossAdaptersSharingHerdr() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
// Both adapters wrap the same herdr, so each list() returns the same global agent set;
|
||||
// the composite must return each pane once, not once per adapter.
|
||||
assertEquals(1, composite.list().size(),
|
||||
"the single herdr-tracked pane appears once, not duplicated per adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapSumsAcrossAdaptersAndEachAdapterReapsOnlyItsOwnPrefix() {
|
||||
// One foreign opencode orphan + one foreign claude orphan, from a prior daemon (different nonce).
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withAgent("opencode-gemini-ffffff-1", "term_o", "wQ:pO", "wQ:tO")
|
||||
.withAgent("claude-claude-eeeeee-1", "term_c", "wQ:pC", "wQ:tC");
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(2, composite.reapOrphanWorkers(),
|
||||
"both orphans are reaped — one by each adapter, summed by the composite");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopTearsDownAPaneSpawnedThroughTheComposite() {
|
||||
// CB-519: handle.id() is a host-unique opaque UUID, not the herdr pane — stop(id) must
|
||||
// resolve it through the owning adapter down to the actual pane coordinate it spawned.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertNotEquals("w9:pRoot_1", handle.id(), "the id is decoupled from the pane coordinate");
|
||||
|
||||
composite.stop(handle.id());
|
||||
assertTrue(herdr.calls.stream()
|
||||
.anyMatch(c -> c.method().equals("pane.close")
|
||||
&& "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id"))),
|
||||
"stop routes to the spawning adapter and closes exactly that worker's pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void opencodeContextResetIsANoOpAndWarnsOnlyOnce() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertFalse(opencodeAdapter(herdr).capabilities().contains(Capability.CONTEXT_RESET));
|
||||
assertTrue(herdr.calls.stream().noneMatch(c -> "agent.prompt".equals(c.method())),
|
||||
"never type Claude's /clear into an opencode prompt");
|
||||
assertEquals(1, appender.list.stream()
|
||||
.filter(e -> e.getFormattedMessage().contains("context reset is unsupported"))
|
||||
.count(), "unsupported reset is logged once per adapter, not once per turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAProfileClaimedByTwoAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// Two opencode adapters both declaring "gemini" — a profile-name collision.
|
||||
OpenCodeLauncher a = opencodeAdapter(herdr);
|
||||
OpenCodeLauncher b = opencodeAdapter(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(a, b), "gemini"),
|
||||
"a profile two adapters both claim is a configuration error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAnEmptyAdapterList() {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(), "claude"),
|
||||
"at least one adapter must be configured");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedDefaultIsNoOpForUnqualifiedSpawns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"fixed placement still routes an unqualified spawn to the default profile");
|
||||
assertEquals("claude", h.profile(), "the returned handle carries the resolved default profile");
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyGatesProfileAtMaxLoad() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
for (int i = 0; i < 5; i++) {
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "profile a is at maxLoad, so every spawn must land on b");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyDistributesAccordingToWeightRatio() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.75f, null),
|
||||
"b", stubWorker("b", 0.25f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 40; i++) {
|
||||
String p = composite.spawn(new SpawnRequest(null, null, null)).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(30, a, "weighted distribution should hold the 3:1 ratio");
|
||||
assertEquals(10, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverRetriesNextCandidateWhenProfileIsUnreachable() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "the spawn must fail over from unreachable a to b");
|
||||
assertEquals(1, adapter.spawnCount("a"), "a was tried once and failed");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b was tried once and succeeded");
|
||||
}
|
||||
|
||||
/**
|
||||
* Definition order — not hash order — decides an exact-weight tie. Paired with the test above
|
||||
* (same two profiles, opposite declaration order, opposite expected first attempt) this pins the
|
||||
* ordering contract from both sides: under a salted map one of the two must fail on every run.
|
||||
*/
|
||||
@Test
|
||||
void reversingDefinitionOrderReversesWhichProfileIsTriedFirst() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"b", stubWorker("b"),
|
||||
"a", stubWorker("a"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("a", h.profile(), "b is declared first and unreachable, so the spawn lands on a");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b, declared first, is the one tried first");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverBoundedByCandidateCount() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a", "b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerUnreachableException e = assertThrows(PeerUnreachableException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("no reachable worker profile"), e.getMessage());
|
||||
assertEquals(1, adapter.spawnCount("a"));
|
||||
assertEquals(1, adapter.spawnCount("b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnAtMaxLoadThrowsPlacementExceptionNamingProfileLiveAndCap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 2 : 0);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("a", null, null)));
|
||||
assertTrue(e.getMessage().contains("'a'"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 live"), "message names the live count: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 cap"), "message names the cap: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("a"), "at cap, the spawn is refused before any delegation");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnUnderMaxLoadStillSucceeds() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile under its cap accepts an explicit spawn");
|
||||
assertEquals(1, adapter.spawnCount("a"), "the under-cap spawn is delegated");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnWithNullMaxLoadIsNeverCapped() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, null),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
// A deliberately absurd live count: an unset maxLoad means unlimited, so it must never refuse.
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 1000);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile with no maxLoad is never capped, however many live workers");
|
||||
}
|
||||
|
||||
@Test
|
||||
void emptyCandidateSetThrowsClearException() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, 1));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 1);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
// ── CB-557: an unqualified spawn is placed inside its role's pool ─────────────────────────
|
||||
|
||||
/** Three profiles in definition order — pools are carved out of this set. */
|
||||
private static Map<String, BridgedConfig.Profile> threeProfiles() {
|
||||
Map<String, BridgedConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("opus", stubWorker("opus", 1.0f, null));
|
||||
m.put("sonnet", stubWorker("sonnet", 1.0f, null));
|
||||
m.put("terra", stubWorker("terra", 1.0f, null));
|
||||
return m;
|
||||
}
|
||||
|
||||
private static Map<String, BridgedConfig.Slot> pool(String... names) {
|
||||
Map<String, BridgedConfig.Slot> m = new LinkedHashMap<>();
|
||||
for (String n : names) {
|
||||
m.put(n, new BridgedConfig.Slot(n));
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
private static CompositePeerLauncher withPools(FakeHerdr herdr, BridgedConfig.Fleet fleet) {
|
||||
Map<String, BridgedConfig.Profile> profiles = threeProfiles();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
return new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* The point of the pools: a role is placed only on a backend its pool names. Before CB-557 an
|
||||
* unqualified spawn ranged over every configured profile, so a reviewer could land on the
|
||||
* architect-only one.
|
||||
*/
|
||||
@Test
|
||||
void anUnqualifiedSpawnIsPlacedInsideItsRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), pool("sonnet"), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.ARCHITECT)).profile());
|
||||
assertEquals("terra", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile());
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.REVIEWER)).profile());
|
||||
}
|
||||
|
||||
/** Under `fixed`, the pool's first entry wins — not the global defaultProfile. */
|
||||
@Test
|
||||
void theRolePoolOutranksTheGlobalDefaultProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), Map.of(), pool("sonnet", "terra"), Map.of(), null));
|
||||
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"the dev pool starts at sonnet, so the global default 'opus' must not win");
|
||||
}
|
||||
|
||||
/**
|
||||
* A role with no pool is unconstrained, not blocked. A config that declares pools for some roles
|
||||
* and not others must keep spawning the rest.
|
||||
*/
|
||||
@Test
|
||||
void aRoleWithNoPoolFallsBackToEveryProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("sonnet"), Map.of(), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"no dev pool ⇒ all profiles are candidates, so `fixed` takes the first one");
|
||||
}
|
||||
|
||||
/** No fleet at all is the pre-CB-557 wiring, and must behave exactly as it did. */
|
||||
@Test
|
||||
void noFleetConfiguredKeepsTheOldWholeProfileListBehaviour() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, null);
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest(null, null, null)).profile());
|
||||
}
|
||||
|
||||
/**
|
||||
* An explicit profile is the operator overriding and is NOT judged against the pool. It must
|
||||
* stay that way: an unrolled `bridge_spawn{profile:"opus"}` carries no role, so it defaults to
|
||||
* DEV, and enforcing the pool here would refuse a spawn the operator asked for by name.
|
||||
*/
|
||||
@Test
|
||||
void anExplicitProfileIsNotConfinedToTheRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest("opus", null, null)).profile(),
|
||||
"naming opus explicitly must work even though the dev pool holds only terra");
|
||||
}
|
||||
|
||||
/** Placement still respects maxLoad, but only across the pool — never by escaping it. */
|
||||
@Test
|
||||
void aFullPoolIsRefusedRatherThanSpilledOntoAnotherRolesProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("opus", stubWorker("opus", 1.0f, null)); // architect-only, uncapped
|
||||
profiles.put("terra", stubWorker("terra", 1.0f, 1)); // the sole dev, capped at 1
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.weighted(), name -> "terra".equals(name) ? 1 : 0,
|
||||
new BridgedConfig.Fleet(Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
}
|
||||
@@ -1,328 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The opencode adapter's launch build: a file-based MCP mount + reply-charter instructions (no
|
||||
* inline flags, no {@code ANTHROPIC_*}, no guard), the {@code -m} model flag, and the shared base
|
||||
* transport (naming, reap, readiness gate) proving the {@link HerdrPeerLauncher} SPI is neutral.
|
||||
*/
|
||||
class OpenCodeLauncherTest {
|
||||
|
||||
private static BridgedConfig.Profile opencodeCfg(String model, String mcpUrl, String gitTokenEnv) {
|
||||
return new BridgedConfig.Profile("gemini", null, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, gitTokenEnv, null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
/** Gate-disabled launcher whose per-spawn config dirs land under an inspectable temp root. */
|
||||
private OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, BridgedConfig.Profile cfg) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), k -> "GITEA_ACCESS_TOKEN".equals(k) ? "tok" : null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, Object> lastStart(FakeHerdr herdr) {
|
||||
return (Map<String, Object>) herdr.lastCall("agent.start").params();
|
||||
}
|
||||
|
||||
/** Protocol 19: the worker's env is injected at pane creation (tab.create), not agent.start. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, String> startEnv(FakeHerdr herdr) {
|
||||
Map<String, String> env =
|
||||
(Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||
return env == null ? Map.of() : env;
|
||||
}
|
||||
|
||||
/** Protocol 19: agent.start carries only the args after the kind-resolved executable. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static List<String> startArgs(FakeHerdr herdr) {
|
||||
return (List<String>) lastStart(herdr).get("args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void writesRemoteMcpConfigAndCharterInstructionsWhenMcpUrlSet(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null))
|
||||
.spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"), "opencode carries no ANTHROPIC_* / subscription boundary");
|
||||
String cfgPath = env.get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "OPENCODE_CONFIG points the worker at the generated config file");
|
||||
assertTrue(Path.of(cfgPath).startsWith(root), "config file is generated under the injected root");
|
||||
|
||||
// Assert on parsed structure, not substrings: the generated config is real JSON and its
|
||||
// whitespace is the formatter's business, not the contract's.
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
assertTrue(json.path("compaction").path("auto").asBoolean(),
|
||||
"spawned opencode peers explicitly enable automatic compaction");
|
||||
JsonNode bridge = json.path("mcp").path("bridge");
|
||||
assertEquals("remote", bridge.path("type").asText(), "bridge is mounted as a remote MCP server");
|
||||
assertEquals("http://127.0.0.1:8765/mcp", bridge.path("url").asText(),
|
||||
"the profile's bridge MCP url is present");
|
||||
assertTrue(bridge.path("enabled").asBoolean(), "the bridge server is enabled");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the reply charter is mounted via instructions");
|
||||
|
||||
// The instructions entry is a real file path holding the reply charter.
|
||||
Path charter = Path.of(cfgPath).resolveSibling("reply-charter.md");
|
||||
assertTrue(Files.exists(charter), "the charter file the config references was written");
|
||||
assertTrue(Files.readString(charter).contains("bridge_reply"),
|
||||
"the charter instructs the worker to answer via bridge_reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noConfigFileWhenMcpUrlAbsent(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
assertNull(startEnv(herdr).get("OPENCODE_CONFIG"),
|
||||
"no bridge MCP url → no config file and no OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void passesTheModelAsDashMFlagAlongsideAutoApprove(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
assertTrue(args.contains("--auto"),
|
||||
"--auto is present alongside -m so a spawned peer never blocks on approval");
|
||||
int m = args.indexOf("-m");
|
||||
assertTrue(m >= 0, "model is selected with -m");
|
||||
assertEquals("google/gemini-2.5-pro", args.get(m + 1), "the provider/model selector follows -m");
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoApproveIsUnconditionalWhenModelBlank(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null)).spawn();
|
||||
assertEquals(List.of("--auto"), startArgs(herdr),
|
||||
"--auto is unconditional: a model-less worker still must never block on approval");
|
||||
}
|
||||
|
||||
@Test
|
||||
void injectsForgeTokenWhenProfileGrantsIt(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN")).spawn();
|
||||
assertEquals("tok", startEnv(herdr).get("GITEA_TOKEN"),
|
||||
"a git-token profile gets the peer-neutral GITEA_TOKEN grant, same as Claude");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesDeclareOrphanReapAndMcpAskAndConditionalSelfPr(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
assertEquals(java.util.Set.of(Capability.MID_TURN_ASK, Capability.WORKTREE, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_RESUME),
|
||||
service(herdr, root, opencodeCfg(null, null, null)).capabilities(),
|
||||
"opencode can be resumed by its own session id, so SESSION_RESUME is always declared");
|
||||
assertFalse(service(herdr, root, opencodeCfg(null, null, null))
|
||||
.capabilities().contains(Capability.SESSION_NAME),
|
||||
"opencode has no display-name flag, so SESSION_NAME must NOT be declared");
|
||||
assertTrue(service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN"))
|
||||
.capabilities().contains(Capability.SELF_PR),
|
||||
"a git-token profile adds SELF_PR");
|
||||
}
|
||||
|
||||
// --- CB-547: resume + post-hoc session discovery --------------------------------------------
|
||||
|
||||
@Test
|
||||
void aResumeSpawnPassesTheSessionIdAsDashS(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, "ses_41b79fc90ffeI9E8uZv6VprUn2"));
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
int s = args.indexOf("-s");
|
||||
assertTrue(s >= 0, "a resumed spawn carries opencode's -s flag");
|
||||
assertEquals("ses_41b79fc90ffeI9E8uZv6VprUn2", args.get(s + 1),
|
||||
"the resume target id follows -s");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFreshSpawnCarriesNoSessionFlag(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, null));
|
||||
|
||||
assertFalse(startArgs(herdr).contains("-s"),
|
||||
"no resume target → a fresh session with no -s flag");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theHandleDiscoversTheSessionIdForTheWorkersCwdOnlyAfterItAppears(@TempDir Path root,
|
||||
@TempDir Path discRoot)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = new OpenCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), Map.of("gemini", opencodeCfg(null, null, null)),
|
||||
"gemini", _ -> null, 0, System::currentTimeMillis, () -> { }, root, discRoot);
|
||||
|
||||
PeerHandle handle = launcher.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
|
||||
// opencode writes the record only when the session is first persisted — the instant the
|
||||
// pane is ready it does not exist, so agentSessionId() is null (never a spawn failure).
|
||||
assertNull(handle.agentSessionId(), "no record yet → null, not a spawn-time block");
|
||||
// Once the record appears (here: same cwd), lazy discovery resolves it — the handle's
|
||||
// session id matches its own worktree, not another's.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "p1", "ses_a.json",
|
||||
"ses_resolved", "/work/dir", 1000L);
|
||||
assertEquals("ses_resolved", handle.agentSessionId(),
|
||||
"agentSessionId() re-scans and picks up a record that has since been written");
|
||||
}
|
||||
|
||||
@Test
|
||||
void foreignWorkerMatchesOpencodePrefixButNotClaude() {
|
||||
String nonce = "abc123";
|
||||
assertTrue(OpenCodeLauncher.isForeignWorker("opencode-gemini-def456-1", nonce),
|
||||
"an opencode pane from another process is foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("opencode-gemini-" + nonce + "-1", nonce),
|
||||
"our own opencode pane (same nonce) is not foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("claude-ltms-local-def456-1", nonce),
|
||||
"a claude pane is never reaped by the opencode adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionConstructorsWireThroughToTheBase() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = opencodeCfg(null, null, null);
|
||||
// 5-arg (gate disabled) and 7-arg (gate enabled) production constructors both expose the profile.
|
||||
OpenCodeLauncher disabled = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
OpenCodeLauncher gated = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null, 5000, 100);
|
||||
assertEquals(java.util.Set.of("gemini"), disabled.profiles());
|
||||
assertEquals("gemini", gated.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnGateThrowsPeerUnreachableWhenNeverInjectable(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never injectable
|
||||
long[] clock = {0};
|
||||
OpenCodeLauncher svc = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", opencodeCfg(null, null, null)), "gemini", _ -> null,
|
||||
1000, () -> clock[0], () -> clock[0] += 50, root, root);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(clock[0] >= 1000, "the fake clock advanced past the timeout: " + clock[0]);
|
||||
long closes = herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("pane.close"))
|
||||
.filter(c -> "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
assertEquals(1, closes, "the worker pane was reaped on timeout (no orphan)");
|
||||
assertNotNull(ex.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsHandleWhenGateDisabled(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerHandle handle = service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null));
|
||||
assertNotNull(handle, "spawn returns a handle when the gate is disabled");
|
||||
assertFalse(herdr.called("agent.get"), "no polling when the gate is disabled");
|
||||
}
|
||||
|
||||
// --- CB-508: pinned OpenAI-compatible endpoint (e.g. a local vLLM) ---------------------------
|
||||
|
||||
/** A profile with a baseUrl but no model provider prefix cannot be resolved — fail loudly. */
|
||||
private static BridgedConfig.Profile pinnedCfg(String model, String baseUrl, String mcpUrl) {
|
||||
return new BridgedConfig.Profile("local", baseUrl, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, null, null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
@Test
|
||||
void baseUrlDeclaresACustomOpenAiCompatibleProvider(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash", "http://127.0.0.1:8000", null))
|
||||
.spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a pinned endpoint needs a config file even with no bridge MCP url");
|
||||
JsonNode provider = new ObjectMapper().readTree(Path.of(cfgPath).toFile())
|
||||
.path("provider").path("local-vllm");
|
||||
|
||||
assertFalse(provider.isMissingNode(), "the provider id comes from the model selector");
|
||||
assertEquals("@ai-sdk/openai-compatible", provider.path("npm").asText());
|
||||
assertEquals("http://127.0.0.1:8000/v1", provider.path("options").path("baseURL").asText(),
|
||||
"a bare host:port gets /v1 appended — that is where these servers mount the API");
|
||||
assertFalse(provider.path("options").path("apiKey").asText().isBlank(),
|
||||
"the AI SDK requires a non-empty key even when the server ignores it");
|
||||
assertFalse(provider.path("models").path("deepseek-v4-flash").isMissingNode(),
|
||||
"the model half of the selector is declared under the provider");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBaseUrlThatAlreadyCarriesAPathIsUsedVerbatim(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/m", "http://127.0.0.1:8000/openai/v1", null)).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("http://127.0.0.1:8000/openai/v1",
|
||||
json.path("provider").path("local-vllm").path("options").path("baseURL").asText(),
|
||||
"an endpoint mounted on a custom path must not have /v1 bolted on");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointRejectsAModelWithNoProviderPrefix(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher =
|
||||
service(herdr, root, pinnedCfg("deepseek-v4-flash", "http://127.0.0.1:8000", null));
|
||||
|
||||
// Silently falling back to the default gateway would point the worker at the wrong LLM
|
||||
// while looking healthy — the one failure mode worth being loud about.
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, launcher::spawn);
|
||||
assertTrue(e.getMessage().contains("<provider>/<model>"), "the error says how to fix it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointAndTheBridgeMcpCoexistInOneConfig(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash",
|
||||
"http://127.0.0.1:8000", "http://127.0.0.1:8766/mcp")).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("remote", json.path("mcp").path("bridge").path("type").asText(),
|
||||
"pinning an endpoint must not drop the bridge MCP mount");
|
||||
assertFalse(json.path("provider").path("local-vllm").isMissingNode(),
|
||||
"and the provider block is still declared alongside it");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the reply charter survives too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noBaseUrlDeclaresNoProviderSoTheDefaultGatewayIsUsed(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("opencode/some-free-model", "http://127.0.0.1:8766/mcp", null))
|
||||
.spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertTrue(json.path("provider").isMissingNode(),
|
||||
"without a baseUrl opencode resolves its own provider as before");
|
||||
}
|
||||
}
|
||||
@@ -1,90 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.FileTime;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* {@link OpenCodeSessionDiscovery} matches an opencode session record by the worker's cwd (its
|
||||
* {@code directory}) against opencode's on-disk storage. These tests populate a TEMP storage root
|
||||
* themselves — never the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
class OpenCodeSessionDiscoveryTest {
|
||||
|
||||
/**
|
||||
* Write a session record {@code {"id":..., "directory":...}} under
|
||||
* {@code <root>/session/<projectID>/<fileName>} and stamp it with a known last-modified time,
|
||||
* so "most recently modified wins" is deterministic. Static so the launcher test can reuse it.
|
||||
*/
|
||||
static void writeRecord(Path root, String projectId, String fileName, String id,
|
||||
String directory, long lastModifiedEpochMillis) throws Exception {
|
||||
Path dir = root.resolve("session").resolve(projectId);
|
||||
Files.createDirectories(dir);
|
||||
Path file = dir.resolve(fileName);
|
||||
Files.writeString(file, "{\"id\":\"" + id + "\",\"directory\":\"" + directory
|
||||
+ "\",\"projectID\":\"" + projectId + "\",\"version\":\"1.1.31\"}");
|
||||
Files.setLastModifiedTime(file, FileTime.fromMillis(lastModifiedEpochMillis));
|
||||
}
|
||||
|
||||
@Test
|
||||
void findsTheRecordWhoseDirectoryEqualsTheCwd(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "ses_b.json", "ses_bbb", "/w/b", 2000L);
|
||||
|
||||
assertEquals("ses_bbb", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/b"),
|
||||
"the record whose directory equals the cwd is the one found");
|
||||
assertEquals("ses_aaa", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingDirectoryYieldsNullRatherThanAMismatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/other"),
|
||||
"no record for this cwd yet → null, not a wrong session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void prefersTheMostRecentlyModifiedRecordWhenSeveralMatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "old.json", "ses_old", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "new.json", "ses_new", "/w/a", 5000L);
|
||||
|
||||
assertEquals("ses_new", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"the freshest record for the cwd wins");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingOrEmptyStorageRootYieldsNullWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
// Missing: no session dir at all under the root.
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
|
||||
// Present but empty: a session dir with nothing in it produces no match, not a throw.
|
||||
Path emptyRoot = root.resolve("empty");
|
||||
Files.createDirectories(emptyRoot.resolve("session"));
|
||||
assertNull(new OpenCodeSessionDiscovery(emptyRoot).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankOrNullDirectoryYieldsNull(@TempDir Path root) {
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertNull(discovery.sessionIdForDirectory(null));
|
||||
assertNull(discovery.sessionIdForDirectory(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMalformedRecordIsSkippedRatherThanFatal(@TempDir Path root) throws Exception {
|
||||
// A record that fails to parse must not abort the scan of its siblings.
|
||||
Path dir = root.resolve("session").resolve("p1");
|
||||
Files.createDirectories(dir);
|
||||
Files.writeString(dir.resolve("broken.json"), "{not valid json");
|
||||
writeRecord(root, "p1", "good.json", "ses_good", "/w/a", 1000L);
|
||||
|
||||
assertEquals("ses_good", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"an unreadable record is skipped; a later valid one still matches");
|
||||
}
|
||||
}
|
||||
@@ -1,612 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.inject.CompletionResolver;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* The message layer's resolution paths (CB-104 reply + CB-106 completion fallback). The turn is
|
||||
* driven deterministically by feeding {@code onStatus} rather than running a real poller.
|
||||
*/
|
||||
class MessageServiceTest {
|
||||
|
||||
private static final String T = "term_a";
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr().readText("BUILD GREEN: 391 files");
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private final Rendezvous rendezvous = new Rendezvous();
|
||||
private final CompletionResolver completion = new CompletionResolver(agents, rendezvous);
|
||||
private final Injector injector = new Injector(agents, completion);
|
||||
private final InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
private final MessageService messages = new MessageService(agents, injector, rendezvous, inbox);
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
// CB-520: the inbox only peeks/acks targets it owns.
|
||||
inbox.own(T);
|
||||
}
|
||||
|
||||
/** Run {@code send} on a background thread; the current thread drives the worker's turn. */
|
||||
private CompletableFuture<MessageService.Reply> sendAsync() {
|
||||
return CompletableFuture.supplyAsync(() -> messages.send(T, "do the task", 5000));
|
||||
}
|
||||
|
||||
private void awaitWaiting() throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while (!rendezvous.isWaiting(T) && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting(T), "send should have opened its rendezvous waiter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackResolvesATurnThatNeverCalledBridgeReply() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
herdr.readText("$ prompt"); // pre-turn pane: no answer yet (baseline reference)
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver the task (baselines the pre-turn content)
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up and works
|
||||
herdr.readText("BUILD GREEN: 391 files"); // the worker's turn produced new output
|
||||
injector.onStatus(T, AgentStatus.IDLE); // working → idle: turn complete, no bridge_reply
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.COMPLETED_UNREPLIED, reply.outcome(),
|
||||
"an unreplied but finished turn resolves via the completion fallback");
|
||||
assertEquals("BUILD GREEN: 391 files", reply.text(), "the scraped transcript tail is returned");
|
||||
assertTrue(reply.completed(), "a scraped completion still counts as completed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitBridgeReplyResolvesAsReplied() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker working
|
||||
assertTrue(rendezvous.resolve(T, "LGTM ship it"), "an explicit reply resolves the send");
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, reply.outcome());
|
||||
assertEquals("LGTM ship it", reply.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWedgedWorkerResolvesTheSendAsFailedWithTheErrorContext() throws Exception {
|
||||
herdr.readText("API Error: Unable to connect to API (ENOTFOUND)");
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker starts the turn
|
||||
for (int i = 0; i < 130; i++) injector.onStatus(T, AgentStatus.UNKNOWN); // then wedges (CB-109)
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.WORKER_FAILED, reply.outcome());
|
||||
assertFalse(reply.completed(), "a wedge is terminal but not a successful completion");
|
||||
assertTrue(reply.text().contains("ENOTFOUND"), "the error screen is carried as the failure reason");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerThatVanishesMidTurnResolvesTheSendAsFailed() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker starts the turn
|
||||
// The worker's pane crashes — the poller sees a *_not_found and drops it (CB-110).
|
||||
injector.drop(T, new HerdrException("worker gone", "pane_not_found", null));
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.WORKER_FAILED, reply.outcome(),
|
||||
"a delivered send whose worker vanishes fails instead of hanging to the timeout");
|
||||
assertFalse(reply.completed());
|
||||
}
|
||||
|
||||
// --- bridge_ask reverse rendezvous (CB-205) ------------------------------------------------
|
||||
|
||||
@Test
|
||||
void askSurfacesAsAQuestionAndTheAnswerResumesTheSameTurn() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
// The worker asks mid-turn on its own thread; the call blocks for the primary's answer.
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
|
||||
// The primary's blocking send unblocks with the question and a turnId to answer on.
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertEquals("which config file?", q.text());
|
||||
assertNotNull(q.turnId(), "a question carries a turnId to answer on");
|
||||
|
||||
// The primary answers via bridge_send(turnId); this blocks again for the worker's reply.
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||
|
||||
// The worker's ask returns the answer — it resumes the same turn.
|
||||
MessageService.AskResult a = ask.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a.outcome());
|
||||
assertEquals("config.yaml", a.answer());
|
||||
|
||||
// The resumed worker finishes with a structured reply, resolving the answering send.
|
||||
awaitWaiting(); // the answering send has (re)opened its forward waiter
|
||||
assertTrue(rendezvous.resolve(T, "done"), "the worker's final reply resolves the answering send");
|
||||
MessageService.Reply done = answer.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, done.outcome());
|
||||
assertEquals("done", done.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void duplicateAsksFromTheSameSessionCoalesceToOneTurn() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
// A transport retry: two concurrent bridge_ask calls from the same worker session.
|
||||
CompletableFuture<MessageService.AskResult> ask1 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
CompletableFuture<MessageService.AskResult> ask2 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
|
||||
// The primary's single blocked send surfaces exactly ONE question (one turnId).
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertEquals("which config file?", q.text());
|
||||
assertNotNull(q.turnId(), "only one turnId should be minted");
|
||||
|
||||
// The primary answers that one turnId; both asks unblock with the same answer.
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||
|
||||
MessageService.AskResult a1 = ask1.get(5, TimeUnit.SECONDS);
|
||||
MessageService.AskResult a2 = ask2.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a1.outcome());
|
||||
assertEquals("config.yaml", a1.answer());
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a2.outcome());
|
||||
assertEquals("config.yaml", a2.answer());
|
||||
|
||||
// The resumed worker finishes with a structured reply, resolving the answering send.
|
||||
awaitWaiting();
|
||||
assertTrue(rendezvous.resolve(T, "done"), "the worker's final reply resolves the answering send");
|
||||
MessageService.Reply done = answer.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, done.outcome());
|
||||
assertEquals("done", done.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void askWithNoOpenDelegationReturnsNoWaiter() {
|
||||
MessageService.AskResult r = messages.ask(T, "anyone listening?", 500);
|
||||
assertEquals(MessageService.AskOutcome.NO_WAITER, r.outcome(),
|
||||
"a question with no blocked send has no primary to answer it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void askTimesOutWhenThePrimaryNeverAnswers() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
MessageService.AskResult r = messages.ask(T, "still there?", 200); // primary never answers
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, r.outcome());
|
||||
|
||||
// The send itself already unblocked with the question the instant the ask surfaced.
|
||||
MessageService.Reply q = send.get(2, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
}
|
||||
|
||||
@Test
|
||||
void answeringAnUnknownTurnIsStale() {
|
||||
MessageService.Reply r = messages.answer(T + "#999", "too late", 500);
|
||||
assertEquals(MessageService.Outcome.STALE_TURN, r.outcome(),
|
||||
"an answer to a turn that never existed (or already lapsed) is stale, not a hang");
|
||||
}
|
||||
|
||||
// --- timeout, answer, poll, and lock-contention edges ----------------------------------
|
||||
|
||||
@Test
|
||||
void sendTimesOutBeforeDeliveryIsQueuedNotWorking() {
|
||||
// Nothing ever delivers the message and nothing resolves the send, so the reply future
|
||||
// times out with delivery still incomplete — the message is still queued for the worker.
|
||||
MessageService.Reply r = messages.send(T, "never delivered", 50);
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_QUEUED, r.outcome(),
|
||||
"an undelivered send that times out is still queued, not working");
|
||||
assertNull(r.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendTimesOutAfterDeliveryIsStillWorking() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "do the task", 300));
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver — the delivered future now completes
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker starts but never replies
|
||||
// No rendezvous.resolve(T, ...) — the reply future rides out its short timeout.
|
||||
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, r.outcome(),
|
||||
"a delivered send whose worker never replies times out as still working");
|
||||
}
|
||||
|
||||
@Test
|
||||
void answerTimesOutWhenTheResumedWorkerNeverReplies() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertNotNull(q.turnId());
|
||||
|
||||
// The primary answers, unblocking the worker; but the worker never sends the follow-up
|
||||
// bridge_reply, so the answering send rides out its short window as still-working.
|
||||
MessageService.Reply answer = messages.answer(q.turnId(), "config.yaml", 200);
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answer.outcome(),
|
||||
"an answered worker that never replies times out as still working");
|
||||
|
||||
MessageService.AskResult a = ask.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a.outcome());
|
||||
assertEquals("config.yaml", a.answer());
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollReturnsNullForAnUnknownTicket() {
|
||||
assertNull(messages.poll("task-999999"), "a ticket that was never minted is unknown");
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollReportsACompletedTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker works
|
||||
assertTrue(rendezvous.resolve(T, "async result"), "a reply resolves the async send");
|
||||
|
||||
// Wait for the background send to finish and publish a DONE view.
|
||||
MessageService.TaskView view = null;
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while (view == null || view.phase() != MessageService.Phase.DONE) {
|
||||
if (System.currentTimeMillis() >= deadline) break;
|
||||
view = messages.poll(ticket);
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertNotNull(view, "a resolved async send must become DONE");
|
||||
assertEquals(MessageService.Phase.DONE, view.phase());
|
||||
assertEquals("async result", view.reply(), "the completed ticket reports the reply");
|
||||
assertEquals("reply", view.replySource(), "a structured bridge_reply is sourced from 'reply'");
|
||||
}
|
||||
|
||||
@Test
|
||||
void concurrentSendToSameSessionWhileFirstHoldsItIsBusy() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> first =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "first", 5000));
|
||||
awaitWaiting(); // the first send now holds the session lock, blocked on its reply
|
||||
|
||||
// A second send to the SAME session cannot take the lock within its short window.
|
||||
MessageService.Reply busy = messages.send(T, "second", 100);
|
||||
assertEquals(MessageService.Outcome.BUSY, busy.outcome(),
|
||||
"a second send while another holds the session is busy, not a hang");
|
||||
assertNull(busy.text());
|
||||
|
||||
// Release the first send so it resolves cleanly and the test thread is not left pinned.
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver the first message
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up
|
||||
assertTrue(rendezvous.resolve(T, "first done"), "the first send resolves with a reply");
|
||||
MessageService.Reply firstReply = first.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, firstReply.outcome());
|
||||
assertEquals("first done", firstReply.text());
|
||||
}
|
||||
|
||||
// --- CB-548: delegator ownership is recorded only on an ACCEPTED send ----------------------
|
||||
|
||||
private static final String LEAD_L = "term_lead_l";
|
||||
private static final String LEAD_A = "term_lead_a";
|
||||
|
||||
/**
|
||||
* The bug CB-548 fixes: L holds worker W, then architect A attempts W and times out BUSY. With
|
||||
* delegator ownership recorded at {@code bridge_send} <em>request</em> time, A's rejected call
|
||||
* would overwrite L — and W's late no-waiter reply would be pushed to A, who never owned the
|
||||
* turn. The accepted-delivery hook must not fire for a BUSY send, so L stays the delegator.
|
||||
*/
|
||||
@Test
|
||||
void busySenderDoesNotBecomeTheDelegatingOwner() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
// L accepts a delegation to W: the send wins the lock and queues delivery → L is recorded.
|
||||
CompletableFuture<MessageService.Reply> first = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "first", 5000, () -> reg.recordDelegation(T, LEAD_L)));
|
||||
awaitWaiting();
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"an accepted send owns the delegation");
|
||||
|
||||
// A attempts W while L holds it → BUSY (lock never taken) → its hook never fires.
|
||||
MessageService.Reply busy = messages.send(T, "second", 100, () -> reg.recordDelegation(T, LEAD_A));
|
||||
assertEquals(MessageService.Outcome.BUSY, busy.outcome());
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"a BUSY send must not steal the delegator ownership it never earned");
|
||||
|
||||
// L completes so the test thread is not left pinned.
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
assertTrue(rendezvous.resolve(T, "first done"));
|
||||
MessageService.Reply firstReply = first.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, firstReply.outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* Once L's accepted delegation is fully done, a later <em>accepted</em> send from A may
|
||||
* legitimately become the new delegator — ownership follows the turn, not the first caller.
|
||||
*/
|
||||
@Test
|
||||
void anAcceptedSendAfterThePriorOwnerFinishesBecomesTheNewOwner() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
CompletableFuture<MessageService.Reply> first = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "first", 5000, () -> reg.recordDelegation(T, LEAD_L)));
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
assertTrue(rendezvous.resolve(T, "first done"));
|
||||
MessageService.Reply firstReply = first.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, firstReply.outcome());
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(), "L owned the first turn");
|
||||
|
||||
// L finished; A's later accepted send takes the delegation over.
|
||||
CompletableFuture<MessageService.Reply> second = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "second", 5000, () -> reg.recordDelegation(T, LEAD_A)));
|
||||
awaitWaiting();
|
||||
assertEquals(LEAD_A, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"an accepted send after the owner finished becomes the new delegator");
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
assertTrue(rendezvous.resolve(T, "second done"));
|
||||
MessageService.Reply secondReply = second.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, secondReply.outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548 requirement: answering an existing {@code bridge_ask} is the SAME delegation, so it must
|
||||
* not rewrite ownership. L accepted the send (owned), the worker paused to ask, and L answers via
|
||||
* turnId — ownership stays L throughout; the answer path never touches the registry.
|
||||
*/
|
||||
@Test
|
||||
void answeringAnAskDoesNotRewriteDelegatorOwnership() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
CompletableFuture<MessageService.Reply> send = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "do X", 5000, () -> reg.recordDelegation(T, LEAD_L)));
|
||||
awaitWaiting();
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(), "L owns the delegation");
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertNotNull(q.turnId());
|
||||
|
||||
// L answers the ask on the same turn; the answer path must not touch ownership.
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer());
|
||||
awaitWaiting(); // the answering send reopened its forward waiter
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"answering an ask keeps L as the delegator — ownership is not rewritten");
|
||||
|
||||
assertTrue(rendezvous.resolve(T, "done"));
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548: {@code onAccepted} is a public callback, so a throwing one must not orphan the turn.
|
||||
* The waiter is opened first, then the hook runs BEFORE delivery is queued — so a throw fails
|
||||
* the send loudly, closes its waiter, and never enqueues a message the worker would pick up and
|
||||
* reply into the void.
|
||||
*/
|
||||
@Test
|
||||
void aThrowingAcceptedHookLeavesNoStaleWaiterOrQueuedOrphan() {
|
||||
assertThrows(IllegalStateException.class,
|
||||
() -> messages.send(T, "doomed", 500,
|
||||
() -> { throw new IllegalStateException("ownership hook failed"); }),
|
||||
"a throwing ownership hook fails the send loudly");
|
||||
|
||||
assertFalse(rendezvous.isWaiting(T), "the failed send must not leave a stale rendezvous waiter");
|
||||
// Give the injector a delivery window: with nothing enqueued, nothing may reach the worker.
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
boolean doomedQueued = herdr.calls.stream()
|
||||
.anyMatch(c -> c.method().equals("agent.prompt")
|
||||
&& String.valueOf(c.params()).contains("doomed"));
|
||||
assertFalse(doomedQueued, "a throwing ownership hook must not leave a queued, orphanable message");
|
||||
}
|
||||
|
||||
/**
|
||||
* The async (fire-and-poll) path runs the same {@code send} on a background thread, so the
|
||||
* accepted-delivery hook must thread through it — ownership is recorded exactly as blocking sends.
|
||||
*/
|
||||
@Test
|
||||
void asyncSendRecordsOwnershipOnAcceptance() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
messages.sendAsync(T, "async task", () -> reg.recordDelegation(T, LEAD_L));
|
||||
awaitWaiting(); // the background send won the lock, queued, and opened its waiter
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"the async path records delegator ownership on acceptance, like the blocking path");
|
||||
}
|
||||
|
||||
// --- CB-307 reply inbox ----------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void replyQueuesInInboxWhenNoSendIsOpen() {
|
||||
// No send is open for this session — reply should queue in the inbox.
|
||||
assertTrue(messages.reply(T, "queued-text"), "reply should succeed (queued)");
|
||||
|
||||
var drained = messages.drainReplies(T);
|
||||
assertEquals(1, drained.size());
|
||||
assertEquals("queued-text", drained.getFirst().content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyResolvesOpenSendDoesNotQueue() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitUninterruptibly(T);
|
||||
|
||||
// An explicit reply resolves the open send.
|
||||
assertTrue(messages.reply(T, "send-resolved"), "reply should succeed (resolved live send)");
|
||||
|
||||
// The inbox should be empty — the reply went to the send, not the inbox.
|
||||
assertTrue(messages.drainReplies(T).isEmpty(), "no reply in the inbox");
|
||||
|
||||
MessageService.Reply r = send.get(3, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, r.outcome());
|
||||
assertEquals("send-resolved", r.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainRepliesReturnsAllPendingThenEmptyOnNextCall() {
|
||||
messages.reply(T, "msg-1");
|
||||
messages.reply(T, "msg-2");
|
||||
|
||||
var first = messages.drainReplies(T);
|
||||
assertEquals(2, first.size());
|
||||
|
||||
var second = messages.drainReplies(T);
|
||||
assertTrue(second.isEmpty(), "second drain should be empty (acked)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aQuestionIsNeverQueuedInTheInbox() {
|
||||
// No send is open — bridge_ask with no delegation returns NO_WAITER,
|
||||
// and the question text MUST NOT appear in the reply inbox.
|
||||
// The inbox is only fed by MessageService.reply(), not by bridge_ask.
|
||||
MessageService.AskResult r = messages.ask(T, "anyone there?", 500);
|
||||
assertEquals(MessageService.AskOutcome.NO_WAITER, r.outcome(),
|
||||
"bridge_ask with no open delegation must return NO_WAITER, never queued");
|
||||
|
||||
assertTrue(messages.drainReplies(T).isEmpty(), "questions must never be queued");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackIsNeverQueued() throws Exception {
|
||||
// The fallback resolves a captured waiter, never the inbox.
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitUninterruptibly(T);
|
||||
injectDelivery();
|
||||
|
||||
// The worker never sends bridge_reply, but the turn completes.
|
||||
herdr.readText("done-scraped");
|
||||
completion.onTurnComplete(T); // The fallback arms and resolves the captured waiter.
|
||||
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.COMPLETED_UNREPLIED, r.outcome());
|
||||
|
||||
// The inbox should be empty — the reply went to the captured waiter.
|
||||
assertTrue(messages.drainReplies(T).isEmpty(), "completion fallback must not queue");
|
||||
}
|
||||
|
||||
// --- helpers ---------------------------------------------------------------------------
|
||||
|
||||
/** Like {@link #awaitWaiting()} but rethrows as unchecked. */
|
||||
private void awaitUninterruptibly(String session) {
|
||||
try {
|
||||
awaitWaiting();
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException(e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Set up a delivered turn so the worker is working, ready for an ask or completion. */
|
||||
private void injectDelivery() {
|
||||
herdr.readText("$ prompt"); // pre-turn content baseline
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver the task
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up
|
||||
}
|
||||
|
||||
// --- CB-516: a released session must not leave a send hanging ------------------------------
|
||||
|
||||
/**
|
||||
* The bug this fixes: tearing a worker down left its rendezvous waiter open, so a blocking send
|
||||
* kept blocking and an async one kept reporting PENDING until the 30-minute async timeout —
|
||||
* even though the worker provably no longer existed.
|
||||
*/
|
||||
@Test
|
||||
void abandonFailsASendThatIsStillWaitingOnAReleasedSession() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "work", 30_000));
|
||||
awaitWaiting();
|
||||
|
||||
assertTrue(messages.abandon(T, "session released"), "a live waiter is abandoned");
|
||||
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.WORKER_FAILED, r.outcome(),
|
||||
"an abandoned send fails rather than riding out its timeout");
|
||||
assertEquals("session released", r.text(), "the caller is told why");
|
||||
}
|
||||
|
||||
@Test
|
||||
void abandonIsANoOpWhenNobodyIsWaiting() {
|
||||
assertFalse(messages.abandon(T, "session released"),
|
||||
"no open send ⇒ nothing to abandon");
|
||||
}
|
||||
|
||||
@Test
|
||||
void abandonDoesNotOverwriteAnAlreadyResolvedSend() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "work", 30_000));
|
||||
awaitWaiting();
|
||||
assertTrue(rendezvous.resolve(T, "the real answer"));
|
||||
|
||||
assertFalse(messages.abandon(T, "session released"),
|
||||
"a send already answered by the worker must not be clobbered");
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals("the real answer", r.text());
|
||||
}
|
||||
|
||||
/** The async path is the one that hung: poll must report FAILED, not PENDING forever. */
|
||||
@Test
|
||||
void anAbandonedAsyncTaskPollsAsFailedNotPending() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase());
|
||||
|
||||
messages.abandon(T, "session released");
|
||||
|
||||
MessageService.TaskView view = null;
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (System.currentTimeMillis() < deadline) {
|
||||
view = messages.poll(ticket);
|
||||
if (view.phase() != MessageService.Phase.PENDING) break;
|
||||
Thread.sleep(10);
|
||||
}
|
||||
assertNotNull(view);
|
||||
assertEquals(MessageService.Phase.FAILED, view.phase(),
|
||||
"a delegation whose worker is gone must not keep reporting PENDING");
|
||||
assertTrue(view.detail() != null && view.detail().contains("released"),
|
||||
"and the detail says why, rather than 'worker unknown'");
|
||||
}
|
||||
}
|
||||
@@ -1,318 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Unit tests for {@link ReplyPushLoop}: decision logic, nudge injection, idempotency,
|
||||
* bounded reminders, and stop conditions.
|
||||
*
|
||||
* <p>Uses a {@link RecordingHerdrClient} that synchronizes access to its call list so the
|
||||
* scheduler thread and test thread never have memory ordering issues. The {@code decide()}
|
||||
* tests use a simple client with no concurrency concern.
|
||||
*/
|
||||
class ReplyPushLoopTest {
|
||||
|
||||
private static final String PRIMARY = "term_primary";
|
||||
private static final String WORKER = "term_worker";
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
private PrimaryRegistry registry;
|
||||
private AgentControl agents;
|
||||
private InMemoryReplyInbox inbox;
|
||||
private ScheduledExecutorService scheduler;
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
registry = new PrimaryRegistry(PRIMARY);
|
||||
inbox = new InMemoryReplyInbox();
|
||||
inbox.own(WORKER); // CB-520: the inbox only peeks/acks targets it owns
|
||||
scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
scheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// --- decide() logic ------------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void decideWithoutPrimaryIsStop() {
|
||||
agents = agentWithStatus("idle");
|
||||
var loop = new ReplyPushLoop(
|
||||
new PrimaryRegistry(null), agents, inbox, scheduler, 5, 100);
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop.decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideWithEmptyInboxIsStop() {
|
||||
agents = agentWithStatus("idle");
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideAtCapIsStop() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop(2, 100).decide(WORKER, 2));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithInjectablePrimaryIsInject() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithBlockedPrimaryIsInject() {
|
||||
agents = agentWithStatus("blocked");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0),
|
||||
"BLOCKED is injectable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithDonePrimaryIsInject() {
|
||||
agents = agentWithStatus("done");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0),
|
||||
"DONE is injectable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithBusyPrimaryIsWaitBusy() {
|
||||
agents = agentWithStatus("working");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.WAIT_BUSY, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithUnknownPrimaryIsWaitBusy() {
|
||||
agents = agentWithStatus("unknown");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.WAIT_BUSY, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideStopsAfterInboxIsEmptied() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0));
|
||||
inbox.ack(WORKER, "m1");
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
// --- onReplyQueued integration -------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void injectablePrimaryCausesExactlyOneNudge() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
|
||||
loop(1, 50).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"one nudge (1 agent.prompt call) should have been sent");
|
||||
|
||||
// Exactly one nudge = exactly 1 agent.prompt call (it submits itself)
|
||||
assertEquals(1, rec.sendCount());
|
||||
assertTrue(rec.sentParams().stream()
|
||||
.anyMatch(e -> e.getValue().toString().contains("bridge_poll")),
|
||||
"nudge text should contain bridge_poll");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onReplyQueuedIsIdempotentPerTarget() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
|
||||
var loop = loop(1, 100);
|
||||
loop.onReplyQueued(WORKER);
|
||||
loop.onReplyQueued(WORKER); // second call — should be a no-op
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"expected exactly one nudge (1 prompt)");
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, rec.sendCount(),
|
||||
"second onReplyQueued must not trigger another nudge");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendsUpToCapThenStops() throws Exception {
|
||||
int cap = 2;
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
rec.sendLatch = new CountDownLatch(cap);
|
||||
|
||||
loop(cap, 50).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(5, TimeUnit.SECONDS),
|
||||
cap + " nudges (" + cap + " prompts) should have fired");
|
||||
Thread.sleep(300);
|
||||
assertEquals(cap, rec.sendCount(),
|
||||
"exactly " + cap + " agent.prompt calls (cap=" + cap + ")");
|
||||
}
|
||||
|
||||
// --- nudge format --------------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void isActiveReflectsALiveScheduleForTheHeartbeatStandDown() {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
ReplyPushLoop loop = loop(1, 100_000); // long backoff so the tick cannot fire mid-test
|
||||
|
||||
loop.onReplyQueued(WORKER);
|
||||
assertTrue(loop.isActive(), "CB-551: the heartbeat must stand aside while a reminder is live");
|
||||
|
||||
loop.stop();
|
||||
assertFalse(loop.isActive(), "stopping clears the active schedule");
|
||||
}
|
||||
|
||||
@Test
|
||||
void nudgeFormatIsCorrect() {
|
||||
String nudge = ReplyPushLoop.NUDGE_FORMAT.formatted(WORKER, WORKER);
|
||||
assertTrue(nudge.contains("Worker term_worker"));
|
||||
assertTrue(nudge.contains("bridge_poll(target=term_worker)"));
|
||||
}
|
||||
|
||||
// --- metrics (CB-512) ----------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void successfulNudgeIncrementsDelivered() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
Metrics metrics = new Metrics();
|
||||
|
||||
loop(1, 50, metrics).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"one nudge (1 agent.prompt call) should have been sent");
|
||||
// The delivered count is bumped on the scheduler thread right after the send that releases
|
||||
// the latch — settle briefly so the counter is published before we read it.
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, metrics.count(BridgedMetrics.PUSH_NUDGES, "outcome", "delivered"),
|
||||
"a successfully sent nudge must count as delivered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reminderCapIncrementsExhausted() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
Metrics metrics = new Metrics();
|
||||
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop(2, 100, metrics).decide(WORKER, 2));
|
||||
|
||||
assertEquals(1, metrics.count(BridgedMetrics.PUSH_NUDGES, "outcome", "exhausted"),
|
||||
"hitting the reminder cap must count as exhausted");
|
||||
assertEquals(0, metrics.count(BridgedMetrics.PUSH_NUDGES, "outcome", "delivered"));
|
||||
}
|
||||
|
||||
// --- helpers -------------------------------------------------------------------------------
|
||||
|
||||
private ReplyPushLoop loop() {
|
||||
return loop(5, 100);
|
||||
}
|
||||
|
||||
private ReplyPushLoop loop(int maxReminders, long backoffMs) {
|
||||
return new ReplyPushLoop(registry, agents, inbox, scheduler, maxReminders, backoffMs);
|
||||
}
|
||||
|
||||
private ReplyPushLoop loop(int maxReminders, long backoffMs, Metrics metrics) {
|
||||
return new ReplyPushLoop(registry, agents, inbox, scheduler, maxReminders, backoffMs, metrics);
|
||||
}
|
||||
|
||||
private static AgentControl agentWithStatus(String status) {
|
||||
return new AgentControl(new FakeHerdrClient(status));
|
||||
}
|
||||
|
||||
/** Non-recording (single-threaded) fake — safe for decide() tests. */
|
||||
private static final class FakeHerdrClient implements HerdrClient {
|
||||
private final String agentStatus;
|
||||
|
||||
FakeHerdrClient(String agentStatus) {
|
||||
this.agentStatus = agentStatus;
|
||||
}
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode()
|
||||
.set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", PRIMARY)
|
||||
.put("agent_status", agentStatus));
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Thread-safe recording fake that counts agent.prompt calls (protocol 19: one nudge = one
|
||||
* prompt). Uses synchronized access so the scheduler thread and test thread never race.
|
||||
*/
|
||||
private static final class RecordingHerdrClient implements HerdrClient {
|
||||
private final List<Map.Entry<String, Object>> calls =
|
||||
Collections.synchronizedList(new ArrayList<>());
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode()
|
||||
.set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", PRIMARY)
|
||||
.put("agent_status", "idle")); // recording double is always injectable
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
calls.add(Map.entry(method, params));
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
long sendCount() {
|
||||
return calls.size();
|
||||
}
|
||||
|
||||
List<Map.Entry<String, Object>> sentParams() {
|
||||
return List.copyOf(calls);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static RecordingHerdrClient recordingClient() {
|
||||
return new RecordingHerdrClient();
|
||||
}
|
||||
}
|
||||
@@ -1,183 +0,0 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Unit tests for the placement policies. They run with no herdr and no launcher — pure selection
|
||||
* logic exercised through the descriptor type so CB-308 host expansion will not need to rewrite
|
||||
* these assertions.
|
||||
*/
|
||||
class PlacementPolicyTest {
|
||||
|
||||
private static Function<String, Integer> noSessions() {
|
||||
return name -> 0;
|
||||
}
|
||||
|
||||
private static PlacementContext ctx(List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable) {
|
||||
return new PlacementContext("b", candidates, liveCount, unreachable);
|
||||
}
|
||||
|
||||
private static PlacementContext ctx(List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount) {
|
||||
return ctx(candidates, liveCount, Set.of());
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedReturnsDefaultEvenIfOtherProfilesExist() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = ctx(List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b")), noSessions());
|
||||
assertEquals("b", policy.select(ctx).profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedFallsBackToFirstCandidateWhenNoDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null,
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of());
|
||||
assertEquals("a", policy.select(ctx).profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedThrowsWhenNoProfilesAndNoDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null, List.of(), noSessions(), Set.of());
|
||||
assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
}
|
||||
|
||||
@Test
|
||||
void roundRobinCyclesThroughAvailableProfiles() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b"),
|
||||
PlacementCandidate.profile("c"));
|
||||
assertEquals("a", policy.select(ctx(candidates, noSessions())).profile());
|
||||
assertEquals("b", policy.select(ctx(candidates, noSessions())).profile());
|
||||
assertEquals("c", policy.select(ctx(candidates, noSessions())).profile());
|
||||
assertEquals("a", policy.select(ctx(candidates, noSessions())).profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void roundRobinSkipsProfilesAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 2),
|
||||
PlacementCandidate.profile("b", 1.0f, null));
|
||||
Function<String, Integer> liveCount = Map.of("a", 2)::get;
|
||||
for (int i = 0; i < 5; i++) {
|
||||
assertEquals("b", policy.select(ctx(candidates, liveCount)).profile());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void roundRobinThrowsWhenAllAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, 1));
|
||||
Function<String, Integer> liveCount = Map.of("a", 1, "b", 1)::get;
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, liveCount)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedAlternatesEvenlyWithEqualWeights() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 0.5f, null),
|
||||
PlacementCandidate.profile("b", 0.5f, null));
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 100; i++) {
|
||||
String p = policy.select(ctx(candidates, noSessions())).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(50, a, "equal weights should split 50/50");
|
||||
assertEquals(50, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedHoldsThreeToOneRatio() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 0.75f, null),
|
||||
PlacementCandidate.profile("b", 0.25f, null));
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 40; i++) {
|
||||
String p = policy.select(ctx(candidates, noSessions())).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(30, a, "0.75/0.25 should yield a 3:1 ratio over a multiple of 4");
|
||||
assertEquals(10, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedSkipsProfileAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, null));
|
||||
Function<String, Integer> liveCount = name -> "a".equals(name) ? 1 : 0;
|
||||
for (int i = 0; i < 5; i++) {
|
||||
assertEquals("b", policy.select(ctx(candidates, liveCount)).profile());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedThrowsWhenAllAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, 1));
|
||||
Function<String, Integer> liveCount = name -> 1;
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, liveCount)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedThrowsWhenAllUnreachable() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b"));
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, noSessions(), Set.of("a", "b"))));
|
||||
assertTrue(e.getMessage().contains("unreachable"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void mixedExclusionMessageNamesBothReasons() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, null));
|
||||
Function<String, Integer> liveCount = name -> "a".equals(name) ? 1 : 0;
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
unreachable.add("b");
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, liveCount, unreachable)));
|
||||
assertTrue(e.getMessage().contains("1 at maxLoad"), e.getMessage());
|
||||
assertTrue(e.getMessage().contains("1 unreachable"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void unknownPolicyNameThrows() {
|
||||
assertThrows(IllegalArgumentException.class, () -> PlacementPolicies.fromName("random"));
|
||||
}
|
||||
}
|
||||
@@ -1,208 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-525 acceptance test for tool-surface isolation. This is one of the few tests that drives real
|
||||
* {@code git} — the behaviour under test is precisely what {@link GitWorktrees} does to a checkout,
|
||||
* so a fake would assert nothing. Everything happens inside a {@link TempDir} throwaway repo.
|
||||
*/
|
||||
class GitWorktreesTest {
|
||||
|
||||
/** A project MCP config with servers in it — what this repo actually commits. */
|
||||
private static final String WITH_SERVERS = """
|
||||
{
|
||||
"mcpServers": {
|
||||
"jetbrains": { "type": "sse", "url": "http://localhost:64342/sse" }
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** An opencode config carrying a {@code {file:.secrets/...}} reference — CB-543's crash repro. */
|
||||
private static final String OPENCODE_WITH_FILE_REF = """
|
||||
{
|
||||
"env": {
|
||||
"CONTEXT7_TOKEN": "{file:.secrets/context7-token}"
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** A non-empty autoenv file — the form that would prompt for authorization in a worktree. */
|
||||
private static final String AUTOENV_WITH_DIRECTIVE = "export HELLO=world\n";
|
||||
|
||||
private static Path initRepo(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
git(dir, "init", "-q", "-b", "main");
|
||||
git(dir, "config", "user.email", "test@example.invalid");
|
||||
git(dir, "config", "user.name", "Test");
|
||||
Files.writeString(dir.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||
git(dir, "add", ".mcp.json", "README.md");
|
||||
git(dir, "commit", "-q", "-m", "seed");
|
||||
return dir;
|
||||
}
|
||||
|
||||
private static void git(Path cwd, String... args) throws Exception {
|
||||
List<String> cmd = new java.util.ArrayList<>(List.of("git"));
|
||||
cmd.addAll(List.of(args));
|
||||
Process p = new ProcessBuilder(cmd).directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git timed out: " + String.join(" ", cmd));
|
||||
assertEquals(0, p.exitValue(), "git " + String.join(" ", args) + " failed:\n" + out);
|
||||
}
|
||||
|
||||
/** Pending changes to {@code file} in {@code cwd}, empty when git considers it unmodified. */
|
||||
private static String status(Path cwd, String file) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "status", "--porcelain", "--", file)
|
||||
.directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git status timed out");
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* The heart of CB-525: a provisioned worktree must not inherit the primary's MCP servers. Without
|
||||
* the isolation step the checked-out {@code .mcp.json} carries them in, and a worker navigating
|
||||
* through the primary's IDE servers edits the primary's tree while building its own.
|
||||
*/
|
||||
@Test
|
||||
void aProvisionedWorktreeInheritsNoMcpServers(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-a", "HEAD");
|
||||
|
||||
Path mcp = Path.of(wt).resolve(".mcp.json");
|
||||
assertTrue(Files.exists(mcp), ".mcp.json must still exist — present and explicitly empty");
|
||||
String body = Files.readString(mcp);
|
||||
assertFalse(body.contains("jetbrains"), "worktree inherited the primary's MCP servers:\n" + body);
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
"expected an explicitly empty server map, got:\n" + body);
|
||||
}
|
||||
|
||||
/** Neutralizing must not look like work in progress, or a worker would commit it into its PR. */
|
||||
@Test
|
||||
void theNeutralizedConfigIsNotAPendingLocalModification(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-b", "HEAD");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"),
|
||||
"the neutralized .mcp.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** Isolation is the worktree's business only; the primary's own checkout must be untouched. */
|
||||
@Test
|
||||
void thePrimaryCheckoutIsLeftAlone(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-525-c", "HEAD");
|
||||
|
||||
assertEquals(WITH_SERVERS, Files.readString(repo.resolve(".mcp.json")),
|
||||
"the primary's .mcp.json was rewritten — isolation reached out of the worktree");
|
||||
}
|
||||
|
||||
/** A repo that commits no {@code .mcp.json} still gets one, so nothing can be inherited later. */
|
||||
@Test
|
||||
void aRepoWithoutAnMcpConfigStillGetsANeutralOne(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-d", "HEAD");
|
||||
|
||||
// Untracked is the normal case here, so the --skip-worktree branch must be skipped rather
|
||||
// than run and fail: `update-index --skip-worktree` on an unknown path exits non-zero.
|
||||
String body = Files.readString(Path.of(wt).resolve(".mcp.json"));
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"), body);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-543's repro: a tracked {@code opencode.json} carries a {@code {file:.secrets/...}} reference
|
||||
* to a gitignored secret that never reaches a worktree, and opencode refuses to start on it. The
|
||||
* worktree's copy must be neutralized and hidden like {@code .mcp.json}.
|
||||
*/
|
||||
@Test
|
||||
void aTrackedOpencodeConfigIsNeutralizedAndHidden(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-a", "HEAD");
|
||||
|
||||
String body = Files.readString(Path.of(wt).resolve("opencode.json"));
|
||||
assertFalse(body.contains(".secrets"),
|
||||
"worktree kept a dangling {file:...} secret reference:\n" + body);
|
||||
assertEquals("{}", body.replaceAll("\\s+", ""),
|
||||
"expected an empty JSON object stub, got:\n" + body);
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"),
|
||||
"the neutralized opencode.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** A config the repo does not carry must be skipped — no stub invented, provisioning still succeeds. */
|
||||
@Test
|
||||
void anAbsentConfigIsSkippedWithoutError(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo")); // only .mcp.json + README are committed
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-b", "HEAD");
|
||||
|
||||
assertFalse(Files.exists(Path.of(wt).resolve("opencode.json")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
assertFalse(Files.exists(Path.of(wt).resolve(".autoenv")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
// .mcp.json's long-standing create-always behaviour must be unchanged.
|
||||
assertTrue(Files.exists(Path.of(wt).resolve(".mcp.json")), ".mcp.json stub was dropped");
|
||||
}
|
||||
|
||||
/** All three protected configs are covered: each one present in a worktree is neutralized and hidden. */
|
||||
@Test
|
||||
void allThreeConfigsAreNeutralizedWhenPresent(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve(".autoenv"), AUTOENV_WITH_DIRECTIVE);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", ".autoenv", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-c", "HEAD");
|
||||
|
||||
assertTrue(Files.readString(Path.of(wt).resolve(".mcp.json"))
|
||||
.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
".mcp.json was not neutralized");
|
||||
assertEquals("{}", Files.readString(Path.of(wt).resolve("opencode.json")).replaceAll("\\s+", ""),
|
||||
"opencode.json was not neutralized");
|
||||
assertEquals("", Files.readString(Path.of(wt).resolve(".autoenv")),
|
||||
".autoenv was not neutralized");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"), ".mcp.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"), "opencode.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), ".autoenv"), ".autoenv still shows as modified");
|
||||
}
|
||||
}
|
||||
@@ -1,496 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-301 / CB-303 acceptance tests for the authoritative session registry, one-shot lifecycle FSM,
|
||||
* and configurable lifecycle limits (idle TTL, context cap, drain).
|
||||
* No live herdr — everything runs against the same {@link FakeHerdr} the rest of the project uses.
|
||||
*/
|
||||
class SessionManagerTest {
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock) {
|
||||
return sessionManager(herdr, clock, 0);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap) {
|
||||
return sessionManager(herdr, clock, contextCap, false);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap,
|
||||
boolean clearAfterTurn) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers, new GitWorktrees(), clock, contextCap, clearAfterTurn);
|
||||
}
|
||||
|
||||
@Test
|
||||
void primaryContactWithNoTerminalIsNotAReadinessSignal() {
|
||||
// The MCP context extractor calls presence.markPresent(p.terminal()) on EVERY request,
|
||||
// and the primary's terminal is null — the presence bridge must treat that as a no-op,
|
||||
// not feed it into the READY transition (which NPEd on the first real primary contact).
|
||||
SessionManager sessions = sessionManager(new FakeHerdr());
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null));
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireRegistersSpawningSessionWithDistinctPaneId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession a = sessions.acquire("ltms-local", "/work/a", "/caller/a", "term_primary");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/work/b", "/caller/b", "term_primary");
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, a.state(), "fresh session starts spawning");
|
||||
assertEquals("ltms-local", a.profile());
|
||||
assertEquals("/work/a", a.cwd(), "explicit requested cwd is recorded");
|
||||
assertEquals("term_primary", a.ownerTerminal());
|
||||
assertTrue(a.spawnedAtNanos() > 0);
|
||||
assertNotNull(a.paneId());
|
||||
assertNotNull(a.terminalId());
|
||||
|
||||
assertNotEquals(a.paneId(), b.paneId(), "no pane reuse");
|
||||
assertNotEquals(a.terminalId(), b.terminalId(), "no terminal reuse");
|
||||
assertEquals(2, sessions.roster().size(), "both sessions are registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullTerminalFromThePrimaryIsANoOpEvenWithSessionsRegistered() {
|
||||
// The primary resolves to a Principal with no terminal, and BridgeMcp's context extractor
|
||||
// forwards that null into markPresent on EVERY MCP call. It only reached the registry scan
|
||||
// once a session existed, so this NPE'd the primary's second spawn while the first passed.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null),
|
||||
"the primary's null terminal must not blow up an unrelated tool call");
|
||||
assertDoesNotThrow(() -> sessions.onDelivered(null));
|
||||
assertDoesNotThrow(() -> sessions.onTurnComplete(null));
|
||||
assertDoesNotThrow(() -> sessions.onTurnFailed(null));
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"and must not transition any registered session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void presenceMovesSpawningToReadyAndDeliveredTurnMovesToDone() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
assertEquals(MemberSession.State.READY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"MCP presence moves SPAWNING → READY");
|
||||
assertTrue(sessions.asPresence().isPresent(terminal), "presence is also recorded");
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
assertEquals(MemberSession.State.BUSY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"delivery moves READY → BUSY");
|
||||
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"turn completion moves BUSY → DONE");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseTearsDownWorkerAndRemovesFromRosterAndIsIdempotent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
String paneId = session.paneId();
|
||||
|
||||
sessions.release(paneId);
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release tears the worker pane down");
|
||||
assertTrue(sessions.get(paneId).isEmpty(), "released session is no longer retrievable");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session is no longer in the roster");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.release(paneId), "a second release is harmless");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedMovesSessionToFailed() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
|
||||
sessions.onTurnFailed(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.FAILED, updated.state(), "turn failure moves to FAILED");
|
||||
assertTrue(sessions.roster().contains(updated), "FAILED is still in acquired-minus-released roster");
|
||||
}
|
||||
|
||||
@Test
|
||||
void recycleProducesNewPaneIdAndOldOneIsGone() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession oldSession = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String oldPane = oldSession.paneId();
|
||||
String oldTerminal = oldSession.terminalId();
|
||||
|
||||
MemberSession fresh = sessions.recycle(oldPane);
|
||||
|
||||
assertNotEquals(oldPane, fresh.paneId(), "recycle yields a new pane id");
|
||||
assertNotEquals(oldTerminal, fresh.terminalId(), "recycle yields a new terminal id");
|
||||
assertEquals(oldSession.profile(), fresh.profile(), "profile is preserved");
|
||||
assertEquals(oldSession.cwd(), fresh.cwd(), "cwd is preserved");
|
||||
assertEquals(oldSession.ownerTerminal(), fresh.ownerTerminal(), "owner is preserved");
|
||||
|
||||
assertTrue(sessions.get(oldPane).isEmpty(), "old pane is deregistered");
|
||||
assertEquals(1, sessions.roster().size(), "only the fresh session remains");
|
||||
assertEquals(fresh.paneId(), sessions.roster().getFirst().paneId());
|
||||
|
||||
// The old session was the first spawn → pane w9:pRoot_1 (CB-519: the registry key is the
|
||||
// uuid id, so teardown is asserted on the real pane coordinate).
|
||||
long paneCloseCount = herdr.calls.stream()
|
||||
.filter(c -> "pane.close".equals(c.method()))
|
||||
.filter(c -> "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
assertEquals(1, paneCloseCount, "the old worker was torn down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void rosterReflectsAcquiredMinusReleased() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession a = sessions.acquire("ltms-local", "/a", "/caller", "ownerA");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/b", "/caller", "ownerB");
|
||||
|
||||
assertEquals(2, sessions.roster().size());
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(a.paneId())));
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(b.paneId())));
|
||||
|
||||
sessions.release(a.paneId());
|
||||
|
||||
assertEquals(1, sessions.roster().size());
|
||||
assertEquals(b.paneId(), sessions.roster().getFirst().paneId());
|
||||
}
|
||||
|
||||
// --- CB-303 lifecycle limits ----------------------------------------------------
|
||||
|
||||
@Test
|
||||
void reapIdleDoesNothingWhenNoSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L);
|
||||
|
||||
assertEquals(0, sessions.reapIdle(10));
|
||||
assertTrue(sessions.roster().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 11;
|
||||
assertEquals(1, sessions.reapIdle(10), "READY session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "reaped session is removed from registry");
|
||||
assertTrue(herdr.called("pane.close"), "reaped session tears the pane down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionWithinIdleTtlSurvives() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 5;
|
||||
assertEquals(0, sessions.reapIdle(10), "READY session within TTL is not reaped");
|
||||
assertEquals(MemberSession.State.READY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"READY session survives");
|
||||
}
|
||||
|
||||
@Test
|
||||
void busySessionPastIdleTtlIsNotReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
|
||||
clock[0] = 100;
|
||||
assertEquals(0, sessions.reapIdle(10), "BUSY session past TTL is never reaped");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"BUSY session remains");
|
||||
}
|
||||
|
||||
@Test
|
||||
void doneSessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
clock[0] = 21;
|
||||
assertEquals(1, sessions.reapIdle(20), "DONE session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "DONE session is removed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIdleReturnsCorrectCountAndSkipsBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "owner1");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "owner2");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId());
|
||||
|
||||
clock[0] = 50;
|
||||
assertEquals(1, sessions.reapIdle(30), "only READY past TTL is reaped");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "READY session is gone");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(busy.paneId()).orElseThrow().state(),
|
||||
"BUSY session is still registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapDisabledSessionSurvivesMultipleTurns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.DONE, updated.state(), "session finishes second turn");
|
||||
assertEquals(2, updated.turnCount(), "turn count tracks both deliveries");
|
||||
long releaseCloseCount = paneCloseCallsFor(herdr, "w9:pRoot_1"); // the real pane coordinate
|
||||
assertEquals(0, releaseCloseCount, "cap disabled — no forced release of the worker pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapTwoReleasesAfterSecondComplete() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 2);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"first turn completes without release");
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "session released after cap reached");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session leaves roster");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"forced release tears the worker pane down exactly once");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnResetsContextWithoutDoubleCountingTheTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
|
||||
sessions.onDelivered(session.terminalId());
|
||||
assertTrue(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(1, updated.turnCount(), "the reset is housekeeping, not a second delegation");
|
||||
assertEquals(List.of("/clear"), promptTexts(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapReleaseWinsOverClearAfterTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 1, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId());
|
||||
|
||||
assertFalse(sessions.hasPostTurnAction(session.terminalId()),
|
||||
"a session at its cap will be released, not reset for reuse");
|
||||
assertFalse(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty());
|
||||
assertTrue(promptTexts(herdr).isEmpty(), "never send /clear into a worker being torn down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnFalsePreservesCompletionWithoutAControlPrompt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, false);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId());
|
||||
|
||||
sessions.onTurnComplete(session.terminalId());
|
||||
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state());
|
||||
assertTrue(promptTexts(herdr).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllReleasesBusyAndReadySessionsAndWaitsForBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "ownerR");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "ownerB");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId());
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(sessions.roster().isEmpty(), "drain clears the roster");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "ready session is released");
|
||||
assertTrue(sessions.get(busy.paneId()).isEmpty(), "busy session is released after timeout");
|
||||
// ready is the first spawn → pane w9:pRoot_1, busy the second → w9:pRoot_2 (FakeHerdr order).
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"ready worker pane is torn down");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_2"),
|
||||
"busy worker pane is torn down");
|
||||
}
|
||||
|
||||
private static long paneCloseCallsFor(FakeHerdr herdr, String paneId) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "pane.close".equals(c.method()))
|
||||
.filter(c -> paneId.equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
}
|
||||
|
||||
private static List<String> promptTexts(FakeHerdr herdr) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "agent.prompt".equals(c.method()))
|
||||
.map(c -> String.valueOf(((Map<?, ?>) c.params()).get("text")))
|
||||
.toList();
|
||||
}
|
||||
|
||||
// --- CB-306 spawn-readiness gate: no half-registered session on timeout ----------------
|
||||
|
||||
@Test
|
||||
void acquireThrowsPeerUnreachableWhenGateTimesOutAndRegistersNoSession() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never becomes injectable
|
||||
long[] clock = {0};
|
||||
|
||||
// Gate-enabled launcher (1 ms timeout + no-op sleeper that advances clock past deadline)
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
1, () -> clock[0], () -> clock[0] += 10);
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(), () -> 0L, 0);
|
||||
|
||||
assertThrows(PeerUnreachableException.class,
|
||||
() -> sessions.acquire("ltms-local", null, "/caller", "term_primary"),
|
||||
"acquire must throw PeerUnreachableException when spawn times out");
|
||||
|
||||
// No half-registered session — the error happened inside spawn, before
|
||||
// SessionManager could put() anything into the registry.
|
||||
assertTrue(sessions.roster().isEmpty(),
|
||||
"no session is registered when spawn times out (roster empty)");
|
||||
}
|
||||
|
||||
// --- CB-516: release must notify, so a blocked send can be failed --------------------------
|
||||
|
||||
@Test
|
||||
void releaseNotifiesTheListenerWithTheReleasedTerminal() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(released::add);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(java.util.List.of(s.terminalId()), released,
|
||||
"every teardown path funnels through release, so one hook must see the terminal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releasingAnUnknownPaneNotifiesNobody() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(released::add);
|
||||
|
||||
sessions.release("w9:p404"); // idempotent teardown of something already gone
|
||||
|
||||
assertTrue(released.isEmpty(), "no session removed ⇒ no send was waiting on it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aThrowingReleaseListenerDoesNotBlockTheTeardown() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
sessions.onRelease(_ -> {
|
||||
throw new IllegalStateException("listener blew up");
|
||||
});
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||
"a listener failure must never prevent the teardown it is reacting to");
|
||||
assertTrue(sessions.get(s.paneId()).isEmpty(), "and the session is still deregistered");
|
||||
}
|
||||
}
|
||||
@@ -1,264 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-301-ext acceptance tests for worktree provisioning and config-parity overlay.
|
||||
* No live git — every Worktrees call is handled by {@link FakeWorktrees} and every herdr
|
||||
* call by {@link FakeHerdr}, matching the project's fake-based test style.
|
||||
*/
|
||||
class WorktreeSessionManagerTest {
|
||||
|
||||
private static ClaudeCodeLauncher workerService(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
}
|
||||
|
||||
private static String startCwd(FakeHerdr herdr) {
|
||||
@SuppressWarnings("unchecked")
|
||||
// Protocol 19: the worker's cwd rides on pane creation (tab.create), not agent.start.
|
||||
Map<String, Object> start = (Map<String, Object>) herdr.lastCall("tab.create").params();
|
||||
Object cwd = start.get("cwd");
|
||||
return cwd == null ? null : cwd.toString();
|
||||
}
|
||||
|
||||
@Test
|
||||
void sharedTreeAcquireMakesNoWorktreesCallsAndRecordsNullWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary");
|
||||
|
||||
assertTrue(worktrees.addCalls().isEmpty(), "shared-tree acquire never adds a worktree");
|
||||
assertTrue(worktrees.repoRootCalls().isEmpty(), "shared-tree acquire never resolves a repo root");
|
||||
assertTrue(worktrees.overlayCalls().isEmpty(), "shared-tree acquire never overlays parity");
|
||||
assertNull(s.worktree(), "shared-tree session has no worktree");
|
||||
assertNull(s.branch(), "shared-tree session has no branch");
|
||||
assertEquals("/caller/proj", s.cwd(), "shared-tree cwd is the caller's cwd");
|
||||
assertEquals("/caller/proj", startCwd(herdr), "spawn receives the caller's cwd");
|
||||
}
|
||||
|
||||
@Test
|
||||
void worktreeAcquireProvisionsAndRecordsPathAndBranch() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary",
|
||||
new WorktreeRequest("cb-999", null));
|
||||
|
||||
assertEquals(1, worktrees.addCalls().size(), "one worktree was added");
|
||||
FakeWorktrees.AddCall add = worktrees.lastAdd();
|
||||
assertNotNull(add);
|
||||
assertEquals("/repo", add.repoRoot());
|
||||
assertTrue(add.branch().startsWith("worker/cb-999-"), "branch is worker/<slug>-<nonce>: " + add.branch());
|
||||
assertNull(add.baseRef(), "null baseRef is passed through (HEAD default)");
|
||||
|
||||
String expectedPath = "/wt/" + add.branch().replace('/', '_');
|
||||
assertEquals(expectedPath, s.worktree(), "session records the returned worktree path");
|
||||
assertEquals(add.branch(), s.branch(), "session records the branch");
|
||||
assertEquals(expectedPath, startCwd(herdr), "spawn receives the worktree path as cwd");
|
||||
assertEquals(expectedPath, s.cwd(), "session cwd is the worktree path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void worktreeAcquireRunsParityOverlayWithProfileDefaults() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt")
|
||||
.track(".envrc")
|
||||
.exists(".claude/settings.local.json");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-888", null));
|
||||
|
||||
assertEquals(1, worktrees.overlayCalls().size());
|
||||
FakeWorktrees.OverlayCall overlay = worktrees.lastOverlay();
|
||||
assertNotNull(overlay);
|
||||
assertEquals("/repo", overlay.repoRoot());
|
||||
assertEquals(List.of(".claude/settings.local.json", ".env", ".envrc"),
|
||||
overlay.requested(), "default parity overlay is used when unset");
|
||||
assertFalse(overlay.requested().contains(".mcp.json"),
|
||||
"CB-525: replicating the primary's MCP config gives a worker the primary's IDE "
|
||||
+ "servers, which navigate its edits out of its own worktree");
|
||||
assertEquals(List.of(".claude/settings.local.json", ".envrc"), overlay.copied(),
|
||||
"existing paths are copied; missing paths are skipped");
|
||||
assertEquals(List.of(".envrc"), overlay.skipWorktree(),
|
||||
"tracked copied paths are --skip-worktree'd");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseRemovesWorktreeButDoesNotDeleteBranch() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-666", null));
|
||||
String paneId = s.paneId();
|
||||
|
||||
sessions.release(paneId);
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release still tears the worker pane down");
|
||||
assertEquals(1, worktrees.removeCalls().size(), "worktree session triggers one remove");
|
||||
FakeWorktrees.RemoveCall remove = worktrees.lastRemove();
|
||||
assertNotNull(remove);
|
||||
assertEquals("/repo", remove.repoRoot());
|
||||
assertEquals(s.worktree(), remove.worktreePath());
|
||||
// The fake records no branch-delete calls because Worktrees.remove only removes the checkout.
|
||||
assertTrue(sessions.get(paneId).isEmpty(), "released session is no longer retrievable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllPreservesWorktreeOfIdleSession() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-544", null));
|
||||
sessions.asPresence().markPresent(s.terminalId()); // READY (idle)
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "shutdown drain still stops the worker pane");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"shutdown drain must NOT remove the worktree — it is the only copy of the work");
|
||||
assertTrue(sessions.roster().isEmpty(), "shutdown drain still deregisters the session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllPreservesWorktreeOfSessionStillBusyAtTimeout() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-544", null));
|
||||
String terminal = s.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal); // BUSY, never completes → still BUSY when the timeout hits
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "shutdown drain stops the worker pane");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"a session still BUSY at timeout must preserve its worktree unconditionally");
|
||||
assertTrue(sessions.roster().isEmpty(), "shutdown drain still deregisters the session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sharedTreeReleaseMakesNoWorktreesCalls() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null);
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release tears the worker pane down");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(), "shared-tree release never removes a worktree");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failedWorktreeAddUnwindsWithoutRegisteringSessionOrSpawning() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().failAdd("worktree add failed");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
assertThrows(WorktreeException.class, () ->
|
||||
sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-555", null)));
|
||||
|
||||
assertEquals(0, sessions.size(), "failed acquire leaves no registry entry");
|
||||
assertFalse(herdr.called("agent.start"), "spawn is never reached when add fails");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(), "no worktree was added, so none is removed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoWorktreeAcquiresYieldDistinctBranchesAndPaths() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
MemberSession a = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-444", null));
|
||||
MemberSession b = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-444", null));
|
||||
|
||||
assertNotEquals(a.branch(), b.branch(), "branches are distinct");
|
||||
assertNotEquals(a.worktree(), b.worktree(), "paths are distinct");
|
||||
assertEquals(2, worktrees.addCalls().size());
|
||||
assertEquals(2, sessions.roster().size());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-507 regression. A plain REST spawn supplies neither a requested nor a caller cwd
|
||||
* ({@code BridgedApp} hardcodes {@code callerCwd = null}), and the worktree branch used to
|
||||
* resolve the repo root from just those two — yielding {@code null}, which the real
|
||||
* {@code GitWorktrees} turns into {@code git -C null} and an NPE out of {@code ProcessBuilder}
|
||||
* (HTTP 500).
|
||||
*
|
||||
* <p>Note this asserts on the <em>recorded</em> cwd rather than expecting a throw:
|
||||
* {@link FakeWorktrees#repoRoot} only records its argument and returns a canned root, so a
|
||||
* null flows through the fake harmlessly. That permissiveness is precisely why the whole
|
||||
* suite stayed green while the feature was broken in production — so the assertion has to be
|
||||
* "a usable cwd was passed down", not "an exception was raised".
|
||||
*/
|
||||
@Test
|
||||
void worktreeAcquireWithNoRequestedOrCallerCwdStillResolvesANonNullRepoRoot() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
sessions.acquire("ltms-local", null, null, null, new WorktreeRequest("cb-507", null));
|
||||
|
||||
assertFalse(worktrees.repoRootCalls().isEmpty(),
|
||||
"repoRoot should have been called to resolve the repo root");
|
||||
String cwd = worktrees.repoRootCalls().getFirst().cwd();
|
||||
assertNotNull(cwd, "a null cwd here becomes `git -C null` and NPEs in the real GitWorktrees");
|
||||
assertFalse(cwd.isBlank(), "a blank cwd is as unusable as a null one");
|
||||
}
|
||||
|
||||
/**
|
||||
* The same line carried a second, quieter bug: it never consulted the profile's configured
|
||||
* {@code cwd:}, so a worktree spawn silently ignored a pinned per-profile working directory.
|
||||
* Routing through {@code effectiveCwd} honours it.
|
||||
*/
|
||||
@Test
|
||||
void worktreeAcquireHonoursTheProfileConfiguredCwd() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
// Argument order matters: configDir is the 4th parameter, cwd the 11th (after mcpUrl).
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, "/pinned/dir", null);
|
||||
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
SessionManager sessions = new SessionManager(launcher, worktrees);
|
||||
|
||||
sessions.acquire("ltms-local", null, null, null, new WorktreeRequest("cb-507b", null));
|
||||
|
||||
assertEquals(1, worktrees.repoRootCalls().size());
|
||||
assertEquals("/pinned/dir", worktrees.repoRootCalls().getFirst().cwd(),
|
||||
"the profile's configured cwd must reach repoRoot, not be ignored");
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,80 +0,0 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<!--
|
||||
CB-504 — launchd agent for bridged (macOS).
|
||||
|
||||
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
|
||||
no systemd. A systemd unit ships alongside (deploy/bridged.service) for the Linux gateways
|
||||
CB-308 introduces.
|
||||
|
||||
Install:
|
||||
cp deploy/dev.ltms.bridged.plist ~/Library/LaunchAgents/
|
||||
# edit the paths + JAVA_HOME below to match this host, then:
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.bridged.plist
|
||||
launchctl list | grep bridged
|
||||
|
||||
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
|
||||
does systemd in a way that survives a socket appearing late. bridged retries the herdr socket
|
||||
on startup instead, so an agent that comes up before herdr converges rather than dying — that
|
||||
retry is the actual fix; KeepAlive below is the backstop.
|
||||
-->
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>dev.ltms.bridged</string>
|
||||
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged/target/bridged.jar</string>
|
||||
<string>bridged.yaml</string>
|
||||
</array>
|
||||
|
||||
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
|
||||
<key>WorkingDirectory</key>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged</string>
|
||||
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>JAVA_HOME</key>
|
||||
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home</string>
|
||||
<key>HERDR_SOCKET_PATH</key>
|
||||
<string>/Users/CHANGEME/.config/herdr/herdr.sock</string>
|
||||
<!--
|
||||
PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker
|
||||
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
|
||||
source .zprofile/.zshrc, so without this the daemon (and therefore every worker) gets a
|
||||
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
|
||||
-->
|
||||
<key>PATH</key>
|
||||
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home/bin:/Users/CHANGEME/Tool/apache-maven-3.9.16/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
|
||||
<!--
|
||||
Worker/API tokens are NOT set here: this file is committed. Export them from a private
|
||||
launchd override or a wrapper script. bridged reads the API token from the env var named
|
||||
by auth.tokenEnv (default BRIDGED_API_TOKEN) and only in auth.mode: token.
|
||||
-->
|
||||
</dict>
|
||||
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
|
||||
<!-- Restart on crash, but not in a tight loop if the config is bad (bridged fails fast on a
|
||||
non-loopback bind without token auth — that is a config error, not a transient one). -->
|
||||
<key>KeepAlive</key>
|
||||
<dict>
|
||||
<key>SuccessfulExit</key>
|
||||
<false/>
|
||||
</dict>
|
||||
<key>ThrottleInterval</key>
|
||||
<integer>10</integer>
|
||||
|
||||
<key>StandardOutPath</key>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged/logs/bridged.out.log</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged/logs/bridged.err.log</string>
|
||||
|
||||
<key>ProcessType</key>
|
||||
<string>Background</string>
|
||||
</dict>
|
||||
</plist>
|
||||
@@ -0,0 +1,127 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<!--
|
||||
CB-504 / CB-594 — launchd agent for fleetd (macOS).
|
||||
|
||||
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
|
||||
no systemd. A systemd unit ships alongside (deploy/fleetd.service) for the Linux gateways
|
||||
CB-308 introduces.
|
||||
|
||||
Install:
|
||||
cp deploy/dev.ltms.fleetd.plist ~/Library/LaunchAgents/
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist
|
||||
launchctl list | grep fleetd
|
||||
|
||||
The paths below are already filled in for this host (resolved 2026-08-16 from
|
||||
`/usr/libexec/java_home`... except that reported the system Applet-plugin JVM, not the jenv-
|
||||
managed JDK 25 actually used to build/run fleetd, so JAVA_HOME here is the real one:
|
||||
`JENV_VERSION=25.0.3 java -XshowSettings:properties -version 2>&1 | grep java.home`; `which mvn`;
|
||||
`echo $HOME`). If this file is copied to a different host, re-resolve all three paths and check
|
||||
no placeholder path is left behind; scripts/redeploy-fleetd.sh's check mode does not (and
|
||||
cannot) check this file for you.
|
||||
|
||||
CB-594 — launchd cannot run a login shell (see the PATH comment on EnvironmentVariables below,
|
||||
and scripts/fleetd-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
|
||||
not java directly, so WORKER_GITEA_TOKEN and AI_GATEWAY_TOKEN still get sourced from
|
||||
${SHARED_ENV}/tools/secrets.sh even though launchd itself never sources anything.
|
||||
|
||||
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
|
||||
does systemd in a way that survives a socket appearing late. fleetd retries the herdr socket
|
||||
on startup instead, so an agent that comes up before herdr converges rather than dying — that
|
||||
retry is the actual fix; KeepAlive below is the backstop.
|
||||
|
||||
CB-594 — KeepAlive vs. scripts/redeploy-fleetd.sh: a bare SIGTERM makes this JVM exit 143 even
|
||||
with its shutdown hook running to completion (measured, see the CB-594 report), which
|
||||
SuccessfulExit:false below reads as a crash and races to restart the OLD jar. The redeploy
|
||||
script now detects a loaded agent and uses `launchctl unload`/`load` instead of a raw kill, so
|
||||
only one supervisor ever touches the process at a time — read that script's own output on a
|
||||
redeploy for the confirmation.
|
||||
-->
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>dev.ltms.fleetd</string>
|
||||
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/target/fleetd.jar</string>
|
||||
<string>fleetd.yaml</string>
|
||||
</array>
|
||||
|
||||
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
|
||||
<key>WorkingDirectory</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd</string>
|
||||
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>JAVA_HOME</key>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home</string>
|
||||
<key>HERDR_SOCKET_PATH</key>
|
||||
<string>/Users/dai.ha/.config/herdr/herdr.sock</string>
|
||||
<!--
|
||||
PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker
|
||||
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
|
||||
source .zprofile/.zshrc, so without this the daemon (and therefore every worker) gets a
|
||||
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
|
||||
-->
|
||||
<key>PATH</key>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin:/Users/dai.ha/Softwares/apache-maven/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
|
||||
<!--
|
||||
Worker/API tokens are NOT set here: this file is committed. CB-594 —
|
||||
scripts/fleetd-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
|
||||
them, by execing a login shell that sources ${SHARED_ENV}/tools/secrets.sh before the
|
||||
daemon itself starts. fleetd also reads the API token from the env var named by
|
||||
auth.tokenEnv (default FLEETD_API_TOKEN) and only in auth.mode: token — the wrapper
|
||||
covers that one too, since it is the same login shell.
|
||||
-->
|
||||
</dict>
|
||||
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
|
||||
<!--
|
||||
CB-600 — read this before assuming ThrottleInterval bounds anything. It paces restarts to at
|
||||
most one per 10s; it does NOT cap how many times launchd retries. If fleetd fails fast on
|
||||
every start — a bad fleetd.yaml, for example auth.mode: token with the token env var unset,
|
||||
which throws in main() before the daemon ever binds a port — launchd restarts it forever,
|
||||
once every 10s, until a human intervenes. LaunchAgents have no "give up after N attempts"
|
||||
primitive, so this is not something a config change here can fix.
|
||||
|
||||
That loop stops only two ways: (1) `launchctl unload -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist`,
|
||||
or (2) the underlying cause gets fixed, so the process starts successfully and stays up (no
|
||||
more exits to restart). scripts/redeploy-fleetd.sh does not add a third way — it does not
|
||||
make fleetd self-disable on a config error, on purpose: a fail-fast exit path that
|
||||
sometimes decides "this is unrecoverable, stop trying" is one more thing that can misfire,
|
||||
and a wrongly self-disabled daemon needs the exact same manual `launchctl load -w` recovery
|
||||
this comment already names — so it buys nothing an operator watching for the crash loop
|
||||
doesn't already have, at the cost of a new way to be silently down. Watch for it with
|
||||
`launchctl list dev.ltms.fleetd` (a high restart count) or by tailing fleetd.out for the
|
||||
same startup error repeating every ~10s.
|
||||
-->
|
||||
<key>KeepAlive</key>
|
||||
<dict>
|
||||
<key>SuccessfulExit</key>
|
||||
<false/>
|
||||
</dict>
|
||||
<key>ThrottleInterval</key>
|
||||
<integer>10</integer>
|
||||
|
||||
<!--
|
||||
CB-594 — same file scripts/redeploy-fleetd.sh already tails ($BRIDGED/fleetd.out), and both
|
||||
streams point at it, not two separate log files: the script's fresh-line / ERROR-count checks
|
||||
after a restart read this one path regardless of whether launchd or the script started the
|
||||
process, and a stdout/stderr split would make half of what happens during a launchd-driven
|
||||
restart invisible to it.
|
||||
-->
|
||||
<key>StandardOutPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
|
||||
<key>ProcessType</key>
|
||||
<string>Background</string>
|
||||
</dict>
|
||||
</plist>
|
||||
@@ -1,45 +1,45 @@
|
||||
# CB-504 — systemd unit for bridged (Linux).
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.bridged.plist) is the supervision target for the
|
||||
# The macOS launchd agent (deploy/dev.ltms.fleetd.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
#
|
||||
# Install (user service — bridged drives the user's herdr, not a system daemon):
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/bridged.service ~/.config/systemd/user/
|
||||
# cp deploy/fleetd.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now bridged
|
||||
# journalctl --user -u bridged -f
|
||||
# systemctl --user enable --now fleetd
|
||||
# journalctl --user -u fleetd -f
|
||||
|
||||
[Unit]
|
||||
Description=bridged — claude-bridge message server
|
||||
Documentation=https://git.ltms.dev/lms/claude-bridge/wiki
|
||||
Description=fleetd — claude-bridge message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# bridged retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take bridged down with it.
|
||||
# fleetd retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down with it.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/bridged
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/bridged.jar bridged.yaml
|
||||
WorkingDirectory=%h/src/claude-bridge/fleetd
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/fleetd.jar fleetd.yaml
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker it
|
||||
# PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit bridged → [Service] / Environment=BRIDGED_API_TOKEN=...
|
||||
# systemctl --user edit fleetd → [Service] / Environment=FLEETD_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/bridged/env
|
||||
# EnvironmentFile=%h/.config/fleetd/env
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes bridged fail fast by design.
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes fleetd fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
@@ -55,7 +55,7 @@ RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=bridged
|
||||
SyslogIdentifier=fleetd
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -1,8 +1,8 @@
|
||||
# LavinMQ — the AMQP broker behind bridged's durable ReplyInbox (CB-307 Stage 2).
|
||||
# LavinMQ — the AMQP broker behind fleetd's durable ReplyInbox (CB-307 Stage 2).
|
||||
#
|
||||
# Why this file exists: the broker was previously run ad hoc and simply vanished from the host,
|
||||
# which takes bridged down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Bridged.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# which takes fleetd down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Fleetd.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# degraded mode. This pins the version, keeps the data, and brings itself back after a reboot.
|
||||
#
|
||||
# Usage:
|
||||
@@ -14,27 +14,27 @@
|
||||
#
|
||||
# Management UI: http://127.0.0.1:15672 (guest / guest)
|
||||
#
|
||||
# This is bridged's OWN broker. Do not point bridged at any other AMQP server on this host —
|
||||
# This is fleetd's OWN broker. Do not point fleetd at any other AMQP server on this host —
|
||||
# notably not the `local-rabbitmq` container, which belongs to a different project and would end
|
||||
# up carrying this project's queues.
|
||||
|
||||
name: bridged-broker
|
||||
name: fleetd-broker
|
||||
|
||||
services:
|
||||
lavinmq:
|
||||
# Pinned deliberately: :latest silently moves the broker under a running daemon.
|
||||
image: cloudamqp/lavinmq:2.9.1
|
||||
container_name: bridged-lavinmq
|
||||
container_name: fleetd-lavinmq
|
||||
|
||||
# The failure this deployment exists to prevent — survive reboots and Docker restarts, but
|
||||
# stay down if it was stopped on purpose.
|
||||
restart: unless-stopped
|
||||
|
||||
# Loopback-bound on purpose. LavinMQ ships a default guest/guest account, which is only
|
||||
# acceptable because nothing off-host can reach it. bridged connects over 127.0.0.1, and
|
||||
# acceptable because nothing off-host can reach it. fleetd connects over 127.0.0.1, and
|
||||
# binding 0.0.0.0 here would expose a broker with default credentials to the network.
|
||||
ports:
|
||||
- "127.0.0.1:5672:5672" # AMQP — bridged.yaml broker.uri points here
|
||||
- "127.0.0.1:5672:5672" # AMQP — fleetd.yaml broker.uri points here
|
||||
- "127.0.0.1:15672:15672" # HTTP management API + UI
|
||||
|
||||
# The whole point of Stage 2. Held-but-unacked replies live here; without a named volume a
|
||||
@@ -57,4 +57,4 @@ services:
|
||||
|
||||
volumes:
|
||||
lavinmq-data:
|
||||
name: bridged-lavinmq-data
|
||||
name: fleetd-lavinmq-data
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
**Status:** design spec for review → delegate implementation.
|
||||
**Grounded in:** `WorkerService`, `Injector`/`StatusPoller`/`TurnListener`, `MessageService`,
|
||||
`BridgeMcp`, `BridgedApp` (see [wiki 9. Implementation](../wiki/9-Implementation.md)).
|
||||
`FleetMcp`, `FleetApp` (see [wiki 9. Implementation](../wiki/9-Implementation.md)).
|
||||
|
||||
## Problem
|
||||
|
||||
@@ -15,7 +15,7 @@ Consequences today:
|
||||
since when?" without shelling to herdr for a raw agent list (no state, no ownership, no age).
|
||||
- Cleanup of a worker that outlived its owning process depends entirely on the boot-time
|
||||
name-nonce **reaper** (CB-117) — there is no live, authoritative roster during a run.
|
||||
- `bridge_list` (CB-304) can only surface herdr's view, not a bridge-owned roster.
|
||||
- `fleet_list` (CB-304) can only surface herdr's view, not a bridge-owned roster.
|
||||
- There is no seam for per-session policy (checkpoint on teardown → CB-302; idle_ttl /
|
||||
context_cap / drain → CB-303).
|
||||
|
||||
@@ -35,7 +35,7 @@ build on.
|
||||
- **No checkpoint content.** Writing `STATE.md` + commit on teardown is CB-302; CB-301 only exposes
|
||||
the release hook it will attach to.
|
||||
|
||||
"Recycle" under no-reuse is simply **release + fresh acquire** — a helper, not a pool operation.
|
||||
Under no-reuse, a released session is terminal. A new `acquire` always creates a fresh session.
|
||||
|
||||
## Design
|
||||
|
||||
@@ -43,14 +43,9 @@ build on.
|
||||
subscription-guarded spawn/teardown mechanics; `SessionManager` adds the registry, lifecycle, and
|
||||
ownership on top.
|
||||
|
||||
**Package:** new `dev.ltms.bridged.session` — keeps the registry/lifecycle concern separate from
|
||||
**Package:** new `dev.ltms.fleet.session` — keeps the registry/lifecycle concern separate from
|
||||
the `worker` spawn mechanics. Holds `SessionManager` + `WorkerSession`.
|
||||
|
||||
**`recycle` is IN SCOPE for CB-301** (decided): implement `recycle(paneId, …)` = `release` the old
|
||||
session then `acquire` a fresh one, asserting a new distinct paneId (the no-reuse invariant). It is
|
||||
a thin convenience over the two primitives, shipped now so the no-reuse teardown+respawn path is
|
||||
covered by a test from day one.
|
||||
|
||||
### `WorkerSession` (record or small mutable holder)
|
||||
|
||||
| Field | Source | Notes |
|
||||
@@ -88,7 +83,6 @@ SPAWNING|READY|BUSY|DONE --vanished/drop--> FAILED
|
||||
final class SessionManager {
|
||||
WorkerSession acquire(String profile, String requestedCwd, String callerCwd, String ownerTerminal);
|
||||
void release(String paneId); // deterministic teardown + deregister
|
||||
WorkerSession recycle(String paneId, ...); // release + acquire (no-reuse convenience)
|
||||
Optional<WorkerSession> get(String paneId);
|
||||
List<WorkerSession> roster(); // bridge-owned view (CB-304 consumes this)
|
||||
// lifecycle hooks (package-private): onReady/onDelivered/onComplete/onFailed(target)
|
||||
@@ -102,13 +96,13 @@ final class SessionManager {
|
||||
|
||||
### Integration points
|
||||
|
||||
- **`Bridged.main`** — construct `SessionManager(workerService, ...)`; wire it as/decorating the
|
||||
- **`Fleetd.main`** — construct `SessionManager(workerService, ...)`; wire it as/decorating the
|
||||
`TurnListener` alongside `CompletionResolver` so it sees turn boundaries, and give it the
|
||||
`WorkerPresence` signal for `READY`.
|
||||
- **`BridgeMcp.spawn` / `BridgedApp.spawnWorker`** — route spawn through `SessionManager.acquire`
|
||||
(carry `callerTerminal` as `ownerTerminal`). **`bridge_stop` / `DELETE /workers/{paneId}`** →
|
||||
- **`FleetMcp.spawn` / `FleetApp.spawnWorker`** — route spawn through `SessionManager.acquire`
|
||||
(carry `callerTerminal` as `ownerTerminal`). **`fleet_stop` / `DELETE /workers/{paneId}`** →
|
||||
`SessionManager.release`.
|
||||
- **`bridge_list` / `GET /sessions` (CB-304 later)** — read `SessionManager.roster()`.
|
||||
- **`fleet_list` / `GET /sessions` (CB-304 later)** — read `SessionManager.roster()`.
|
||||
- **`MessageService`** — no change required for one-shot; a later CB-303 auto-release hook can call
|
||||
`release` from `onTurnComplete` under policy.
|
||||
|
||||
@@ -120,12 +114,11 @@ final class SessionManager {
|
||||
3. `release` tears the worker down via `WorkerService.stop` and removes it from `roster()`;
|
||||
a second `release` on the same paneId is a harmless no-op.
|
||||
4. `onTurnFailed` / drop moves the session to `FAILED` and it is absent from the live roster.
|
||||
5. `recycle` produces a new paneId and the old one is gone (no-reuse invariant).
|
||||
6. `roster()` reflects exactly the sessions acquired-minus-released, joined with live status.
|
||||
5. `roster()` reflects exactly the sessions acquired-minus-released, joined with live status.
|
||||
|
||||
## Seams left open (deliberately)
|
||||
|
||||
- **CB-302** — attach a checkpoint step (`STATE.md` + commit) to the `release` path.
|
||||
- **CB-303** — a policy loop over `roster()` using `spawnedAtNanos`/state to auto-`release` on
|
||||
`idle_ttl`, or drain on `context_cap`.
|
||||
- **CB-304** — `bridge_list` reads `roster()` for a bridge-owned roster + live join.
|
||||
- **CB-304** — `fleet_list` reads `roster()` for a bridge-owned roster + live join.
|
||||
|
||||
@@ -2,12 +2,12 @@
|
||||
|
||||
**Status:** ✅ shipped — implemented at commit `97ecc71` (per-worker git worktree + config-parity
|
||||
overlay). As-built: `session/GitWorktrees.java` behind the `Worktrees` port, wired in
|
||||
`Bridged.main` and configurable via `worktreeRoot` / per-profile `parityOverlay`
|
||||
(see `bridged.example.yaml`). Branch/worktree surface in `bridge_list` landed with CB-304
|
||||
`Fleetd.main` and configurable via `worktreeRoot` / per-profile `parityOverlay`
|
||||
(see `fleetd.example.yaml`). Branch/worktree surface in `fleet_list` landed with CB-304
|
||||
(`9fe04bf`); the worker-opened-PR checkpoint landed as CB-302 (`64e70ef`).
|
||||
**Extends:** [CB-301 Session Manager](CB-301-Session-Manager.md) (shipped, commit `54d907c`).
|
||||
**Realizes:** the config-parity requirement in [Worker Git Workflow](Worker-Git-Workflow.md).
|
||||
**Grounded in:** `SessionManager`, `WorkerService.spawn/effectiveCwd`, `BridgedConfig.Worker`,
|
||||
**Grounded in:** `SessionManager`, `WorkerService.spawn/effectiveCwd`, `FleetConfig.Worker`,
|
||||
`inject/…LsofPeerPidLookup` (the `ProcessBuilder` exec pattern).
|
||||
|
||||
## Problem
|
||||
@@ -43,7 +43,7 @@ provider.
|
||||
### `WorktreeRequest` (new, nullable = "no worktree")
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.session;
|
||||
package dev.ltms.fleet.session;
|
||||
/** Ask acquire() to provision an isolated worktree. null ⇒ run in the shared primary tree. */
|
||||
public record WorktreeRequest(String ticketSlug, String baseRef) {
|
||||
// ticketSlug seeds the branch name; baseRef null/blank ⇒ current HEAD of the repo.
|
||||
@@ -63,7 +63,7 @@ untouched.
|
||||
### `Worktrees` seam (new)
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.session;
|
||||
package dev.ltms.fleet.session;
|
||||
public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree add <path> -b <branch> <baseRef|HEAD>. Returns the worktree path. */
|
||||
String add(String repoRoot, String branch, String baseRef);
|
||||
@@ -112,7 +112,7 @@ public void release(String paneId) {
|
||||
}
|
||||
```
|
||||
|
||||
### Config — `BridgedConfig.Worker.parityOverlay` + a `worktreeRoot`
|
||||
### Config — `FleetConfig.Worker.parityOverlay` + a `worktreeRoot`
|
||||
|
||||
- Add `List<String> parityOverlay` to the `Worker` record (12th field). Compact-constructor default
|
||||
when null/empty: `[".mcp.json", ".claude/settings.local.json", ".env", ".envrc"]` (missing paths are
|
||||
@@ -122,7 +122,7 @@ public void release(String paneId) {
|
||||
|
||||
### Surface: MCP + REST
|
||||
|
||||
- `bridge_spawn` gains an optional `worktree` arg: `true`, or a ticket slug string. Truthy ⇒ build a
|
||||
- `fleet_spawn` gains an optional `worktree` arg: `true`, or a ticket slug string. Truthy ⇒ build a
|
||||
`WorktreeRequest(slug, null)` and call the 5-arg `acquire`.
|
||||
- `POST /workers` gains `worktree` (+ optional `ticket`) in the body/query, same mapping.
|
||||
- `workerView`/`view(WorkerSession)` include `worktree` and `branch` **when non-null** (omit for
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
# CB-306 — Spawn-Readiness Gate (launcher-owned terminal readiness)
|
||||
|
||||
**Status:** design note / delegation spec (branch `worker/cb-306-readiness`)
|
||||
**Issue:** gitea `lms/claude-bridge` #4
|
||||
**Issue:** gitea `fleet/fleetd` #4
|
||||
**Owner of the behaviour:** `ClaudeCodeLauncher` (the `PeerLauncher` adapter) — NOT core.
|
||||
|
||||
## 1. Problem
|
||||
|
||||
`bridge_spawn` today returns a session the instant the herdr pane is started. The pane is not
|
||||
`fleet_spawn` today returns a session the instant the herdr pane is started. The pane is not
|
||||
yet a usable Claude REPL — it may still be sitting at the folder-trust prompt, or the CLI may
|
||||
never come up at all. Nothing blocks or times out on that. Consequences:
|
||||
|
||||
- A `bridge_send` to a not-yet-ready worker surfaces as a **~60 s MCP-client timeout** (the send
|
||||
- A `fleet_send` to a not-yet-ready worker surfaces as a **~60 s MCP-client timeout** (the send
|
||||
blocks waiting for a turn that can't start) instead of a fast, explicit spawn failure.
|
||||
- A worker stuck at the folder-trust prompt lingers in `SPAWNING` forever; nothing fails it.
|
||||
|
||||
@@ -55,7 +55,7 @@ a crash, and a slow start all present as "never becomes injectable" and all corr
|
||||
or `spawnReadyTimeoutMs` elapses.
|
||||
3. **Ready** → return the `WorkerHandle(paneId, terminalId)` as today.
|
||||
4. **Timeout** → the launcher **closes the pane it started** (and its tab, via the same path
|
||||
`release`/`stop` uses) and throws **`PeerUnreachableException`** (new, in `dev.ltms.bridged.peer`).
|
||||
`release`/`stop` uses) and throws **`PeerUnreachableException`** (new, in `dev.ltms.fleet.peer`).
|
||||
No orphan pane is left behind — the launcher cleans up its own failed birth.
|
||||
|
||||
`spawnReadyTimeoutMs == 0` (or unset) **disables** the gate = legacy non-blocking behaviour, so the
|
||||
@@ -79,8 +79,8 @@ Unit tests (add to the existing `ClaudeCodeLauncher` test):
|
||||
|
||||
## 5. Config
|
||||
|
||||
Add to the launcher-level config (a bridged-level knob, not per-profile) in `bridged.yaml` +
|
||||
`BridgedConfig`:
|
||||
Add to the launcher-level config (a fleetd-level knob, not per-profile) in `fleetd.yaml` +
|
||||
`FleetConfig`:
|
||||
|
||||
```yaml
|
||||
spawn_ready_timeout_ms: 20000 # 0 disables the gate (legacy non-blocking spawn)
|
||||
@@ -88,7 +88,7 @@ spawn_ready_poll_ms: 300
|
||||
```
|
||||
|
||||
Jackson ignores unknown keys, so omitting them in existing YAML is safe; pick sane defaults in code
|
||||
(`20000` / `300`). Keep the names consistent with existing config field style in `BridgedConfig`.
|
||||
(`20000` / `300`). Keep the names consistent with existing config field style in `FleetConfig`.
|
||||
|
||||
## 6. Core / MCP propagation
|
||||
|
||||
@@ -99,11 +99,11 @@ Jackson ignores unknown keys, so omitting them in existing YAML is safe; pick sa
|
||||
`spawn` throws — verify the new exception flows through it (worktree removed, nothing registered).
|
||||
- The **non-worktree** path registers the session only *after* `spawn` returns, so a throw means no
|
||||
half-live `SPAWNING` session is ever registered — confirm this and add a test.
|
||||
- `bridge_spawn` (MCP verb) must return an **error result** carrying the exception message, not a
|
||||
success with a dead session. Trace `BridgeMcp`/`BridgedApp` spawn handlers and make sure the
|
||||
- `fleet_spawn` (MCP verb) must return an **error result** carrying the exception message, not a
|
||||
success with a dead session. Trace `FleetMcp`/`FleetApp` spawn handlers and make sure the
|
||||
exception becomes a clean tool error, not an uncaught 500 with a stack trace.
|
||||
|
||||
**Out of scope (do NOT do here):** gating `bridge_send` on session `READY` (existing status-gate +
|
||||
**Out of scope (do NOT do here):** gating `fleet_send` on session `READY` (existing status-gate +
|
||||
this spawn gate already close the window), MCP-handshake-as-readiness signal, the CB-307 broker,
|
||||
any config `kind:` discriminator, any second adapter.
|
||||
|
||||
@@ -111,7 +111,7 @@ any config `kind:` discriminator, any second adapter.
|
||||
|
||||
- `ClaudeCodeLauncher.spawn` blocks until injectable or throws `PeerUnreachableException` +
|
||||
self-reaps the pane; gate disabled when timeout is 0.
|
||||
- New `PeerUnreachableException` in `dev.ltms.bridged.peer`.
|
||||
- New `PeerUnreachableException` in `dev.ltms.fleet.peer`.
|
||||
- Config knobs wired (`spawn_ready_timeout_ms`, `spawn_ready_poll_ms`) with safe defaults.
|
||||
- Existing `SPAWNING→READY` MCP-contact transition untouched.
|
||||
- New unit tests (ready / timeout+reap / disabled) green; **all existing tests still pass unchanged**.
|
||||
|
||||
@@ -13,28 +13,28 @@ Stage 2 (the AMQP/LavinMQ adapter behind the same port) is explicitly **out of s
|
||||
## 1. The bug this fixes (grounded in current code)
|
||||
|
||||
The reverse (worker→primary) path is `Rendezvous` — a `ConcurrentHashMap<session, CompletableFuture<Resolution>>`
|
||||
of **live blocking waiters only**. No queue, no store. When a worker calls `bridge_reply` and **no send
|
||||
of **live blocking waiters only**. No queue, no store. When a worker calls `fleet_reply` and **no send
|
||||
is currently open** for that worker:
|
||||
|
||||
- `Rendezvous.resolve(session, content)` → `complete(...)` → `waiters.get(session) == null` →
|
||||
returns `false` (`msg/Rendezvous.java:212-215`).
|
||||
- The `content` string is **never retained** — it is dropped. The worker is told it failed:
|
||||
`BridgeMcp.reply` returns `error("no send is awaiting a reply for this worker")` (`mcp/BridgeMcp.java:270-272`);
|
||||
REST returns `409 no_pending_send` (`rest/BridgedApp.java:339-345`).
|
||||
`FleetMcp.reply` returns `error("no send is awaiting a reply for this worker")` (`mcp/FleetMcp.java:270-272`);
|
||||
REST returns `409 no_pending_send` (`rest/FleetApp.java:339-345`).
|
||||
|
||||
This is the observed "communication break": a worker that finishes just after its `bridge_send` timed
|
||||
This is the observed "communication break": a worker that finishes just after its `fleet_send` timed
|
||||
out (the ~60s sync window) replies into the void. There is **no message-id, dedup, or ack** anywhere in
|
||||
the message path today.
|
||||
|
||||
## 2. What to build
|
||||
|
||||
### 2.1 The port — `dev.ltms.bridged.msg.ReplyInbox`
|
||||
### 2.1 The port — `dev.ltms.fleet.msg.ReplyInbox`
|
||||
|
||||
A thin interface owned by the `msg` layer. The in-memory adapter is Stage 1; the AMQP adapter (Stage 2)
|
||||
implements the **same** interface, so keep it broker-agnostic.
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.msg;
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@@ -70,7 +70,7 @@ public interface ReplyInbox {
|
||||
seen-set — your call; preserve insertion order).
|
||||
- `peek` returns an immutable copy; `ack` removes by `msgId`. Thread-safe (concurrent publish vs. drain).
|
||||
- **This is soft-state, NOT persistence.** Lost on a `java -jar` bounce — that is correct and consistent
|
||||
with "bridged stays soft-state." Do **not** add any file/DB backing.
|
||||
with "fleetd stays soft-state." Do **not** add any file/DB backing.
|
||||
|
||||
### 2.3 Publish seam — route reply through the service layer
|
||||
|
||||
@@ -79,7 +79,7 @@ in `MessageService`, which already owns the `Rendezvous` and will own the `Reply
|
||||
|
||||
- Add `MessageService.reply(String session, String content)`:
|
||||
```java
|
||||
/** Route a worker's explicit bridge_reply: resolve an open send, or queue it in the inbox if none. */
|
||||
/** Route a worker's explicit fleet_reply: resolve an open send, or queue it in the inbox if none. */
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
return true; // a live send took it — unchanged fast path
|
||||
@@ -89,12 +89,12 @@ in `MessageService`, which already owns the `Rendezvous` and will own the `Reply
|
||||
}
|
||||
```
|
||||
- Repoint the two callers off the bare `rendezvous.resolve(...)` onto `messages.reply(...)`:
|
||||
- `BridgeMcp.reply` (`mcp/BridgeMcp.java:262-273`) — on success return a normal ack; **remove** the
|
||||
- `FleetMcp.reply` (`mcp/FleetMcp.java:262-273`) — on success return a normal ack; **remove** the
|
||||
`error("no send is awaiting a reply…")` branch (that case is now a successful queue).
|
||||
- `BridgedApp.replyMessage` (`rest/BridgedApp.java:330-346`) — return `200` (queued) instead of
|
||||
- `FleetApp.replyMessage` (`rest/FleetApp.java:330-346`) — return `200` (queued) instead of
|
||||
`409 no_pending_send`.
|
||||
|
||||
**DO NOT touch the QUESTION path.** `bridge_ask` / `rendezvous.resolveQuestion` must keep today's
|
||||
**DO NOT touch the QUESTION path.** `fleet_ask` / `rendezvous.resolveQuestion` must keep today's
|
||||
`NO_WAITER` behaviour — a mid-turn question is **interactive** (the worker blocks synchronously and cannot
|
||||
consume a late answer), so it must **never** be queued. Only terminal `REPLY`s go to the inbox.
|
||||
|
||||
@@ -110,27 +110,27 @@ The primary re-checks a worker it delegated to. Expose a drain keyed by **worker
|
||||
- Add `MessageService.drainReplies(String target)`: `peek` the inbox, `ack` each returned `msgId`, hand
|
||||
back the `List<InboxMessage>` (or just the contents). At-least-once: peek→deliver→ack (ack only after
|
||||
the caller has them, so an in-flight failure re-surfaces them).
|
||||
- **DECISION (required default): expose via the existing poll verb, keyed by target.** Extend `bridge_poll`
|
||||
- **DECISION (required default): expose via the existing poll verb, keyed by target.** Extend `fleet_poll`
|
||||
to accept an optional `target` (worker session) and, when present, return that worker's drained replies —
|
||||
alongside a matching REST route `GET /sessions/{id}/replies`. Do **not** change `send`/`answer` semantics
|
||||
(do not drain inside `send` — that conflates "deliver to worker" with "collect its mail"). Keep the
|
||||
existing ticket-based `bridge_poll(ticket)` path working unchanged. If you see a cleaner surface, still
|
||||
existing ticket-based `fleet_poll(ticket)` path working unchanged. If you see a cleaner surface, still
|
||||
ship this default and note the alternative for review.
|
||||
|
||||
## 3. Config
|
||||
|
||||
**None for Stage 1.** The in-memory adapter is the unconditional default — wire `new InMemoryReplyInbox()`
|
||||
into `MessageService` in `Bridged.main`. Do **not** add a `broker:` config block (that arrives with the
|
||||
into `MessageService` in `Fleetd.main`. Do **not** add a `broker:` config block (that arrives with the
|
||||
Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 4. Acceptance criteria (what the primary will verify)
|
||||
|
||||
1. New `ReplyInbox` + `InboxMessage` + `InMemoryReplyInbox` in `dev.ltms.bridged.msg`.
|
||||
2. `bridge_reply` with **no open send** now **succeeds and queues** (no more `error` / `409`); the reply is
|
||||
1. New `ReplyInbox` + `InboxMessage` + `InMemoryReplyInbox` in `dev.ltms.fleet.msg`.
|
||||
2. `fleet_reply` with **no open send** now **succeeds and queues** (no more `error` / `409`); the reply is
|
||||
later retrievable and identical.
|
||||
3. The queued reply is drainable by the primary keyed by target; draining **acks** it (a second drain
|
||||
returns nothing); dedup by `msgId` (re-publishing the same id does not double-queue).
|
||||
4. **QUESTION path unchanged** — `bridge_ask` with no open send still returns `NO_WAITER` (add/keep a test
|
||||
4. **QUESTION path unchanged** — `fleet_ask` with no open send still returns `NO_WAITER` (add/keep a test
|
||||
proving a question is never queued).
|
||||
5. Completion/failure fallbacks unchanged.
|
||||
6. Unit tests covering: `InMemoryReplyInbox` publish/peek/ack/dedup/FIFO/concurrency; `MessageService.reply`
|
||||
@@ -140,7 +140,7 @@ Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 5. Build & verification (worker side)
|
||||
|
||||
- Build with Maven from the worktree's `bridged/` dir. **Capture the exit code without a masking pipe**
|
||||
- Build with Maven from the worktree's `fleetd/` dir. **Capture the exit code without a masking pipe**
|
||||
(`mvn clean install; echo "MVN_EXIT=$?"` — never `mvn … | tail`, which hides failures).
|
||||
- Read the real test totals from `target/surefire-reports/TEST-*.xml`, not from stdout scroll.
|
||||
- You do **not** have IDE MCP access — do not claim `ide_diagnostics` results. The **primary** runs the
|
||||
@@ -152,7 +152,7 @@ Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
- **`.mcp.json` is `--skip-worktree` in your worktree — never edit, `git add`, or commit it.**
|
||||
- **`wiki/` is a submodule — never run git in it; never touch it.**
|
||||
- Commit only your feature changes (the new port/adapter, the `msg`/`mcp`/`rest` wiring, tests, and if
|
||||
you add config wiring in `Bridged.java`). Nothing else.
|
||||
you add config wiring in `Fleetd.java`). Nothing else.
|
||||
- Work only inside your assigned worktree on your feature branch. The primary fast-forwards `main` after
|
||||
re-gating — do not touch `main`.
|
||||
- Java 25 idioms are welcome (unnamed `_` params, records). Keep the diff minimal and match surrounding style.
|
||||
@@ -171,7 +171,7 @@ removal, msgId dedup, cross-restart redelivery). It is `@Tag("contract")`, so th
|
||||
untouched. Run it explicitly when Docker (or a broker) is available:
|
||||
|
||||
```bash
|
||||
cd bridged
|
||||
cd fleetd
|
||||
mvn -Pcontract test -Dtest=AmqpReplyInboxContractTest # local: spins a RabbitMQ Testcontainers fixture
|
||||
```
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ The design rests on three pieces (the shape this ticket proposes):
|
||||
1. **Dedicated per-agent channels** — every agent has its own addressable inbox on the broker.
|
||||
2. **A federated agent directory** — a global "who/where/status" lookup, assembled from per-host
|
||||
presence, not a central database.
|
||||
3. **A per-host gateway** — each host runs a `bridged` that owns its local herdr, registers/manages
|
||||
3. **A per-host gateway** — each host runs a `fleetd` that owns its local herdr, registers/manages
|
||||
its own sessions, and proxies messages to/from other hosts over the broker.
|
||||
|
||||
## 2. What is single-host today (the assumptions to break)
|
||||
@@ -27,7 +27,7 @@ The design rests on three pieces (the shape this ticket proposes):
|
||||
flowchart TB
|
||||
subgraph host["Single host (today)"]
|
||||
primary["primary<br/>(MCP client)"]
|
||||
daemon["bridged daemon<br/>127.0.0.1:8765"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765"]
|
||||
reg["in-process registry<br/>keyed by PeerHandle.id() == paneId"]
|
||||
herdr["herdr<br/>(local unix-socket PTY mux)"]
|
||||
w1["worker pane wQ:p1"]
|
||||
@@ -48,21 +48,21 @@ Three concrete bake-ins assume one host:
|
||||
|---|---|---|
|
||||
| **herdr is local** | `herdr/` unix socket `~/.config/herdr/herdr.sock` | You cannot drive another host's PTYs → each host **must** own its herdr. This is why a per-host gateway is mandatory. |
|
||||
| **registry is in-process, keyed by `paneId`** | `session/SessionManager` | `paneId` (e.g. `wQ:p2B`) is a herdr-local coordinate — meaningless off-host. Routing needs a host-unique id. |
|
||||
| **loopback, no authn** | `rest/BridgedApp` binds `127.0.0.1:8765` | Fine on one host; the moment a second host can talk to a gateway, that link is a trust boundary. |
|
||||
| **loopback, no authn** | `rest/FleetApp` binds `127.0.0.1:8765` | Fine on one host; the moment a second host can talk to a gateway, that link is a trust boundary. |
|
||||
|
||||
## 3. Target architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A"]
|
||||
gA["gateway = bridged A"]
|
||||
gA["gateway = fleetd A"]
|
||||
regA["local registry + herdr"]
|
||||
primary["primary (MCP client)"]
|
||||
gA --- regA
|
||||
primary --- gA
|
||||
end
|
||||
subgraph hostB["HOST B"]
|
||||
gB["gateway = bridged B"]
|
||||
gB["gateway = fleetd B"]
|
||||
regB["local registry + herdr"]
|
||||
wb["worker panes"]
|
||||
gB --- regB
|
||||
@@ -96,13 +96,13 @@ host's terminals.*
|
||||
delayed-message exchange (the remind/backoff loop for free) — the same reasons CB-307 picked it.
|
||||
|
||||
- **Federated agent directory** = a **soft-state, bridge-owned** roster, *not* a broker-stored
|
||||
database. Per the persistence-boundary decision (bridged is soft-state; the broker owns *message*
|
||||
database. Per the persistence-boundary decision (fleetd is soft-state; the broker owns *message*
|
||||
durability, not *who/where/status*), each gateway announces its local agents `(globalId, host,
|
||||
status, capabilities)` on a `roster.*` presence topic with periodic heartbeats. Every gateway
|
||||
builds an eventually-consistent **union view** — literally CB-304's `rosterView`, federated. A
|
||||
stale entry expires by missed heartbeat (reuses CB-303's idle/TTL thinking).
|
||||
|
||||
- **Per-host gateway** = today's `bridged` daemon, evolved. It already registers/manages sessions
|
||||
- **Per-host gateway** = today's `fleetd` daemon, evolved. It already registers/manages sessions
|
||||
and controls its local herdr; multi-host adds exactly two responsibilities: (a) a broker client
|
||||
that consumes its agents' inboxes and injects into local herdr, and (b) presence announce +
|
||||
union-roster assembly. Evolution, not rewrite.
|
||||
@@ -111,7 +111,7 @@ host's terminals.*
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
send["bridge_send(globalId, msg)"] --> lookup{"directory:<br/>is globalId local?"}
|
||||
send["fleet_send(globalId, msg)"] --> lookup{"directory:<br/>is globalId local?"}
|
||||
lookup -->|"yes"| local["inject via local herdr<br/>(today's Injector path)"]
|
||||
lookup -->|"no"| pub["publish agent.<id>.inbox<br/>(broker routes to owning gateway)"]
|
||||
pub --> consume["owning gateway consumes<br/>→ injects into its local herdr"]
|
||||
@@ -123,7 +123,7 @@ broker. A sender is oblivious to which branch it took.*
|
||||
## 4. What CB-307 already provides vs. what is net-new
|
||||
|
||||
**CB-307 delivers the transport half** and is independently valuable on a single host: the AMQP
|
||||
broker fabric, the `bridged → broker` client/adapter, at-least-once + idempotent (dedup-by-id)
|
||||
broker fabric, the `fleetd → broker` client/adapter, at-least-once + idempotent (dedup-by-id)
|
||||
delivery, DLQ, and delayed-retry (remind). That *is* the "proxy cross-host message" backbone;
|
||||
extending the same broker from "worker→primary reliability" to "gateway↔gateway" is incremental.
|
||||
|
||||
@@ -158,17 +158,17 @@ sequenceDiagram
|
||||
participant BR as broker
|
||||
participant GA as gateway A
|
||||
participant P as primary (host A, MCP client)
|
||||
W->>GB: bridge_reply
|
||||
W->>GB: fleet_reply
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route to A's primary inbox
|
||||
Note over GA: held durably until the primary pulls
|
||||
P->>GA: blocking bridge_send resolves / bridge_poll
|
||||
P->>GA: blocking fleet_send resolves / fleet_poll
|
||||
GA-->>P: reply (then ACK to broker)
|
||||
```
|
||||
|
||||
*Figure 4 — the broker makes the middle hop lossless, ordered, and idempotent; the **final** hop
|
||||
into the primary is still a **pull** (gateway A holds the message until the primary's blocking
|
||||
`bridge_send` or `bridge_poll`). Cross-host neither improves nor worsens this — it just spans hosts.
|
||||
`fleet_send` or `fleet_poll`). Cross-host neither improves nor worsens this — it just spans hosts.
|
||||
This is precisely the gap CB-307 closes on one host and CB-308 stretches across hosts.*
|
||||
|
||||
## 6. Staging & dependencies
|
||||
@@ -207,8 +207,8 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
gateway's host. This extends the single-host invariant — *identity comes from the connection,
|
||||
never an argument* — across the broker: cross-host, identity comes from the key. Complements
|
||||
(not replaces) per-gateway broker logins over TLS.
|
||||
2. **Profiles are owned by the worker's host.** `bridge_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `bridged.yaml`. Gateways advertise their profile names in presence
|
||||
2. **Profiles are owned by the worker's host.** `fleet_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `fleetd.yaml`. Gateways advertise their profile names in presence
|
||||
heartbeats, so a leader sees what each host offers before spawning; an unknown name is a clear
|
||||
error from the target. Secrets (base URLs, tokens) never leave the host that uses them.
|
||||
3. **Repo provisioning — clone from the forge, pinned.** A cross-host spawn names the repo URL and
|
||||
@@ -259,7 +259,7 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
serialize acks fleet-wide. Ordering caveat: a *return* (unroutable) arrives **before** the
|
||||
confirm, so "confirmed" ≠ "routed"; the sender checks the returned-set at confirm time.
|
||||
`mandatory` is false only for `BROADCAST`, where an empty group is legal silence.
|
||||
10. **Queue lifecycle is session lifecycle.** `bridge_stop`/reap deletes the worker's inbox queue
|
||||
10. **Queue lifecycle is session lifecycle.** `fleet_stop`/reap deletes the worker's inbox queue
|
||||
(its `broadcast.*` bindings die with it — no broadcasts to the dead); `x-expires` collects
|
||||
queues orphaned by a crashed gateway (long for main/orchestrator inboxes, short for workers).
|
||||
Queue names carry a version suffix (`.v2`): AMQP refuses to redeclare an existing durable
|
||||
@@ -275,11 +275,11 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
- **Gateway discovery:** how gateways find the broker and each other (static config vs. discovery).
|
||||
- **Control authorization — THE GATE ON U4.** Signing (§7.1) settles *who sent it*; authorization
|
||||
is *who may do what*. **Cross-host spawn must not land before the minimal version exists**: a
|
||||
per-host allowlist in `bridged.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
per-host allowlist in `fleetd.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
publish control to this host, checked against the verified signature. A few lines of config and
|
||||
check; without them, any principal holding broker credentials can start processes on every host
|
||||
in the fleet.
|
||||
- **Key distribution & rotation:** static config (host → public key in each `bridged.yaml`) is
|
||||
- **Key distribution & rotation:** static config (host → public key in each `fleetd.yaml`) is
|
||||
fine at the current 2–3 host scale; rotation is manual. A refinement, not a blocker.
|
||||
- **Gateway death mid-turn:** the roster reaps it by missed heartbeat, and in-flight primary-bound
|
||||
messages survive by broker durability; still open is reconciling *worker* state when the dead
|
||||
|
||||
@@ -7,7 +7,7 @@ the core learning any one peer's environment.
|
||||
## 1. Why
|
||||
|
||||
`claude-bridge` is a **communication bus between heterogeneous AI agents** — its stable surface is
|
||||
the protocol (`bridge_spawn / send / poll / reply / ask / list / stop / status`), and that surface
|
||||
the protocol (`fleet_spawn / send / poll / reply / ask / list / stop / status`), and that surface
|
||||
should stay provider-neutral. Today the daemon can only materialize one kind of peer: an
|
||||
off-subscription Claude Code CLI over herdr. Everything specific to *how that peer is set up*
|
||||
(`ANTHROPIC_BASE_URL`, the subscription guard, `--mcp-config`/system-prompt flags, `claude-*`
|
||||
@@ -60,7 +60,7 @@ Where Claude/herdr specifics actually live today:
|
||||
| `claude-<profile>-<nonce>-<seq>` naming, `WORKER_NAME` regex, orphan reap (CB-117) | `WorkerService` | **→ adapter** (naming is a herdr-label detail) |
|
||||
| `GITEA_TOKEN` / `GITEA_HOST` injection (CB-302 checkpoint) | `WorkerService.spawn` | **→ adapter** + a **capability** (§6) |
|
||||
| tab/pane placement, worker space, tab labels | `WorkerService.spawnInTab/spawnAsPane` via herdr `WorkspaceControl` | **→ adapter** (herdr transport detail) |
|
||||
| `BridgedConfig.Worker` profile shape (`baseUrl`, `model`, `configDir`, …) | `config` | **mostly adapter-shaped** — see §5 |
|
||||
| `FleetConfig.Worker` profile shape (`baseUrl`, `model`, `configDir`, …) | `config` | **mostly adapter-shaped** — see §5 |
|
||||
| FSM, registry, roster, `reapIdle`/`drainAll`/`contextCap`, `rosterView` | `SessionManager` | **stays core** |
|
||||
| turn/completion detection (`TurnListener`, `CompletionResolver`, `StatusPoller`, `WorkerPresence`) | `inject/` | **stays core**, but reads herdr terminal output → transport-coupled (§4b) |
|
||||
| message store & routing | `msg/` | **stays core** |
|
||||
@@ -118,7 +118,7 @@ hard-wire "turns come from herdr".
|
||||
|
||||
## 5. Config shape
|
||||
|
||||
`BridgedConfig.Worker` is Claude-shaped (`baseUrl`, `model`, `configDir`, `tokenEnv`). Rather than
|
||||
`FleetConfig.Worker` is Claude-shaped (`baseUrl`, `model`, `configDir`, `tokenEnv`). Rather than
|
||||
break existing YAML, CB-401 keeps `workers:` exactly as-is and treats those fields as the
|
||||
**ClaudeCodeLauncher's** profile schema. A future peer kind adds a `kind:` discriminator
|
||||
(default `"claude-code"`) selecting the launcher; unknown-kind → clear config error. No migration of
|
||||
@@ -126,7 +126,7 @@ existing configs. (Jackson already ignores unknown keys, so adding `kind` is bac
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MCP as bridge_spawn (MCP/REST)
|
||||
participant MCP as fleet_spawn (MCP/REST)
|
||||
participant SM as SessionManager
|
||||
participant L as PeerLauncher (by profile.kind)
|
||||
participant T as transport (herdr)
|
||||
@@ -150,7 +150,7 @@ degrades gracefully when a launcher lacks one:
|
||||
|
||||
| Capability | Meaning | Claude Code | Codex (likely) | Human |
|
||||
|---|---|---|---|---|
|
||||
| `MID_TURN_ASK` | supports `bridge_ask` rendezvous | ✓ | ? | ✗ |
|
||||
| `MID_TURN_ASK` | supports `fleet_ask` rendezvous | ✓ | ? | ✗ |
|
||||
| `SELF_PR` | can open its own PR at checkpoint (CB-302) | ✓ (opt-in token) | ? | ✗ |
|
||||
| `WORKTREE` | can run in a provisioned git worktree | ✓ | ✓ | ✗ |
|
||||
| `ORPHAN_REAP` | spawner can reconcile orphaned peers on boot | ✓ | ? | ✗ |
|
||||
@@ -185,13 +185,13 @@ messages, not verified facts.
|
||||
Deliverable for CB-401 Stage A — mechanical, behaviour-preserving:
|
||||
|
||||
1. `PeerLauncher` interface + `PeerHandle` (opaque id) + `SpawnRequest` (profile, requestedCwd,
|
||||
callerCwd) + `Capability` enum, new package `dev.ltms.bridged.peer`.
|
||||
callerCwd) + `Capability` enum, new package `dev.ltms.fleet.peer`.
|
||||
2. `ClaudeCodeLauncher implements PeerLauncher` = today's `WorkerService`, adapted: `spawn(...)`
|
||||
returns a `PeerHandle` (id = paneId), `capabilities()` declares
|
||||
`MID_TURN_ASK, SELF_PR(when token), WORKTREE, ORPHAN_REAP`.
|
||||
3. `SessionManager` depends on `PeerLauncher`, not `WorkerService` concretely; routing keys on
|
||||
`PeerHandle.id()` (== paneId today, so zero value change).
|
||||
4. `Bridged.main` wires the concrete `ClaudeCodeLauncher` behind the interface.
|
||||
4. `Fleetd.main` wires the concrete `ClaudeCodeLauncher` behind the interface.
|
||||
5. **No behaviour change, no config change.** Full green gate: `ide_sync` → `ide_diagnostics`
|
||||
(0 errors/0 warnings) → `mvn clean install` with `MVN_EXIT` captured (no masking pipe). All
|
||||
existing tests pass unchanged; add tests only for the new `PeerHandle` indirection.
|
||||
|
||||
@@ -11,7 +11,7 @@ All five increments of §4 are done, including increment 5 (the §5 live checkli
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Prove the [`PeerLauncher`](../bridged/src/main/java/dev/ltms/bridged/peer/PeerLauncher.java) SPI
|
||||
Prove the [`PeerLauncher`](../fleetd/src/main/java/dev/ltms/fleet/peer/PeerLauncher.java) SPI
|
||||
actually holds for a **non-Claude** coding agent by shipping a second, first-class in-tree
|
||||
adapter: **opencode** (`opencode` 1.1.31, a provider-agnostic terminal coding agent).
|
||||
|
||||
@@ -56,8 +56,8 @@ flowchart TB
|
||||
|
||||
*Figure 1 — the two concerns tangled inside today's single launcher; CB-402 splits them.*
|
||||
|
||||
There is also a **Stage-A deferral** to finish: `Bridged.main` still casts
|
||||
`(ClaudeCodeLauncher) workers` at the `BridgeMcp` and `BridgedApp` constructors. Those two
|
||||
There is also a **Stage-A deferral** to finish: `Fleetd.main` still casts
|
||||
`(ClaudeCodeLauncher) workers` at the `FleetMcp` and `FleetApp` constructors. Those two
|
||||
callers only invoke `profiles()`, `defaultProfile()`, and `list()` — **all already on the
|
||||
`PeerLauncher` interface**. The cast survives for one reason only: `PeerLauncher.list()`
|
||||
returns `List<?>` (element type erased) while the callers use `Agent` element methods in their
|
||||
@@ -68,7 +68,7 @@ roster join. Finishing the migration is therefore small and contained (§4.D).
|
||||
## 3. Target design
|
||||
|
||||
Template-Method base + two thin adapters + a routing composite that keeps the Stage-A seam
|
||||
(one `PeerLauncher` reference held by `SessionManager` / `BridgeMcp` / `BridgedApp`) intact.
|
||||
(one `PeerLauncher` reference held by `SessionManager` / `FleetMcp` / `FleetApp`) intact.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
@@ -114,7 +114,7 @@ Two adapter **hooks** (abstract):
|
||||
protected abstract String namePrefix(); // "claude" | "opencode"
|
||||
|
||||
/** Build the peer-specific launch: env map + argv. Runs any pre-spawn guard here. */
|
||||
protected abstract Launch buildLaunch(BridgedConfig.Worker cfg, SpawnRequest req);
|
||||
protected abstract Launch buildLaunch(FleetConfig.Worker cfg, SpawnRequest req);
|
||||
record Launch(Map<String,String> env, List<String> argv) {}
|
||||
```
|
||||
|
||||
@@ -126,7 +126,7 @@ so the opencode adapter never reaps a `claude-*` pane and vice-versa. The compos
|
||||
|
||||
### B. `kind:` config discriminator
|
||||
|
||||
Add one field to `BridgedConfig.Worker`:
|
||||
Add one field to `FleetConfig.Worker`:
|
||||
|
||||
```java
|
||||
String kind // "claude-code" (default) | "opencode"
|
||||
@@ -138,7 +138,7 @@ String kind // "claude-code" (default) | "opencode"
|
||||
so the record stays declarative.)
|
||||
- Keep the existing back-compat constructors; `kind` is additive and optional.
|
||||
|
||||
`bridged.example.yaml` documents a two-kind `workers:` block.
|
||||
`fleetd.example.yaml` documents a two-kind `workers:` block.
|
||||
|
||||
### C. `OpenCodeLauncher` — the adapter hooks for opencode
|
||||
|
||||
@@ -164,13 +164,13 @@ String kind // "claude-code" (default) | "opencode"
|
||||
|
||||
### D. `CompositePeerLauncher` + finish the Stage-A migration
|
||||
|
||||
- `Bridged.main` groups configured profiles by `kind`, instantiates one launcher per kind
|
||||
- `Fleetd.main` groups configured profiles by `kind`, instantiates one launcher per kind
|
||||
present, and wraps them in `CompositePeerLauncher implements PeerLauncher`.
|
||||
- Routing methods (`spawn(req)`, `effectiveCwd(req)`, `parityOverlay(name)`) dispatch by the
|
||||
profile's kind. Fan-out methods (`list()`, `reapOrphanWorkers()`, `capabilities()`,
|
||||
`profiles()`, `defaultProfile()`) merge across sub-launchers. `stop(id)` tries each (teardown
|
||||
only knows the pane id) — already best-effort/idempotent.
|
||||
- **Migrate `BridgeMcp` + `BridgedApp` to the `PeerLauncher` interface**, dropping both
|
||||
- **Migrate `FleetMcp` + `FleetApp` to the `PeerLauncher` interface**, dropping both
|
||||
`(ClaudeCodeLauncher)` casts. Only friction is `list()`'s `List<?>`; resolve by giving the SPI
|
||||
a typed roster element (small neutral `PeerAgent` view exposing `id()`/`name()`/status) that
|
||||
the CB-304 roster join consumes — or, minimally, narrow at the callsite. Prefer the typed view.
|
||||
@@ -179,12 +179,12 @@ String kind // "claude-code" (default) | "opencode"
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant P as Primary
|
||||
participant M as BridgeMcp / REST
|
||||
participant M as FleetMcp / REST
|
||||
participant C as CompositePeerLauncher
|
||||
participant O as OpenCodeLauncher
|
||||
participant B as HerdrPeerLauncher (base)
|
||||
participant H as herdr
|
||||
P->>M: bridge_spawn(profile="oc-impl")
|
||||
P->>M: fleet_spawn(profile="oc-impl")
|
||||
M->>C: spawn(SpawnRequest)
|
||||
C->>C: kind(profile)=="opencode"
|
||||
C->>O: spawn(req)
|
||||
@@ -208,7 +208,7 @@ sequenceDiagram
|
||||
`ClaudeCodeLauncher` extend it with `namePrefix()="claude"` and `buildLaunch()` wrapping
|
||||
today's guard+env+argv logic. Green build, identical tests — pure refactor. *(IDE
|
||||
`refactor` where possible; the primary re-runs the gate workers can't.)*
|
||||
2. **`kind:` discriminator.** Add the field + normalization + `bridged.example.yaml`. Default
|
||||
2. **`kind:` discriminator.** Add the field + normalization + `fleetd.example.yaml`. Default
|
||||
path unchanged (`kind=claude-code`).
|
||||
3. **`OpenCodeLauncher`.** Implement the three hooks; unit-test `buildLaunch` (env has no
|
||||
`ANTHROPIC_BASE_URL`; `OPENCODE_CONFIG` points at a file carrying the bridge MCP block +
|
||||
@@ -226,11 +226,11 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
- **opencode TUI ⇄ herdr injection.** herdr drives a pane by typing into a TUI. Must confirm
|
||||
opencode's TUI accepts injected keystrokes/submit the way `claude` does, and reaches an
|
||||
`injectable` status the CB-306 gate recognizes. *Validation:* spawn one opencode worker,
|
||||
watch the readiness gate pass, `bridge_send` a trivial task.
|
||||
watch the readiness gate pass, `fleet_send` a trivial task.
|
||||
- **Bridge MCP visibility in opencode.** Confirm `OPENCODE_CONFIG` (or `opencode mcp add`)
|
||||
actually surfaces the `bridge_*` tools inside the opencode session, and that `bridge_reply`
|
||||
actually surfaces the `fleet_*` tools inside the opencode session, and that `fleet_reply`
|
||||
is callable — the reply-charter is worthless if the tool isn't mounted. *Validation:* the
|
||||
worker completes a task by calling `bridge_reply`; the reply lands via the CB-307 path.
|
||||
worker completes a task by calling `fleet_reply`; the reply lands via the CB-307 path.
|
||||
- **opencode MCP/config schema drift.** opencode is fast-moving (1.1.31 today). Pin the config
|
||||
schema we generate against the installed version; treat the exact keys (`type: "remote"` vs
|
||||
`"http"`, `instructions` shape) as a dogfood-verified fact, not an assumption.
|
||||
@@ -244,7 +244,7 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
|
||||
- **Unit (hermetic):** base-extraction regression (existing `ClaudeCodeLauncher` tests pass
|
||||
unchanged); `OpenCodeLauncher.buildLaunch` env/argv/config assertions; `kind` normalization
|
||||
in `BridgedConfigTest`; `CompositePeerLauncher` routing + fan-out (merge of `profiles()`,
|
||||
in `FleetConfigTest`; `CompositePeerLauncher` routing + fan-out (merge of `profiles()`,
|
||||
summed `reapOrphanWorkers()`, per-kind reap isolation) with fake sub-launchers.
|
||||
- **Live (dogfood, manual):** the §5 checklist on the running daemon.
|
||||
- **Gate (primary):** IDE diagnostics 0/0 on every changed file, `mvn clean install` green with
|
||||
@@ -269,7 +269,7 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
|
||||
## 8. As-built — live dogfood (2026-07-29)
|
||||
|
||||
Run against `bridged` on `127.0.0.1:8766` at main `19cdf8d`, with opencode **1.18.5** installed
|
||||
Run against `fleetd` on `127.0.0.1:8766` at main `19cdf8d`, with opencode **1.18.5** installed
|
||||
via Homebrew. Every §5 risk is now a verified fact rather than an assumption.
|
||||
|
||||
**The version-drift risk was the real one, and it did not bite.** This adapter was designed against
|
||||
@@ -281,7 +281,7 @@ as a dogfood-verified fact for 1.18.5.
|
||||
| §5 risk | Result |
|
||||
|---|---|
|
||||
| opencode TUI ⇄ herdr injection; CB-306 gate | ✅ `peer pane=wD:p3 reached injectable state` ~0.6s after `agent.start` |
|
||||
| Bridge MCP visible + `bridge_reply` callable | ✅ MCP `initialize` from `Implementation[name=opencode, version=1.18.5]`; worker replied through the tool |
|
||||
| Bridge MCP visible + `fleet_reply` callable | ✅ MCP `initialize` from `Implementation[name=opencode, version=1.18.5]`; worker replied through the tool |
|
||||
| Config schema drift (1.1.31 → 1.18.5) | ✅ unchanged, see above |
|
||||
| Provider credentials | ✅ free tier, zero credentials |
|
||||
|
||||
@@ -291,7 +291,7 @@ Full lifecycle exercised through the REST surface:
|
||||
to `OpenCodeLauncher` (`spawning opencode profile=opencode-free`), pane `wD:p3`.
|
||||
2. Readiness: `{"ready":true,"status":"idle"}`, roster state `ready`.
|
||||
3. `POST /sessions/{id}/message` → **`{"replySource":"reply","reply":"391"}`** — a *structured*
|
||||
`bridge_reply`, not the CB-115 completion-fallback transcript scrape. The clean path.
|
||||
`fleet_reply`, not the CB-115 completion-fallback transcript scrape. The clean path.
|
||||
4. `DELETE /workers/wD:p3` → `204`, roster empty, tolerant teardown (`tab_not_found` ignored —
|
||||
opencode had already closed its own tab).
|
||||
|
||||
@@ -301,5 +301,5 @@ identity (loopback peer PID → herdr pane) classified an **opencode** process a
|
||||
opencode-specific handling — confirming the identity model is peer-kind-agnostic, which is exactly
|
||||
what CB-308 needs when it stretches the roster across hosts.
|
||||
|
||||
CB-502 counters for the same run: `bridged_sends_total{outcome="replied"} 1`,
|
||||
`bridged_replies_total{path="rendezvous"} 1`, `bridged_inbox_depth{...} 0`.
|
||||
CB-502 counters for the same run: `fleet_sends_total{outcome="replied"} 1`,
|
||||
`fleet_replies_total{path="rendezvous"} 1`, `fleet_inbox_depth{...} 0`.
|
||||
|
||||
@@ -38,14 +38,14 @@ never toolchain ownership (§7).
|
||||
flowchart TB
|
||||
human["human (types)"]
|
||||
primary["PRIMARY (Opus)<br/>MCP client — pull-only"]
|
||||
daemon["bridged daemon<br/>127.0.0.1:8765 (single host)"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765 (single host)"]
|
||||
comp["CompositePeerLauncher<br/>routes by kind"]
|
||||
cc["ClaudeCodeLauncher"]
|
||||
oc["OpenCodeLauncher"]
|
||||
w1["worker pane (gx00 vLLM)"]
|
||||
w2["worker pane (ollama)"]
|
||||
human --> primary
|
||||
primary -->|"bridge_send / spawn / ask"| daemon
|
||||
primary -->|"fleet_send / spawn / ask"| daemon
|
||||
daemon --> comp
|
||||
comp --> cc
|
||||
comp --> oc
|
||||
@@ -80,7 +80,7 @@ flowchart TB
|
||||
m1["main A: Opus<br/>MCP client"]
|
||||
m2["main B: cloud module<br/>MCP client"]
|
||||
end
|
||||
subgraph bus["bridged fabric (CB-307/308 substrate)"]
|
||||
subgraph bus["fleetd fabric (CB-307/308 substrate)"]
|
||||
chan["per-agent inbox channels<br/>agent.<globalId>.inbox"]
|
||||
roster["federated roster (union view)"]
|
||||
end
|
||||
@@ -146,11 +146,11 @@ env-manager" rule (§7).*
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant M as main (delegator)
|
||||
participant D as bridged
|
||||
participant D as fleetd
|
||||
participant SL as SandboxLauncher
|
||||
participant SB as sandbox (peer-owned)
|
||||
participant A as agent in sandbox
|
||||
M->>D: bridge_spawn(profile=backend, role=backend)
|
||||
M->>D: fleet_spawn(profile=backend, role=backend)
|
||||
D->>SL: spawn(SpawnRequest)
|
||||
SL->>SB: start entrypoint (image = backend role)
|
||||
Note over SL,SB: bridge injects guarded ANTHROPIC_BASE_URL,<br/>mounts bridge MCP url + reply charter
|
||||
@@ -189,7 +189,7 @@ flowchart TB
|
||||
m2["main B: cloud module<br/>MCP client (pull-only)"]
|
||||
end
|
||||
reg["PrimaryRegistry → multi-slot<br/>(terminal per main)"]
|
||||
subgraph fabric["bridged"]
|
||||
subgraph fabric["fleetd"]
|
||||
ca["agent.A.inbox"]
|
||||
cb["agent.B.inbox"]
|
||||
push["ReplyPushLoop → N terminals"]
|
||||
@@ -211,14 +211,14 @@ transport is already peer-neutral — what was missing is N pull endpoints).*
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MA as main A (Opus)
|
||||
participant BR as bridged / broker
|
||||
participant BR as fleetd / broker
|
||||
participant MB as main B (cloud)
|
||||
MA->>BR: bridge_send(to = main B, msg)
|
||||
MA->>BR: fleet_send(to = main B, msg)
|
||||
BR->>BR: publish agent.B.inbox (durable, msg id)
|
||||
Note over BR: held until B pulls (B is a client too)
|
||||
MB->>BR: blocking bridge_send / poll resolves
|
||||
MB->>BR: blocking fleet_send / poll resolves
|
||||
BR-->>MB: msg (then ACK)
|
||||
MB->>BR: bridge_reply(to = main A)
|
||||
MB->>BR: fleet_reply(to = main A)
|
||||
BR->>BR: publish agent.A.inbox
|
||||
MA->>BR: poll resolves
|
||||
BR-->>MA: reply
|
||||
@@ -233,7 +233,7 @@ push-loop fan-out; relax "orchestration tools only the primary calls" to "any re
|
||||
## 6. Development C — Orchestrator tier
|
||||
|
||||
> **SUPERSEDED — do not implement this model.** The operator rejected a supervisor above the lead.
|
||||
> The human continues to drive the pre-existing lead directly; bridged neither spawns nor resumes that
|
||||
> The human continues to drive the pre-existing lead directly; fleetd neither spawns nor resumes that
|
||||
> lead. What replaced this proposal is **one lead, two short-lived advisory architects, and N workers**:
|
||||
> the lead engages architects sideways for a strong-model assessment, then discards them. Architect
|
||||
> slots are declared in `architects:` (see Gitea issue #16), rather than making leads managed sessions.
|
||||
@@ -243,7 +243,7 @@ push-loop fan-out; relax "orchestration tools only the primary calls" to "any re
|
||||
|
||||
> **Historical alternative retained.** The text and figures below record the considered model and why it
|
||||
> was rejected: it re-rooted the human-facing session above the lead, violating the still-true premise
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by bridged.
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by fleetd.
|
||||
|
||||
The orchestrator is **`SessionManager` recursed one tier up**: today it spawns/names/reaps *worker*
|
||||
sessions; the orchestrator does the same for *main* sessions, and adds **context scoping**.
|
||||
@@ -287,7 +287,7 @@ sequenceDiagram
|
||||
H->>O: high-level goal (large context)
|
||||
O->>O: name/resume main A session
|
||||
O->>MA: task + SCOPED context slice (not the whole history)
|
||||
MA->>W: bridge_send(delegation, carrying only the relevant slice)
|
||||
MA->>W: fleet_send(delegation, carrying only the relevant slice)
|
||||
W-->>MA: result
|
||||
MA-->>O: rollup
|
||||
O->>O: fold into orchestrator context, pick next main/turn
|
||||
@@ -385,7 +385,7 @@ ticket is an extension of an existing pattern (CB-402 for A, CB-308 for B/C).
|
||||
|
||||
The follow-up question — *"clarify the architecture when we have distributed agents in sandboxes"* —
|
||||
resolves the fork left open in §4 and §9. **Decision: Development A (sandbox launcher) and CB-308
|
||||
(per-host federation) *compose*, not compete — each host runs a `bridged` gateway whose launcher
|
||||
(per-host federation) *compose*, not compete — each host runs a `fleetd` gateway whose launcher
|
||||
spawns agents into that host's *local* sandboxes.** A sandbox is never reached across the network; it
|
||||
is reached by the gateway sitting next to it.
|
||||
|
||||
@@ -393,7 +393,7 @@ is reached by the gateway sitting next to it.
|
||||
|
||||
The bus delivers a turn by **herdr keystroke-injection** — `Injector → AgentControl.send` writes into
|
||||
a PTY that its **local** herdr owns. The broker moves *messages and presence*, **never keystrokes**.
|
||||
So an agent's PTY must live in a herdr that *some* `bridged` instance drives locally: a remote
|
||||
So an agent's PTY must live in a herdr that *some* `fleetd` instance drives locally: a remote
|
||||
container with no local herdr **cannot be injected into**. That rules out a central daemon reaching
|
||||
remote PTYs, and collapses the design to a single identity:
|
||||
|
||||
@@ -403,7 +403,7 @@ remote PTYs, and collapses the design to a single identity:
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A — gateway"]
|
||||
mA["main / orchestrator<br/>MCP client → LOCAL gateway"]
|
||||
gA["bridged A<br/>herdr + CompositePeerLauncher<br/>(incl. SandboxLauncher)"]
|
||||
gA["fleetd A<br/>herdr + CompositePeerLauncher<br/>(incl. SandboxLauncher)"]
|
||||
cBEa["sandbox: backend<br/>(local container)"]
|
||||
cFEa["sandbox: frontend<br/>(local container)"]
|
||||
mA --- gA
|
||||
@@ -415,7 +415,7 @@ flowchart TB
|
||||
roster["roster.* (federated presence)"]
|
||||
end
|
||||
subgraph hostB["HOST B — gateway"]
|
||||
gB["bridged B<br/>herdr + SandboxLauncher"]
|
||||
gB["fleetd B<br/>herdr + SandboxLauncher"]
|
||||
cBEb["sandbox: backend<br/>(local container)"]
|
||||
gB -->|"spawn → PTY in B's herdr"| cBEb
|
||||
end
|
||||
@@ -440,13 +440,13 @@ sequenceDiagram
|
||||
participant BR as broker
|
||||
participant GB as gateway B
|
||||
participant SB as sandbox agent (host B, container)
|
||||
MA->>GA: bridge_send(globalId on B, msg)
|
||||
MA->>GA: fleet_send(globalId on B, msg)
|
||||
GA->>GA: directory lookup - is globalId local? NO
|
||||
GA->>BR: publish agent.ID.inbox (durable)
|
||||
BR->>GB: route to the owning gateway
|
||||
GB->>SB: inject via B's LOCAL herdr (keystrokes)
|
||||
Note over GB,SB: SandboxLauncher already spawned the container -<br/>its PTY is in B's herdr, CB-306 readiness passed
|
||||
SB-->>GB: bridge_reply (to B's LOCAL MCP endpoint)
|
||||
SB-->>GB: fleet_reply (to B's LOCAL MCP endpoint)
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route back to A
|
||||
Note over GA: held until the main pulls (the main is a client)
|
||||
@@ -470,7 +470,7 @@ unchanged — the sandbox is transparent to it.*
|
||||
|
||||
| Concern | Provided by |
|
||||
|---|---|
|
||||
| Per-host gateway (owns local herdr + sessions) | **CB-308** (today's `bridged`, evolved) |
|
||||
| Per-host gateway (owns local herdr + sessions) | **CB-308** (today's `fleetd`, evolved) |
|
||||
| Spawn into a local sandbox / role→image | **Development A** `SandboxLauncher` (§4), routed by `CompositePeerLauncher` |
|
||||
| Addressing a remote sandboxed agent | **CB-308** global id + federated roster (host + role as metadata) |
|
||||
| Orphan reap after a gateway restart | **CB-117** per-gateway, summed by the composite — each reaps only its **local** herdr |
|
||||
|
||||
@@ -0,0 +1,467 @@
|
||||
# CB-591 — move the fleet onto the LLM and MCP gateway
|
||||
|
||||
**Status: DONE — the fleet is on the gateway as of 2026-08-15.** `local` runs on `/anthropic` and
|
||||
`gx` on `/v1`, both at `weight: 100`; `local-direct` stays at `weight: 0` as the escape hatch. Getting
|
||||
here took a revert and two upstream fixes — see §7.1, which is the useful part of this document. One
|
||||
risk is **accepted rather than solved**: a stream cut by any mid-response timer arrives as HTTP 200
|
||||
with no terminator, and our third-party members cannot detect it (§7.2).
|
||||
· **Upstream:** [systems/vms wiki → LLM and MCP Gateway](https://git.ltms.dev/systems/vms/wiki/LLM-and-MCP-Gateway)
|
||||
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
||||
|
||||
The gateway went live on 2026-08-15 and replaced Bifrost. This plan says what that means for a
|
||||
**member definition** in `fleetd.yaml`, because that is the part of this repo the change actually
|
||||
touches.
|
||||
|
||||
---
|
||||
|
||||
## 1. What changed upstream
|
||||
|
||||
One front door for every LLM and MCP client: `https://llm.ltms.dev`, one token per consumer.
|
||||
|
||||
| Surface | URL |
|
||||
|---|---|
|
||||
| OpenAI chat | `https://llm.ltms.dev/v1/chat/completions` |
|
||||
| OpenAI models | `https://llm.ltms.dev/v1/models` |
|
||||
| **Anthropic messages** | `https://llm.ltms.dev/anthropic/v1/messages` |
|
||||
| MCP, all servers multiplexed | `https://llm.ltms.dev/mcp` |
|
||||
|
||||
Anything outside that list returns **404 before any token is checked**, on purpose — the gateway must
|
||||
never become a blanket proxy.
|
||||
|
||||
The model backend is unchanged: GX10 vLLM at `10.10.10.26:8000` (`gx00.gw`), model name exactly
|
||||
`deepseek-v4-flash`. The direct LAN path stays open on purpose as an escape hatch.
|
||||
|
||||
---
|
||||
|
||||
## 2. Where claude-bridge sits today
|
||||
|
||||
We do **not** use the gateway. The `local` profile talks straight to the vLLM:
|
||||
|
||||
```yaml
|
||||
local:
|
||||
kind: claude-code
|
||||
baseUrl: http://gx00.gw:8000 # direct vLLM — no auth, LAN only
|
||||
model: deepseek-v4-flash
|
||||
configDir: /Users/dai.ha/.ccs/instances/gx10
|
||||
```
|
||||
|
||||
Three facts about our side that decide the shape of this work:
|
||||
|
||||
1. **`baseUrl` becomes `ANTHROPIC_BASE_URL`** in the member's environment, and `tokenEnv` becomes
|
||||
`ANTHROPIC_AUTH_TOKEN` (the value is read from a host env var and never stored in config).
|
||||
`local` sets no `tokenEnv` today, because a direct vLLM needs no token.
|
||||
2. **`SubscriptionGuard` refuses any host not on an allowlist**, and that allowlist is
|
||||
`guard.offSubscriptionHosts: [gx00.gw]`. It is built once in `Fleetd.java:93` and handed to the
|
||||
launcher, so **it is a restart-required key**, not a hot one. Changing `baseUrl` without changing
|
||||
this makes every `local` spawn throw.
|
||||
3. **The wiki names us as a blocker.** Under *Not done yet*: retiring the shared `legacy` token is
|
||||
blocked because "kb, brain, **claude-bridge** and the workstation still share it. Each needs its
|
||||
own consumer first."
|
||||
|
||||
Context7 is mounted twice today, both times straight at `https://ct7.ltms.dev/mcp` — once in
|
||||
`.mcp.json` (the primary) and once in `opencode.json` (the `sol` and `terra` members).
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph now["Today"]
|
||||
M1["local member<br/>claude-code"] -->|"ANTHROPIC_BASE_URL"| V1["vLLM gx00.gw:8000<br/>no auth, LAN only"]
|
||||
M2["sol / terra<br/>opencode"] --> CT1["ct7.ltms.dev/mcp"]
|
||||
P1["primary"] --> CT1
|
||||
end
|
||||
subgraph after["Proposed"]
|
||||
M3["local member"] -->|"ANTHROPIC_BASE_URL<br/>+ ANTHROPIC_AUTH_TOKEN"| G["llm.ltms.dev/anthropic<br/>consumer: claude-bridge"]
|
||||
G --> V2["vLLM gx00.gw:8000"]
|
||||
M4["local-direct<br/>weight 0, escape hatch"] --> V2
|
||||
end
|
||||
```
|
||||
|
||||
*The member definition is the only thing that moves. The model behind it does not.*
|
||||
|
||||
---
|
||||
|
||||
## 3. The member definition change
|
||||
|
||||
The gateway serves an Anthropic surface *and* an OpenAI surface, so **both member kinds can point at
|
||||
it**. That is the main opportunity here, and it is bigger than the `local` profile alone.
|
||||
|
||||
### 3a. `local` — claude-code, on `/anthropic`
|
||||
|
||||
| Key | Today | After | Note |
|
||||
|---|---|---|---|
|
||||
| `baseUrl` | `http://gx00.gw:8000` | `https://llm.ltms.dev/anthropic` | see the schema warning below |
|
||||
| `tokenEnv` | *(unset)* | `AI_GATEWAY_TOKEN` | new consumer token, `llmk-claude-bridge-<32 hex>` |
|
||||
| `model` | `deepseek-v4-flash` | unchanged | must stay **exact**; a regex match returns an empty `/v1/models` while completions keep working |
|
||||
| `guard.offSubscriptionHosts` | `[gx00.gw]` | `[gx00.gw, llm.ltms.dev]` | **restart required** |
|
||||
|
||||
### 3b. A new opencode profile on `/v1` — no code needed
|
||||
|
||||
`OpenCodeLauncher` already supports a pinned OpenAI-compatible endpoint (CB-508). Given `baseUrl` it
|
||||
writes a custom provider block into the worker's opencode config:
|
||||
|
||||
- `baseUrl` → `options.baseURL`. `openAiBaseUrl` appends `/v1` to a bare host, and takes a URL that
|
||||
already has a path **as-is** — so `https://llm.ltms.dev/v1` works unchanged.
|
||||
- `tokenEnv` → `options.apiKey` (falls back to a placeholder when unset, since a local vLLM ignores it).
|
||||
- `model:` **must** be `<provider>/<model>` when `baseUrl` is set — a bare name is rejected loudly
|
||||
rather than silently falling back to opencode's default gateway.
|
||||
|
||||
So the profile is pure config:
|
||||
|
||||
```yaml
|
||||
gx:
|
||||
kind: opencode
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
model: gx/deepseek-v4-flash # provider id is ours to choose; the half after / is the model
|
||||
argv: ["opencode"]
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
weight: 100 # same tier as `local` — free
|
||||
maxLoad: 2
|
||||
# deliberately NO credentialId — this is our own box, not the shared OpenAI account
|
||||
```
|
||||
|
||||
**Why this matters more than it looks.** Today every opencode member is `sol` or `terra`, and those
|
||||
are two models on **one** OpenAI account sharing `credentialId: openai-shared` — so an exhaustion on
|
||||
either locks out both, and half the fleet's opencode capacity dies at once. A gateway-backed opencode
|
||||
profile is free, is not on that credential, and therefore is not in that quarantine pair. It removes
|
||||
a single point of failure rather than just adding capacity.
|
||||
|
||||
**Note the asymmetry, it is deliberate:** `SubscriptionGuard` does not apply to opencode at all — the
|
||||
guard exists to stop a *Claude* worker borrowing the operator's subscription, and opencode reads its
|
||||
own provider credentials. So 3b needs **no allowlist change**; only 3a does.
|
||||
|
||||
**Both still need a restart, for a different reason.** `tokenEnv` is resolved by
|
||||
`HerdrPeerLauncher.resolveEnv` → `env.apply(name)`, which reads the **daemon's own process
|
||||
environment**. The running `fleetd` inherited its environment when it started, so a variable added to
|
||||
`secrets.sh` afterwards is simply not there — the launcher would inject an empty token and the
|
||||
gateway would answer 401. This is the same failure as trap 1 in `scripts/redeploy-fleetd.sh`
|
||||
(`WORKER_GITEA_TOKEN`), and it has the same fix: **restart from a login shell**, and use
|
||||
`scripts/redeploy-fleetd.sh --check` to confirm the name resolves before restarting anything.
|
||||
|
||||
### 3c. What this does to `ccs`
|
||||
|
||||
Once a profile carries `baseUrl`, `tokenEnv` and `model` itself, the ccs instance stops being what
|
||||
routes a member. Be precise about what is left, though: `configDir` still supplies **folder trust**
|
||||
and `settings.json`, and dropping it is what produced the trust dialog and the wrong-model error
|
||||
recorded in `fleetd.yaml`. So ccs goes from *deciding where the tokens go* to *holding client-side
|
||||
state*. Less load-bearing, not removable.
|
||||
|
||||
### Why `/anthropic` and never `/v1/chat/completions`
|
||||
|
||||
The gateway declares its Anthropic backend as `schema.name: Anthropic`, which means **no
|
||||
translation** — streaming, tool use and thinking blocks pass through exactly as they do against vLLM
|
||||
directly.
|
||||
|
||||
Declared as `OpenAI`, Envoy's translator looks for a `thinking_blocks` field that our vLLM does not
|
||||
send (it sends `reasoning_content`), and **every thinking delta disappears silently**. Claude Code
|
||||
speaks the Anthropic protocol, so `/anthropic` is both correct and the only safe choice.
|
||||
|
||||
This is the exact failure shape this repo keeps hitting: it compiles, it answers, it looks healthy,
|
||||
and a capability is quietly off. Treat it as a `silent-default` risk, not a config preference.
|
||||
|
||||
**Open question for 3b — ANSWERED, 2026-08-15.** The worry was that the OpenAI surface might drop
|
||||
reasoning the way the wiki documents for a mis-declared Anthropic backend. It does not. Checked at
|
||||
the API before any profile was switched:
|
||||
|
||||
| surface | request | result |
|
||||
|---|---|---|
|
||||
| `/anthropic/v1/messages` | `deepseek-v4-flash`, 64 tokens | 200, response carries a real `"type":"thinking"` block |
|
||||
| `/v1/chat/completions` | same | 200, message carries a populated `reasoning_content` (and a `reasoning` field) |
|
||||
| `/v1/models` | — | 200, exactly `["deepseek-v4-flash"]` — the exact-name trap is clear |
|
||||
| `/v1/models`, **no token** | — | **401** — Caddy is gating, as designed |
|
||||
|
||||
So reasoning survives on **both** surfaces, and the `/anthropic` choice for `local` is about protocol
|
||||
correctness rather than a repair for a known loss. The last row matters on its own: the wiki warns
|
||||
the gateway's own `SecurityPolicy` fails open, so it is worth knowing the proxy in front really does
|
||||
refuse an unauthenticated request here.
|
||||
|
||||
---
|
||||
|
||||
## 4. Decisions
|
||||
|
||||
### D1 — switch, but keep the direct path as an explicit profile · **recommended**
|
||||
|
||||
Switching buys four things we do not have:
|
||||
|
||||
- **Free opencode capacity, off the shared credential.** The largest single win. See §3b — it retires
|
||||
a real single point of failure, not just a cost line.
|
||||
- **Per-consumer usage figures.** The cockpit counts requests per consumer. That is the first real
|
||||
measurement of what the fleet consumes, and it feeds [CB-589](https://git.ltms.dev/fleet/fleetd/issues/74) Gap 2 directly.
|
||||
- **Our own revocable token.** One consumer to revoke if a worker ever leaks it, instead of a shared
|
||||
`legacy` token used by four systems.
|
||||
- **It works off-LAN.** `gx00.gw` resolves on the LAN only.
|
||||
|
||||
The cost is honest and worth stating: we add a TLS edge, an auth proxy and a gateway to the path of
|
||||
every member spawn. The wiki keeps the direct route open precisely because "if the gateway breaks,
|
||||
nothing that matters is blocked."
|
||||
|
||||
So keep it. Add a second profile `local-direct` pointing at `http://gx00.gw:8000` with **`weight: 0`**
|
||||
— never auto-selected, still spawnable with an explicit `fleet_spawn{profile: "local-direct"}`.
|
||||
That is exactly what CB-554 made `weight: 0` mean, and it turns the escape hatch into something the
|
||||
lead can actually reach during an incident.
|
||||
|
||||
### D2 — do members also mount the gateway's `/mcp`? · **OPEN, operator's call**
|
||||
|
||||
Not a detail. `CLAUDE.md` states in two places that a member mounts **only** the bridge MCP, and a
|
||||
worker's honesty rule leans on it ("never claim the result of a check you had no way to run").
|
||||
|
||||
- **Keep bridge-only.** The invariant stays true and simple. Workers stay cheap and narrow.
|
||||
- **Add the gateway MCP.** Implementers get context7 documentation lookups, which is genuinely useful
|
||||
for library work. But `mcpUrl` in `FleetConfig.Profile` is a **single `String`**, so a
|
||||
claude-code member can mount exactly one MCP — this needs a code change, not a config edit.
|
||||
|
||||
Note the invariant is **already inaccurate**: `opencode.json` gives `sol` and `terra` both context7
|
||||
and gitea. So the choice is really "make the rule true" or "make the rule match reality". Either is
|
||||
defensible; picking one is not mine to do.
|
||||
|
||||
### D3 — token scope
|
||||
|
||||
One consumer, `claude-bridge`, its token in `${SHARED_ENV}/tools/secrets.sh` as `AI_GATEWAY_TOKEN`,
|
||||
referenced by name only. Never the literal value in `fleetd.yaml` — `tokenEnv` exists for this.
|
||||
|
||||
---
|
||||
|
||||
## 5. Units of work
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
U1["U1 · consumer token<br/>issue via cockpit, add to secrets.sh"]
|
||||
U2["U2 · profile + guard<br/>fleetd.yaml, restart"]
|
||||
U3["U3 · verify live<br/>spawn, prove thinking survives"]
|
||||
U4["U4 · context7 via gateway<br/>.mcp.json + opencode.json"]
|
||||
U5["U5 · docs<br/>CLAUDE.md, wiki 11-Features"]
|
||||
U1 --> U2 --> U3
|
||||
U4 --> U5
|
||||
U3 --> U5
|
||||
```
|
||||
|
||||
| # | Scope | Who | Why |
|
||||
|---|---|---|---|
|
||||
| U1 | Issue the `claude-bridge` consumer at `auth.ltms.dev`; store as `AI_GATEWAY_TOKEN` | **operator** | touches secrets and a host we do not own |
|
||||
| U2a | New `gx` opencode profile on `/v1` — pure config, no guard change | **lead** | `fleetd.yaml` is gitignored, so a worker cannot see or edit it |
|
||||
| U2b | `local` → `/anthropic`; add `local-direct` weight 0; add `llm.ltms.dev` to the guard allowlist | **lead** | same |
|
||||
| U2c | One restart from a **login shell**, after U2a and U2b | **lead** | picks up `AI_GATEWAY_TOKEN` into the daemon env *and* the guard allowlist, in one stop |
|
||||
| U3 | Live spawn on both new profiles; confirm reasoning survives on each surface | **lead** | needs real spawns and the running daemon |
|
||||
| U4 | Point `.mcp.json` and `opencode.json` context7 at the gateway `/mcp`; rename pinned tools | delegatable | tracked files, self-contained |
|
||||
| U5 | Fix the "members mount only the bridge" claim; add a `wiki/11-Features.md` entry | delegatable | writing, clear criteria |
|
||||
|
||||
U1 blocks U2a, U2b and U3. U4 and U5 do not depend on it.
|
||||
|
||||
**Write U2a and U2b, then restart once (U2c), then verify `gx` before `local`.** Since both profiles
|
||||
need the same restart there is no reason to do two, but there is still a reason to *verify* in order:
|
||||
`gx` exercises the token and the gateway with no guard involved, so if it fails the cause is upstream.
|
||||
`local` adds the guard allowlist on top, so a failure there points at our config instead. Testing them
|
||||
in that order separates the two causes instead of confusing them.
|
||||
|
||||
> **U1 status, 2026-08-15:** the operator issued the consumer and exported it as `AI_GATEWAY_TOKEN`
|
||||
> (one key for every agent and MCP client behind `llm.ltms.dev`). Confirmed: it resolves in a login
|
||||
> shell, is 48 characters and carries the documented `llmk-` prefix. The value was never printed.
|
||||
|
||||
---
|
||||
|
||||
## 6. Traps carried over from the wiki
|
||||
|
||||
Each of these cost someone real debugging time upstream. They apply to us.
|
||||
|
||||
1. **Rotating a token restarts the auth proxy, which drops in-flight streaming responses.** For us
|
||||
that means rotating `AI_GATEWAY_TOKEN` kills every live member mid-turn, and an async ticket's
|
||||
report goes with it. This is the same rule as a daemon redeploy: **drain the fleet first**
|
||||
(`fleet_list` → `fleet_poll` anything wanted → `fleet_stop`), then rotate.
|
||||
2. **The gateway's own `SecurityPolicy` fails open.** Standalone `aigw run` accepts it and silently
|
||||
ignores it — an unauthenticated request returned **200**. Auth is the Caddy proxy in front, and
|
||||
nothing else. Never reason as if the gateway authenticates.
|
||||
3. **Exact model name.** A regex match routes fine but returns an **empty** `/v1/models` list while
|
||||
completions keep working. A wrong name returns a bare 404 that reads exactly like a dead gateway.
|
||||
4. **MCP tool names changed prefix separator.** Bifrost used one dash (`ct7-resolve-library-id`); the
|
||||
gateway uses **two underscores** (`ct7__resolve-library-id`). Relevant only if U4 is done.
|
||||
5. **`/v1/models` 404 vs empty list are different faults.** 404 means no route loaded at all; empty
|
||||
means the model match is a regex. Do not conflate them when diagnosing.
|
||||
|
||||
---
|
||||
|
||||
## 7. Verification — what would prove this works
|
||||
|
||||
Merging config is not proving it. The checks, in order:
|
||||
|
||||
1. `fleet_spawn{profile: "gx"}` succeeds and the member completes a real turn ending in
|
||||
`fleet_reply`. This is the first proof of the token, the URL and the model name, and it risks
|
||||
nothing the fleet depends on.
|
||||
2. `fleet_spawn{profile: "local"}` succeeds. If the guard allowlist was missed, this **throws** — a
|
||||
loud, self-correcting failure, which is the good kind. If the restart was missed, it also throws,
|
||||
for the same reason.
|
||||
3. A `local` member completes a turn. That exercises streaming through two TLS edges, the auth proxy
|
||||
and the gateway.
|
||||
4. **Reasoning survives, checked separately on each surface.** For `local` on `/anthropic` this is
|
||||
the check that catches the `/v1` versus `/anthropic` mistake, and it is the only one that does —
|
||||
nothing else distinguishes a working passthrough from a translator quietly dropping thinking
|
||||
deltas. For `gx` on `/v1`, this answers the open question in §3 rather than assuming it.
|
||||
5. The cockpit at `auth.ltms.dev` shows requests counted against the `claude-bridge` consumer, not
|
||||
`legacy`. That is the whole point of taking our own token.
|
||||
6. `fleet_spawn{profile: "local-direct"}` still works, so the escape hatch is real rather than
|
||||
theoretical.
|
||||
7. `fleet_list` shows `gx` carrying no `credentialId`, so a `sol`/`terra` exhaustion cannot
|
||||
quarantine it. This is the single-point-of-failure claim in §3b, checked rather than asserted.
|
||||
|
||||
---
|
||||
|
||||
## 7.1 What the live run actually found — 2026-08-15
|
||||
|
||||
U1–U2c were done, the daemon restarted onto them, and both new profiles were spawned for real. The
|
||||
migration was then **reverted**. This section is the result, so none of it has to be re-derived.
|
||||
|
||||
### The blocker
|
||||
|
||||
`llm.ltms.dev` answers **HTTP 413 Request Entity Too Large** above **32 KiB (32768 bytes)**, on both
|
||||
surfaces:
|
||||
|
||||
```
|
||||
/v1 32695 bytes -> 200 /anthropic 32095 bytes -> 200
|
||||
/v1 32795 bytes -> 413 /anthropic 32855 bytes -> 413
|
||||
```
|
||||
|
||||
32 KiB is far below one real agent turn.
|
||||
|
||||
**Root cause — confirmed by the systems/vms side, 2026-08-15.** My guess that it was a Caddy
|
||||
`request_body max_size` was **wrong**. It is Envoy, inside `aigw` on `llm.vm`. Envoy Gateway defaults
|
||||
a listener's `per_connection_buffer_limit_bytes` to **32768**, and the AI Gateway buffers the *whole*
|
||||
request body before it can route on the model name — so that default is not a network tuning knob
|
||||
here, it is a hard ceiling on prompt size. Read out of the live Envoy `config_dump`:
|
||||
|
||||
```
|
||||
listener default/llm/http per_connection_buffer_limit_bytes: 32768
|
||||
```
|
||||
|
||||
Nobody chose 32 KiB; it was inherited from the default. Both TLS edges are innocent: the same
|
||||
boundary reproduces on the LAN path and the internet path, and both 413s carry an `x-llm-consumer`
|
||||
header their auth proxy sets only *after* authenticating — so the body cleared both edges and the
|
||||
auth. Directly on `llm.vm`, `aigw` 413s at 39 KB while the vLLM backend accepts the same 39 KB and
|
||||
answers 200.
|
||||
|
||||
**Do not plan around 32 KiB.** The intended ceiling is far higher. Their fix — a `ClientTrafficPolicy`
|
||||
setting `bufferLimit: 8Mi` — is written but **not deployed** as of this note, pending their operator's
|
||||
approval. I have not re-tested and will not until they confirm, so as not to measure a half-changed
|
||||
system. Fixed in **systems/vms**, not here.
|
||||
|
||||
### The part worth remembering
|
||||
|
||||
Two members were spawned at the same moment with the same message:
|
||||
|
||||
| | `local` (claude-code, `/anthropic`) | `gx` (opencode, `/v1`) |
|
||||
|---|---|---|
|
||||
| READY → BUSY | 19:07:26 | 19:07:45 |
|
||||
| BUSY → DONE | **19:08:51 (66s)** | **never — 10+ min, ticket FAILED** |
|
||||
|
||||
**`local` passed.** It passed only because the probe was three trivial questions in a fresh session,
|
||||
so the request fit under 32 KiB. The profile looked healthy and was a landmine set to fire on the
|
||||
first turn that reads a file.
|
||||
|
||||
So §7's checklist was not wrong, it was **too easy**. Any future run of it must use a task that reads
|
||||
a real file. A liveness probe proves the token and the URL; it does not prove the path.
|
||||
|
||||
`gx` did not fail loudly either. Reproduced outside the bridge by running `opencode` by hand with the
|
||||
launcher's own generated config:
|
||||
|
||||
```
|
||||
Error: Request Entity Too Large
|
||||
...compacts context, retries...
|
||||
Error: Request Entity Too Large
|
||||
```
|
||||
|
||||
opencode **catches the 413, compacts, and retries — indefinitely**. A member that fails loudly costs
|
||||
one turn; this one costs the whole task and is indistinguishable from a slow worker.
|
||||
|
||||
> **Diagnosing a stuck opencode member.** Do not read its pane. The launcher writes its config to a
|
||||
> temp dir and passes it as `OPENCODE_CONFIG` — find it with
|
||||
> `ls -dt /var/folders/*/*/T/fleetd-opencode-* | head -1`, check the provider block and the key's
|
||||
> length and prefix (never its value), then reproduce with `opencode run --auto -m <provider>/<model>`
|
||||
> using the same `OPENCODE_CONFIG`. That is what turned "it hangs" into a one-line error.
|
||||
|
||||
### What checked out, and needs no re-testing
|
||||
|
||||
- Token accepted on both surfaces. **Unauthenticated → 401**, so the Caddy proxy really does gate —
|
||||
the wiki's "SecurityPolicy fails open" warning is about the gateway itself, not the edge.
|
||||
- `/v1/models` returns exactly `["deepseek-v4-flash"]`, so trap 3 is clear.
|
||||
- **Reasoning survives both surfaces** — see §3b above.
|
||||
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
||||
key rather than the `fleetd-local-noauth` placeholder.
|
||||
- `SubscriptionGuard` accepted `llm.ltms.dev` after the allowlist edit and the restart: `local`
|
||||
spawned without throwing, which is the check that catches a missed restart.
|
||||
|
||||
### Resolution — both ceilings fixed, migration completed
|
||||
|
||||
systems/vms fixed both, and each was re-checked from this side rather than taken on trust:
|
||||
|
||||
| ceiling | was | now | our own check |
|
||||
|---|---|---|---|
|
||||
| listener buffer | 32 KiB | 32 Mi | 1.2 MB body → **200** (was 413) |
|
||||
| LLM route timeout | 60s | 86400s | the request that truncated: **101s, `message_stop` present, 4000/4000** |
|
||||
|
||||
The timeout moved in two steps on 2026-08-15: 60s → 1800s, then 1800s → **86400s (24 hours)** after
|
||||
the truncation risk below was discussed. They tried `request: 0s` first, which removes the
|
||||
total-duration timer completely. It works, but on an `AIGatewayRoute` the **idle timeout is derived
|
||||
from the request timeout**, so `0s` also removed any bound on a stalled connection. 86400s keeps a
|
||||
reaper for dead connections while putting the truncation timer out of practical reach.
|
||||
|
||||
Neither was deliberate. The 32 KiB was Envoy Gateway's default `per_connection_buffer_limit_bytes`;
|
||||
the 60s was Envoy AI Gateway's own documented default. The 60s bounded **generation** as well as
|
||||
prompt size — a tiny prompt with a long answer returned 504 at 60.05s.
|
||||
|
||||
Two configuration facts worth keeping, from their bisection:
|
||||
|
||||
- **`ClientTrafficPolicy` is honoured in standalone `aigw run`; `BackendTrafficPolicy` is NOT.** A
|
||||
`BackendTrafficPolicy` setting `requestTimeout` is accepted, logs nothing, and leaves the routes
|
||||
unchanged (upstream `envoyproxy/gateway#9513`). What works is `timeouts: {request: …}` on each
|
||||
`AIGatewayRoute` rule. Nothing from the outside distinguishes the two — the same silent-default
|
||||
shape as their `SecurityPolicy` caveat.
|
||||
- In that stack, "the config was accepted" proves nothing. Read the live `config_dump`.
|
||||
|
||||
## 7.2 The risk we accepted, and why we could not remove it
|
||||
|
||||
Raising the timeout made the failure **rare, not impossible**, and the residual failure is silent.
|
||||
|
||||
On a mid-response timeout over chunked HTTP/1.1, Envoy ends the chunked encoding *cleanly* instead of
|
||||
resetting the connection, so the client receives what looks like a complete transfer
|
||||
(`envoyproxy/envoy#17186` — acknowledged as a bug in 2021, closed by a stale bot, never fixed). The
|
||||
December 2025 fix `envoyproxy/envoy#42269` changes locally-originated resets from `NO_ERROR` to
|
||||
`INTERNAL_ERROR`, but it is **HTTP/2 only** and SSE clients here speak HTTP/1.1.
|
||||
|
||||
Measured on our side while the timeout was still 60s:
|
||||
|
||||
```
|
||||
HTTP 200 61.07s 141992 bytes
|
||||
message_stop 0 message_delta 0 error events 0
|
||||
emitted 2473 of 4000, ending on a WELL-FORMED SSE frame
|
||||
```
|
||||
|
||||
A syntactically valid stream that simply stops. Any timer firing mid-stream — route timeout, idle
|
||||
timeout, `max_stream_duration` — fails this same way.
|
||||
|
||||
**The recommended defence does not transfer to us.** The right fix is to treat a stream with no
|
||||
`message_stop` / `[DONE]` / `finish_reason` as failed. We cannot: our members are Claude Code and
|
||||
opencode, third-party clients whose SSE parsing we do not own, and there is no seam to insert the
|
||||
check. Whether either detects a missing terminator is unverified — and opencode's handling of the 413
|
||||
(swallow, compact, retry forever, never surface an error) does not suggest it is strict.
|
||||
|
||||
So the honest statement of our position:
|
||||
|
||||
> Gateway traffic is acceptable at 86400s because a single request would have to run for 24 hours to
|
||||
> trip the bug — **not** because we could detect it if it did.
|
||||
|
||||
At 86400s our **own** limit binds first, which is the ordering we want. `MessageService.ASYNC_TIMEOUT_MS`
|
||||
caps a turn at 30 minutes, so a runaway request ends as a clean `FAILED` ticket that we raised, rather
|
||||
than as a silently truncated `200` that we cannot see. While the gateway sat at 1800s the two numbers
|
||||
were equal and did not nest, so a gateway-side stall could have been misread as a bug in our own ticket
|
||||
handling. That ambiguity is now gone.
|
||||
|
||||
**If a member ever returns a confident but truncated answer, suspect this before anything in our own
|
||||
code.** That is the whole reason this section exists.
|
||||
|
||||
---
|
||||
|
||||
## 8. Related
|
||||
|
||||
- [CB-589 / #74](https://git.ltms.dev/fleet/fleetd/issues/74) — cost-first placement and a
|
||||
gateway that reports live capacity. The per-consumer figures this migration unlocks are the first
|
||||
input that ticket actually needs.
|
||||
- `docs/CB-500-Multi-Tier-Coordination.md` §11 — the distributed-sandbox topology this gateway is
|
||||
part of.
|
||||
+18
-18
@@ -12,12 +12,12 @@ not after it.
|
||||
|
||||
## 1. Why this stage is not optional bookkeeping
|
||||
|
||||
`bridged` today has **exactly one security control: the loopback bind**. Every other guarantee
|
||||
`fleetd` today has **exactly one security control: the loopback bind**. Every other guarantee
|
||||
rests on it.
|
||||
|
||||
The identity model (`mcp/ConnectionIdentity.java`) resolves a caller from the connection alone —
|
||||
the OS reports the connecting PID, herdr owns the PID→pane map, so a worker cannot forge another
|
||||
worker. Its own javadoc is explicit: *"Single-host only (the herd shares the `bridged` host); the
|
||||
worker. Its own javadoc is explicit: *"Single-host only (the herd shares the `fleetd` host); the
|
||||
token path is the split-host fallback."* The token path does not exist yet.
|
||||
|
||||
That leaves a seam that is **latent today and load-bearing the moment the bind moves**:
|
||||
@@ -67,7 +67,7 @@ it will be disabled and the stage is wasted. So:
|
||||
auth:
|
||||
mode: loopback-trust # default — behaves exactly like today: loopback ⇒ PRIMARY, no token needed
|
||||
# mode: token # every non-worker caller must present a valid bearer token
|
||||
# tokenEnv: BRIDGED_API_TOKEN # host env var holding the token; never the literal value
|
||||
# tokenEnv: FLEETD_API_TOKEN # host env var holding the token; never the literal value
|
||||
```
|
||||
|
||||
`mode: loopback-trust` is the current behaviour, named honestly and now *chosen* rather than
|
||||
@@ -112,9 +112,9 @@ The roadmap says "systemd unit". **This host is macOS — there is no systemd on
|
||||
not found), and the daemon that has been dogfooded for weeks runs as a bare foreground
|
||||
`java -jar`. Ship **both**:
|
||||
|
||||
- `deploy/dev.ltms.bridged.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
- `deploy/dev.ltms.fleet.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
ordered start after herdr.
|
||||
- `deploy/bridged.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
- `deploy/fleetd.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
|
||||
Ordering after herdr is advisory in both: the herdr socket may not exist at boot, so the daemon
|
||||
must **retry the socket rather than exit** — supervision ordering is a nicety, socket-retry is the
|
||||
@@ -158,10 +158,10 @@ configured. It already leaks nothing but herdr's version and up/down.
|
||||
### 3.1 There are TWO entry paths, and only one of them has identity today
|
||||
|
||||
The wiki describes MCP as "a thin adapter over the REST core". **At the code level that is not
|
||||
literally true, and the difference is security-relevant.** `BridgeMcp` calls `MessageService` /
|
||||
literally true, and the difference is security-relevant.** `FleetMcp` calls `MessageService` /
|
||||
`SessionManager` *directly*; it never issues an HTTP request against a Javalin route. And `/mcp` is
|
||||
mounted as a raw servlet on Jetty's `ServletContextHandler`
|
||||
(`BridgedApp.build → cfg.jetty.modifyServletContextHandler`), so it does **not** pass through
|
||||
(`FleetApp.build → cfg.jetty.modifyServletContextHandler`), so it does **not** pass through
|
||||
Javalin's `before` filters at all.
|
||||
|
||||
The current split is the mirror image of what you'd expect:
|
||||
@@ -172,7 +172,7 @@ The current split is the mirror image of what you'd expect:
|
||||
| REST routes | ❌ **none at all** — the session id is taken from the URL path and trusted | ❌ none |
|
||||
|
||||
So REST is the *more* exposed surface: `POST /sessions/{id}/reply` accepts any `{id}` from the
|
||||
path, whereas the MCP `bridge_reply` derives the worker from the connection and refuses to read it
|
||||
path, whereas the MCP `fleet_reply` derives the worker from the connection and refuses to read it
|
||||
from an argument. Loopback-only bind is what makes this safe today.
|
||||
|
||||
**Therefore CB-505 must enforce on both paths against one shared resolver** — not at a single
|
||||
@@ -188,15 +188,15 @@ Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
|
||||
| Metric | Type | Why it exists |
|
||||
|---|---|---|
|
||||
| `bridged_sends_total{outcome}` | counter | outcome ∈ replied\|completion_fallback\|timeout\|failed — the completion-fallback rate is the health signal for turn detection (CB-115/116/118) |
|
||||
| `bridged_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `bridged_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `bridged_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `bridged_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `bridged_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `bridged_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `bridged_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
| `bridged_auth_failures_total{reason}` | counter | only meaningful once CB-501 lands; catches misconfigured workers |
|
||||
| `fleet_sends_total{outcome}` | counter | outcome ∈ replied\|completion_fallback\|timeout\|failed — the completion-fallback rate is the health signal for turn detection (CB-115/116/118) |
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
| `fleet_auth_failures_total{reason}` | counter | only meaningful once CB-501 lands; catches misconfigured workers |
|
||||
|
||||
---
|
||||
|
||||
@@ -226,7 +226,7 @@ was answered from the forge itself. Kept here as the decision record.*
|
||||
|
||||
1. ✅ **TLS scope (D3) — confirmed.** Bearer auth + the fail-fast bind guard ship in the daemon;
|
||||
TLS terminates at a reverse proxy, documented with a worked example. AMQP gets TLS via an
|
||||
`amqps://` URI. No keystore handling in `bridged`.
|
||||
`amqps://` URI. No keystore handling in `fleetd`.
|
||||
2. ✅ **Micrometer (D4) — confirmed dropped.** Zero-dependency Prometheus text renderer, for the
|
||||
reasons in D4 (pom reconciliation burden + the mandated CVE gate being un-runnable this
|
||||
session). Revisit if a push-gateway or JVM-metrics requirement appears; the endpoint is the
|
||||
|
||||
@@ -0,0 +1,972 @@
|
||||
# M4 - Fleet health, recovery, routing, and capacity
|
||||
|
||||
**Status:** Design accepted on 2026-08-15. CB-573 part 1 has shipped the classification model and
|
||||
the `fleet_list` capacity view; the remaining M4 units are not yet shipped. See
|
||||
[Unit 2 — what has landed so far](#unit-2---what-has-landed-so-far) before planning Unit 2 work:
|
||||
some of its criteria were met by separate CB tickets, and one of them contradicts the unit text.
|
||||
**Scope:** Fleet evidence, safe mechanical repair, lead routing, capacity reporting, and optional
|
||||
human notification.
|
||||
**Grounded in:** `health/FleetHealth`, `health/PaneBudget`, `inject/StatusPoller`,
|
||||
`inject/StatusRefiner`, `inject/CompletionResolver`, `inject/Injector`, `session/SessionManager`,
|
||||
`msg/MessageService`, `msg/ReplyInbox`, `msg/ReplyPushLoop`, `msg/LeadHeartbeatLoop`,
|
||||
`mcp/PrimaryRegistry`, and `herdr/AgentControl`.
|
||||
|
||||
## 1. Problem and decision boundary
|
||||
|
||||
The operator asked the bridge to detect idle agents, exceptions, stopped work, and broken
|
||||
communication. The bridge may read an agent pane from time to time. It must notify a person when
|
||||
the fleet cannot move forward.
|
||||
|
||||
The four operator terms are not four equal health states. `IDLE` is a normal mode. An exception is
|
||||
sometimes visible only as pane text. Stopped work may look the same as slow work. Broken
|
||||
communication can occur on several links.
|
||||
|
||||
M4 uses this boundary:
|
||||
|
||||
- The bridge detects facts and joins evidence.
|
||||
- The bridge repairs only mechanical failures with no judgement.
|
||||
- The lead decides whether to stop, retry, replace, or reassign a member.
|
||||
- A human is notified only when no healthy lead can act.
|
||||
- n8n may route an outbound incident. It never classifies state or chooses recovery.
|
||||
|
||||
An inbound n8n decider would need bridge authority. No narrow machine-decider role exists. Giving a
|
||||
workflow engine lead authority is unsafe, while adding a new role is a separate authorization
|
||||
design. An outbound sink needs no bridge role.
|
||||
|
||||
The bridge must never replay a delivered task. That task may already have changed files, pushed a
|
||||
branch, opened a pull request, or changed external state. A replay can run those side effects twice.
|
||||
This rule must remain true even if later code stores delivered prompt text.
|
||||
|
||||
## 2. Evidence model
|
||||
|
||||
A health state is mainly a comparison between two views:
|
||||
|
||||
- **herdr view:** current agents and raw live status from one `AgentControl.list()` call.
|
||||
- **bridge view:** session FSM, MCP presence, accepted turns, tasks, inbox state, and lead ownership.
|
||||
|
||||
A strong fault often appears as a disagreement between those views. For example, `BUSY` in the
|
||||
session FSM and `DONE` in herdr means the bridge missed a turn boundary. Pane reads support this
|
||||
model, but they are not the main monitor.
|
||||
|
||||
`SessionManager.rosterView` already joins session state and live status. `AgentControl.list()`
|
||||
already gets the whole live fleet in one call. M4 makes that join persistent and adds timers,
|
||||
accepted-turn state, and incident state.
|
||||
|
||||
### 2.1 Real traces behind the design
|
||||
|
||||
The first trace was an architect that stopped making progress:
|
||||
|
||||
```text
|
||||
profile=opus role=architect state=busy liveStatus=done
|
||||
```
|
||||
|
||||
The session moved from `DONE` to `BUSY` for turn 2. Eighteen minutes later, the session still said
|
||||
`BUSY`, herdr still said `DONE`, the async task still said `PENDING`, and no completion fallback had
|
||||
run. This is `TURN_BOUNDARY_LOST`, not a general slow-turn guess.
|
||||
|
||||
The second trace had two async sends to the same pane, one second apart. The pane was then stopped.
|
||||
One ticket became failed. The other stayed `pending - worker unknown`. Current
|
||||
`MessageService.abandon` resolves only `Rendezvous.currentWaiter(target)`, while async tasks live in
|
||||
a separate ticket map. CB-568 is intended to fix that bug. M4 still keeps an independent
|
||||
post-teardown invariant so a later regression becomes `DELEGATION_ORPHANED`.
|
||||
|
||||
### 2.2 Corrections made during design
|
||||
|
||||
The first state table missed `BUSY` in fleetd plus `IDLE` or `DONE` in herdr. It would have found
|
||||
the real trace only through a late, weak stall timer. The final model adds
|
||||
`TURN_BOUNDARY_LOST` as a strong disagreement state.
|
||||
|
||||
The first notification design also required a webhook before `health.enabled` could turn on. That
|
||||
removed useful local detection to avoid a narrower human-notification gap. The final design splits
|
||||
detection from notification. Missing human escalation is shown as partial coverage instead of
|
||||
disabling health.
|
||||
|
||||
## 3. Classification precedence
|
||||
|
||||
Evidence is applied in this order. A lower rule cannot hide a higher one.
|
||||
|
||||
1. **Control link:** failed fleet list plus failed ping becomes `CONTROL_LINK_DOWN`.
|
||||
2. **Definitive target loss:** `_not_found` becomes `GONE` or `LEAD_UNREACHABLE` when the control
|
||||
link is healthy.
|
||||
3. **Startup and teardown invariants:** readiness expiry becomes `NEVER_READY`; surviving tasks
|
||||
after teardown become `DELEGATION_ORPHANED`.
|
||||
4. **Bridge/live disagreement:** `BUSY` plus stable raw `IDLE` or `DONE` becomes
|
||||
`TURN_BOUNDARY_LOST`.
|
||||
5. **Known screen evidence:** a tested fatal signature becomes `ERROR_ON_SCREEN`.
|
||||
6. **Timed suspicion:** unchanged sparse pane probes may become `STALL_SUSPECTED`.
|
||||
7. **Communication quality:** completion fallback becomes `MUTE`; an old inbox entry becomes
|
||||
`REPLY_STRANDED`.
|
||||
8. **Normal mode:** `STARTING`, `IDLE`, `WORKING`, `WORK_PENDING`, or `BLOCKED_AMBIGUOUS`.
|
||||
|
||||
The member flow in Figure 1 shows lifecycle states and the main fault exits. Fault states are
|
||||
reported beside the session FSM; most are not new FSM values.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Registered["Member registered"] --> Starting["STARTING"]
|
||||
Starting -->|"MCP presence"| Idle["IDLE"]
|
||||
Starting -->|"Readiness grace expires"| NeverReady["NEVER_READY"]
|
||||
Idle -->|"Accepted delivery"| Working["WORKING"]
|
||||
Working -->|"Trusted turn boundary"| Idle
|
||||
Working -->|"Bridge BUSY and herdr IDLE or DONE"| Lost["TURN_BOUNDARY_LOST"]
|
||||
Working -->|"Known fatal screen"| Error["ERROR_ON_SCREEN"]
|
||||
Working -->|"Long age and unchanged sparse probes"| Stall["STALL_SUSPECTED"]
|
||||
Working -->|"Target not found"| Gone["GONE"]
|
||||
Idle -->|"Inbox or queued delivery exists"| Pending["WORK_PENDING"]
|
||||
Pending -->|"Delivery or collection finishes"| Idle
|
||||
Idle -->|"Raw BLOCKED with an open turn"| Blocked["BLOCKED_AMBIGUOUS"]
|
||||
Lost -->|"Strict guarded repair"| Repaired["DONE with reconciled completion"]
|
||||
Lost -->|"Repair refused"| LeadDecision["Lead decision required"]
|
||||
```
|
||||
|
||||
*Figure 1. The member lifecycle and the main health exits. Pane-based states never authorise an
|
||||
automatic retry of the task.*
|
||||
|
||||
## 4. State model
|
||||
|
||||
### 4.1 Normal and transitional member states
|
||||
|
||||
| State | Exact evidence | Meaning and certainty |
|
||||
|---|---|---|
|
||||
| `STARTING` | Session is `SPAWNING`; MCP presence is absent | Normal inside the startup grace. MCP contact is the readiness signal. |
|
||||
| `IDLE` | Session is `READY` or `DONE`; live status is `IDLE` or `DONE`; no open turn or inbox item exists | Normal. Idle is not a fault. |
|
||||
| `WORKING` | Session is `BUSY`; raw live status is `WORKING`; the accepted turn is open | Certain that herdr sees work. It does not prove useful progress. |
|
||||
| `WORK_PENDING` | Queued delivery or inbox content exists while the target is injectable | Transitional. Existing injector or push logic should move it. |
|
||||
| `BLOCKED_AMBIGUOUS` | An open turn exists and raw live status is `BLOCKED` | The bridge cannot tell whether this is permission, input, or a settled screen. |
|
||||
|
||||
Idle may drive configured resource cleanup. It never opens an incident and never pages a person.
|
||||
|
||||
### 4.2 Member fault and quality states
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `NEVER_READY` | `SPAWNING`, no MCP presence, and an accepted delivery waits through the existing readiness grace | Delivery never became possible. The exact cause is unknown. Fail the send, stop the process, and preserve a provisioned worktree. |
|
||||
| `GONE` | Per-target herdr call returns `_not_found` while fleet list or ping works | Certain target loss. Fail all target work. Do not replay it. |
|
||||
| `TURN_BOUNDARY_LOST` | Same session turn stays `BUSY`; same accepted task stays open; two raw snapshots show `IDLE` or `DONE` | Strong disagreement. Strict reconciliation may repair it. |
|
||||
| `ERROR_ON_SCREEN` | Suspicious non-working state survives grace; `detection` matches a tested adapter-specific fatal signature | Certain only for the matched signature. A bare word such as `Exception` is not enough. |
|
||||
| `STALL_SUSPECTED` | Open turn is older than the configured threshold; two normalised `recent_unwrapped` digests are unchanged; no boundary or reply occurs | Not certain. A long valid API call can look the same. Lead decides. |
|
||||
| `MUTE` | Turn resolves through completion fallback instead of `fleet_reply` | Certain that no structured reply won. It does not prove an MCP failure. A single event is a metric, not an incident. |
|
||||
| `REPLY_STRANDED` | Typed reply or health message remains after owning-lead push reaches its cap | Collection failed. This does not explain whether the lead is busy, dead, or ignoring the nudge. |
|
||||
| `DELEGATION_ORPHANED` | Target is gone, failed, or released, but one or more tasks remain `PENDING` after reconciliation grace | Certain bridge invariant failure. This is not an inbox-drain fault. |
|
||||
| `WORK_PRODUCT_AT_RISK` | Provisioned branch has commits after its recorded base; member is `DONE`, `FAILED`, or preserved after release; no turn or inbox item remains; long-idle threshold passed | A warning, not proof of loss. Work may already have an open pull request or a squash merge. |
|
||||
|
||||
`MUTE` opens an incident only after a small fixed rate threshold for one target or profile, or when
|
||||
it appears with another fault.
|
||||
|
||||
`WORK_PRODUCT_AT_RISK` must not become `WORK_PRODUCT_UNCOLLECTED`. The bridge does not know pull
|
||||
request or merge state. If committed work appears with `REPLY_STRANDED` or
|
||||
`DELEGATION_ORPHANED`, the existing incident gains `committedWorkAtRisk: true`.
|
||||
|
||||
### 4.3 Control-link state
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `CONTROL_LINK_DOWN` | Two full-fleet `agent.list` calls fail across the grace, and herdr `ping` also fails | Certain for the fleetd-to-herdr link. Retry calls, record the incident, and use human escalation if no lead can be reached. |
|
||||
|
||||
A failed fleet list alone is not a dead-member claim. A single `_not_found` with a healthy global
|
||||
link is a target fault, not a control-link fault.
|
||||
|
||||
### 4.4 Lead states
|
||||
|
||||
| State | Exact evidence | Meaning and action |
|
||||
|---|---|---|
|
||||
| `LEAD_IDLE` | Expected lead is present with raw injectable status; no actionable state waits | Normal. Existing heartbeat may run under its own policy. |
|
||||
| `LEAD_WORKING` | Expected lead is present with raw `WORKING`; stall threshold is not met | Reachable and busy. Never inject into the live turn. |
|
||||
| `LEAD_STATUS_UNKNOWN` | Expected lead is present with raw `UNKNOWN` | Neither dead nor a healthy routing target. Retain evidence and retry. |
|
||||
| `LEAD_UNREACHABLE` | Expected lead is absent from two successful live-agent snapshots while ping works, or targeted lookup returns `_not_found` with a healthy control link | Route to a healthy peer. If none exists, use human escalation. |
|
||||
| `LEAD_UNRESPONSIVE` | Actionable state waits; lead stays injectable; bounded nudges exhaust; inbox remains uncollected | Route to a healthy peer or a person. |
|
||||
| `LEAD_STALL_SUSPECTED` | Lead stays `WORKING` past threshold; two sparse pane probes show no progress | Not certain. Never kill or restart automatically. Route to peer or person. |
|
||||
|
||||
The monitor retains the lead name and terminal, last successful sighting, raw status and age,
|
||||
consecutive list absences, targeted errors, pane-probe facts, pending incident age, and nudge
|
||||
outcomes. Current heartbeat and push loops discard much of this history.
|
||||
|
||||
Expected lead identity comes from the same supplier used by `CallerResolver`. It is not liveness
|
||||
evidence. `LeadTabScanner` keeps cached identity after a failed scan, so the health monitor compares
|
||||
that identity with a fresh successful agent list. A dynamic identity also survives a two-successful-
|
||||
snapshot retirement grace. This stops a dead lead from escaping health by disappearing from one map.
|
||||
|
||||
### 4.5 Evidence limits
|
||||
|
||||
M4 cannot tell these cases apart with current evidence:
|
||||
|
||||
- A valid long call and a hung call may have the same status and pane digest.
|
||||
- `BLOCKED` does not explain which input is needed.
|
||||
- An idle prompt after failure may look like an idle prompt after success.
|
||||
- A missing structured reply does not prove a broken MCP connection.
|
||||
- An undrained inbox does not explain why the lead did not collect it.
|
||||
- Arbitrary pane text cannot safely classify arbitrary exceptions.
|
||||
- A branch ahead of its base does not prove that work was not collected.
|
||||
|
||||
Logs are outputs, not classifier inputs. The monitor never parses its own logs.
|
||||
|
||||
## 5. Automatic action and lead action
|
||||
|
||||
### 5.1 Actions the bridge may take
|
||||
|
||||
The bridge may:
|
||||
|
||||
- retry transient herdr status, list, ping, and pane-read failures with bounded backoff;
|
||||
- re-submit Enter after the existing paste/submit race;
|
||||
- fail queued delivery after `NEVER_READY`;
|
||||
- stop a never-ready process while preserving its provisioned worktree;
|
||||
- fail all queued, accepted, and async tasks for a gone or released target;
|
||||
- reconcile one lost boundary when every strict gate in Section 8 passes;
|
||||
- hold typed messages, nudge the owning lead, and stop at the configured cap;
|
||||
- use the existing bounded idle-lead heartbeat;
|
||||
- deduplicate, route, update, and resolve incidents.
|
||||
|
||||
These actions do not choose new work and do not replay old work.
|
||||
|
||||
### 5.2 Decisions reserved for the lead
|
||||
|
||||
Only the lead may:
|
||||
|
||||
- stop or continue `BLOCKED_AMBIGUOUS`;
|
||||
- stop, inspect, or wait on `ERROR_ON_SCREEN`;
|
||||
- kill or continue `STALL_SUSPECTED`;
|
||||
- spawn a replacement or reassign work;
|
||||
- retry a delivered task;
|
||||
- choose how to use partial work in a worktree;
|
||||
- restart herdr or change network, model, credentials, backend, or configuration.
|
||||
|
||||
Reports include literal safe tool calls such as `fleet_status(sessionId="...")`,
|
||||
`fleet_poll(ticket="...")`, `fleet_list()`, and optional `fleet_stop(paneId="...")`. A judgement
|
||||
state never presents stop as the only action.
|
||||
|
||||
### 5.3 Release causes and worktree safety
|
||||
|
||||
| Release cause | Process action | Provisioned worktree |
|
||||
|---|---|---|
|
||||
| `SPAWN_ROLLBACK` before registration or delivery | Stop and clean up | Remove |
|
||||
| `COMPLETED` for `READY` or `DONE` without pending work, idle TTL, or successful context-cap completion | Stop | Remove only if clean; preserve a dirty worktree (CB-576) |
|
||||
| `NEVER_READY` | Stop | Preserve |
|
||||
| `GONE` | Best-effort stop | Preserve |
|
||||
| `TURN_FAILED` or lead abort while `BUSY` or `FAILED` | Stop | Preserve |
|
||||
| `RELEASE_WITH_PENDING_TASKS` | Stop | Preserve |
|
||||
| `SHUTDOWN` | Stop | Preserve |
|
||||
|
||||
Explicit stop is state-aware. `SPAWNING`, `BUSY`, `FAILED`, or any target with pending tasks uses a
|
||||
preserving cause.
|
||||
|
||||
Before abnormal release removes the live session, M4 writes an atomic manifest under the worktree
|
||||
root. It records session identity, owner, role, profile, repository, path, branch, base commit,
|
||||
release cause, release time, state, and pending task ids. `fleet_list.preservedWorktrees` loads these
|
||||
manifests after restart. Stop output and WARN logs also name the path and cause. M4 never
|
||||
auto-deletes a preserved worktree.
|
||||
|
||||
## 6. Fleet health monitor
|
||||
|
||||
Add `FleetHealthMonitor`. Do not widen `StatusPoller` into a policy loop.
|
||||
|
||||
`StatusPoller` has a 250 ms delivery cadence and samples only injector targets with outstanding
|
||||
work. Health needs all sessions, all leads, task state, inbox age, and global control evidence. One
|
||||
loop cannot serve both cadences safely.
|
||||
|
||||
Build the monitor like `LeadHeartbeatLoop`:
|
||||
|
||||
- pure `decide(snapshot, priorState, now)` logic;
|
||||
- a thin scheduler;
|
||||
- an injected clock;
|
||||
- edge-triggered state changes;
|
||||
- no network work in the pure function;
|
||||
- no sleeping in tests.
|
||||
|
||||
Each enabled fleet tick reads:
|
||||
|
||||
- one `AgentControl.list()` result for the whole fleet;
|
||||
- one in-memory `SessionManager.roster()` snapshot;
|
||||
- accepted turns and async task state;
|
||||
- typed inbox depth, kind, and age;
|
||||
- push and heartbeat outcomes;
|
||||
- configured and discovered leads.
|
||||
|
||||
Existing failure paths publish structured evidence to the monitor. The monitor does not infer events
|
||||
from log text.
|
||||
|
||||
### 6.1 Pane budget
|
||||
|
||||
Healthy idle members, recent working members, and quiet leads cause no pane reads.
|
||||
|
||||
A pane is eligible only for a stable lost boundary, sustained `BLOCKED` or `UNKNOWN`, work older
|
||||
than the suspect threshold, or one final evidence read for a confirmed fault when the pane exists.
|
||||
|
||||
Compiled brakes apply even if config asks for more:
|
||||
|
||||
- per-target pane cooldown is at least 60 seconds;
|
||||
- working age before the first progress probe is at least 300 seconds;
|
||||
- at most two pane reads occur in one fleet tick;
|
||||
- targets rotate fairly;
|
||||
- only a normalised digest and optional clipped local excerpt are stored;
|
||||
- no pane excerpt leaves fleetd in a human webhook.
|
||||
|
||||
Use `detection` for tested screen signatures. Use normalised `recent_unwrapped` only for progress
|
||||
comparison.
|
||||
|
||||
## 7. Typed inbox and routing
|
||||
|
||||
### 7.1 Semantic record
|
||||
|
||||
The typed inbox record carries:
|
||||
|
||||
```text
|
||||
schemaVersion
|
||||
kind: reply | health
|
||||
msgId, target, subjectTerminal, recipientLead
|
||||
severity, state, evidence
|
||||
createdAtEpochMillis, firstSeenEpochMillis, lastSeenEpochMillis
|
||||
recoveryTried, suggestedToolCalls, content
|
||||
```
|
||||
|
||||
A health message never calls `Rendezvous.resolve`. It cannot look like the member's task result.
|
||||
|
||||
Both inbox adapters share field preservation, first-id-wins dedup, FIFO among decoded messages,
|
||||
explicit ownership, ack, and release rules. The in-memory adapter stores typed records directly. It
|
||||
does not copy AMQP migration logic.
|
||||
|
||||
### 7.2 AMQP migration
|
||||
|
||||
The reader uses AMQP `content_type`, never body sniffing:
|
||||
|
||||
```text
|
||||
Legacy v0: text/plain
|
||||
Typed family: application/vnd.ltms.fleet.inbox-message+json
|
||||
```
|
||||
|
||||
A legacy reply may begin with `{`. It remains plain text because its media type is `text/plain`.
|
||||
Legacy text becomes `kind=reply` with exact UTF-8 content and absent typed metadata.
|
||||
|
||||
Typed JSON has required integer `schemaVersion: 1`. Version 1 ignores unknown optional fields.
|
||||
Missing required fields, invalid enums, malformed UTF-8 or JSON, and property/body identity mismatch
|
||||
are invalid data.
|
||||
|
||||
An unknown schema version is not partly decoded. It remains unacknowledged on the original queue and
|
||||
creates one operator-visible `unsupported_version` failure. A newer daemon may read it later.
|
||||
|
||||
Invalid known-format data is copied byte-for-byte to durable queue
|
||||
`agent.<target>.inbox.quarantine`. A dedicated confirm-mode publisher confirms the persistent copy
|
||||
before the original is acknowledged. A failed quarantine handoff leaves the original unacknowledged.
|
||||
The raw body never enters logs.
|
||||
|
||||
Decode failure creates a redacted WARN, metric, `fleet_list` summary, and routed health incident.
|
||||
One bad entry never escapes the consumer callback and never stops later valid messages.
|
||||
|
||||
Safe downgrade is not supported. The previous build ignores `content_type` and would show typed JSON
|
||||
as ordinary reply text. If drained, it would acknowledge the message and lose typed meaning. Typed
|
||||
queues must be drained or preserved before an old jar runs.
|
||||
|
||||
The existing contract suite uses RabbitMQ. Production uses LavinMQ. The migration and lead-key
|
||||
ownership cases must run once against production LavinMQ before release, or the release must state
|
||||
that LavinMQ was not checked.
|
||||
|
||||
### 7.3 Member routing
|
||||
|
||||
A member incident first goes to the exact lead that owns its accepted delegation.
|
||||
`PrimaryRegistry` needs a no-fallback `delegatingLeadFor(memberTarget)` query. Health routing must not
|
||||
use the old singular-primary fallback when several leads exist.
|
||||
|
||||
Publish the incident under the affected member target. Trigger the existing bounded push route. The
|
||||
push waits until the owning lead is injectable, so it does not interrupt a live lead turn.
|
||||
|
||||
### 7.4 Peer lead routing
|
||||
|
||||
Figure 2 shows the route from incident to lead, peer, or person.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Incident["Open incident"] --> Member{"Member incident?"}
|
||||
Member -->|"yes"| Known{"Exact delegation owner known?"}
|
||||
Known -->|"no"| Sink{"Human webhook enabled and healthy?"}
|
||||
Known -->|"yes"| Owner{"Owner lead healthy?"}
|
||||
Owner -->|"yes"| OwnerInbox["Publish to owner lead path"]
|
||||
Owner -->|"no"| PeerSet["Build healthy peer candidate set"]
|
||||
Member -->|"no, lead incident"| PeerSet
|
||||
PeerSet --> Peer{"Healthy peer exists?"}
|
||||
Peer -->|"yes"| Select["Choose fewest assigned incidents<br/>then stable name and terminal id"]
|
||||
Select --> PeerInbox["Publish to peer lead inbox<br/>and status-gated push"]
|
||||
Peer -->|"no"| Sink
|
||||
Sink -->|"yes"| Webhook["Send classified outbound incident"]
|
||||
Sink -->|"no"| Passive["Keep incident open<br/>show partial coverage on local surfaces"]
|
||||
```
|
||||
|
||||
*Figure 2. Routing keeps delegation ownership separate from temporary peer fallback.*
|
||||
|
||||
Peer candidates exclude the incident subject, failed owner, absent leads, raw-unknown leads, and
|
||||
leads with an open unhealthy state. A reachable `WORKING` peer may be selected; its push waits for an
|
||||
injectable window.
|
||||
|
||||
Choose the candidate with the fewest assigned foreign incidents. Break ties by stable lead name,
|
||||
then terminal id. Pin the recipient. Reassign only if that peer becomes unhealthy or retires. A
|
||||
routing generation marks a reassignment, and old pending assignments become superseded.
|
||||
|
||||
`fleet_list` lead rows show health, health age, assigned foreign incident count, and a bounded list
|
||||
of incident id, subject, state, severity, age, and routing generation. The top-level view also shows
|
||||
owner, recipient, and routing reason.
|
||||
|
||||
A peer incident is published under the recipient lead's inbox key, not the failed subject's key. Its
|
||||
status-gated nudge names the failed lead and gives the exact
|
||||
`fleet_poll(target="<recipient-terminal>")` call.
|
||||
|
||||
### 7.5 Lead inbox ownership
|
||||
|
||||
Add `LeadInboxRegistry`, driven by the same expected-lead supplier as `CallerResolver`.
|
||||
|
||||
It calls `replyInbox.own(leadTerminal)` at startup for configured leads, after successful discovery,
|
||||
after config adds a lead, and before publication. Ownership is not an authorization side effect.
|
||||
|
||||
A missing lead keeps its key owned. Release happens only after confirmed retirement, all incidents
|
||||
are reassigned or resolved, typed health messages move or ack, and the queue is empty. Own a
|
||||
replacement terminal before moving messages from the old key. Never release a non-empty in-memory
|
||||
lead key, because in-memory release clears local data.
|
||||
|
||||
### 7.6 Single-lead deployment
|
||||
|
||||
One lead and no peer is a normal mode, not an edge case.
|
||||
|
||||
An idle, reachable lead may receive the existing bounded nudge. An unreachable or stalled sole lead
|
||||
has no safe in-loop recovery. The bridge must not restart or replace it. A new lead would not have the
|
||||
failed lead's plan or context, and an uncertain relaunch could create two orchestrators.
|
||||
|
||||
With no webhook, only `fleet_list`, `/healthz`, metrics, WARN logs, and the incident journal remain.
|
||||
These are passive surfaces. They are not a human notification.
|
||||
|
||||
## 8. Lost-boundary reconciliation
|
||||
|
||||
This is the only M4 path that reconstructs a result. It must prefer a visible stall over a fabricated
|
||||
reply.
|
||||
|
||||
### 8.1 Why normal completion rules are not enough
|
||||
|
||||
Current `CompletionResolver.resolve` has two fail-open rules. It resolves when the delivery baseline
|
||||
is missing. It also resolves an empty completion when the pane read fails. Those choices are valid
|
||||
after a trusted `WORKING -> IDLE` boundary because the bridge knows the turn ran. They are unsafe
|
||||
when health only guesses that a boundary was lost.
|
||||
|
||||
M4 gives each accepted send an internal `TurnToken`. It ties target, exact waiter, session turn,
|
||||
delivery baseline, and task outcome together.
|
||||
|
||||
### 8.2 Delivery baseline
|
||||
|
||||
Capture the baseline immediately after prompt send and before the delivery future completes. Store:
|
||||
|
||||
```text
|
||||
TurnToken
|
||||
exact waiter identity
|
||||
capture time and pane source
|
||||
normalised assistant block clipped to MAX_SCRAPE_CHARS
|
||||
whether a supported assistant marker was recognised
|
||||
capture result: PRESENT | READ_FAILED | UNRECOGNISED
|
||||
```
|
||||
|
||||
A failed or missing baseline never authorises repair. A late baseline is not valid evidence. After a
|
||||
daemon restart, the old waiter, task, token, and baseline are gone, so the old turn cannot be
|
||||
repaired.
|
||||
|
||||
Automatic repair is enabled only for agent kinds with tested assistant-block fixtures. Current
|
||||
extraction is Claude Code-specific and falls back to arbitrary raw text without `⏺`. That raw fallback
|
||||
cannot authorise repair. OpenCode repair stays disabled until live pane fixtures exist.
|
||||
|
||||
### 8.3 Strict gates and resolver result
|
||||
|
||||
Figure 3 shows the repair gates. Any failed gate keeps the waiter unchanged.
|
||||
|
||||
The two raw snapshots must describe the same `TurnToken` and session turn. No `WORKING`,
|
||||
`BLOCKED`, `UNKNOWN`, missing-agent, reply, failure, or new-delivery observation may occur between
|
||||
them.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Candidate["TURN_BOUNDARY_LOST candidate"] --> Stable{"Same TurnToken and BUSY turn<br/>across two raw IDLE or DONE snapshots?"}
|
||||
Stable -->|"no"| Resnapshot["Take a fresh snapshot"]
|
||||
Stable -->|"yes"| Waiter{"Exact captured waiter<br/>still open by identity?"}
|
||||
Waiter -->|"no"| Stale["STALE_TURN or ALREADY_RESOLVED"]
|
||||
Waiter -->|"yes"| Baseline{"Successful recognised<br/>delivery baseline exists?"}
|
||||
Baseline -->|"no"| Refuse["Refuse repair<br/>leave ticket pending"]
|
||||
Baseline -->|"yes"| Read{"Fresh pane read succeeds?"}
|
||||
Read -->|"no"| Refuse
|
||||
Read -->|"yes"| Output{"Recognised non-blank assistant block<br/>differs from clipped baseline?"}
|
||||
Output -->|"no"| Refuse
|
||||
Output -->|"yes"| Resolve["Shared CompletionResolver guard core<br/>resolves exact waiter"]
|
||||
Resolve -->|"won race"| Repaired["RECONCILED_COMPLETION<br/>same turn becomes DONE"]
|
||||
Resolve -->|"lost race"| Resnapshot
|
||||
```
|
||||
|
||||
*Figure 3. Repair needs stronger evidence than a normal observed turn boundary.*
|
||||
|
||||
Refactor the current resolver into one guard core with two policies:
|
||||
|
||||
```text
|
||||
resolveCaptured(target, inFlight, OBSERVED_BOUNDARY)
|
||||
resolveCaptured(target, inFlight, LOST_BOUNDARY_REPAIR)
|
||||
```
|
||||
|
||||
The health monitor calls only:
|
||||
|
||||
```text
|
||||
CompletionResolver.reconcileLostBoundary(target, expectedTurnToken)
|
||||
```
|
||||
|
||||
It returns `REPAIRED`, `ALREADY_RESOLVED`, `REFUSED_NO_CAPTURE`, `REFUSED_NO_BASELINE`,
|
||||
`REFUSED_UNREADABLE`, `REFUSED_UNCHANGED`, `REFUSED_AMBIGUOUS_OUTPUT`, `STALE_TURN`, or
|
||||
`RACE_LOST`.
|
||||
|
||||
Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move that turn from `BUSY` to `DONE`. A
|
||||
per-target reconciliation gate stops a queued second send from being accepted between waiter
|
||||
resolution and the FSM transition.
|
||||
|
||||
### 8.4 Lead-visible marker and refusal
|
||||
|
||||
A repaired result uses distinct `RECONCILED_COMPLETION` values in `Rendezvous`, `MessageService`,
|
||||
task poll source, and metrics. The lead sees:
|
||||
|
||||
```text
|
||||
[repaired completion - fleetd detected a lost turn boundary. The member did not call
|
||||
fleet_reply; pane-derived text follows and may be partial]
|
||||
```
|
||||
|
||||
Clipped text also keeps the existing clipped-tail marker.
|
||||
|
||||
A refused repair leaves `TURN_BOUNDARY_LOST` open and the ticket pending. The report states that no
|
||||
reply was reconstructed and no task was replayed. `UNCHANGED`, `UNREADABLE`, and
|
||||
`AMBIGUOUS_OUTPUT` get at most one delayed retry for the same token. Missing capture or baseline gets
|
||||
no retry. After two refused scrapes, automatic repair stops for that token.
|
||||
|
||||
### 8.5 Target-wide teardown invariant
|
||||
|
||||
CB-568 owns the multi-ticket cancellation mechanism. M4 routes every terminal cause through that one
|
||||
idempotent operation and checks this independent invariant after teardown:
|
||||
|
||||
- no injector entry exists for the target;
|
||||
- no accepted turn or completion record exists;
|
||||
- no rendezvous waiter or ask exists;
|
||||
- every async task is terminal or was already terminal;
|
||||
- no thread waiting for the target send lock can later accept it;
|
||||
- new sends fail immediately;
|
||||
- each old task has one terminal outcome and one metric count.
|
||||
|
||||
A violation becomes `DELEGATION_ORPHANED`. The monitor may call the same idempotent target-wide
|
||||
failure operation once. It never recreates the task.
|
||||
|
||||
## 9. Capacity and utilisation
|
||||
|
||||
Capacity is a view, not a health state.
|
||||
|
||||
`fleet_list` adds one block per profile:
|
||||
|
||||
```text
|
||||
profile, maxLoad, live, free, reclaimable
|
||||
```
|
||||
|
||||
For an unlimited profile, `maxLoad` and `free` are null. `free` is
|
||||
`max(0, maxLoad - live)` for a capped profile.
|
||||
|
||||
The view must use the exact live-count function used by placement. A second calculation could show a
|
||||
free slot that placement then refuses. Member rows add `idleForSeconds` only when state is `READY` or
|
||||
`DONE`, no accepted turn exists, and the inbox is empty. `reclaimable` means only that the member
|
||||
holds capacity without open bridge work.
|
||||
|
||||
The existing idle-lead nudge gains a bounded capacity summary. It lists per-profile live, cap, free,
|
||||
and reclaimable counts, plus at most three long-idle members. Capacity does not make
|
||||
`FleetState.hasPending()` true. A changed capacity fingerprint may re-arm one capped heartbeat
|
||||
sequence. The fingerprint excludes changing idle durations, so a static idle fleet cannot reset the
|
||||
cap forever. Reply-push stand-down remains first.
|
||||
|
||||
The bridge must never:
|
||||
|
||||
- spawn a member because a slot is free;
|
||||
- generate a task or acceptance criteria;
|
||||
- move queued work to another member or profile;
|
||||
- treat a free slot or idle member as an incident;
|
||||
- stop an idle member only to improve utilisation.
|
||||
|
||||
The bridge knows capacity facts but has no work list. Only the lead has the plan, task context,
|
||||
side-effect history, and acceptance criteria.
|
||||
|
||||
Capacity calculation is in memory and adds no pane reads. Work-product checks run on a terminal
|
||||
session edge, not every fleet tick.
|
||||
|
||||
This capacity design adds no automatic stop. The accepted `NEVER_READY` cleanup can still stop a
|
||||
very slow startup after the existing grace, which is a known risk. Free capacity and long idle time
|
||||
never trigger that path.
|
||||
|
||||
## 10. Human escalation and notification
|
||||
|
||||
### 10.1 Escalation rule
|
||||
|
||||
Notify a person only when no healthy lead can act:
|
||||
|
||||
- `CONTROL_LINK_DOWN` survives grace;
|
||||
- a lead is unhealthy and no healthy peer can receive the incident;
|
||||
- a member incident has no known owning lead;
|
||||
- the only owning lead becomes unreachable, unresponsive, or stalled;
|
||||
- incident publication or routing itself fails.
|
||||
|
||||
Do not page a person for a member fault while a healthy owning lead exists. An uncollected member
|
||||
incident feeds lead-health evidence. If the lead then becomes unhealthy, peer or human routing starts.
|
||||
|
||||
### 10.2 Detection and notification switches
|
||||
|
||||
`health.enabled` controls detection and bridge-local reporting. It does not require a webhook.
|
||||
|
||||
`health.notifications.mode` is `disabled` or `webhook`. Disabled is valid and is the default.
|
||||
Webhook mode requires a resolved environment variable. Turning notification off stops outbound
|
||||
attempts but keeps incidents. Turning it back on resumes still-open human incidents.
|
||||
|
||||
Without a sink, `fleet_list.healthCoverage` states that human escalation is unavailable. `/healthz`
|
||||
keeps its existing HTTP liveness result and adds a nested `fleetHealth.status=partial` component.
|
||||
Metrics and one startup or reload WARN expose the same limit.
|
||||
|
||||
### 10.3 Incident and delivery deduplication
|
||||
|
||||
One open incident uses this key:
|
||||
|
||||
```text
|
||||
(scope, subjectStableId, state, causeFingerprint)
|
||||
```
|
||||
|
||||
The cause fingerprint includes stable error codes, dependency names, signature ids, or invariant
|
||||
names. It excludes times, ages, retry counts, pane text, and changing digests. A later recurrence
|
||||
after resolution gets a new generation and incident id.
|
||||
|
||||
Each outbound event uses:
|
||||
|
||||
```text
|
||||
Idempotency-Key = hash(incidentId, eventType, eventRevision)
|
||||
```
|
||||
|
||||
Event types are `open`, `severity_changed`, `reminder`, and `resolved`. Transport retries keep the
|
||||
same key.
|
||||
|
||||
An atomic owner-only journal beside the active config stores open incidents, routing, delivered
|
||||
revisions, retry state, and resolution state. It stores no pane or task content. Journal failure does
|
||||
not stop detection, but notification coverage becomes degraded.
|
||||
|
||||
### 10.4 Retry, reminder, and resolve
|
||||
|
||||
Send the first event immediately. Retry network errors, timeouts, HTTP 408, HTTP 429, and HTTP 5xx
|
||||
with full-jitter exponential backoff:
|
||||
|
||||
```text
|
||||
base: 5 seconds
|
||||
factor: 3
|
||||
maximum delay: 15 minutes
|
||||
one outstanding attempt per event
|
||||
```
|
||||
|
||||
Respect `Retry-After` up to 15 minutes. Other HTTP 4xx responses are permanent for that event until
|
||||
config changes or a person requests replay.
|
||||
|
||||
Transport retry is not an incident reminder. `humanRepeatSeconds` creates a new reminder revision
|
||||
for an unresolved critical incident after the last successful human event. Disabled mode does not
|
||||
build an unbounded reminder queue.
|
||||
|
||||
Send `resolved` only if at least one human event for that incident was delivered. If an incident
|
||||
resolves before its first successful delivery, cancel the pending open event and record local
|
||||
resolution.
|
||||
|
||||
### 10.5 Outbound payload boundary
|
||||
|
||||
An outbound payload may contain incident id and event type, severity, state, scope, stable bridge
|
||||
ids, role or profile, times, duration, structured evidence type and counts, recovery attempted,
|
||||
routing reason, coverage, and safe tool calls.
|
||||
|
||||
It must never contain:
|
||||
|
||||
- raw pane text, pane excerpts, or pane digests;
|
||||
- task briefs, prompts, or member reply content;
|
||||
- source files, diffs, or worktree file content;
|
||||
- worktree paths;
|
||||
- environment values, tokens, credentials, headers, or webhook URL;
|
||||
- raw exception messages or stack traces;
|
||||
- arbitrary model output.
|
||||
|
||||
The sink response body is ignored. A webhook cannot direct recovery. n8n remains outbound-only.
|
||||
|
||||
### 10.6 Metrics
|
||||
|
||||
M4 adds bounded-label series:
|
||||
|
||||
```text
|
||||
fleet_health_incidents{scope,state,severity}
|
||||
fleet_health_incidents_total{event}
|
||||
fleet_health_notifications_total{event,outcome}
|
||||
fleet_health_notification_queue_depth
|
||||
fleet_health_notification_last_success_seconds
|
||||
fleet_health_notification_capability{mode,status}
|
||||
fleet_lead_health{lead,state}
|
||||
fleet_lead_assigned_incidents{lead}
|
||||
```
|
||||
|
||||
Metric labels never include terminal ids, incident ids, URLs, or error text.
|
||||
|
||||
## 11. Configuration
|
||||
|
||||
The optional `health:` block is absent or disabled by default. The dormant monitor scheduler does no
|
||||
herdr or pane work while disabled. Every listed key is hot because the monitor reads `ConfigRef` on
|
||||
each tick or notification.
|
||||
|
||||
| Key | Class | Default and hard bound | Purpose |
|
||||
|---|---|---|---|
|
||||
| `health.enabled` | Hot | `false` | Enable detection and bridge-local reporting. |
|
||||
| `health.snapshotIntervalSeconds` | Hot | default 30, minimum 15 | Whole-fleet comparison cadence. |
|
||||
| `health.workingSuspectAfterSeconds` | Hot | default 600, minimum 300 | Age before working-pane probes. |
|
||||
| `health.paneProbeIntervalSeconds` | Hot | default 60, minimum 60 | Per-target pane cooldown. |
|
||||
| `health.leadUnresponsiveAfterSeconds` | Hot | default 300, minimum 120 | Delay after exhausted actionable nudges before lead fault. |
|
||||
| `health.humanRepeatSeconds` | Hot | default 3600, minimum 900 | Minimum repeat period for one open human incident. |
|
||||
| `health.capacityLongIdleAfterSeconds` | Hot | default 900, minimum 300 | Long-idle threshold for capacity summaries. |
|
||||
| `health.includePaneExcerpt` | Hot | `false` | Allow a clipped excerpt in local lead reports only. Human payloads still exclude it. |
|
||||
| `health.notifications.mode` | Hot | `disabled` | Select `disabled` or `webhook`. |
|
||||
| `health.notifications.webhookUrlEnv` | Hot | required in webhook mode | Name of the environment variable that holds the sink URL. |
|
||||
| `health.notifications.requestTimeoutMs` | Hot | default 10000, range 1000-30000 | Whole webhook request limit. |
|
||||
|
||||
Two consecutive snapshots are compiled floors for lost boundary, lead disappearance, and control
|
||||
link failure. The two-pane-reads-per-tick limit is also compiled and cannot be weakened by config.
|
||||
|
||||
## 12. Delivery units and acceptance
|
||||
|
||||
### Unit 1 - Evidence model and fleet snapshot
|
||||
|
||||
Scope: health state model, fleet join, clocks, evidence retention, and pane budget.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. One `agent.list` call covers one enabled fleet tick.
|
||||
2. Pure decision tests cover every state and every evidence limit in Section 4.
|
||||
3. `BUSY` plus stable raw `DONE` opens `TURN_BOUNDARY_LOST` after two snapshots.
|
||||
4. Healthy fleet snapshots perform zero pane reads.
|
||||
5. Pane cooldown, two-read fleet budget, and fair rotation cannot be disabled by config.
|
||||
6. Logs are outputs only; no log parsing exists.
|
||||
7. Fleet snapshots expose the same profile live-count calculation that placement uses.
|
||||
8. Capacity rows report cap, live, free, and reclaimable values without opening incidents.
|
||||
|
||||
### Unit 2 - Lost boundary and task reconciliation
|
||||
|
||||
Scope: accepted-turn identity, guarded repair, target-wide teardown, release causes, and preserved
|
||||
worktree discovery.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Every accepted send receives a stable `TurnToken` tied to target, exact waiter, and delivery
|
||||
baseline.
|
||||
|
||||
**Corrected during implementation (2026-08-15).** This criterion first also required the session
|
||||
turn number and the task outcome. That is not implementable at this layer, and the implementer
|
||||
refused it three times rather than fabricate a value — correctly. The reason is an ordering fact
|
||||
that is invisible from any single class: `MessageService` owns acceptance and holds the waiter and
|
||||
the async `Task`, but it learns nothing about delivery, because the delivery event goes to
|
||||
`CompletionResolver` through `TurnListener.onDelivered`. And `CompletionResolver.onDelivered` runs
|
||||
*before* `SessionManager.onDelivered`, so the session turn number does not exist yet at the only
|
||||
point where the token could capture it.
|
||||
|
||||
Two ways out were rejected. A shared registry keyed by target reintroduces exactly the "whichever
|
||||
send happens to be waiting" ambiguity the token exists to remove — the same weak claim
|
||||
`Rendezvous.currentWaiter` warns about. Injecting a turn counter into `MessageService` adds a
|
||||
required cross-layer dependency to populate a field that nothing in this slice reads, which is
|
||||
speculative coupling across a boundary already shown to be fragile.
|
||||
|
||||
So the token identifies the **accepted send**, and `SessionManager` keeps verifying its own
|
||||
delivery separately. Repair (criterion 2) does need the session turn; binding it means resolving
|
||||
that acceptance-versus-delivery ordering first, and that work belongs to the repair unit, not
|
||||
here. The token record carries a comment saying the field is deliberately absent.
|
||||
2. Repair requires the same `BUSY` token, two raw `IDLE` or `DONE` snapshots, no conflicting
|
||||
observation, exact open waiter, successful baseline, and new recognised assistant output.
|
||||
3. Missing, failed, late, or post-restart baseline never authorises repair.
|
||||
4. Repair is enabled only for agent kinds with tested assistant-block extraction. Raw-text fallback
|
||||
without a recognised marker refuses repair.
|
||||
5. Normal completion and repair use one resolver guard core. Waiter, scrape, clipping, unchanged, and
|
||||
exact-turn guards are not duplicated.
|
||||
6. `reconcileLostBoundary` returns every typed result named in Section 8.3.
|
||||
7. Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move the same turn to `DONE`.
|
||||
8. A per-target reconciliation gate blocks a queued second send during repair and FSM update.
|
||||
9. Repaired completion has distinct rendezvous kind, message outcome, poll source, lead marker, and
|
||||
metric. Clipping keeps its extra marker.
|
||||
10. Unchanged, unreadable, or ambiguous evidence gets at most one delayed retry. Missing capture or
|
||||
baseline gets none.
|
||||
11. Refusal leaves the ticket pending and tells the lead that no result was rebuilt or replayed.
|
||||
12. Release, gone, never-ready, and abnormal stop use CB-568's one idempotent target-wide failure
|
||||
operation.
|
||||
13. The post-teardown invariant in Section 8.5 is tested independently of CB-568 internals.
|
||||
14. A violated teardown invariant creates `DELEGATION_ORPHANED` and retries only the idempotent
|
||||
failure operation.
|
||||
15. `SPAWN_ROLLBACK` and normal `COMPLETED` remove worktrees. Abnormal and shutdown causes preserve
|
||||
them.
|
||||
16. Explicit stop is state-aware. Any pending task or non-terminal state preserves the worktree.
|
||||
17. Atomic preserved-worktree manifests reload after restart and appear in lead-only
|
||||
`fleet_list.preservedWorktrees`.
|
||||
18. Manifest failure preserves the worktree and opens an operator-visible health failure.
|
||||
19. Provision records the base commit. Terminal, long-idle worktrees report
|
||||
`WORK_PRODUCT_AT_RISK` only under the evidence in Section 4.2 and never auto-delete work.
|
||||
20. No path replays a delivered task, rebuilds its brief, or retargets it, even when prompt text is
|
||||
available.
|
||||
21. Tests cover both real traces, all repair refusals, clipping, explicit-reply and next-turn races,
|
||||
restart without capture, concurrent send and release, and preserved discovery after restart.
|
||||
|
||||
#### Unit 2 - what has landed so far
|
||||
|
||||
Checked against `main` at `e09cac6` on 2026-08-15. Unit 2 was written as one block, but parts of it
|
||||
have since been built by separate CB tickets. Read this before planning the rest, or that work gets
|
||||
done twice.
|
||||
|
||||
The check was a symbol survey of `fleetd/src/main/java` plus the merge history. It tells you whether
|
||||
the machinery exists at all. It is **not** a line-by-line audit of whether each criterion is fully
|
||||
met, and I did not run one.
|
||||
|
||||
| Criterion | Marker searched for | Found in main source | Reading |
|
||||
|---|---|---|---|
|
||||
| 1 | `TurnToken` | 8 files | **Done** — unit 2a, merged as `fec284e`. Criterion 1 was corrected first; see the note under it. |
|
||||
| 2-5, 9 | `REPAIRED` | 0 files | Not started. The whole guarded-repair path is absent. |
|
||||
| 6, 7, 10 | `reconcileLostBoundary` | 0 files | Not started. |
|
||||
| 12 | CB-568 failure operation | via CB-580 | **Partial.** CB-580 (`0af902e`) routes `GONE` and `NEVER_READY` into the one idempotent target-wide failure. I did not check that release and abnormal stop go through the same call. |
|
||||
| 14 | `DELEGATION_ORPHANED` | 3 files | **Partial.** The health state exists. The teardown-invariant check that creates it, and the retry rule, do not. |
|
||||
| 15 | `SPAWN_ROLLBACK` | 0 files | **Contradicted — see below.** |
|
||||
| 16 | — | — | Partial at best. CB-576 made release preserve a dirty worktree; whether explicit stop is state-aware is not checked. |
|
||||
| 17, 18 | `preservedWorktrees` | 0 files | Not started. No manifest, and no lead-only `fleet_list` field. |
|
||||
| 19 | `WORK_PRODUCT_AT_RISK` | 0 files | Not started. |
|
||||
|
||||
**Criterion 15 no longer matches the code, and the code is right.** It says "normal `COMPLETED`
|
||||
remove worktrees". Since CB-576 (`500bfa2`) that is false on purpose: a `COMPLETED` release now
|
||||
preserves the worktree when it still holds uncommitted work, because deleting it destroys work
|
||||
nobody can get back. CB-576 was filed after exactly that loss. CB-581 goes further — if the
|
||||
dirty-check itself fails, the worktree is preserved rather than removed, since "we could not tell"
|
||||
must not be treated as "it is clean".
|
||||
|
||||
So criterion 15 should be rewritten as: `SPAWN_ROLLBACK` and a `COMPLETED` release with a **clean**
|
||||
worktree remove it; abnormal causes, shutdown, a dirty worktree, and a failed dirty-check all
|
||||
preserve it. `SPAWN_ROLLBACK` itself does not exist yet.
|
||||
|
||||
### Unit 3 - Typed inbox and member routing
|
||||
|
||||
Scope: semantic record, AMQP migration, both adapters, member routing, polling, and member health in
|
||||
`fleet_list`.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. AMQP selects legacy or typed decoding only from `content_type`; it never sniffs the body.
|
||||
2. Persistent `text/plain` from the old build becomes `kind=reply` with exact UTF-8 content,
|
||||
including content beginning with `{`.
|
||||
3. New entries use the vendor media type, `schemaVersion: 1`, UTF-8, persistent delivery, and AMQP
|
||||
message ids.
|
||||
4. Version 1 ignores unknown optional fields but rejects missing fields and identity mismatch.
|
||||
5. Unknown versions are not decoded or acked. They remain on the original queue and create one
|
||||
deduplicated failure.
|
||||
6. Invalid known data never escapes the callback, appears as a reply, or blocks later valid messages.
|
||||
7. Invalid data reaches durable per-target quarantine before original ack. Failed handoff leaves the
|
||||
original unacked.
|
||||
8. Decode failures create redacted WARN, metric, `fleet_list` summary, and routed incident without
|
||||
raw content.
|
||||
9. Both adapters pass one semantic contract for fields, FIFO, dedup, ownership, ack, and release.
|
||||
10. Lead keys require explicit ownership. Publication never claims a queue.
|
||||
11. Unit codec tests cover legacy `{`, Unicode, malformed UTF-8, typed round trip, additive fields,
|
||||
malformed JSON, missing fields, identity mismatch, media type, version, and dedup.
|
||||
12. A live broker contract writes old wire data and reads it with the new adapter after reconnect.
|
||||
13. Live contract tests cover mixed entries, quarantine confirm-before-ack, unsupported redelivery,
|
||||
later progress past poison, property persistence, lead ownership, and ack removal.
|
||||
14. Safe downgrade is documented as unsupported.
|
||||
15. RabbitMQ contract tests pass with `mvn test -Pcontract`. The same cases run once on production
|
||||
LavinMQ, or the release states that LavinMQ was not checked.
|
||||
16. Member incidents route to the exact delegating lead and never resolve a task rendezvous.
|
||||
17. `fleet_list` shows compact member health and capacity without pane content. Member
|
||||
`idleForSeconds` is present only when no accepted turn or inbox item exists.
|
||||
|
||||
### Unit 4 - Lead health and peer routing
|
||||
|
||||
Scope: lead evidence, exact ownership, peer selection, explicit-recipient push, and lead inbox
|
||||
lifecycle.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Lead identity uses the `CallerResolver` supplier. Liveness uses successful current agent data.
|
||||
2. Two successful-list absences with healthy ping become `LEAD_UNREACHABLE`; global link failure does
|
||||
not.
|
||||
3. Raw `WORKING`, raw `UNKNOWN`, first-seen time, failures, last success, and error class persist
|
||||
across ticks.
|
||||
4. Heartbeat and push publish status and nudge outcomes before safe no-injection decisions.
|
||||
5. Dynamic lead identity survives a two-successful-snapshot retirement grace.
|
||||
6. Member incidents first use exact delegation ownership with no singular-primary fallback.
|
||||
7. Peer selection follows the exclusions, load rule, and stable tie break in Section 7.4.
|
||||
8. A selected working peer is not interrupted. Its push waits for an injectable window.
|
||||
9. Recipient assignment stays pinned. Reassignment increments generation and supersedes old pending
|
||||
assignment.
|
||||
10. `fleet_list` shows bounded foreign assignments, recipient, reason, and generation without pane
|
||||
content.
|
||||
11. `LeadInboxRegistry` owns configured and discovered lead keys before publication.
|
||||
12. Missing leads keep ownership. Retirement needs an empty queue and handled incidents.
|
||||
13. Replacement owns the new key before messages move. Non-empty in-memory keys are not released.
|
||||
14. Tests cover dead versus busy, unknown, global failure, stale scan cache, disappearance, one peer,
|
||||
several peers, reassignment, and no peer.
|
||||
15. Adapter tests cover lead ownership, restart re-ownership, retirement, and terminal replacement.
|
||||
LavinMQ is checked or named as unchecked.
|
||||
16. A sole unreachable or stalled lead is never restarted or replaced. Without a sink, only passive
|
||||
evidence remains and every coverage surface says so.
|
||||
|
||||
### Unit 5 - Human sink, hot config, metrics, and operator coverage
|
||||
|
||||
Scope: generic webhook, config split, incident journal, retry, resolve, metrics, example config, and
|
||||
operator documentation.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. `health.enabled` works without a human sink.
|
||||
2. Notification mode is hot, defaults to disabled, and supports disabled or webhook.
|
||||
3. Webhook mode requires a resolved environment value. Bad notification config does not disable an
|
||||
already valid detector.
|
||||
4. Mode changes keep open incidents. Re-enable resumes eligible incidents.
|
||||
5. `fleet_list`, `/healthz`, metrics, and one WARN show partial coverage without a sink. HTTP
|
||||
liveness behavior stays unchanged.
|
||||
6. One-lead, no-sink coverage states that lead failure has no active notification or recovery.
|
||||
7. Incident and outbound dedupe use the stable keys in Section 10.3.
|
||||
8. The owner-only local journal survives restart and contains no pane or task content.
|
||||
9. Journal failure keeps detection running but marks notification coverage degraded.
|
||||
10. Retry tests cover network failure, timeout, 408, 429, `Retry-After`, 5xx, permanent 4xx, jitter,
|
||||
delay cap, config re-arm, and one outstanding attempt.
|
||||
11. Reminders and transport retries remain separate. Disabled mode does not build an unbounded queue.
|
||||
12. Resolve sends only after an earlier human event succeeded. Resolve-before-delivery cancels stale
|
||||
open delivery.
|
||||
13. Metrics use bounded labels and exclude ids, URLs, and error text.
|
||||
14. Payload tests reject every content type forbidden in Section 10.5.
|
||||
15. Webhook response bodies are ignored and cannot direct recovery.
|
||||
16. Tests cover disabled mode, one lead without sink, open/update/reminder/resolve, restart, dedup,
|
||||
reassignment, disable/re-enable, and sink failure while local health continues.
|
||||
17. `fleetd.example.yaml` documents all hot keys and compiled floors.
|
||||
18. The operator Features wiki is updated separately. The portable `CLAUDE.md` block is checked and
|
||||
changed only if shipped tool or inbox semantics make it untrue.
|
||||
19. `mvn clean install` passes.
|
||||
|
||||
## 13. Not checked and release gates
|
||||
|
||||
These limits are part of the design, not optional follow-up notes.
|
||||
|
||||
- **OpenCode pane status and assistant markers were not checked.** OpenCode lost-boundary repair is
|
||||
disabled until live fixtures exist.
|
||||
- **Permission-prompt status was not checked** for Claude Code or OpenCode. `BLOCKED` remains
|
||||
ambiguous and has no automatic action.
|
||||
- **`recent_unwrapped` stability was not checked** across all supported agent kinds. If normalisation
|
||||
is not stable, `STALL_SUSPECTED` must say its evidence is weaker.
|
||||
- **The real `BUSY + DONE` trace was not replayed against live herdr.** The design uses the observed
|
||||
production trace and current poller behavior.
|
||||
- **CB-568 was not present when Unit 2 was designed.** Unit 2 must inspect the landed API and keep its
|
||||
independent teardown invariant.
|
||||
- **Production LavinMQ was not checked.** Existing durable-inbox contracts use RabbitMQ. Migration,
|
||||
quarantine, redelivery, lead ownership, and reassignment must run on LavinMQ before release or be
|
||||
recorded as unchecked.
|
||||
- **Live multi-lead routing was not checked.** Peer choice and reassignment are design rules backed by
|
||||
fake-clock and adapter tests until a live exercise runs.
|
||||
- **A live sole-lead failure with a webhook was not checked.** The no-peer path is a design result,
|
||||
not a tested recovery.
|
||||
- **No n8n, Slack, PagerDuty, or other receiver was checked.** The webhook remains generic and
|
||||
outbound-only.
|
||||
- **Deployment supervisor behavior for nested `/healthz` fields was not checked.** HTTP liveness
|
||||
status stays unchanged to reduce this risk.
|
||||
- **Incident-journal crash behavior was not checked** because the journal does not exist yet. Unit 5
|
||||
must test atomic replacement and restart recovery.
|
||||
- **Worktree merge state cannot be checked reliably** without forge or explicit collection evidence.
|
||||
`WORK_PRODUCT_AT_RISK` stays a warning.
|
||||
|
||||
## 14. Locked exclusions
|
||||
|
||||
M4 does not expose `agent.read` as a bridge tool. It does not add a workflow engine, inbound n8n
|
||||
authority, automatic task assignment, task replay, automatic lead replacement, or automatic member
|
||||
spawn for free capacity.
|
||||
|
||||
The bridge remains a message bus with evidence and bounded mechanical repair. The lead remains the
|
||||
place where judgement and work planning happen.
|
||||
+95
-80
@@ -1,11 +1,26 @@
|
||||
# MCP Contract — `bridged`'s unified gateway
|
||||
# MCP Contract — `fleetd`'s unified gateway
|
||||
|
||||
> **Status:** 🟡 Design (2026-07-14). Greenfield — no MCP code exists yet; the pom carries
|
||||
> only Javalin/Jackson. This page defines the tool surface that CB-104 and its followers
|
||||
> implement. It supersedes nothing; it fills the "MCP server face" left open by the
|
||||
> [Architecture](1-Architecture) page.
|
||||
> **Status: 🔴 HISTORICAL DESIGN — do NOT use as the tool reference.** Written 2026-07-14, before
|
||||
> any MCP code existed. The system shipped and this page never caught up, so **its tool names,
|
||||
> parameter names and REST paths are wrong today**. Audited 2026-08-17; the specific drift:
|
||||
>
|
||||
> - **Tools it names that do not exist:** `fleet_read`, `fleet_cancel`.
|
||||
> - **Shipped tools it omits:** `fleet_poll`, `fleet_ack`, `fleet_profiles`, `fleet_whoami`.
|
||||
> - **Parameter names are wrong nearly everywhere** — it says `message`/`target`/`timeout_seconds`/
|
||||
> `block` where the code takes `content`/`sessionId`/`timeoutMs`/`wait`; `text` where
|
||||
> `fleet_reply` takes `content`; `target` where `fleet_stop` takes `paneId`.
|
||||
> - **REST paths are wrong:** it says `POST /workers` and `DELETE /workers/{paneId}`; the daemon
|
||||
> serves `POST /members` and `DELETE /members/{paneId}`.
|
||||
>
|
||||
> **The authoritative tool surface is the live MCP schema** (each tool's own description and
|
||||
> parameters, as mounted), with the intent→tool table in `CLAUDE.md` as the short form. Both were
|
||||
> checked against `mcp/FleetMcp.java` on 2026-08-17 and are accurate.
|
||||
>
|
||||
> What is still worth reading here is **§6 — the flows and the error model** (rendezvous,
|
||||
> `fleet_ask`, detached delivery, the turn-done fallback). The shapes it describes are the ones
|
||||
> that shipped; only the names around them drifted. Rewriting this page is tracked as **CB-609**.
|
||||
|
||||
`bridged` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
`fleetd` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
@@ -20,18 +35,18 @@ semantics, and how it maps onto the code already in the tree.
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `bridged` must decide *who is calling* from the connection —
|
||||
catalog must serve both, and `fleetd` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `bridged` holds open — never a cross-turn poll loop that would burn
|
||||
*single* MCP call that `fleetd` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`bridged` owns policy; herdr owns PTYs.** MCP tools express *intent*; `bridged`
|
||||
5. **`fleetd` owns policy; herdr owns PTYs.** MCP tools express *intent*; `fleetd`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
@@ -44,8 +59,8 @@ face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients an
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["bridged — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>bridge_send · bridge_reply<br/>bridge_ask · bridge_status · lifecycle"]
|
||||
subgraph BD["fleetd — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>fleet_send · fleet_reply<br/>fleet_ask · fleet_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
@@ -57,8 +72,8 @@ flowchart LR
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"bridge_send (blocks)"| MCP
|
||||
W -.->|"bridge_reply / bridge_ask"| MCP
|
||||
OPUS -->|"fleet_send (blocks)"| MCP
|
||||
W -.->|"fleet_reply / fleet_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
@@ -72,18 +87,18 @@ flowchart LR
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `bridged` resolves the caller's role on every
|
||||
Because the same server is mounted by everyone, `fleetd` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `bridged` spawns every worker
|
||||
- **Workers are known.** `fleetd` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `bridge_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `bridge_reply` / `bridge_ask` on the same identity resolves that
|
||||
- **Turn correlation.** A blocking `fleet_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `fleet_reply` / `fleet_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
@@ -91,13 +106,13 @@ request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`bridged` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
`fleetd` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http bridged http://127.0.0.1:8080/mcp
|
||||
claude mcp add --transport http fleetd http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
@@ -108,65 +123,65 @@ This adds an MCP-server dependency the pom does not yet carry. See [Open decisio
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`bridge_send`](#bridge_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`bridge_reply`](#bridge_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`bridge_ask`](#bridge_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`bridge_status`](#bridge_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`bridge_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`bridge_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`bridge_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`bridge_read`](#bridge_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`bridge_cancel`](#bridge_cancel) | primary | no | — ❌ (future) |
|
||||
| [`fleet_send`](#fleet_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`fleet_reply`](#fleet_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`fleet_ask`](#fleet_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`fleet_status`](#fleet_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`fleet_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`fleet_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`fleet_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`fleet_read`](#fleet_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`fleet_cancel`](#fleet_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `bridge_send`
|
||||
#### `fleet_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `bridge_ask`).
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `fleet_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `bridge_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `bridge_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker calls `fleet_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `fleet_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `bridge_status` on a split-host primary.
|
||||
via `fleet_status` on a split-host primary.
|
||||
|
||||
#### `bridge_reply`
|
||||
#### `fleet_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `bridged` **injects the primary's idle pane** instead.
|
||||
waiter exists (detached delegation), `fleetd` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `bridge_ask`
|
||||
#### `fleet_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `bridge_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `bridge_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
open `fleet_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `fleet_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`bridge_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
- **`fleet_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`bridge_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`bridge_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
- **`fleet_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`fleet_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `bridge_status`
|
||||
#### `fleet_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
@@ -175,7 +190,7 @@ Thin adapters over [`WorkerService`](1-Architecture) — parity with the existin
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `bridge_read`
|
||||
#### `fleet_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
@@ -184,7 +199,7 @@ Thin adapters over [`WorkerService`](1-Architecture) — parity with the existin
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `bridge_cancel`
|
||||
#### `fleet_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
@@ -201,38 +216,38 @@ One blocking call, zero polls.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as bridged (MCP + Injector)
|
||||
participant B as fleetd (MCP + Injector)
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: bridge_reply("result")
|
||||
W->>B: fleet_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`bridge_ask`)
|
||||
### 6.2 Clarification — reverse rendezvous (`fleet_ask`)
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: bridge_ask("which config?") — worker blocks
|
||||
W->>B: fleet_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: bridge_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve bridge_ask → { answer:"config.yaml" }
|
||||
P->>B: fleet_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve fleet_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: bridge_reply("done")
|
||||
W->>B: fleet_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
```
|
||||
|
||||
@@ -243,13 +258,13 @@ The primary does not block; the reply arrives later in its idle pane.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
|
||||
P->>B: bridge_send("do X", target=w, block=false)
|
||||
P->>B: fleet_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: bridge_reply("result")
|
||||
W->>B: fleet_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
@@ -257,18 +272,18 @@ sequenceDiagram
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
|
||||
A worker that never calls `bridge_reply` still returns a result: `bridged` reads its terminal
|
||||
A worker that never calls `fleet_reply` still returns a result: `fleetd` reads its terminal
|
||||
tail when the turn completes.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls bridge_reply
|
||||
W->>W: works, never calls fleet_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
@@ -279,7 +294,7 @@ sequenceDiagram
|
||||
## 7. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `bridge_send` is simply its producer.
|
||||
enforces via `AgentStatus.injectable()`; MCP `fleet_send` is simply its producer.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
@@ -315,17 +330,17 @@ touching this state machine.
|
||||
|
||||
## 8. Error model
|
||||
|
||||
| Condition | `bridge_send` result | Notes |
|
||||
| Condition | `fleet_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `bridge_send(turn_id)` |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `fleet_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`bridge_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
`fleet_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
@@ -337,27 +352,27 @@ seam. Only the **rendezvous registry** and the **caller-identity resolver** are
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `bridge_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `bridge_reply` / `bridge_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `bridge_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `bridge_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `bridge_read` | `AgentControl.read` | MCP adapter only |
|
||||
| `fleet_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `fleet_reply` / `fleet_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `fleet_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `fleet_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `fleet_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `BridgedApp` already exercise the collaborators, MCP tools are
|
||||
Because the REST routes in `FleetApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`bridge_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
1. **`fleet_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `bridge_send` (keeps the catalog
|
||||
small) vs. a separate `bridge_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `bridge_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `bridge_spawn` first. **Recommend
|
||||
2. **Detached delivery shape.** A `block:false` param on `fleet_send` (keeps the catalog
|
||||
small) vs. a separate `fleet_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `fleet_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `fleet_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
@@ -366,9 +381,9 @@ validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `bridge_send` + rendezvous registry + caller-identity resolver
|
||||
- **CB-104** — blocking `fleet_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `bridge_reply` / `bridge_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`bridge_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — `fleet_reply` / `fleet_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`fleet_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `bridge_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
- **Later** — `fleet_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
# v1.0.0 — One leader, one host, complete
|
||||
|
||||
This is the first release of **`bridged`**.
|
||||
This is the first release of **`fleetd`**.
|
||||
|
||||
`bridged` lets one main Claude Code session (the **leader**, on your Pro/Max subscription)
|
||||
`fleetd` lets one main Claude Code session (the **leader**, on your Pro/Max subscription)
|
||||
run a team of **workers** — extra Claude Code sessions on a cheaper or local model, and
|
||||
non-Claude agents too. The leader's own session is never touched: it stays on subscription,
|
||||
with a clean environment.
|
||||
@@ -19,12 +19,12 @@ of this one.
|
||||
## One gateway for all messages
|
||||
|
||||
- **Everyone talks through the same door.** The leader and every worker connect to the same
|
||||
MCP server and use only its tools: `bridge_whoami` · `bridge_profiles` · `bridge_spawn` ·
|
||||
`bridge_list` · `bridge_status` · `bridge_send` · `bridge_reply` · `bridge_ask` ·
|
||||
`bridge_poll` · `bridge_ack` · `bridge_stop`.
|
||||
MCP server and use only its tools: `fleet_whoami` · `fleet_profiles` · `fleet_spawn` ·
|
||||
`fleet_list` · `fleet_status` · `fleet_send` · `fleet_reply` · `fleet_ask` ·
|
||||
`fleet_poll` · `fleet_ack` · `fleet_stop`.
|
||||
- **You are who your connection says you are.** The bridge finds out who is calling from the
|
||||
connection itself, never from a name the caller sends. So a worker cannot pretend to be
|
||||
someone else, and `bridge_whoami` tells each agent its own role — no guessing.
|
||||
someone else, and `fleet_whoami` tells each agent its own role — no guessing.
|
||||
- **The subscription line cannot be crossed.** Only a spawned worker gets
|
||||
`ANTHROPIC_BASE_URL`; the leader never does. Each worker profile has a list of allowed
|
||||
model hosts, checked before anything starts.
|
||||
@@ -59,7 +59,7 @@ had to ask for its replies. That gap is now closed on a single machine:
|
||||
- **The leader gets a tap on the shoulder.** When a reply lands, the bridge nudges the
|
||||
leader's own pane — only when the leader is free, and only a few times. If the leader is on
|
||||
another machine, this quietly falls back to pick-up mode; the reply still waits.
|
||||
- **Workers can ask questions.** With `bridge_ask`, a worker can pause mid-task, ask the
|
||||
- **Workers can ask questions.** With `fleet_ask`, a worker can pause mid-task, ask the
|
||||
leader something, and continue the *same* task with the answer.
|
||||
|
||||
## More than one kind of worker
|
||||
|
||||
+21
-21
@@ -1,9 +1,9 @@
|
||||
# Team — lead orchestrating a mixed Claude + local-LLM fleet
|
||||
|
||||
The message server (`bridged`) delivers **one turn into one worker**. A **team** is the
|
||||
The message server (`fleetd`) delivers **one turn into one worker**. A **team** is the
|
||||
layer above it: a **Claude team-lead** that fans a job out across a **mixed fleet** of
|
||||
workers — some on Claude, some on the remote local LLM — and reduces their replies. Same
|
||||
`bridged` delivery, same subscription boundary; this doc is only about **orchestration** —
|
||||
`fleetd` delivery, same subscription boundary; this doc is only about **orchestration** —
|
||||
who the workers are, how the lead picks one, and how it runs many at once.
|
||||
|
||||
> Delivery mechanics (blocking `POST /message`, status-gated reply envelope) live in the
|
||||
@@ -12,12 +12,12 @@ who the workers are, how the lead picks one, and how it runs many at once.
|
||||
## The team
|
||||
|
||||
- **Team-lead** — the primary **Opus** (Claude Code, env **CLEAN**, on Pro/Max). Not a
|
||||
worker; a **thin client of `bridged`**. It plans, routes, dispatches, and integrates, and
|
||||
worker; a **thin client of `fleetd`**. It plans, routes, dispatches, and integrates, and
|
||||
never sets `ANTHROPIC_BASE_URL`.
|
||||
- **Workers** — a herd of `claude` panes in herdr, each an addressable `bridged` session
|
||||
- **Workers** — a herd of `claude` panes in herdr, each an addressable `fleetd` session
|
||||
with its **own model/env**:
|
||||
- **Claude workers** (clean env, e.g. Sonnet) — reasoning-heavy or high-accuracy subtasks.
|
||||
- **Local workers** (`ANTHROPIC_BASE_URL=https://ollama.ltms.dev`) — bulk, cheap, or
|
||||
- **Local workers** (`ANTHROPIC_BASE_URL=https://llm.ltms.dev/anthropic`) — bulk, cheap, or
|
||||
embarrassingly parallel subtasks.
|
||||
|
||||
Every worker is still a *real Claude Code process* (inherits `CLAUDE.md`, hooks, skills,
|
||||
@@ -28,14 +28,14 @@ MCP) — only its model differs. Scale each kind horizontally by adding panes.
|
||||
```mermaid
|
||||
flowchart TB
|
||||
LEAD["lead — Opus<br/>(Claude Code, env CLEAN)"]
|
||||
BD["bridged<br/>message server + router"]
|
||||
BD["fleetd<br/>message server + router"]
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
WC1["w-claude-1<br/>Sonnet · CLEAN"]
|
||||
WC2["w-claude-2<br/>Sonnet · CLEAN"]
|
||||
WL1["w-local-1<br/>ANTHROPIC_BASE_URL set"]
|
||||
WL2["w-local-2<br/>ANTHROPIC_BASE_URL set"]
|
||||
ANT["api.anthropic.com<br/>(Pro/Max)"]
|
||||
OLL["ollama.ltms.dev<br/>(local model)"]
|
||||
OLL["llm.ltms.dev<br/>(gateway to the local model)"]
|
||||
|
||||
LEAD -->|"blocking POST /message (target role)"| BD
|
||||
BD -->|"Unix socket · send_text · events.subscribe"| HERDR
|
||||
@@ -62,14 +62,14 @@ flowchart TB
|
||||
| `w-local-*` | `ANTHROPIC_BASE_URL` set | local LLM | task is bulk / cheap / embarrassingly parallel |
|
||||
|
||||
The lead applies this rubric itself, guided by its `CLAUDE.md` team charter (below). Worker
|
||||
selection is **policy in the lead**, not a `bridged` concern — `bridged` just delivers to
|
||||
selection is **policy in the lead**, not a `fleetd` concern — `fleetd` just delivers to
|
||||
the session the lead names.
|
||||
|
||||
## Subscription boundary in a team
|
||||
|
||||
Unchanged from the base architecture, and it scales with the fleet: **only local-worker
|
||||
panes** launch with `ANTHROPIC_BASE_URL`. The lead and every Claude worker stay env-clean on
|
||||
the subscription. `bridged` enforces which panes may carry the off-subscription env, so
|
||||
the subscription. `fleetd` enforces which panes may carry the off-subscription env, so
|
||||
adding workers never widens the boundary.
|
||||
|
||||
## Parallel fan-out (map / reduce)
|
||||
@@ -80,7 +80,7 @@ different workers at once, then results are gathered.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant L as lead (Opus)
|
||||
participant B as bridged
|
||||
participant B as fleetd
|
||||
participant WC as w-claude-1
|
||||
participant WL as w-local-1
|
||||
|
||||
@@ -100,26 +100,26 @@ sequenceDiagram
|
||||
```
|
||||
|
||||
- **Map:** the lead issues N concurrent blocking `POST /message` calls (one per subtask → its
|
||||
chosen worker). Each call blocks only *that* request; `bridged` holds it open until the
|
||||
chosen worker). Each call blocks only *that* request; `fleetd` holds it open until the
|
||||
worker's turn completes (status-gated) and returns the reply envelope.
|
||||
- **Reduce:** the lead collects the N envelopes and integrates. A slow local worker never
|
||||
blocks a fast Claude worker — wall-clock ≈ the slowest single subtask, not the sum.
|
||||
- **Detached / long jobs** use the async broker path instead of a held request (Channel 2 in
|
||||
the base architecture), so the lead never busy-polls across turns.
|
||||
|
||||
Fan-out is bounded by the herd size (pane count) and `bridged`'s concurrency policy, not by
|
||||
Fan-out is bounded by the herd size (pane count) and `fleetd`'s concurrency policy, not by
|
||||
the lead.
|
||||
|
||||
## Knowing the roster
|
||||
|
||||
The lead discovers its team from `bridged` (session list / roles) rather than hard-coding
|
||||
The lead discovers its team from `fleetd` (session list / roles) rather than hard-coding
|
||||
pane ids, so workers can be added or restarted without editing the lead. A minimal charter
|
||||
in the lead's `CLAUDE.md` turns Opus into the orchestrator:
|
||||
|
||||
```markdown
|
||||
## Your team (via bridged)
|
||||
You are the team-lead. Delegate through the bridged client — never launch workers yourself.
|
||||
Roster: ask bridged for current sessions/roles.
|
||||
## Your team (via fleetd)
|
||||
You are the team-lead. Delegate through the fleetd client — never launch workers yourself.
|
||||
Roster: ask fleetd for current sessions/roles.
|
||||
- w-claude-* — Claude Sonnet. Reasoning-heavy / high-accuracy subtasks.
|
||||
- w-local-* — remote local LLM. Bulk, cheap, or parallelizable subtasks.
|
||||
|
||||
@@ -133,22 +133,22 @@ tool instead of hand-rolling the HTTP request.
|
||||
|
||||
## What this layer does NOT change
|
||||
|
||||
- **Delivery** is still `bridged` → herdr `pane.send_text` + status events (Message-Server).
|
||||
- **Delivery** is still `fleetd` → herdr `pane.send_text` + status events (Message-Server).
|
||||
- **Completion timing** is still the worker status event; **reply content** still rides the
|
||||
worker `Stop`-hook envelope.
|
||||
- **Single-host** still applies: herdr's socket is local, so the whole herd lives on the
|
||||
`bridged` host. The lead may be remote — it only needs HTTP to `bridged`.
|
||||
`fleetd` host. The lead may be remote — it only needs HTTP to `fleetd`.
|
||||
|
||||
## Open questions
|
||||
|
||||
- **Routing intelligence:** rubric-in-`CLAUDE.md` (lead decides) vs. a `bridged` role-router
|
||||
- **Routing intelligence:** rubric-in-`CLAUDE.md` (lead decides) vs. a `fleetd` role-router
|
||||
(label-based). Start with the former; promote to the latter if routing logic grows.
|
||||
- **Backpressure:** per-role concurrency caps in `bridged` so a fan-out can't exhaust the
|
||||
- **Backpressure:** per-role concurrency caps in `fleetd` so a fan-out can't exhaust the
|
||||
local gateway.
|
||||
- **Result schema:** whether reply envelopes should carry structured metadata (worker, model,
|
||||
tokens) to help the lead's reduce step.
|
||||
|
||||
## Status
|
||||
|
||||
🟡 Design (2026-07-11). Orchestration layer over the selected `bridged` server; inherits
|
||||
🟡 Design (2026-07-11). Orchestration layer over the selected `fleetd` server; inherits
|
||||
herdr (chosen) + AgentAPI (fallback). Delivery unchanged — see the Message-Server design.
|
||||
|
||||
+10
-10
@@ -20,7 +20,7 @@ requirement, not a nice-to-have.**
|
||||
- **Worktree provisioned by the daemon** — `SessionManager` creates a dedicated git worktree +
|
||||
branch per session, **hydrates it to full config parity** (below), and tears it down on release.
|
||||
- **Worker opens its own PR** — the worker commits, pushes its branch, and opens the PR/MR itself,
|
||||
returning the PR URL in its `bridge_reply`.
|
||||
returning the PR URL in its `fleet_reply`.
|
||||
|
||||
## Why worktrees (the hazard being fixed)
|
||||
|
||||
@@ -76,7 +76,7 @@ worktree checks out anyway. Amber is the real gap — untracked local config the
|
||||
3. **Never overlay the git plumbing** — the worktree's own `.git` file/branch is what gives
|
||||
isolation; that's the *one* thing that must differ from the main tree.
|
||||
|
||||
The overlay set lives in config (`BridgedConfig.Worker.parityOverlay` — a list of repo-relative
|
||||
The overlay set lives in config (`FleetConfig.Worker.parityOverlay` — a list of repo-relative
|
||||
paths, with sane defaults) so it's auditable and per-repo tunable.
|
||||
|
||||
> **Trust note (deliberate).** Hydrating local config means the primary's local secrets/tokens
|
||||
@@ -103,7 +103,7 @@ sequenceDiagram
|
||||
Note over W: implement in the isolated worktree
|
||||
W->>G: git commit + git push (SSH, same user)
|
||||
W->>G: open PR (branch to main)
|
||||
W-->>P: bridge_reply (prUrl, branch, summary, tests)
|
||||
W-->>P: fleet_reply (prUrl, branch, summary, tests)
|
||||
P->>SM: release(paneId)
|
||||
SM->>G: git worktree remove wt
|
||||
Note over G: branch + PR persist for review/merge
|
||||
@@ -115,9 +115,9 @@ earlier `STATE.md` idea — a PR is reviewable, mergeable, and self-describing.*
|
||||
|
||||
## Infra facts (verified this session)
|
||||
|
||||
- **Remote:** `ssh://git@git.ltms.dev:2224/lms/claude-bridge.git` (gitea). Push is over **SSH** —
|
||||
- **Remote:** `ssh://git@git.ltms.dev:2224/fleet/fleetd.git` (gitea). Push is over **SSH** —
|
||||
a worker running as the same user with the same keys can `git push` **with no extra credential**.
|
||||
- **gitea is NOT in the project `.mcp.json`** (only `jetbrains`, `intellij-index`, `bridged`). The
|
||||
- **gitea is NOT in the project `.mcp.json`** (only `jetbrains`, `intellij-index`, `fleetd`). The
|
||||
primary's gitea MCP comes from a global/user config, so **workers do not inherit it**. A worker
|
||||
gets only the `bridge` MCP mounted (via `--mcp-config` launch flag).
|
||||
- **No gitea CLI** (`tea`) installed; `glab` is present but is the GitLab CLI (wrong backend).
|
||||
@@ -128,7 +128,7 @@ earlier `STATE.md` idea — a PR is reviewable, mergeable, and self-describing.*
|
||||
|
||||
| Option | Mechanism | Trade-off |
|
||||
|---|---|---|
|
||||
| **A. gitea REST + token** | Worker `curl`s `POST /api/v1/repos/lms/claude-bridge/pulls` with a scoped token injected by the daemon into the worker env | Minimal, no new server; token lives in the off-subscription worker's env (scope it tightly) |
|
||||
| **A. gitea REST + token** | Worker `curl`s `POST /api/v1/repos/fleet/fleetd/pulls` with a scoped token injected by the daemon into the worker env | Minimal, no new server; token lives in the off-subscription worker's env (scope it tightly) |
|
||||
| **B. mount gitea MCP into workers** | Add the gitea MCP to the worker's `--mcp-config` alongside `bridge` | Clean tool call, but the gitea MCP's own auth/token must be provisioned per worker; more moving parts |
|
||||
| **C. install `tea` CLI** | Worker runs `tea pr create` with a token | Another dependency to install + configure; same token question as A |
|
||||
|
||||
@@ -140,7 +140,7 @@ and the token is a single scoped secret the daemon injects like it already injec
|
||||
|
||||
- Off-subscription workers already *could* push (SSH, same user). The **incremental grant is
|
||||
PR-create**, i.e. a gitea API token.
|
||||
- Scope the token **minimally**: the `lms/claude-bridge` repo, `write:repository` (create branch +
|
||||
- Scope the token **minimally**: the `fleet/fleetd` repo, `write:repository` (create branch +
|
||||
PR), **not** merge/admin/org. A leaked token can open PRs, not merge them — the primary/human is
|
||||
still the merge gate.
|
||||
- Inject via the daemon (env var, e.g. `GITEA_TOKEN`), never written to the worker's config dir —
|
||||
@@ -153,11 +153,11 @@ and the token is a single scoped secret the daemon injects like it already injec
|
||||
|---|---|---|
|
||||
| Worktree provision/teardown | **CB-301 ext** — `SessionManager.acquire`/`release`; `WorkerSession` gains `worktree`, `branch` | daemon shells out to `git worktree add/remove` |
|
||||
| **Config-parity overlay** | **CB-301 ext** — `SessionManager.acquire`, after `git worktree add` | symlink/copy the `parityOverlay` set into the worktree so the worker is a full peer; **this is what makes worktrees viable, not a dead-end** |
|
||||
| Overlay config | `BridgedConfig.Worker.parityOverlay` — repo-relative paths, sane defaults | auditable, per-repo tunable; keep explicit + minimal (trust) |
|
||||
| Overlay config | `FleetConfig.Worker.parityOverlay` — repo-relative paths, sane defaults | auditable, per-repo tunable; keep explicit + minimal (trust) |
|
||||
| Branch naming | `worker/<ticket-slug>-<nonce>` off `main` (or a configured base) | one branch per session |
|
||||
| Commit + push + PR handoff | **CB-302** — worker-driven, guided by the skill | push = SSH; PR = option A |
|
||||
| Implementer skill | `.claude/skills/implementer/SKILL.md` | worktree-aware playbook (see below); mounts automatically since workers inherit repo cwd |
|
||||
| gitea token injection | `WorkerService` env + `BridgedConfig` | repo-scoped, minimal perms |
|
||||
| gitea token injection | `WorkerService` env + `FleetConfig` | repo-scoped, minimal perms |
|
||||
| PR review + merge | Primary (has gitea MCP + judgment) | merge on green; the human/primary gate stays |
|
||||
|
||||
## Implementer skill (outline)
|
||||
@@ -170,7 +170,7 @@ A worker-facing playbook (sibling to the existing `reviewer` skill):
|
||||
3. **Push** your branch (`git push -u origin HEAD`).
|
||||
4. **Open a PR** to `main` (option A `curl`, or the decided mechanism) with a title/body describing
|
||||
the change and referencing the ticket.
|
||||
5. **Reply** via `bridge_reply` with the **PR URL**, branch name, files changed, and test names —
|
||||
5. **Reply** via `fleet_reply` with the **PR URL**, branch name, files changed, and test names —
|
||||
that reply is the whole handoff.
|
||||
6. Do **not** merge; do **not** touch `.mcp.json` or `wiki/`.
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Worker startup: working directory & the folder-trust prompt
|
||||
|
||||
When `bridged` spawns a worker, the worker CLI may show an **interactive startup prompt** before it
|
||||
When `fleetd` spawns a worker, the worker CLI may show an **interactive startup prompt** before it
|
||||
is ready to accept a task — most importantly a *"Do you trust the files in this folder?"* dialog. An
|
||||
unattended worker parked on that prompt never becomes injectable: the status-gated injector waits for
|
||||
`idle`/`blocked`, the task is never delivered, and (worst case) a stray Enter answers the dialog
|
||||
@@ -14,11 +14,11 @@ rule that **a worker inherits the primary's directory** (never `$HOME`), and how
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A["bridge_spawn / POST /workers"] --> B{"explicit cwd?<br/>(profile cwd or spawn arg)"}
|
||||
A["fleet_spawn / POST /workers"] --> B{"explicit cwd?<br/>(profile cwd or spawn arg)"}
|
||||
B -->|"yes — told otherwise"| C["use that cwd"]
|
||||
B -->|"no"| D{"caller PID resolvable?<br/>(MCP peer PID)"}
|
||||
D -->|"yes"| E["cwd = the primary's cwd<br/>lsof -a -p PID -d cwd"]
|
||||
D -->|"no (REST / off-host)"| F["cwd = bridged daemon cwd<br/>(never $HOME by assumption)"]
|
||||
D -->|"no (REST / off-host)"| F["cwd = fleetd daemon cwd<br/>(never $HOME by assumption)"]
|
||||
C --> G["ensureWorkspace → tab.create → agent.start {cwd}"]
|
||||
E --> G
|
||||
F --> G
|
||||
@@ -51,10 +51,10 @@ only affect the seed shell, which the bridge closes).
|
||||
| # | Source | When |
|
||||
|---|--------|------|
|
||||
| 1 | Explicit `cwd` — a per-profile `cwd:` in config, or a spawn argument | "told otherwise" — pin a fixed workdir |
|
||||
| 2 | The **primary's cwd**, auto-detected from the `bridge_spawn` caller | normal MCP spawn from the primary |
|
||||
| 3 | The `bridged` daemon's own cwd | REST spawn / off-host caller — **never `$HOME`** |
|
||||
| 2 | The **primary's cwd**, auto-detected from the `fleet_spawn` caller | normal MCP spawn from the primary |
|
||||
| 3 | The `fleetd` daemon's own cwd | REST spawn / off-host caller — **never `$HOME`** |
|
||||
|
||||
The primary's cwd (source 2) is discoverable with no new plumbing: `bridged` already resolves the MCP
|
||||
The primary's cwd (source 2) is discoverable with no new plumbing: `fleetd` already resolves the MCP
|
||||
caller's loopback **peer PID** for connection identity (`ConnectionIdentity` → `LsofPeerPidLookup`);
|
||||
the same PID yields its cwd via `lsof -a -p <pid> -d cwd -Fn` (the `n…` line). The primary maps to no
|
||||
worker pane (it is not a worker), but its PID and cwd are still readable.
|
||||
@@ -62,10 +62,10 @@ worker pane (it is not a worker), but its PID and cwd are still readable.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as "Primary (main)"
|
||||
participant B as "bridged"
|
||||
participant B as "fleetd"
|
||||
participant O as "OS (lsof)"
|
||||
participant H as "herdr"
|
||||
P->>B: "bridge_spawn {profile} (no cwd)"
|
||||
P->>B: "fleet_spawn {profile} (no cwd)"
|
||||
B->>O: "peer PID for this connection's port"
|
||||
O-->>B: "pid"
|
||||
B->>O: "cwd of pid (lsof -d cwd)"
|
||||
@@ -77,9 +77,9 @@ sequenceDiagram
|
||||
|
||||
*Figure 2 — a no-cwd spawn inherits the primary's directory from the caller's PID.*
|
||||
|
||||
> **Status:** implemented (CB-112). `bridged` threads the resolved `cwd` onto **`agent.start {cwd}`**
|
||||
> **Status:** implemented (CB-112). `fleetd` threads the resolved `cwd` onto **`agent.start {cwd}`**
|
||||
> (verified: the worker process is rooted there), keeping the single shared worker space. On an MCP
|
||||
> `bridge_spawn` the primary's cwd is auto-detected from the caller's PID; over REST (no MCP caller)
|
||||
> `fleet_spawn` the primary's cwd is auto-detected from the caller's PID; over REST (no MCP caller)
|
||||
> it is the explicit `cwd` param else the daemon's cwd. Both placements (`tab` and legacy `pane`)
|
||||
> carry it, since it rides `agent.start`.
|
||||
|
||||
@@ -133,5 +133,5 @@ unattended.
|
||||
|
||||
## See also
|
||||
|
||||
- `docs/MCP-Contract.md` — the tool surface (`bridge_spawn`, `bridge_profiles`, …).
|
||||
- `docs/MCP-Contract.md` — the tool surface (`fleet_spawn`, `fleet_profiles`, …).
|
||||
- `wiki/2-Message-Server.md` — the herdr `agent.*` / `workspace.*` schema (`workspace.create {cwd}`).
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
# Wiki audit for #168
|
||||
|
||||
**Source checked:** `.wiki-snapshot/` at `68e32c6` (2026-08-31). I did not use
|
||||
`wiki/`. Code references below are from the current `fleetd` source tree. A quoted
|
||||
line is a concrete claim that needs correction, unless the table says `KEEP`.
|
||||
|
||||
| Page | Verdict | One-line reason |
|
||||
|---|---|---|
|
||||
| `Home.md` | REVISE | Good overview, but it still names the retired product. |
|
||||
| `_Sidebar.md` | REVISE | The heading still says `claude-bridge`. |
|
||||
| `1-Architecture.md` | REBUILD | Its component contract mixes current names with removed tools, routes, and planned backends. |
|
||||
| `2-Message-Server.md` | REBUILD | The claimed MCP schema, mount command, REST/SSE surface, and fallback paths are pre-build design. |
|
||||
| `3-Approaches.md` | REVISE | Useful research history, but it presents unbuilt AgentAPI as a selectable fallback. |
|
||||
| `4-Setup.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `5-Operations.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `6-Team.md` | REBUILD | It teaches role-addressed sends and a Claude-only team model that the shipped API does not have. |
|
||||
| `7-Use-Cases.md` | REBUILD | Its flagship flow depends on removed `ccs` profiles and removed send parameters. |
|
||||
| `8-Roadmap.md` | REBUILD | It is a historical plan, but it presents old implementation choices and planned work as the current stack. |
|
||||
| `9-Implementation.md` | REBUILD | Its package, class, endpoint, and outcome map has drifted from the source. |
|
||||
| `10-Cross-Host-Messaging.md` | REVISE | It labels most federation work proposed, but misses the shipped `coordinator:` lead channel. |
|
||||
| `11-Features.md` | REVISE | It is the right catalogue, but code-path names are old and it misses the second-herdr-daemon capability. |
|
||||
| `12-Claude-to-OpenCode.md` | REVISE | The porting guide is mostly current, but calls the product and spawned-member path a bridge. |
|
||||
| `13-User-Guide.md` | REVISE | It is the best operator page, but needs the product rename and the second-herdr-daemon setup. |
|
||||
|
||||
## Pages needing work
|
||||
|
||||
### `Home.md` — REVISE
|
||||
|
||||
- Quote: `# claude-bridge` (line 1) and `` `claude-bridge` keeps`` (line 11).
|
||||
The product is `fleet` / `fleetd`. The MCP server identifies itself as `fleet` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:313-315`.
|
||||
- Quote: `AgentAPI ... swappable fallback injector` (lines 73-76).
|
||||
There is no AgentAPI implementation under `fleetd/src/main/java`; the actual
|
||||
launchers are selected by `Profile.kind` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java:265-270`.
|
||||
|
||||
### `_Sidebar.md` — REVISE
|
||||
|
||||
- Quote: `### 📖 claude-bridge` (line 1).
|
||||
Rename it to `fleet`. `FleetMcp` registers the current product-facing tool set at
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:301-326`.
|
||||
|
||||
### `1-Architecture.md` — REBUILD
|
||||
|
||||
- Quote: `` `claude-bridge` lets`` (line 3). The product was renamed; the MCP
|
||||
server name is `fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: ``fleet_read`` in the tool list (line 102). No such tool is registered.
|
||||
The complete registered list is `fleet_send` through `fleet_whoami` at
|
||||
`FleetMcp.java:301-326`; `fleet_read` is absent.
|
||||
- Quote: `SSE (GET /events)` (line 143). `FleetApp.build()` registers no `/events`
|
||||
route; its routes are listed at `FleetApp.java:143-159`.
|
||||
- Quote: `Redis Streams / NATS JetStream, or an embedded queue` (line 106).
|
||||
The shipped durable inbox is AMQP, configured by `broker`, at
|
||||
`FleetConfig.java:49-50` and `FleetConfig.java:655-714`.
|
||||
- Quote: `AgentAPI (fallback)` (line 107). No AgentAPI adapter exists; shipped
|
||||
launcher kinds are `claude-code` and `opencode` (`FleetConfig.java:265-270`).
|
||||
|
||||
### `2-Message-Server.md` — REBUILD
|
||||
|
||||
- Quote: `claude mcp add --transport http bridge http://127.0.0.1:8080/mcp`
|
||||
(line 67). The daemon defaults to port `8765` in `FleetConfig.java:183-187`,
|
||||
and identifies its server as `fleet` at `FleetMcp.java:313-315`.
|
||||
- Quote: ``fleet_send(message, target?, {block, timeout_seconds, auto_spawn,
|
||||
turn_id})`` (line 80). The real parameters are `sessionId`, `content`,
|
||||
`timeoutMs`, `wait`, `turnId`, and `coordId` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: ``fleet_read(target, source)`` (line 85). It is not registered; see the
|
||||
complete registration at `FleetMcp.java:301-326`.
|
||||
- Quote: `docs/MCP-Contract.md ... normative` (lines 87-88). That is not a valid
|
||||
reference: only §6 is current, as the current operator guide itself says at
|
||||
`.wiki-snapshot/13-User-Guide.md:466`.
|
||||
- Quote: `SSE (GET /events)` (line 45). No route exists in the built REST surface,
|
||||
`FleetApp.java:143-159`.
|
||||
|
||||
### `3-Approaches.md` — REVISE
|
||||
|
||||
- Quote: `AgentAPI ... remains a swappable fallback injector` (lines 78-84).
|
||||
It was never built. The shipped adapter selection is only `claude-code` or
|
||||
`opencode` (`FleetConfig.java:265-270`). Keep it as discarded research, not an
|
||||
operational fallback.
|
||||
- Quote: `claude-bridge` (line 109). Rename the product to `fleet`; the runtime
|
||||
package is `dev.ltms.fleet`, for example `FleetMcp.java:1`.
|
||||
|
||||
### `4-Setup.md` — RETIRE
|
||||
|
||||
It is a 25-line redirect and says its procedure was never written (lines 3-9).
|
||||
Chapter 13 is the maintained install procedure. Keeping a second navigation page
|
||||
adds no working documentation.
|
||||
|
||||
### `5-Operations.md` — RETIRE
|
||||
|
||||
It is a 35-line redirect and says its runbook was never written (lines 3-14).
|
||||
Chapter 13 now owns run and recovery instructions.
|
||||
|
||||
### `6-Team.md` — REBUILD
|
||||
|
||||
- Quote: `fleet_send {role: w-claude, prompt: A}` (line 98). `fleet_send` accepts
|
||||
`sessionId` and `content`, not `role` or `prompt` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: `some on Claude, some on the remote local LLM` (lines 3-5) and `Every
|
||||
worker is ... Claude Code` (line 25). `opencode` is a first-class launcher kind,
|
||||
not a Claude worker (`FleetConfig.java:265-270`).
|
||||
- Quote: `fleetd's concurrency policy` (line 121). The configured capacity control
|
||||
is per-profile `maxLoad` (`FleetConfig.java:251-264`), not the role routing model
|
||||
described here.
|
||||
|
||||
### `7-Use-Cases.md` — REBUILD
|
||||
|
||||
- Quote: `ccs profile` (line 10), `ccs + herdr` (line 22), and `ccs-spawn`
|
||||
(line 45). The configuration has `profiles` and `fleet`, not `ccs`:
|
||||
`FleetConfig.java:34-58` and `FleetConfig.java:81-101`.
|
||||
- Quote: `fleet_send({"to", "kind", "body", "block"})` (lines 55-62).
|
||||
None of those are the shipped send parameters. The schema is
|
||||
`FleetMcp.java:1096-1108`.
|
||||
- Quote: `fleet_list() → { "profiles": ... }` (lines 74-80). `fleet_list` is a
|
||||
roster view; `fleet_profiles` is the configured-backend view, as registered at
|
||||
`FleetMcp.java:307-311` and described at `FleetMcp.java:1176-1182`.
|
||||
|
||||
### `8-Roadmap.md` — REBUILD
|
||||
|
||||
- Quote: `Java 21+` (line 43). The current project guidance and source use Java 25;
|
||||
the `FleetConfig` source itself uses Java 25 unnamed lambda parameters, for
|
||||
example `FleetConfig.java:102`.
|
||||
- Quote: `herdr 0.7.0 / protocol 14` (line 46). The current REST health endpoint
|
||||
reports the live protocol returned by herdr (`FleetApp.java:240-244`), while the
|
||||
current operator guide records protocol 19 at
|
||||
`.wiki-snapshot/13-User-Guide.md:76-85`.
|
||||
- Quote: `ccs <profile> claude` and `ccs env <profile>` (lines 47-48). Shipped
|
||||
configuration uses `Profile` records and launcher `kind`,
|
||||
`FleetConfig.java:313-330` and `FleetConfig.java:265-270`.
|
||||
- Quote: `Redis Streams via Lettuce` (line 50). The actual durable inbox is AMQP
|
||||
`broker`, `FleetConfig.java:655-714`.
|
||||
|
||||
### `9-Implementation.md` — REBUILD
|
||||
|
||||
- Quote: `rest.FleetdApp` and `mcp.BridgeMcp` (lines 29-30). The classes are
|
||||
`rest.FleetApp` and `mcp.FleetMcp` (`FleetApp.java:46`; `FleetMcp.java:67`).
|
||||
- Quote: `dev.ltms.fleetd` (line 67). The source package is `dev.ltms.fleet`
|
||||
(`FleetMcp.java:1`).
|
||||
- Quote: `WorkerPresence` (line 110). The current class is `MemberPresence`, as
|
||||
imported and used by `FleetMcp` at `FleetMcp.java:12` and `465-469`.
|
||||
- Quote: the outcome list ending in `STALE_TURN` (lines 128-131). The code also
|
||||
has `BACKEND_EXHAUSTED` (`FleetMcp.java:550-554`) and async `ASKING` handling
|
||||
(`FleetMcp.java:664-668`).
|
||||
- Quote: `FleetdApp` (line 207) and `FleetdConfig` (line 211). These names do not
|
||||
resolve; current classes are `FleetApp` and `FleetConfig`.
|
||||
|
||||
### `10-Cross-Host-Messaging.md` — REVISE
|
||||
|
||||
- Quote: the chapter says the cross-host fabric is proposed except for the
|
||||
single-host inbox (lines 3-8). Cross-host **lead-to-lead** delivery shipped:
|
||||
`fleet_send` accepts `coordId` (`FleetMcp.java:1094-1107`) and publishes it at
|
||||
`FleetMcp.java:616-641`; configuration has `coordinator` at
|
||||
`FleetConfig.java:74-78` and `99-101`.
|
||||
- Quote: `bridge.dlx` (line 90). This product name is stale. The shipped lead path
|
||||
uses `LeadChannel`, not the proposed exchange flow (`FleetMcp.java:95-96` and
|
||||
`616-641`). Keep the proposed federation design, but add a clear shipped/proposed
|
||||
boundary for CB-637.
|
||||
|
||||
### `11-Features.md` — REVISE
|
||||
|
||||
- Quote: `mcp/BridgeMcp` (line 22), `config/FleetdConfig` (lines 25-27), and other
|
||||
index references. These paths no longer resolve; the source classes are
|
||||
`mcp/FleetMcp` (`FleetMcp.java:67`) and `config/FleetConfig`
|
||||
(`FleetConfig.java:81`).
|
||||
- Quote: `fleet_whoami` returns only `primary` or `worker` (lines 99-100).
|
||||
It also returns `architect` (`FleetMcp.java:1235-1244`).
|
||||
- The page needs the missing separate member-herdr-daemon feature listed below.
|
||||
|
||||
### `12-Claude-to-OpenCode.md` — REVISE
|
||||
|
||||
- Quote: `same bridge mount` (line 5) and `a bridge-spawned worker` (line 94).
|
||||
Rename the product path to `fleet`. The daemon exposes the MCP server as `fleet`
|
||||
(`FleetMcp.java:313-315`), and profiles select OpenCode with `kind: opencode`
|
||||
(`FleetConfig.java:332-335`).
|
||||
- Quote: the sample mount name is `fleetd` (line 67). The server name is `fleet`;
|
||||
update the sample to avoid teaching a second product name.
|
||||
|
||||
### `13-User-Guide.md` — REVISE
|
||||
|
||||
- Quote: `The bridge is the only channel` (line 63). The invariant is correct, but
|
||||
the product term needs the `fleet` rename. The daemon's MCP server name is
|
||||
`fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: it describes one herdr socket (lines 72-85). It needs the optional
|
||||
`memberHerdrSocket` setup and two-daemon health meaning. The config key is in
|
||||
`FleetConfig.java:34-37`, and `/healthz` checks both daemons when configured at
|
||||
`FleetApp.java:210-245`.
|
||||
|
||||
## MISSING
|
||||
|
||||
`11-Features.md` has a body section for **routing members through a separate herdr daemon**
|
||||
(`## memberHerdrSocket`, line 2174), but **no row in the index table** at the top of the page
|
||||
(lines 20-95). That table is how the page is meant to be read, so a capability absent from it is
|
||||
effectively undiscoverable. Lead note: this is my own omission — I added the section on 2026-08-31
|
||||
and did not add the matching row. Fixed in the wiki at `68e32c6`'s successor.
|
||||
|
||||
The original audit stated the feature had no entry at all. That was wrong: the section exists. The
|
||||
gap is the index row. Recorded here rather than silently corrected, because the difference matters —
|
||||
"undocumented" and "documented but unindexed" are different jobs.
|
||||
|
||||
Evidence for the feature itself: `FleetConfig.java:34-37` and `FleetApp.java:103-115`, `210-245`,
|
||||
and `247-263`.
|
||||
|
||||
## Audit method and coverage
|
||||
|
||||
I checked all 15 pages. I checked concrete tool, route, config, class, file, and
|
||||
product-name claims claim-by-claim on 11 pages: Home, Sidebar, 1, 2, 4, 5, 6, 7, 9,
|
||||
11, and 13. I skimmed the remaining four long historical or research pages (3, 8, 10,
|
||||
12), then checked their concrete claims that affect the verdict. This is an audit of
|
||||
the supplied snapshot, not a wiki rewrite.
|
||||
+6
-6
@@ -2,11 +2,11 @@
|
||||
|
||||
A standard, repeatable **live** end-to-end test of the two-way channel: it drives a real
|
||||
multi-turn conversation between a primary and an off-subscription worker **through the
|
||||
running `bridged` daemon**, captures the full transcript, and grades the channel.
|
||||
running `fleetd` daemon**, captures the full transcript, and grades the channel.
|
||||
|
||||
This is the committed form of the ad-hoc channel test that discovered the CB-115 gaps
|
||||
(herdr `unknown` misclassification wedging delivery, dirty completion scrapes, and workers
|
||||
never calling `bridge_reply` in conversation). Run it after any change to the injector,
|
||||
never calling `fleet_reply` in conversation). Run it after any change to the injector,
|
||||
status handling, completion/failure paths, or the worker reply charter.
|
||||
|
||||
## What it exercises
|
||||
@@ -20,7 +20,7 @@ construction** — it only calls the bridge's loopback REST face.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant T as conversation_test.py
|
||||
participant B as bridged (REST)
|
||||
participant B as fleetd (REST)
|
||||
participant W as worker (off-sub)
|
||||
T->>B: POST /workers (spawn)
|
||||
T->>B: GET /sessions/{id}/status (await ready)
|
||||
@@ -28,7 +28,7 @@ sequenceDiagram
|
||||
T->>B: POST /sessions/{id}/message {wait:false}
|
||||
B-->>T: ticket
|
||||
B->>W: inject prompt (status-gated)
|
||||
W-->>B: bridge_reply
|
||||
W-->>B: fleet_reply
|
||||
T->>B: GET /tasks/{ticket} (poll)
|
||||
B-->>T: done + reply
|
||||
end
|
||||
@@ -37,7 +37,7 @@ sequenceDiagram
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- `bridged` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
- `fleetd` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
profile configured and its backend reachable.
|
||||
- herdr is up (the daemon needs it).
|
||||
- Python 3 (standard library only — no pip installs).
|
||||
@@ -68,7 +68,7 @@ Per-turn grade:
|
||||
|
||||
| Grade | Meaning |
|
||||
|------------|---------------------------------------------------------------------|
|
||||
| `OK` | delivered and resolved by an explicit `bridge_reply` (`source=reply`) |
|
||||
| `OK` | delivered and resolved by an explicit `fleet_reply` (`source=reply`) |
|
||||
| `DEGRADED` | delivered and answered, but resolved via completion-scrape fallback |
|
||||
| `EMPTY` | turn completed but the reply was empty |
|
||||
| `FAILED` | the worker's turn ended in failure (`phase=failed`) |
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Sustained back-and-forth bridge test — ONE primary, ONE worker, many dependent turns
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `bridged` daemon.
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `fleetd` daemon.
|
||||
|
||||
Where conversation_test.py proves a handful of turns work and issue_hunt_test.py proves
|
||||
fan-out isolation, this proves the channel stays healthy under a *sustained, stateful*
|
||||
@@ -24,7 +24,7 @@ Usage:
|
||||
--turn-timeout per-turn max wait, seconds (default 150)
|
||||
--keep-worker do not stop the worker at the end
|
||||
|
||||
Exit code: 0 if every turn in the window resolved via a clean bridge_reply with no channel
|
||||
Exit code: 0 if every turn in the window resolved via a clean fleet_reply with no channel
|
||||
break; 1 otherwise. A live per-turn log streams to stdout so the run can be watched.
|
||||
"""
|
||||
import argparse
|
||||
@@ -47,12 +47,12 @@ STEPS = [7, 3, 11, 5, 9, 4, 13, 6, 8, 2]
|
||||
RULES = (
|
||||
"Let's play a running-total game across several messages. The total starts at 0. "
|
||||
"In each message I'll tell you to add a number; keep the running total yourself and "
|
||||
"reply via bridge_reply with ONLY the current total as a plain integer — no words, no "
|
||||
"reply via fleet_reply with ONLY the current total as a plain integer — no words, no "
|
||||
"punctuation, just the number. Do not restate the arithmetic. First move: add {step}."
|
||||
)
|
||||
NEXT = ("Add {step}. Reply via bridge_reply with only the new running total.")
|
||||
NEXT = ("Add {step}. Reply via fleet_reply with only the new running total.")
|
||||
REANCHOR = ("Let's re-sync — the running total is {total}. Now add {step}. Reply via "
|
||||
"bridge_reply with only the new running total.")
|
||||
"fleet_reply with only the new running total.")
|
||||
|
||||
|
||||
def parse_int(reply):
|
||||
@@ -142,15 +142,15 @@ def main():
|
||||
print("=" * 72)
|
||||
print(f"SUSTAINED CONVERSATION SUMMARY — 1 primary <-> 1 worker over {dur}s (~{dur/60:.1f} min)")
|
||||
print(f" turns: {turns}")
|
||||
print(f" clean bridge_reply exchanges: {oks + drifts}/{turns} (channel breaks: {breaks})")
|
||||
print(f" clean fleet_reply exchanges: {oks + drifts}/{turns} (channel breaks: {breaks})")
|
||||
print(f" arithmetic correct (continuity held): {oks}/{turns} (drifts: {drifts})")
|
||||
print(f" latency: avg {avg}s over {turns} turns")
|
||||
ok = breaks == 0 and turns >= 2
|
||||
if ok and drifts == 0:
|
||||
print(" RESULT: PASS — every turn resolved via bridge_reply and the worker held the "
|
||||
print(" RESULT: PASS — every turn resolved via fleet_reply and the worker held the "
|
||||
"running total across the whole window.")
|
||||
elif ok:
|
||||
print(f" RESULT: PASS (channel) — every turn resolved via bridge_reply for the full "
|
||||
print(f" RESULT: PASS (channel) — every turn resolved via fleet_reply for the full "
|
||||
f"window; {drifts} arithmetic drift(s) (worker recovered after re-anchor).")
|
||||
else:
|
||||
print(" RESULT: FAIL — the channel broke on at least one turn (see CHANNEL BREAK above).")
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge conversation test — a multi-turn primary↔worker exchange through
|
||||
the running `bridged` daemon, fully captured, with automatic gap analysis.
|
||||
the running `fleetd` daemon, fully captured, with automatic gap analysis.
|
||||
|
||||
This is the repeatable form of the ad-hoc channel test that surfaced the CB-115 gaps
|
||||
(herdr `unknown` misclassification, dirty completion scrape, workers not calling
|
||||
bridge_reply). It drives a real off-subscription worker over the live gateway exactly
|
||||
fleet_reply). It drives a real off-subscription worker over the live gateway exactly
|
||||
as a primary Opus session would (async fire-and-poll), records every turn, and grades
|
||||
the channel.
|
||||
|
||||
@@ -130,9 +130,9 @@ def grade(rec):
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit bridge_reply"
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call bridge_reply)"
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
@@ -206,7 +206,7 @@ def main():
|
||||
ok = all(g in ("OK", "DEGRADED") for g in grades)
|
||||
reply_clean = all(g == "OK" for g in grades)
|
||||
if reply_clean:
|
||||
print(" RESULT: PASS — every turn delivered and got a clean bridge_reply.")
|
||||
print(" RESULT: PASS — every turn delivered and got a clean fleet_reply.")
|
||||
elif ok:
|
||||
print(" RESULT: PASS (with notes) — every turn delivered & replied, but some via fallback.")
|
||||
else:
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Live bridge_ask test — the REVERSE rendezvous (CB-205), watched end to end.
|
||||
"""Live fleet_ask test — the REVERSE rendezvous (CB-205), watched end to end.
|
||||
|
||||
Every other harness drives the forward path: primary `bridge_send` → worker `bridge_reply`.
|
||||
Every other harness drives the forward path: primary `fleet_send` → worker `fleet_reply`.
|
||||
This drives the one that runs the other way. A worker is told to pause its delegated turn,
|
||||
ask the primary a question via `bridge_ask`, and only finish once it has the answer — so the
|
||||
ask the primary a question via `fleet_ask`, and only finish once it has the answer — so the
|
||||
turn round-trips primary→worker→primary→worker inside a SINGLE delegation.
|
||||
|
||||
The mechanics that only this path exercises:
|
||||
|
||||
• a worker's mid-turn question surfacing on the primary's *own* blocked send (Outcome.QUESTION),
|
||||
• the `turnId` correlation that lets the primary answer the exact paused turn,
|
||||
• the answer resuming that same turn and the worker's final `bridge_reply` landing on the
|
||||
• the answer resuming that same turn and the worker's final `fleet_reply` landing on the
|
||||
re-opened forward waiter (never a stale or cross-wired one).
|
||||
|
||||
It is two blocking REST calls, no polling:
|
||||
@@ -24,7 +24,7 @@ Like the rest of the suite it talks ONLY to the bridge's REST face on loopback
|
||||
ANTHROPIC_BASE_URL and never touches herdr, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 bridge_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
python3 fleet_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
[--send-timeout SECS] [--answer-timeout SECS] [--keep-worker]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
@@ -57,14 +57,14 @@ ANSWER_COLOR = "blue"
|
||||
|
||||
# A task that CANNOT be completed without asking: the worker is not told which color to choose,
|
||||
# only that the primary will name one when asked. So a correct final reply is only reachable by
|
||||
# actually calling bridge_ask and using the answer.
|
||||
# actually calling fleet_ask and using the answer.
|
||||
TASK_PROMPT = (
|
||||
"You are a bridge worker in a quick coordination game. You do NOT know which color to pick — "
|
||||
"only the primary does. Do exactly this, in order:\n"
|
||||
"1. Call the `bridge_ask` tool with EXACTLY this question: \"PICK A COLOR: red or blue?\"\n"
|
||||
"1. Call the `fleet_ask` tool with EXACTLY this question: \"PICK A COLOR: red or blue?\"\n"
|
||||
"2. The primary will answer with one color word. Take that color and uppercase it.\n"
|
||||
"3. Call `bridge_reply` with EXACTLY one line: CHOSEN=<COLOR> (e.g. CHOSEN=GREEN if told green).\n"
|
||||
"Do not guess a color. Do not call bridge_reply before bridge_ask has returned an answer. "
|
||||
"3. Call `fleet_reply` with EXACTLY one line: CHOSEN=<COLOR> (e.g. CHOSEN=GREEN if told green).\n"
|
||||
"Do not guess a color. Do not call fleet_reply before fleet_ask has returned an answer. "
|
||||
"Do nothing else — no file reads, no other tools."
|
||||
)
|
||||
|
||||
@@ -134,7 +134,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
rec.update(tid=tid, pane=pane, spawned=True)
|
||||
await_ready(base, tid)
|
||||
|
||||
# 1) Delegate the ask-forcing task. This blocks until the worker calls bridge_ask, at which
|
||||
# 1) Delegate the ask-forcing task. This blocks until the worker calls fleet_ask, at which
|
||||
# point our own send unblocks carrying the question and the turnId to answer on.
|
||||
print(f"[{now()}] delegating task (blocks until the worker asks; up to {send_timeout}s)…")
|
||||
t0 = time.time()
|
||||
@@ -159,7 +159,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
rec.update(phase="no_turnid", detail="question surfaced without a turnId to answer on")
|
||||
return rec
|
||||
|
||||
# 2) Answer on that exact turn. This blocks again until the resumed worker calls bridge_reply.
|
||||
# 2) Answer on that exact turn. This blocks again until the resumed worker calls fleet_reply.
|
||||
print(f"[{now()}] answering '{ANSWER_COLOR}' on turn {rec['turnId']} (blocks until reply; up to {answer_timeout}s)…")
|
||||
t1 = time.time()
|
||||
rec["phase"] = "awaiting_reply"
|
||||
@@ -188,7 +188,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
def grade(rec):
|
||||
"""PASS only if the worker asked, the turn resumed, and the reply reflects the answer."""
|
||||
if rec["phase"] == "no_question":
|
||||
return "NO_ASK", "the worker finished/stalled without ever calling bridge_ask"
|
||||
return "NO_ASK", "the worker finished/stalled without ever calling fleet_ask"
|
||||
if rec["phase"] in ("send_error", "answer_error", "spawn"):
|
||||
return "ERROR", rec.get("detail") or "transport error before the round-trip completed"
|
||||
if rec["phase"] == "no_turnid":
|
||||
@@ -203,26 +203,26 @@ def grade(rec):
|
||||
return "OK", "asked, resumed the same turn, and the reply reflected the primary's answer"
|
||||
if reflected:
|
||||
return "DEGRADED", f"reply reflected the answer but resolved via {rec['replySource']} " \
|
||||
"(worker did not call bridge_reply cleanly)"
|
||||
"(worker did not call fleet_reply cleanly)"
|
||||
return "WRONG_ANSWER", f"the worker replied but did not reflect '{ANSWER_COLOR}' — " \
|
||||
f"the answer may not have reached the resumed turn: {rec['reply']!r}"
|
||||
return "WEDGE", f"unexpected terminal phase {rec['phase']}: {rec.get('detail')}"
|
||||
|
||||
|
||||
def write_transcript(out_dir, rec, meta):
|
||||
path = out_dir / "bridge_ask_transcript.md"
|
||||
path = out_dir / "fleet_ask_transcript.md"
|
||||
g, note = grade(rec)
|
||||
with path.open("w") as f:
|
||||
f.write(f"# Live bridge_ask — reverse rendezvous — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"# Live fleet_ask — reverse rendezvous — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"One worker paused its delegated turn to ask the primary, then resumed with the "
|
||||
f"answer (profile `{meta['profile']}`). Result: **`{g}`**.\n\n")
|
||||
f.write("## Round-trip\n\n")
|
||||
f.write(f"1. **primary → worker** (delegation): the ask-forcing task.\n")
|
||||
f.write(f"2. **worker → primary** (`bridge_ask`, {rec.get('ask_latency')}s): "
|
||||
f.write(f"2. **worker → primary** (`fleet_ask`, {rec.get('ask_latency')}s): "
|
||||
f"{rec.get('question')!r} — surfaced on the primary's blocked send as a "
|
||||
f"`question` with `turnId={rec.get('turnId')}`.\n")
|
||||
f.write(f"3. **primary → worker** (answer on that turn): `{ANSWER_COLOR}`.\n")
|
||||
f.write(f"4. **worker → primary** (`bridge_reply`, {rec.get('answer_latency')}s, "
|
||||
f.write(f"4. **worker → primary** (`fleet_reply`, {rec.get('answer_latency')}s, "
|
||||
f"source={rec.get('replySource')}): {rec.get('reply')!r}\n\n")
|
||||
f.write(f"> **{g}:** {note}\n")
|
||||
if rec.get("detail"):
|
||||
@@ -231,7 +231,7 @@ def write_transcript(out_dir, rec, meta):
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Live bridge_ask reverse-rendezvous test (CB-205)")
|
||||
ap = argparse.ArgumentParser(description="Live fleet_ask reverse-rendezvous test (CB-205)")
|
||||
ap.add_argument("--base", default="http://127.0.0.1:8765")
|
||||
ap.add_argument("--profile", default=None)
|
||||
ap.add_argument("--repo", default=str(REPO_ROOT))
|
||||
@@ -241,7 +241,7 @@ def main():
|
||||
ap.add_argument("--keep-worker", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
print(f"[{now()}] live bridge_ask: 1 primary, 1 worker "
|
||||
print(f"[{now()}] live fleet_ask: 1 primary, 1 worker "
|
||||
f"(profile={args.profile or 'default'}, repo={args.repo})\n")
|
||||
|
||||
rec = {"spawned": False, "pane": None}
|
||||
@@ -262,7 +262,7 @@ def main():
|
||||
|
||||
print()
|
||||
print("=" * 72)
|
||||
print("LIVE bridge_ask SUMMARY — reverse rendezvous (CB-205)")
|
||||
print("LIVE fleet_ask SUMMARY — reverse rendezvous (CB-205)")
|
||||
print(f" asked: {rec.get('question')!r} (turnId={rec.get('turnId')}, {rec.get('ask_latency')}s)")
|
||||
print(f" answered: {ANSWER_COLOR!r}")
|
||||
print(f" replied: {rec.get('reply')!r} (source={rec.get('replySource')}, {rec.get('answer_latency')}s)")
|
||||
@@ -1,12 +1,12 @@
|
||||
# Live bridge_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
# Live fleet_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
|
||||
One worker paused its delegated turn to ask the primary, then resumed with the answer (profile `default`). Result: **`OK`**.
|
||||
|
||||
## Round-trip
|
||||
|
||||
1. **primary → worker** (delegation): the ask-forcing task.
|
||||
2. **worker → primary** (`bridge_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
2. **worker → primary** (`fleet_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
3. **primary → worker** (answer on that turn): `blue`.
|
||||
4. **worker → primary** (`bridge_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
4. **worker → primary** (`fleet_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
|
||||
> **OK:** asked, resumed the same turn, and the reply reflected the primary's answer
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge fan-out test — ONE primary vs MANY workers, concurrently, for
|
||||
issue hunting through the running `bridged` daemon, fully captured, with gap analysis.
|
||||
issue hunting through the running `fleetd` daemon, fully captured, with gap analysis.
|
||||
|
||||
Where conversation_test.py exercises a single worker over multiple turns, this drives
|
||||
the path that only appears under fan-out: the primary spawns N workers, sends each a
|
||||
@@ -12,7 +12,7 @@ and collects every reply concurrently. That stresses what a single worker never
|
||||
• reply routing under concurrency (worker A's answer must never resolve worker B's send),
|
||||
|
||||
and, as the payload, whether a fleet of off-subscription workers can actually surface
|
||||
real issues in the repo and report them back structurally via bridge_reply.
|
||||
real issues in the repo and report them back structurally via fleet_reply.
|
||||
|
||||
It talks ONLY to the bridge's REST face on loopback — it never sets ANTHROPIC_BASE_URL
|
||||
and never touches herdr directly, so it is subscription-safe by construction.
|
||||
@@ -53,18 +53,18 @@ REPO_ROOT = HERE.parent
|
||||
# hot files this project has been iterating on, so a real issue is plausible to find.
|
||||
DEFAULT_ASSIGNMENTS = [
|
||||
{"id": "completion", "probe": "CompletionResolver",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/inject/CompletionResolver.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java"},
|
||||
{"id": "worker", "probe": "WorkerService",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/worker/WorkerService.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/worker/WorkerService.java"},
|
||||
{"id": "rendezvous", "probe": "Rendezvous",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/msg/Rendezvous.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/msg/Rendezvous.java"},
|
||||
]
|
||||
|
||||
PROMPT_TMPL = (
|
||||
"You are one of several issue-hunting workers in the claude-bridge repo (it is your "
|
||||
"current working directory). Your assignment: inspect the file `{target}` and find the "
|
||||
"SINGLE most important real bug, correctness gap, or risk in it. Read the file before "
|
||||
"answering. Reply via bridge_reply with EXACTLY these four lines:\n"
|
||||
"answering. Reply via fleet_reply with EXACTLY these four lines:\n"
|
||||
"1. {target}:<line>\n"
|
||||
"2. issue: <one sentence>\n"
|
||||
"3. fix: <one line>\n"
|
||||
@@ -147,9 +147,9 @@ def grade(rec):
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit bridge_reply"
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call bridge_reply)"
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
||||
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
||||
fleetd.yaml
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -5,10 +5,10 @@
|
||||
|
||||
## Why this exists
|
||||
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `bridge_send` is
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `fleet_send` is
|
||||
open: it lands in `agent.<target>.inbox` on the broker and survives a daemon bounce. But
|
||||
delivery is still **pull** — the primary only sees the reply if it happens to call
|
||||
`bridge_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
`fleet_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
the primary works on something else.
|
||||
|
||||
This layer makes delivery **active**: the bridge *pushes* a nudge to the primary the moment
|
||||
@@ -26,11 +26,11 @@ pointed at the primary's pane instead.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
W["worker"] -->|"bridge_reply (no open send)"| MS["MessageService.reply"]
|
||||
W["worker"] -->|"fleet_reply (no open send)"| MS["MessageService.reply"]
|
||||
MS -->|"inbox.publish"| INBOX[("agent.<target>.inbox<br/>(durable, LavinMQ)")]
|
||||
MS -->|"notify"| LOOP["ReplyPushLoop"]
|
||||
LOOP -->|"status-gated inject"| PANE["primary's herdr pane"]
|
||||
PANE -->|"primary drains"| DRAIN["bridge_poll(target)<br/>= peek + ack"]
|
||||
PANE -->|"primary drains"| DRAIN["fleet_poll(target)<br/>= peek + ack"]
|
||||
DRAIN -->|"inbox now empty"| LOOP
|
||||
LOOP -.->|"still non-empty →<br/>re-inject on backoff"| PANE
|
||||
classDef store fill:#2c5282,stroke:#1a365d,color:#ffffff;
|
||||
@@ -51,13 +51,13 @@ the caller runs in a herdr pane on this host. Today it's discarded for the prima
|
||||
(`presence.markPresent` is a no-op on it).
|
||||
|
||||
**Plan:** a single-slot `PrimaryRegistry` (thread-safe) holding the primary's `terminal_id`.
|
||||
Populate it from the **orchestration-side** MCP tools — `bridge_send`, `bridge_spawn`,
|
||||
`bridge_poll`, `bridge_list`, `bridge_status`, `bridge_profiles` — capturing
|
||||
Populate it from the **orchestration-side** MCP tools — `fleet_send`, `fleet_spawn`,
|
||||
`fleet_poll`, `fleet_list`, `fleet_status`, `fleet_profiles` — capturing
|
||||
`callerTerminal(exchange)` when it is (a) non-null and (b) **not** a registered worker
|
||||
session in `SessionManager`. That caller is, by construction, the primary. Worker-side tools
|
||||
(`bridge_reply`, `bridge_ask`) never set it.
|
||||
(`fleet_reply`, `fleet_ask`) never set it.
|
||||
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `BridgedConfig`
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `FleetConfig`
|
||||
(nested record, same shape as `Broker`). Lets an operator pin it, or supply it when
|
||||
derivation can't (see degrade case).
|
||||
- **Degrade:** if the primary is off-host or in a non-herdr terminal, `terminalForPid`
|
||||
@@ -71,13 +71,13 @@ A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
(`MessageService.reply` → the `inbox.publish` branch, `MessageService.java:192`).
|
||||
|
||||
- **Inject a nudge, not the payload.** The injected turn tells the primary *to drain*
|
||||
(e.g. "Worker `<target>` returned a reply — run `bridge_poll(target=<target>)` to collect
|
||||
(e.g. "Worker `<target>` returned a reply — run `fleet_poll(target=<target>)` to collect
|
||||
it"), it does **not** carry the reply text. Rationale: replies can be large/multiline and
|
||||
terminal injection would mangle them; the drain response is the clean transport. Keeps the
|
||||
push idempotent — re-nudging is harmless.
|
||||
- **Ack = drain.** The primary draining (`drainReplies` = peek + ack) is the acknowledgement.
|
||||
The loop's **stop condition is `inbox.peek(target).isEmpty()`** — the reply is gone from the
|
||||
inbox because it was acked. No new `bridge_ack` tool needed for v1 (see Increment 3).
|
||||
inbox because it was acked. No new `fleet_ack` tool needed for v1 (see Increment 3).
|
||||
- **Status-gated injection (mechanism (b), chosen).** A dedicated lightweight scheduled loop,
|
||||
**not** the worker `Injector`. It injects via `AgentControl.send(primaryTerminal, nudge)`
|
||||
(the same herdr `agent.send` = `pane send-text` + submit that delivers to workers) only when
|
||||
@@ -93,10 +93,10 @@ A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
the reply remains in the durable inbox and the next natural poll (or a later worker reply's
|
||||
nudge) still surfaces it. Bounded so the bridge never spams the primary.
|
||||
|
||||
### Increment 3 — optional per-`msgId` `bridge_ack` tool (deferred)
|
||||
### Increment 3 — optional per-`msgId` `fleet_ack` tool (deferred)
|
||||
|
||||
Drain-as-ack is coarse: it clears *all* pending replies for a target at once. If finer
|
||||
control is ever needed (ack one reply, leave others held), add a `bridge_ack(msgId)` tool
|
||||
control is ever needed (ack one reply, leave others held), add a `fleet_ack(msgId)` tool
|
||||
mapping to `inbox.ack(target, msgId)` — the port already supports per-`msgId` ack. Not built
|
||||
in v1; the stop-on-empty loop is sufficient.
|
||||
|
||||
@@ -106,7 +106,7 @@ in v1; the stop-on-empty loop is sufficient.
|
||||
resolved terminal is non-null **and not a registered worker session**, seen on an
|
||||
orchestration-side tool. This never mislabels a worker (workers are in `SessionManager`)
|
||||
and needs no new env var or argument (identity stays connection-derived, per the existing
|
||||
`BridgeMcp` invariant).
|
||||
`FleetMcp` invariant).
|
||||
|
||||
2. **Readiness-gate mismatch → dedicated loop.** The existing `Injector` gates delivery on
|
||||
`ready.test(target)` = `WorkerPresence` (the *worker's* MCP connected). The primary is not
|
||||
@@ -132,7 +132,7 @@ boundary**. The bridge is signalling the primary that it has mail — not drivin
|
||||
non-null terminal AND not a registered session" predicate; the loop's stop-on-empty and
|
||||
bounded-reminder logic with an injected clock + a fake injector (no real herdr).
|
||||
- **Live dogfood (primary-side):** with the daemon on the broker jar + a real worker,
|
||||
delegate a task, let the worker reply after the `bridge_send` window closes, and observe the
|
||||
delegate a task, let the worker reply after the `fleet_send` window closes, and observe the
|
||||
bridge inject a drain nudge into *this* primary pane; confirm draining stops the reminders;
|
||||
confirm an unreachable primary (registry empty) degrades to pull with no loss.
|
||||
|
||||
@@ -0,0 +1,722 @@
|
||||
# fleetd configuration (example). Copy to fleetd.yaml and adjust.
|
||||
#
|
||||
# fleetd is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — fleetd REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
|
||||
# API authentication (CB-501). Governs how a caller that is NOT an on-host worker pane proves it
|
||||
# is the primary. Worker identity never depends on this: a loopback peer PID that maps to a herdr
|
||||
# pane is unforgeable and is always honoured, so turning auth on cannot lock the fleet out.
|
||||
#
|
||||
# mode: loopback-trust → DEFAULT, and the historical behaviour: any loopback caller that is not
|
||||
# a worker is the primary, no credential needed. Sound ONLY because the
|
||||
# OS refuses remote connections to a loopback socket.
|
||||
# mode: token → such a caller must send `Authorization: Bearer <token>`; without it it
|
||||
# is anonymous and authorized for nothing. REQUIRED for a non-loopback
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# FLEETD_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
# location / { proxy_pass http://127.0.0.1:8765; proxy_set_header Authorization $http_authorization; }
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: FLEETD_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from fleet_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
# pushReminders: 5 # max nudges before giving up (default 5)
|
||||
# pushBackoffMs: 15000 # delay between nudges (default 15000)
|
||||
|
||||
# CB-530: MORE THAN ONE LEAD. `primary:` above is singular by construction — every other pane
|
||||
# resolves as a worker — which is right for one lead driving a fleet and wrong the moment two leads
|
||||
# (say a Claude lead and an opencode lead) work as peers: the second is silently demoted and refused
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# tab → the ONLY field identity depends on (CB-579); the exact label of the tab hosting the lead.
|
||||
# Label the tab yourself, or let fleetd label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by fleet_whoami
|
||||
#
|
||||
# A lead's tab must already carry its label (or be launched by fleetd, which labels it) — there is
|
||||
# no terminal id to paste in and nothing to re-pin when the session restarts: the tab survives, so
|
||||
# the same label resolves the same lead again on the next scan.
|
||||
# `fleet_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
# destination for its nudges, and is a separate mechanism from lead identity — see `fleet.leaders:`.
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing fleetd places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open fleet_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
# Opt-in on purpose — it SPENDS the operator's subscription on its own initiative (each nudge starts
|
||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||
# block = feature off, exactly as before.
|
||||
#
|
||||
# Three knobs, each with a default that errs on the side of not burning context:
|
||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||
# leadHeartbeat:
|
||||
# idleAfterSeconds: 300
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
# per tick.
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
# Math.max(15, intervalSeconds), so a lower value is silently raised, not
|
||||
# rejected.
|
||||
# workingSuspectAfterSeconds → age before a BUSY member is suspected of a stall (default 600).
|
||||
# ENFORCED floor of 300: a lower value is silently raised.
|
||||
# paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by anything. Setting it changes
|
||||
# nothing right now. It exists so a later build can start honouring it without
|
||||
# another config-shape change.
|
||||
# notifications.mode → "webhook" flips what fleet_list REPORTS (healthCoverage: "full" instead
|
||||
# of "detection-only") — it does NOT make fleetd send any webhook call; no
|
||||
# delivery mechanism is implemented yet. Any other value, or omitting the
|
||||
# block, reports "detection-only".
|
||||
# health:
|
||||
# enabled: true
|
||||
# intervalSeconds: 30 # floor 15
|
||||
# workingSuspectAfterSeconds: 600 # floor 300 — how long BUSY with no activity means STALL_SUSPECTED
|
||||
# paneProbeIntervalSeconds: 60 # parsed, but nothing reads it yet — changing it changes nothing
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# Optional socket for member panes. Omit this to use herdrSocket for both leads and members.
|
||||
# memberHerdrSocket: /Users/member/.config/herdr/herdr.sock
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
# it is for; that is the member's role, and roles live under `fleet:` below. Which profile an
|
||||
# unqualified spawn lands on comes from that role's pool, not from a global default.
|
||||
#
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → fleetd mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# ideMcpUrl → opt-in (CB-634), default off. When set, fleetd mounts the IDE Index MCP as a
|
||||
# second inline server named `intellij`, and adds an IDE charter that pins every
|
||||
# ide_* call to the member's own worktree. A URL, not a boolean — host and port
|
||||
# are host-specific. Set it only on a host where the IDE actually runs.
|
||||
# ideProjectDir → repo-relative module dir the IDE opens and the overlay pins (CB-634). Only read
|
||||
# when ideMcpUrl is set. This repo's Maven pom lives in `fleetd/`, not at the
|
||||
# worktree root, so opening the root imports no module and ide_* resolves nothing;
|
||||
# set this to `fleetd`. Omit for a repo whose project is the worktree root.
|
||||
# ideOpenCommand → host command that opens ideProjectDir in the IDE at spawn (CB-634 auto-open).
|
||||
# Only read when ideMcpUrl is set. `{dir}` is replaced with the absolute module
|
||||
# dir and the command runs through `/bin/sh -c`, so set env inline if needed —
|
||||
# e.g. `env DISPLAY=:10.0 idea {dir}`. Best-effort: a failure is logged, never
|
||||
# fails the spawn. Omit to open the member's module by hand. There is no close
|
||||
# half yet — an opened module stays open until the operator closes it.
|
||||
# autoCompactWindow → opt-in, default off. A bounded token window that forces a spawned member to
|
||||
# compact its context instead of running on the backend's own default and dying
|
||||
# mid-turn (losing its fleet_reply — the whole point of the turn — with it).
|
||||
# Validated at config load to [100000, 1000000] — the band Claude Code's own
|
||||
# --autocompact flag accepts.
|
||||
# CROSS-BACKEND SEMANTICS DIFFER: on claude-code this is a launch-time
|
||||
# `--autocompact <tokens>` flag — the member compacts AT this window. opencode
|
||||
# has no equivalent flag (it only forces `compaction.auto: true`, unconditionally,
|
||||
# already), so this is instead applied as the model's `limit.context` in the
|
||||
# generated opencode.json — the member compacts WITHIN this window, not exactly
|
||||
# at it — and only when this profile's `model:` is in `provider/model` form; if it
|
||||
# isn't, fleetd logs a WARN naming the profile rather than silently doing nothing.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
# cwd on an MCP spawn, else the daemon's cwd — never $HOME. See
|
||||
# docs/Worker-Startup-and-Trust.md.
|
||||
# configDir → CLAUDE_CONFIG_DIR for the worker, so it inherits that profile's
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.env, .envrc]. (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
# handed a worker the primary's IDE servers, which are bound to the primary's
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# fleetd neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
# Opt-in by design — omit and the worker gets no PR-create grant (push over
|
||||
# SSH is unaffected). The token value itself is never stored in this file.
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# exhaustedPattern → regex matched against a completion-fallback scrape (CB-578 stage A) to
|
||||
# classify a turn that ended with no fleet_reply as the backend having
|
||||
# refused on a subscription usage limit, rather than a real answer. Opt-in —
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into fleetd itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
# BACKEND_EXHAUSTED classification fires. Two profiles that set the SAME
|
||||
# credentialId share one quarantine — the case this exists for is two models
|
||||
# on one account (e.g. sol and terra both billing one OpenAI credential): an
|
||||
# exhaustion on either one must lock out both, or the fleet just walks onto
|
||||
# the same dead account under the sibling's name. Opt-in — omit and this
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. fleetd hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# fleetd now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.fleet.plist and deploy/fleetd.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
# entry cannot repoint a worker past the SubscriptionGuard — which is checked
|
||||
# against `baseUrl` alone.
|
||||
# Put `defaultMode: "auto"` in each ccs profile so the worker runs autonomously.
|
||||
profiles:
|
||||
gx10: # ccs profile name (NOT a hostname)
|
||||
kind: claude-code # which adapter spawns this profile (default; may omit)
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: FLEETD_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
# weight: relative selection weight for automatic placement (weighted, round-robin, and
|
||||
# fixed's fallback walk). Absent defaults to 1.0. An explicit 0 or negative value means
|
||||
# "never auto-select this profile" (CB-554) — it stays reachable via an explicit
|
||||
# `fleet_spawn{profile:"gx10"}`, which bypasses placement entirely; only automatic
|
||||
# selection skips it.
|
||||
weight: 0.5
|
||||
# maxLoad: max live workers on this profile. Omit for unlimited. An explicit 0 (CB-585) caps
|
||||
# the profile at zero live members — it is excluded from automatic placement and an explicit
|
||||
# `fleet_spawn{profile:"gx10"}` against it is refused too; a cap holds even when the profile
|
||||
# is named directly. Negative is refused at config load — there is no sane meaning for it.
|
||||
maxLoad: 2
|
||||
# subscription: true
|
||||
# THE KNOB THAT DECIDES WHO PAYS (CB-539). Default false. When true, this profile's members
|
||||
# run on the OPERATOR'S OWN Claude subscription instead of a metered endpoint — every spawn
|
||||
# bills your plan and eats your usage limit. Off-subscription is the whole point of this
|
||||
# daemon, so treat `true` as a deliberate exception, not a convenience.
|
||||
#
|
||||
# What changes when it is set (ClaudeCodeLauncher):
|
||||
# - no ANTHROPIC_BASE_URL and no ANTHROPIC_AUTH_TOKEN are injected — the member inherits
|
||||
# the operator's own Claude Code auth, which is exactly why it bills the plan;
|
||||
# - SubscriptionGuard never vets it, because there is no baseUrl to vet;
|
||||
# - no token is required, so `tokenEnv` is irrelevant here.
|
||||
#
|
||||
# MUTUALLY EXCLUSIVE with `baseUrl` — setting both is refused at config load (CB-542). On the
|
||||
# subscription path no guard would vet the URL, so allowing both would be a way around the
|
||||
# guard rather than a configuration.
|
||||
#
|
||||
# GOTCHA 1 — it is invisible to the startup secret check. `Fleetd.reportRequiredSecrets`
|
||||
# skips subscription profiles on purpose (they need no token), so a boot log that reports
|
||||
# every secret as fine says nothing about these profiles.
|
||||
#
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".env", ".envrc"] # the default; never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
# autoCompactWindow: 250000 # opt-in: bound member context; claude-code compacts AT this, opencode within it (model limit.context)
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
# Pin an auto-compact window BELOW the served model's context ceiling. The global
|
||||
# ~/.claude/settings.json value is shared by every ccs instance and the primary, so the
|
||||
# per-profile override belongs here. Equal to the ceiling means auto-compact never fires
|
||||
# before the server rejects the prompt, which kills a worker mid-turn (CB-523).
|
||||
env:
|
||||
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "280000"
|
||||
# CB-402: a second coding-agent kind, proving the PeerLauncher SPI is provider-neutral.
|
||||
# opencode is provider-agnostic and uses NONE of Claude's private seams: no ANTHROPIC_BASE_URL /
|
||||
# SubscriptionGuard (so it needs no `guard` host entry), no --mcp-config / --append-system-prompt.
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → fleet_send →
|
||||
# structured fleet_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
#
|
||||
# CB-508: point an opencode profile at your OWN OpenAI-compatible endpoint (local vLLM, llama.cpp,
|
||||
# LM Studio, TGI…) instead of opencode's gateway. Setting `baseUrl` on a `kind: opencode` profile
|
||||
# makes the bridge emit a custom `provider` block into the generated opencode.json — opencode has
|
||||
# no ANTHROPIC_BASE_URL seam, so this is how the endpoint is pinned.
|
||||
# baseUrl → a bare host:port gets `/v1` appended (where these servers mount the API); a URL that
|
||||
# already has a path is used verbatim, so a custom mount point still works.
|
||||
# model → MUST be "<provider>/<model>". The provider half names the generated block; the model
|
||||
# half must match an id the server reports at /v1/models. One field drives both the
|
||||
# declaration and the `-m` flag, so they cannot drift apart. A bare model name with a
|
||||
# baseUrl set is rejected at spawn rather than silently using the default gateway.
|
||||
# tokenEnv → optional; its value becomes the provider apiKey. Most local servers ignore the key,
|
||||
# so a placeholder is used when unset (the AI SDK still requires a non-empty one).
|
||||
# NOTE: no `guard` entry is needed even with a baseUrl set. The SubscriptionGuard exists to stop a
|
||||
# worker borrowing the primary's Anthropic subscription, and an opencode process has no Anthropic
|
||||
# credential path at all.
|
||||
# opencode-local:
|
||||
# kind: opencode
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
# How an unqualified spawn chooses a profile: fixed (default, reproduces pre-CB-518 behaviour),
|
||||
# round-robin, or weighted. Omitting this key is a strict no-op for existing configs.
|
||||
#
|
||||
# `weighted` IS NOT "cheapest first" — read this before you set weights (CB-589).
|
||||
# It is smooth weighted round-robin: it spreads spawns across EVERY profile that has a free slot,
|
||||
# in weight ratio. It has no idea which profile costs money. So with local:10 / paid:2 you do not
|
||||
# get "use local, overflow to paid" — you get roughly one spawn in six going to the paid profile
|
||||
# while the local box still has a free slot.
|
||||
#
|
||||
# There is a sharper second effect. The policy's running score map lives for the daemon's whole
|
||||
# life. While a profile is at maxLoad it is filtered out and its score FREEZES, so the paid
|
||||
# profiles keep accumulating against it. When the local slot frees up it returns with a stale
|
||||
# score and can LOSE the next pick — a paid spawn while the free box sits idle.
|
||||
#
|
||||
# Until a real cost-first policy exists, the workaround is to make the ratio decisive rather than
|
||||
# proportional: give the free profile a weight so large that it wins every pick it is eligible
|
||||
# for, and paid profiles only ever take genuine overflow. On this host that is local weight 100
|
||||
# against paid weights of ~1.
|
||||
#
|
||||
# The gotcha with that workaround: it expresses a PREFERENCE ORDER through a RATIO knob. Add a
|
||||
# future profile at weight 150 and it silently outranks the free box, with nothing to warn you.
|
||||
# Re-check the weights whenever you add a profile.
|
||||
placement: weighted
|
||||
|
||||
# How long a credential sits out after a BACKEND_EXHAUSTED classification (CB-578 stage B), in
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded fleetd keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. fleetd checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
# Not every key can move under a running daemon, and the difference is about what already exists
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Fleetd.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
# with nothing in the deferred list — but has NO effect until you restart. Treat it
|
||||
# as deferred in practice, even though today's reload output does not say so.
|
||||
# DEFERRED → accepted into the new config, but the wiring built at startup keeps the old value
|
||||
# until you restart: `lifecycle:`, `leadHeartbeat:`, `guard:`, `worktreeRoot:`,
|
||||
# `spawnReadyTimeoutMs` / `spawnReadyPollMs`, `quarantineCooldownSeconds` (CB-578
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
# bound, the broker connection is open, and the auth mode decides who may reach the
|
||||
# port that is already listening.
|
||||
#
|
||||
# A changed COLD key refuses the WHOLE reload — not the hot half applied and the cold half warned
|
||||
# about. A half-applied reload would leave the daemon matching no file on disk, which is the worst
|
||||
# thing a reload can do to an operator debugging one. A file that fails to parse or fails a startup
|
||||
# validator is refused the same way, and the running config stays live.
|
||||
# configReload:
|
||||
# enabled: true
|
||||
# intervalSeconds: 10
|
||||
|
||||
# THE FLEET (CB-557) — who the daemon may run, and under which role. This one block replaced four
|
||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||
#
|
||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
||||
# file, the playbook skill and the authz row.
|
||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||
# which is why the two cannot be one field.
|
||||
#
|
||||
# The ROLE IS THE CONTAINING KEY, not a `role:` field. That is not only tidier: a misspelled role
|
||||
# used to parse into a member with no contract at all, while a misspelled pool name here simply
|
||||
# declares nothing.
|
||||
#
|
||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
||||
# with being listed here; the entry key just names the entry.
|
||||
fleet:
|
||||
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
||||
# reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put secrets
|
||||
# here: a later launch step writes this text to a world-readable temp file, and ${ENV} interpolation
|
||||
# is deliberately not supported.
|
||||
charters:
|
||||
architect: |-
|
||||
You are an architect in this fleet. You refine work before anyone builds it:
|
||||
scope, acceptance criteria, risks, and a unit split. You read the repo and
|
||||
write analysis. You never commit production code and never open a PR.
|
||||
A design task is worked by two architects. Design alone first, then exchange
|
||||
and say plainly where you disagree. Do not concede just to agree.
|
||||
dev: |-
|
||||
You implement the one unit you were given, and nothing else. You test it,
|
||||
commit it, and open your own pull request. You never merge.
|
||||
reviewer: |-
|
||||
You review the diff you were given. You report bugs, risks and missing tests.
|
||||
You do not change code.
|
||||
|
||||
# Optional. Template for a member tab's label; {role}, {profile}, {model} and {n} are substituted.
|
||||
# {n} counts per role+profile, so `dev: sonnet #2` really is the second sonnet dev. Because {role}
|
||||
# comes from a closed enum, a generated label can never begin with a lead's tabPrefix.
|
||||
# tabLabel: "{role}: {profile} #{n}"
|
||||
|
||||
# Panes that orchestrate rather than are orchestrated. A lead may now be CREATED as well as
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
# inside it restarts — the terminal id changes; the tab, and its label, do not, so no config edit
|
||||
# follows a restart.
|
||||
#
|
||||
# A lead the daemon launches is labelled BY the daemon with this same `tab:` value, so it is found
|
||||
# by the same scan. A lead counts as live only when herdr also reports a running agent in that
|
||||
# tab — a label left behind by a session that died does not block the relaunch, and a tab that is
|
||||
# gone entirely drops out of the next scan rather than being remembered forever.
|
||||
#
|
||||
# An auto-launched lead is NOT a member: it gets no worker reply charter, is never registered with
|
||||
# the session lifecycle (the idle reaper would kill your orchestrator), and stays on the
|
||||
# subscription — ANTHROPIC_BASE_URL/AUTH_TOKEN are stripped from its env whatever the profile says.
|
||||
#
|
||||
# GET THE `tab:` VALUE RIGHT. A pane that does not match any configured `tab:` (a typo, a renamed
|
||||
# tab, a pane no entry names at all) is not recognised as a lead — it resolves as an ordinary
|
||||
# WORKER instead, silently, and every orchestration call it makes (spawn/stop/send/drain) is
|
||||
# refused. There is no error at startup for this: an unmatched pane is simply not a lead. If your
|
||||
# primary suddenly can't spawn or send, check this section first.
|
||||
# leaders:
|
||||
# opus-5.0:
|
||||
# profile: opus # omit to never create this lead, only recognise it
|
||||
# instances: 1 # desired live count; only the shortfall is launched. 0 = off
|
||||
# tab: "lead: opus-5.0" # REQUIRED — the exact tab label this lead lives in
|
||||
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
||||
# # this convention at startup; plays no part in matching a lead
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
||||
# kind: claude # descriptive; reported by fleet_whoami
|
||||
# gpt-sol-5.6:
|
||||
# tab: "lead: gpt-sol-5.6"
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
# architect-2:
|
||||
# profile: sol # a different vendor on purpose — two architects that share a
|
||||
# # model share its blind spots
|
||||
developers:
|
||||
gx10:
|
||||
profile: gx10
|
||||
# reviewers:
|
||||
# gx10:
|
||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Every profile above must have its host listed here.
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- gx01.gw
|
||||
|
||||
# Member credential policy (CB-596, gitea issue #82). A herdr pane runs a LOGIN shell, and that
|
||||
# shell re-sources the operator's own secret store — so a spawned member inherits every credential
|
||||
# the operator's shell holds, not just the ones fleetd means to give it. Measured on this host:
|
||||
# 31 credential names, all set, with only ONE (GITEA_ACCESS_TOKEN) blocked before this — and that
|
||||
# block was a single name hardcoded in HerdrPeerLauncher.java, not driven by this file. This block
|
||||
# replaces that hardcoded shadow with a config-driven list of names.
|
||||
#
|
||||
# ROUND-2 CORRECTION, measured live: the pane-creation env overlay below (applied at tab.create /
|
||||
# pane.split, BEFORE the pane's login shell runs) does NOT survive that login shell for any name
|
||||
# secrets.sh actually exports — the shell re-exports it afterwards and overwrites the sentinel.
|
||||
# Proof: GITEA_ACCESS_TOKEN comes back blocked only because secrets.sh itself carries a guarded
|
||||
# export (`[ -n "${BRIDGED_MEMBER:-}" ] || export GITEA_ACCESS_TOKEN=...`) — that guard, not this
|
||||
# file, is what wins. No other name in `known` below has a matching guard in secrets.sh yet (1
|
||||
# guard measured against 33 export lines there). So today this block's overlay is REAL protection
|
||||
# only for a name secrets.sh does not export, or a peer kind whose pane never runs a login shell —
|
||||
# for everything secrets.sh exports and guards, the guard in secrets.sh (out of scope for this
|
||||
# ticket) is what actually blocks it, not this list. An exec-time fix (winning after the login
|
||||
# shell finishes, before the agent process starts) was attempted and found to have no seam in the
|
||||
# current herdr protocol — AgentControl.start takes a fixed `kind` (herdr resolves the executable)
|
||||
# plus trailing CLI args for that binary, not an arbitrary argv or an env map; only tab.create /
|
||||
# pane.split accept `env`, and that is this same pane-creation overlay. See gitea #82 for the open
|
||||
# design question this leaves.
|
||||
#
|
||||
# DENY-BY-DEFAULT, NOT A DENY-LIST. A deny-list (name the bad ones, let everything else through) is
|
||||
# silently wrong the moment the operator's store gains a new secret — nothing would ever report it.
|
||||
# Deny-by-default inverts that: `known` bounds the blast radius to names actually enumerated below,
|
||||
# and EVERY one of them is blocked UNLESS it is also in `allow`. Omitting this block entirely (the
|
||||
# shipped default) blocks NOTHING — unlike most optional blocks in this file, absence here is a real
|
||||
# gap, not a safe "feature off". A name that is neither `known` nor `allow`-ed is not silently let
|
||||
# through either: the daemon logs a WARN naming any credential-shaped env var it finds on neither
|
||||
# list (never its value), so a secret added to the store later does not go unnoticed forever.
|
||||
#
|
||||
# policy → "deny-by-default" (the default; also accepted spelled "deny-list") overlays each
|
||||
# known-but-not-allowed name BEFORE the pane's login shell runs — real protection only
|
||||
# where that shell does not re-export the name (see ROUND-2 CORRECTION above). An
|
||||
# unrecognized value refuses to start, naming it.
|
||||
# policy → "allow-list" (CB-633) moves the control to a per-spawn ZDOTDIR directory the daemon
|
||||
# generates and passes through tab.create's env map. Each generated startup file sources
|
||||
# its ~/ counterpart FIRST and then runs the scrub, so the scrub happens after the
|
||||
# operator's whole chain and no sourced file can undo it.
|
||||
# The scrub is sourced from BOTH the generated .zshrc and the generated .zlogin, because
|
||||
# herdr does not open the same kind of shell everywhere: macOS panes run a LOGIN zsh (so
|
||||
# .zlogin runs), Linux panes run a plain interactive zsh (so .zlogin never runs at all).
|
||||
# A scrub in .zlogin alone would be a control that silently does nothing on Linux.
|
||||
# The allow-list is DERIVED, never typed:
|
||||
# every profile's tokenEnv/gitTokenEnv/gitHostEnv values and env-map keys, plus an
|
||||
# infrastructure set (PATH HOME SHELL TERM LANG LC_* TMPDIR USER LOGNAME PWD SHLVL EDITOR
|
||||
# PAGER JAVA_HOME XDG_* ZDOTDIR), plus whatever keys this spawn's own env overlay carries.
|
||||
# Adding a profile can therefore only widen the list, never break another spawn's scrub.
|
||||
# Under this policy `known`/`allow` below become REPORTING ONLY — they feed the gap WARN,
|
||||
# they are no longer a control. If the member's login shell is NOT zsh, the daemon logs a
|
||||
# loud WARN saying protection is off and falls back to deny-by-default's overlay.
|
||||
# Each pane writes a scrub-report.txt naming how many variables it kept of how many it
|
||||
# saw; the daemon logs that "allowed N of M" line when the pane stops. If the report is
|
||||
# MISSING the daemon logs a WARN instead — the scrub cannot then be confirmed to have
|
||||
# run, and a silently-dead control is exactly what this policy exists to prevent.
|
||||
# allow → credential names a member legitimately needs. Under deny-by-default, left OUT of the
|
||||
# pane's env overlay entirely, so the value the pane's own (login) shell exports passes
|
||||
# through untouched. Under allow-list: reporting only.
|
||||
# known → every credential name the operator's store is known to export. Under deny-by-default,
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAuthSock → whether SSH_AUTH_SOCK may pass through under allow-list ("allow") or must be
|
||||
# blanked like any other non-derived name ("block", the default). This is a decision you
|
||||
# have to make explicitly: SSH_AUTH_SOCK is a handle to YOUR ssh-agent, and a member
|
||||
# holding it can sign with your keys — it sits in no secret file and looks like no
|
||||
# credential, which is why it slipped past three earlier tickets (gitea #110). Blocking
|
||||
# it breaks git over SSH inside members (push/fetch authenticate as you); use HTTPS
|
||||
# remotes or scoped deploy keys instead of allowing it lightly.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAuthSock: block # allow-list only; see the sshAuthSock note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
# - WORKER_GITEA_TOKEN # the repo-scoped forge token a member needs to open its own PR (CB-302)
|
||||
# - CONTEXT7_TOKEN # already decided as allowed by CB-593
|
||||
# - GITEA_HOST # not a credential — a hostname, paired with the forge token above
|
||||
# known:
|
||||
# - AI_GATEWAY_TOKEN
|
||||
# - BESZEL_ADMIN_EMAIL
|
||||
# - BESZEL_ADMIN_PASSWORD
|
||||
# - BESZEL_HUB_URL
|
||||
# - BESZEL_KEY
|
||||
# - BESZEL_UNIVERSAL_TOKEN
|
||||
# - BRAIN_MCP_TOKEN
|
||||
# - CF_ACCOUNT_ID
|
||||
# - CF_API_TOKEN
|
||||
# - CF_USER_TOKEN
|
||||
# - CONFLUENCE_API_TOKEN
|
||||
# - CONFLUENCE_USERNAME
|
||||
# - CONTEXT7_TOKEN
|
||||
# - GITEA_ACCESS_TOKEN
|
||||
# - GITEA_HOST
|
||||
# - GITLAB_OAUTH_CLIENT_SECRET
|
||||
# - GITLAB_PERSONAL_ACCESS_TOKEN
|
||||
# - GRAFANA_ADMIN_PASSWORD
|
||||
# - GRAFANA_ADMIN_USER
|
||||
# - HASS_TOKEN
|
||||
# - HW_PASSWORD
|
||||
# - HW_USER
|
||||
# - LTMS_API_KEY
|
||||
# - MEMORY_MCP_TOKEN
|
||||
# - METRICS_PUSH_TOKEN
|
||||
# - OPENCODE_AUTOMODE_MODEL
|
||||
# - TELEGRAM_BOT_TOKEN
|
||||
# - TELEGRAM_CHAT_ID
|
||||
# - TS_API_KEY
|
||||
# - TS_AUTHKEY
|
||||
# - WORKER_GITEA_TOKEN
|
||||
|
||||
# Spawn-readiness gate (CB-306). The launcher blocks until the worker's herdr status is
|
||||
# injectable (IDLE/BLOCKED/DONE) or the timeout elapses. 0 disables the gate.
|
||||
# NOTE: keys are camelCase — config is bound by plain Jackson with no naming strategy and
|
||||
# unknown keys are ignored, so a snake_case key would be silently dropped (default kept).
|
||||
# spawnReadyTimeoutMs: 20000
|
||||
# spawnReadyPollMs: 300
|
||||
|
||||
# Worktree provisioning root (CB-301-ext). Where per-worker git worktrees are checked out so
|
||||
# each worker owns an isolated branch instead of sharing the primary's tree. Omit to default
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Worktree group sharing (fleetd #185 stage 3). OPTIONAL, off by default. Names an OS group
|
||||
# that a provisioned worktree's repo is made group-writable for (git config
|
||||
# core.sharedRepository group, plus a one-time chgrp/chmod/setgid fix-up), so a member spawned
|
||||
# under a DIFFERENT OS user (see memberHerdrSocket) can write its own worktree, its
|
||||
# per-worktree git metadata, and its own commit objects — without it, every file GitWorktrees
|
||||
# creates is owned by fleetd's own uid and unwritable by another user.
|
||||
# CAUTION: this isolates credentials, not the repository — a member in the group can still
|
||||
# write the operator's git objects and refs in the shared repo. The operator running fleetd
|
||||
# must already be a member of the named group, or every provisioning spawn fails loudly.
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
# contextCap → force-release a session after this many delegated turns
|
||||
# drainTimeoutSeconds → seconds to wait for BUSY sessions on shutdown before forced teardown
|
||||
# clearAfterTurn → whether a reusable worker discards its conversation context after every
|
||||
# completed delegated turn (default false). Works for claude-code workers
|
||||
# only — any other peer kind (e.g. opencode) logs "context reset is
|
||||
# unsupported for peer kind …" once and the reset is a no-op.
|
||||
# lifecycle:
|
||||
# idleTtlSeconds: 300
|
||||
# contextCap: 10
|
||||
# drainTimeoutSeconds: 5
|
||||
# clearAfterTurn: false
|
||||
|
||||
# Durable reply delivery (CB-307 Stage 2). OMIT this block entirely to keep the default
|
||||
# in-memory, soft-state reply inbox (late worker replies are held only until a daemon bounce).
|
||||
# Set a broker uri to swap in the AMQP-backed inbox: worker replies with no open send are held
|
||||
# on a durable per-target queue (agent.<target>.inbox) and survive a restart — the broker
|
||||
# redelivers anything the primary had not yet drained. Production default is LavinMQ; a stock
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# uriEnv → CB-151: name of a host env var holding the AMQP URI, preferred over `uri` (wins
|
||||
# whenever set). The URI carries `user:pass@` inline, so naming a variable keeps the
|
||||
# password out of fleetd.yaml — same pattern as auth.tokenEnv/Profile.tokenEnv. A
|
||||
# uriEnv that resolves to an unset or blank variable is treated as NOT configured and
|
||||
# the daemon falls back to the in-memory inbox, warning loudly.
|
||||
# prefetch → CB-527: consumer basicQos, capping how many unacked messages the inbox holds
|
||||
# in-heap per owned target (the rest sits on the broker's durable queue instead of
|
||||
# growing the JVM heap). Default 32 when omitted.
|
||||
# broker:
|
||||
# uriEnv: LAVINMQ_URI
|
||||
# prefetch: 32
|
||||
|
||||
# Shared cross-host LEADER coordination broker. OMIT this block to leave lead-to-lead messaging
|
||||
# off entirely (config-only in this ticket — nothing here wires it into a live LeadMailbox yet).
|
||||
# This is a SEPARATE AMQP vhost from `broker:` above: member/worker inboxes always stay on the
|
||||
# per-fleet `broker:` vhost, and this vhost carries only leader-to-leader traffic, so two fleets
|
||||
# whose members must never see each other can still share one coordination vhost for their leads.
|
||||
# uriEnv → name of a host env var holding the coordination AMQP URI, same convention as
|
||||
# broker.uriEnv (keeps the credential out of fleetd.yaml). Wins over `uri` when set.
|
||||
# selfId → this daemon's own lead coord-id — the name its mailbox is owned under
|
||||
# (lead.<selfId>.inbox), e.g. "mac-opus" or "fleet01-lead". Must be globally unique
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
# terminal → pin the primary's herdr terminal id. Omit to learn it from the connection on
|
||||
# the first orchestration-side MCP call (the normal case). An off-host or
|
||||
# non-herdr primary leaves this unresolved → the loop is a no-op and delivery
|
||||
# degrades to pull; the reply is still never lost.
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just fleetd-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off fleet_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
# primary:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
@@ -5,17 +5,17 @@
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
<groupId>dev.ltms</groupId>
|
||||
<artifactId>bridged</artifactId>
|
||||
<artifactId>fleetd</artifactId>
|
||||
<version>1.0.0</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<name>bridged</name>
|
||||
<description>claude-bridge message server: sole gateway between primary/worker Claude sessions and herdr</description>
|
||||
<name>fleetd</name>
|
||||
<description>fleet message server: sole gateway between a lead session, its members, and herdr</description>
|
||||
|
||||
<properties>
|
||||
<maven.compiler.release>25</maven.compiler.release>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<mainClass>dev.ltms.bridged.Bridged</mainClass>
|
||||
<mainClass>dev.ltms.fleet.Fleetd</mainClass>
|
||||
|
||||
<jackson.version>2.19.0</jackson.version>
|
||||
<javalin.version>6.7.0</javalin.version>
|
||||
@@ -28,6 +28,7 @@
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -44,6 +45,12 @@
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
|
||||
fleetd #206: org.xerial:sqlite-jdbc 3.53.4.0 (added for OpenCodeSessionDiscovery) — the
|
||||
only known advisory against this artifact is CVE-2023-32697 (RCE via an attacker-controlled
|
||||
JDBC URL), fixed in 3.41.2.2; 3.53.4.0 is well past that fix and OSV.dev reports no open
|
||||
advisory against it. Checked via the OSV.dev API (no Mend.io/JetBrains IDE MCP mount
|
||||
available from this worktree) on 2026-08-31.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
@@ -106,7 +113,7 @@
|
||||
</dependency>
|
||||
|
||||
<!-- MCP server: the SERVER face. Streamable-HTTP servlet mounted on Javalin's Jetty at
|
||||
/mcp, exposing bridge_send/bridge_reply/bridge_status as thin adapters over REST. -->
|
||||
/mcp, exposing fleet_send/fleet_reply/fleet_status as thin adapters over REST. -->
|
||||
<dependency>
|
||||
<groupId>io.modelcontextprotocol.sdk</groupId>
|
||||
<artifactId>mcp</artifactId>
|
||||
@@ -123,6 +130,17 @@
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #206: opencode moved its session store from a JSON tree to SQLite
|
||||
(opencode.db). This is the JDBC driver OpenCodeSessionDiscovery uses to read it
|
||||
read-only. Ships bundled native libraries (linux/mac/windows, several archs), so it
|
||||
is a heavier jar than most deps here — see the pom's dependency-security note below
|
||||
for the size/CVE tradeoff actually measured. -->
|
||||
<dependency>
|
||||
<groupId>org.xerial</groupId>
|
||||
<artifactId>sqlite-jdbc</artifactId>
|
||||
<version>${sqlite-jdbc.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -161,7 +179,10 @@
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
<finalName>bridged</finalName>
|
||||
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
||||
launchd plist together. The installed plist names fleetd/target/fleetd.jar and
|
||||
KeepAlive is armed, so this name, the plist, and the wrapper must move as one. -->
|
||||
<finalName>fleetd</finalName>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
@@ -203,7 +224,7 @@
|
||||
</configuration>
|
||||
</plugin>
|
||||
|
||||
<!-- Runnable fat jar: java -jar target/bridged.jar -->
|
||||
<!-- Runnable fat jar: java -jar target/fleetd.jar -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-shade-plugin</artifactId>
|
||||
@@ -0,0 +1,913 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.ConfigWatcher;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||
import dev.ltms.fleet.lead.LeadLauncher;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.inject.TurnListener;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.fleet.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.fleet.msg.AmqpReplyInbox;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadMailbox;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.rest.FleetApp;
|
||||
import dev.ltms.fleet.session.GitWorktrees;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.session.SessionReaper;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.member.OpenCodeLauncher;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* {@code fleetd} entry point. Wires the real herdr socket client to the REST app and
|
||||
* starts listening. Before anything else it asserts its own environment is clean —
|
||||
* {@code fleetd} is not a Claude process and must never carry a base_url.
|
||||
*/
|
||||
public final class Fleetd {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Fleetd.class);
|
||||
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
/**
|
||||
* CB-637: how often the lead coordination loop looks for peer messages. A few seconds — slow
|
||||
* enough that an idle fleet is not polling a broker in a tight loop, fast enough that a peer
|
||||
* lead's message is not left sitting once the local lead reaches a turn boundary. The mailbox
|
||||
* pushes into the loop's held set on its own consumer thread, so this interval bounds only the
|
||||
* pane delivery, never the receive.
|
||||
*/
|
||||
private static final long LEAD_COORD_INTERVAL_MS = 3_000L;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
/**
|
||||
* CB-632/CB-634: prefer {@code fleetd.yaml} in {@code dir}; fall back to the legacy
|
||||
* {@code bridged.yaml} when the new name is not there. The product renamed to {@code fleetd},
|
||||
* but a deployment whose local config file is still {@code bridged.yaml} keeps working until
|
||||
* that file is renamed.
|
||||
*/
|
||||
static Path chooseDefaultConfigFile(Path dir) {
|
||||
Path fleetd = dir.resolve("fleetd.yaml");
|
||||
if (Files.exists(fleetd)) {
|
||||
return fleetd;
|
||||
}
|
||||
return dir.resolve("bridged.yaml");
|
||||
}
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = args.length > 0 ? Path.of(args[0]) : chooseDefaultConfigFile(Path.of(""));
|
||||
// The config file was renamed bridged.yaml -> fleetd.yaml. Name the file we actually
|
||||
// loaded, whichever of the two names it carries.
|
||||
log.info("Using configuration file {}", configPath);
|
||||
FleetConfig cfg = FleetConfig.load(configPath);
|
||||
// CB-594: report which secret env vars the config actually needs, by name, before anything
|
||||
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
|
||||
// boots fine either way — this is the only thing that says so out loud.
|
||||
reportRequiredSecrets(cfg);
|
||||
// CB-596: an absent (or empty) memberCredentials: block blocks NOTHING — no credential
|
||||
// name is hardcoded any more to fall back on. Say so loudly, the same way a missing
|
||||
// secret is reported above, so upgrading past this commit never silently drops CB-592's
|
||||
// protection.
|
||||
reportMemberCredentialsGap(cfg);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched fleetd must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
cfg.validateCharters();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
UnixSocketHerdrClient memberHerdr = cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()
|
||||
? UnixSocketHerdrClient.connect(Path.of(cfg.memberHerdrSocket()), new com.fasterxml.jackson.databind.ObjectMapper())
|
||||
: herdr;
|
||||
AtomicReference<Supplier<Map<String, String>>> leadsRef = new AtomicReference<>(Map::of);
|
||||
HerdrRouter router = new HerdrRouter(herdr, memberHerdr,
|
||||
target -> leadsRef.get().get().containsKey(target));
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, FleetConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, FleetConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(router.memberAgents(), router.memberSpaces(), guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials(), null, config::get));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(router.memberAgents(), router.memberSpaces(),
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials(), config::get));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
// The cooldown is deferred (see FleetConfig#quarantineCooldownSeconds): it is read once
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine);
|
||||
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot(), cfg.worktreeGroup()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// The lead and the members now share ONE workspace (the operator asked for a single
|
||||
// "session" with many tabs), so no workspace can be excluded — the lead lives in the
|
||||
// members' space by design. A lead is told from a member by its exact tab label alone:
|
||||
// a lead carries its configured `tab`, a member its `worker: {profile} #{n}` template,
|
||||
// and the two never collide. (The scanner still supports an exclusion set for a split
|
||||
// layout; the fleet's policy here is simply not to use one.)
|
||||
//
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
// This must use the lead daemon: scanning member tabs would demote the lead to a worker.
|
||||
leads = new LeadTabScanner(herdr, tabToName, Set.of(),
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, shared fleet space)",
|
||||
tabToName.keySet(), scanIntervalSeconds);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
leadsRef.set(leads);
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(router.leadAgents(), router.leadSpaces(), cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
// Compiled once at startup, keyed by profile name; a profile with no exhaustedPattern is
|
||||
// simply absent here, so its workers keep today's completion-fallback behaviour unchanged.
|
||||
Map<String, Pattern> exhaustedPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasExhaustedPattern()) {
|
||||
exhaustedPatternsByProfile.put(name, Pattern.compile(profile.exhaustedPattern()));
|
||||
}
|
||||
});
|
||||
ExhaustedPatternLookup exhaustedPatterns = target -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.map(profileName -> config.get().profiles().get(profileName))
|
||||
.ifPresent(profile -> {
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}' classified "
|
||||
+ "BACKEND_EXHAUSTED): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
});
|
||||
AgentControl agents = router.memberAgents();
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns, exhaustionSink);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Predicate<String> deliverable = deliverableTo(presence, leads);
|
||||
Injector injector = new Injector(router, turnListener, deliverable,
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block selects the AMQP-backed durable adapter; absent (or
|
||||
// unusable), fleetd stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox = selectReplyInbox(cfg.broker(), System.getenv(), AmqpReplyInbox::open);
|
||||
// CB-637: this daemon's lead-to-lead mailbox on the SHARED coordination vhost — a separate
|
||||
// broker from the reply inbox by design (see FleetConfig.Coordinator). Absent a coordinator:
|
||||
// block this is null and every lead path below is simply not wired, which is exactly the
|
||||
// behaviour before this ticket. It owns a broker connection, so keep the reference for the
|
||||
// ordered shutdown hook.
|
||||
final LeadMailbox leadMailbox = openLeadMailbox(cfg.coordinator(), System.getenv(), LeadMailbox::open);
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open fleet_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = FleetMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, router.leadAgents(), replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, router.leadAgents(), replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(router, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(),
|
||||
cfg.health().workingSuspectAfterOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(detail -> {
|
||||
// CB-578 stage C, acceptance criterion 10: a failed ticket's detail should tell a lead
|
||||
// where to re-dispatch onto the same tree, not just that the worker vanished.
|
||||
String reason = "the worker session was released before it replied";
|
||||
if (detail.worktreePath() != null) {
|
||||
reason += "; worktree=" + detail.worktreePath() + " branch=" + detail.branch()
|
||||
+ " snapshot=" + (detail.snapshotRef() != null ? detail.snapshotRef() : "none");
|
||||
}
|
||||
// CB-584 (issue #65 criterion 5): also name the agent session, so a lead can resume the
|
||||
// member's conversation instead of only re-dispatching a fresh one onto the same files.
|
||||
if (detail.agentSessionId() != null) {
|
||||
reason += " agentSessionId=" + detail.agentSessionId();
|
||||
}
|
||||
messages.abandon(detail.terminalId(), reason);
|
||||
replyInbox.release(detail.terminalId());
|
||||
primaryRegistry.forgetDelegation(detail.terminalId()); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): fleet_send/fleet_reply/fleet_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
// CB-185: a caller's pane can live on either daemon (a lead's on the lead daemon, a
|
||||
// member's on the member daemon) — search both, lead first. Collapses to one scan when
|
||||
// memberHerdrSocket is unset (herdr == memberHerdr).
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr, memberHerdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting fleetd");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new FleetMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
new FleetMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}),
|
||||
new FleetMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine),
|
||||
leadMailbox);
|
||||
|
||||
// CB-637: the receive half. Only constructed when a lead mailbox actually opened — with no
|
||||
// coordinator (or an unreachable one) there is nothing to deliver, so no scheduler is
|
||||
// created and no thread runs. It reads the SAME live lead supplier the injector's
|
||||
// deliverability gate does, so a lead found by the tab scan after startup is reachable
|
||||
// without a restart.
|
||||
final LeadCoordLoop leadCoordLoop;
|
||||
final ScheduledExecutorService leadCoordSchedulerRef;
|
||||
if (leadMailbox != null) {
|
||||
var leadCoordScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-leadcoord-").unstarted(r));
|
||||
leadCoordLoop = new LeadCoordLoop(leadMailbox, router.leadAgents(), leads, leadCoordScheduler,
|
||||
LEAD_COORD_INTERVAL_MS);
|
||||
leadCoordLoop.start();
|
||||
leadCoordSchedulerRef = leadCoordScheduler;
|
||||
} else {
|
||||
leadCoordLoop = null;
|
||||
leadCoordSchedulerRef = null;
|
||||
}
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (leadCoordLoop != null) leadCoordLoop.close(); // CB-637: stop delivering peer-lead messages
|
||||
if (leadCoordSchedulerRef != null) leadCoordSchedulerRef.shutdownNow();
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
// CB-637: the coordination connection goes with it — after the loop that reads it has
|
||||
// stopped, so no tick can be mid-ack against a closed channel.
|
||||
if (leadMailbox != null) {
|
||||
try {
|
||||
leadMailbox.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
router.close();
|
||||
}));
|
||||
|
||||
// CB-185: give FleetApp both daemons — /healthz must require both to answer and
|
||||
// GET /sessions must merge across both, or a down/unpolled member daemon is invisible.
|
||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a spawned
|
||||
* member whose agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> member's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code FleetMcp} marks presence
|
||||
* for every spawned member (worker and architect), deliberately, since that map doubles as the
|
||||
* member roster's availability signal and a lead counted there would show up as an available
|
||||
* member. So without the second disjunct a lead is permanently un-deliverable: every
|
||||
* lead→lead send sat on the gate for {@code READINESS_GRACE_POLLS} (~60s) and then failed
|
||||
* having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #selectReplyInbox}: production binds {@link AmqpReplyInbox#open}. */
|
||||
@FunctionalInterface
|
||||
interface AmqpOpener {
|
||||
ReplyInbox open(String uri, int prefetch);
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #openLeadMailbox}: production binds {@link LeadMailbox#open}. */
|
||||
@FunctionalInterface
|
||||
interface LeadMailboxOpener {
|
||||
LeadMailbox open(String uri, String selfCoordId, int prefetch);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-637: open this daemon's lead-to-lead mailbox, or return {@code null} to leave the feature
|
||||
* off. Package-private and env-injected for the same reason as {@link #selectReplyInbox}: the
|
||||
* selection is then testable without a broker or a mutable process environment.
|
||||
*
|
||||
* <p>Every "off" path returns {@code null}, and each says why at the level it deserves:
|
||||
*
|
||||
* <ul>
|
||||
* <li>no {@code coordinator:} block — silent. Lead coordination is opt-in; an operator who
|
||||
* never configured it does not need to be told it is off on every boot.</li>
|
||||
* <li>a block whose {@code uriEnv} does not resolve — INFO, the same "you moved to the secret
|
||||
* store and the variable is not there" case {@code selectReplyInbox} warns about.</li>
|
||||
* <li>a configured broker but no {@code selfId} — WARN. This one is a half-finished config: a
|
||||
* mailbox is named after the coord-id that owns it, so with no id there is no queue to own
|
||||
* and no {@code from} to send as. Loud, because the operator plainly intended the feature.</li>
|
||||
* <li>the broker refuses at boot — WARN, and carry on. Mirrors {@code openAmqpOrFallback}: a
|
||||
* coordination broker that is down must never take a whole fleet's daemon with it, and the
|
||||
* fleet still works exactly as it did before this feature existed.</li>
|
||||
* </ul>
|
||||
*/
|
||||
static LeadMailbox openLeadMailbox(FleetConfig.Coordinator coordinator, Map<String, String> env,
|
||||
LeadMailboxOpener opener) {
|
||||
if (coordinator == null) {
|
||||
return null; // opt-in: nothing configured, nothing to say
|
||||
}
|
||||
String uri = coordinator.effectiveUri(env);
|
||||
if (uri == null) {
|
||||
log.info("lead coordination: OFF — coordinator{} has no usable broker uri",
|
||||
coordinator.uriEnv() == null ? "" : ".uriEnv=" + coordinator.uriEnv());
|
||||
return null;
|
||||
}
|
||||
if (coordinator.selfId() == null || coordinator.selfId().isBlank()) {
|
||||
log.warn("coordinator.selfId is unset — lead coordination is OFF. A lead mailbox is the "
|
||||
+ "queue named after the coord-id that owns it, so with no id there is nothing to "
|
||||
+ "own and no sender identity to publish as. Set coordinator.selfId to a name that "
|
||||
+ "is unique across every daemon sharing {} and restart fleetd.",
|
||||
stripCredentials(uri));
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
LeadMailbox mailbox = opener.open(uri, coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
log.info("lead coordination: ON as coord-id {} (prefetch={})",
|
||||
coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
return mailbox;
|
||||
} catch (IllegalStateException e) {
|
||||
log.warn("cannot reach the AMQP coordination broker ({}) — lead-to-lead messaging is OFF "
|
||||
+ "for this process lifetime. fleet_send{{coordId}} will report it as "
|
||||
+ "not configured, and peer messages already queued stay on the broker until "
|
||||
+ "a restart picks them up. Reason: {}",
|
||||
stripCredentials(uri), reasonOf(e));
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-151/152: pick the reply inbox. A usable broker — a literal {@code uri}, or a {@code
|
||||
* uriEnv} whose variable resolves (both read from {@code env}) — selects the durable AMQP inbox.
|
||||
* Everything else falls back to the in-memory inbox: no broker block, a blank {@code uri}, a
|
||||
* {@code uriEnv} whose variable is unset or blank, or a broker unreachable at boot. The two
|
||||
* lossy paths warn <em>loudly</em> — never silently — because what is lost is durable,
|
||||
* cross-restart reply delivery. Package-private and env-injected so the selection is testable
|
||||
* without a real broker or a mutable process environment.
|
||||
*/
|
||||
static ReplyInbox selectReplyInbox(FleetConfig.Broker broker, Map<String, String> env, AmqpOpener amqp) {
|
||||
if (broker == null) {
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
if (broker.hasUriEnv()) {
|
||||
// uriEnv is authoritative whenever set (CB-151): the operator moved off clear text, so
|
||||
// it must not quietly fall back onto a stale literal uri.
|
||||
if (broker.uri() != null && !broker.uri().isBlank()) {
|
||||
log.info("broker.uri is ignored because broker.uriEnv={} is set", broker.uriEnv());
|
||||
}
|
||||
String effectiveUri = broker.effectiveUri(env);
|
||||
if (effectiveUri == null) {
|
||||
log.warn("broker.uriEnv={} is unset or blank — durable AMQP reply inbox DISABLED. "
|
||||
+ "Replies are soft-state and will not survive a restart. Set {} in the "
|
||||
+ "daemon's environment (see scripts/redeploy-fleetd.sh) and restart to "
|
||||
+ "use the durable broker inbox.",
|
||||
broker.uriEnv(), broker.uriEnv());
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
log.info("reply inbox: AMQP broker (durable) via env var {} (prefetch={})",
|
||||
broker.uriEnv(), broker.prefetchOrDefault());
|
||||
return openAmqpOrFallback(effectiveUri, broker.prefetchOrDefault(), "uriEnv " + broker.uriEnv(), amqp);
|
||||
}
|
||||
// No uriEnv: the literal uri path (existing behaviour).
|
||||
if (broker.effectiveUri(env) == null) {
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
log.info("reply inbox: AMQP broker (durable) (prefetch={})", broker.prefetchOrDefault());
|
||||
return openAmqpOrFallback(broker.uri(), broker.prefetchOrDefault(), "uri", amqp);
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the AMQP inbox, falling back to the in-memory inbox for this process lifetime if the
|
||||
* broker cannot be reached at boot (CB-152). Not silent: the warning says durable delivery is
|
||||
* off, replies are soft-state and will not survive a restart, plus the source that failed and
|
||||
* the URI <em>with credentials stripped</em>. Never retries in the background — a broker that
|
||||
* drops <em>after</em> startup already self-heals via the connection factory's automatic
|
||||
* recovery; only the boot path is changed here.
|
||||
*/
|
||||
private static ReplyInbox openAmqpOrFallback(String effectiveUri, int prefetch, String source,
|
||||
AmqpOpener amqp) {
|
||||
try {
|
||||
return amqp.open(effectiveUri, prefetch);
|
||||
} catch (IllegalStateException e) {
|
||||
log.warn("cannot reach AMQP broker ({}, {}) — falling back to the in-memory reply inbox "
|
||||
+ "for this process lifetime. Durable, cross-restart reply delivery is OFF; "
|
||||
+ "replies are soft-state and will not survive a restart. Reason: {}",
|
||||
source, stripCredentials(effectiveUri), reasonOf(e));
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
}
|
||||
|
||||
/** An AMQP URI carries {@code user:pass@} inline — show the host/port, never the credentials. */
|
||||
static String stripCredentials(String uri) {
|
||||
return uri == null ? null : uri.replaceAll("://[^@/]*@", "://");
|
||||
}
|
||||
|
||||
/** The deepest cause's class and message — the outermost {@code IllegalStateException} echoes the URI (with password). */
|
||||
private static String reasonOf(Throwable e) {
|
||||
Throwable t = e;
|
||||
while (t.getCause() != null && t.getCause() != t) {
|
||||
t = t.getCause();
|
||||
}
|
||||
String msg = t.getMessage();
|
||||
return t.getClass().getSimpleName() + (msg == null || msg.isBlank() ? "" : ": " + msg);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: which env vars the loaded config actually needs, and why — every non-{@code
|
||||
* subscription} profile's {@code tokenEnv} (a subscription profile never reads one, see
|
||||
* {@link FleetConfig.Profile#isSubscription()}), plus every profile's {@code gitTokenEnv}
|
||||
* where set (opt-in), plus a configured {@code broker.uriEnv} (CB-151). Derived from the
|
||||
* config, not hard-coded, so a new profile is covered for free. A var required by more than one
|
||||
* profile is one entry naming every profile that needs it. Deliberately excludes {@code
|
||||
* auth.tokenEnv}: that one is already enforced loudly, by a startup throw in {@code main()} —
|
||||
* about 370 lines <em>below</em> this method's call site
|
||||
* ({@link #reportRequiredSecrets(FleetConfig)}), not a few lines above it. That throw only
|
||||
* fires when {@code auth.mode: token} is configured; under the default loopback-trust mode it
|
||||
* never runs, and {@code auth.tokenEnv} is simply not required.
|
||||
*
|
||||
* <p>Package-private and pure (no I/O, no logging) so the derivation is unit-testable without
|
||||
* capturing log output; {@link #reportRequiredSecrets(FleetConfig)} is the logging caller.
|
||||
*/
|
||||
static Map<String, List<String>> requiredSecretEnvVars(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (!profile.isSubscription()) {
|
||||
requiredBy.computeIfAbsent(profile.tokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' tokenEnv");
|
||||
}
|
||||
if (profile.hasGitToken()) {
|
||||
requiredBy.computeIfAbsent(profile.gitTokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' gitTokenEnv");
|
||||
}
|
||||
});
|
||||
FleetConfig.Broker broker = cfg.broker();
|
||||
if (broker != null && broker.hasUriEnv()) {
|
||||
requiredBy.computeIfAbsent(broker.uriEnv(), _ -> new ArrayList<>())
|
||||
.add("broker uriEnv");
|
||||
}
|
||||
return requiredBy;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: log, by name only, which required env vars (see {@link #requiredSecretEnvVars}) are
|
||||
* set in the daemon's own process environment — the environment every profile's {@code
|
||||
* tokenEnv}/{@code gitTokenEnv} is read from at spawn time (see
|
||||
* {@code HerdrPeerLauncher.resolveEnv}). Never logs a value, a prefix, or a length.
|
||||
*
|
||||
* <p>A missing entry only warns — it must never refuse to start. A daemon that boots and says
|
||||
* what is wrong is strictly more useful than one that will not boot at all.
|
||||
*/
|
||||
private static void reportRequiredSecrets(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = requiredSecretEnvVars(cfg);
|
||||
if (requiredBy.isEmpty()) {
|
||||
log.info("startup secrets: no profile references a token env var — nothing to check");
|
||||
return;
|
||||
}
|
||||
Map<String, String> env = System.getenv();
|
||||
requiredBy.forEach((varName, sources) -> {
|
||||
String value = env.get(varName);
|
||||
if (value != null && !value.isBlank()) {
|
||||
log.info("startup secret {}: set ({})", varName, String.join(", ", sources));
|
||||
} else {
|
||||
log.warn("startup secret {}: MISSING ({}) — the daemon will start anyway, and this "
|
||||
+ "failure stays invisible until a worker actually needs it. Fix "
|
||||
+ "${SHARED_ENV}/tools/secrets.sh and restart fleetd from a LOGIN "
|
||||
+ "shell (see scripts/redeploy-fleetd.sh).",
|
||||
varName, String.join(", ", sources));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: {@code known:} empty (block absent entirely, or present but empty) means {@link
|
||||
* FleetConfig.MemberCredentials#blockedSet()} is empty too — every member pane inherits the
|
||||
* operator's whole secret store, unblocked, exactly the defect this ticket fixes. Unlike a
|
||||
* missing token ({@link #reportRequiredSecrets}), there is no name to point at: the point is
|
||||
* that the block itself is missing. Warn once at startup and say what to add; never refuse to
|
||||
* start over it — see {@link #reportRequiredSecrets} for why a daemon that boots and says
|
||||
* what is wrong beats one that will not boot at all.
|
||||
*
|
||||
* <p>Package-private so the test can capture the log directly, the same way {@link
|
||||
* #requiredSecretEnvVars} is exposed for {@link #reportRequiredSecrets}'s own test.
|
||||
*/
|
||||
static void reportMemberCredentialsGap(FleetConfig cfg) {
|
||||
FleetConfig.MemberCredentials creds = cfg.memberCredentials();
|
||||
if (creds != null && !creds.known().isEmpty()) {
|
||||
log.info("memberCredentials: policy={}, {} known name(s), {} allowed — blocking {} on "
|
||||
+ "every spawn{}",
|
||||
creds.policy(), creds.known().size(), creds.allow().size(), creds.blockedSet().size(),
|
||||
creds.isAllowList()
|
||||
? " (allow-list: known/allow are reporting only — the control is the derived ZDOTDIR scrub)"
|
||||
: "");
|
||||
return;
|
||||
}
|
||||
log.warn("memberCredentials: absent or empty — the daemon will start anyway, and every "
|
||||
+ "member pane inherits the operator's WHOLE secret store, unblocked (CB-592's "
|
||||
+ "protection is lost). Add a memberCredentials: block (policy/allow/known) to "
|
||||
+ "fleetd.yaml — see fleetd.example.yaml — and restart.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Fleetd() {
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
+2
-2
@@ -1,9 +1,9 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||
*
|
||||
* <p>Most of these rules are already true de facto — {@code BridgeMcp} derives a worker's identity
|
||||
* <p>Most of these rules are already true de facto — {@code FleetMcp} derives a worker's identity
|
||||
* from the connection rather than reading it from an argument, so a worker has never been able to
|
||||
* reply <em>as</em> another worker over MCP. What was missing is that the REST surface trusted the
|
||||
* session id in the URL path, and neither surface checked role at all. This class makes the
|
||||
+51
-39
@@ -1,16 +1,18 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Resolves every caller to a {@link Principal}, for both entry paths into the core (CB-501).
|
||||
*
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code BridgeMcp}
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code FleetMcp}
|
||||
* calls the service layer directly and is mounted as a raw servlet (so it never passes through a
|
||||
* Javalin filter), while the REST routes historically resolved no identity at all. Both now
|
||||
* delegate here, so the authorization rules are stated once instead of drifting apart.
|
||||
@@ -57,17 +59,19 @@ public final class CallerResolver {
|
||||
* <p>Like {@link #leadTerminals}, a supplier rather than a fixed map, so a binding injected
|
||||
* after startup — when the later spawn lifecycle establishes a live architect session, or an
|
||||
* operator pins one — takes effect without a restart. Consulted per resolve; today's wiring
|
||||
* in {@code Bridged} reads a constant from config, which is the degenerate live case.
|
||||
* in {@code Fleetd} reads a constant from config, which is the degenerate live case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> architectTerminals;
|
||||
private final Function<String, MemberRole> memberSlotRoles;
|
||||
private final Function<String, String> memberSlotNames;
|
||||
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. */
|
||||
public CallerResolver(ConnectionIdentity identity) {
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity) {
|
||||
this(identity, false, null, Map.of());
|
||||
}
|
||||
|
||||
/** As {@link #CallerResolver(ConnectionIdentity, boolean, String, Map)} with no leads pinned. */
|
||||
public CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token) {
|
||||
/** As {@link #CallerResolver(ConnectionIdentity, boolean, String, Map)} with no leads pinned. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token) {
|
||||
this(identity, tokenMode, token, Map.of());
|
||||
}
|
||||
|
||||
@@ -82,8 +86,8 @@ public final class CallerResolver {
|
||||
* @param pinnedPrimaryTerminal the primary's own herdr {@code terminal_id}
|
||||
* ({@code null}/blank = unpinned)
|
||||
*/
|
||||
public static CallerResolver pinnedTo(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token, String pinnedPrimaryTerminal) {
|
||||
static CallerResolver pinnedTo(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token, String pinnedPrimaryTerminal) {
|
||||
return new CallerResolver(identity, tokenMode, token,
|
||||
pinnedPrimaryTerminal == null || pinnedPrimaryTerminal.isBlank()
|
||||
? Map.of() : Map.of(pinnedPrimaryTerminal, "primary"));
|
||||
@@ -98,21 +102,11 @@ public final class CallerResolver {
|
||||
* {@link Role#PRIMARY} — rather than a worker. Empty = nothing pinned,
|
||||
* so every pane resolves as a worker.
|
||||
*/
|
||||
public CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals) {
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals) {
|
||||
this(identity, tokenMode, token, fixed(leadTerminals), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Map-form of both registries (CB-548): lead terminals and the initial architect terminal
|
||||
* bindings, each snapshotted at construction (a handed-over map is not offered as live state).
|
||||
*/
|
||||
public CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals,
|
||||
Map<String, String> architectTerminals) {
|
||||
this(identity, tokenMode, token, fixed(leadTerminals), fixed(architectTerminals));
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-registry form: {@code leadTerminals} is consulted on every resolve, so leads discovered
|
||||
* after startup (CB-531's tab scan) take effect without a restart.
|
||||
@@ -121,26 +115,26 @@ public final class CallerResolver {
|
||||
* {@link #pinnedTo}: {@code Map} and {@code Supplier} overloads are ambiguous for a literal
|
||||
* {@code null}.
|
||||
*/
|
||||
public static CallerResolver withLeads(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token,
|
||||
Supplier<Map<String, String>> leadTerminals) {
|
||||
static CallerResolver withLeads(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token,
|
||||
Supplier<Map<String, String>> leadTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-registry form for both {@code leadTerminals} and the CB-548 architect registry: both
|
||||
* are consulted on every resolve, so a slot binding injected after startup takes effect
|
||||
* without a restart.
|
||||
* Live registry form that can confirm a bound slot is an architect slot.
|
||||
*
|
||||
* <p>A static factory rather than a constructor overload, for the same reason as
|
||||
* {@link #pinnedTo}: too many {@code Map}/{@code Supplier} combinations to make {@code null}
|
||||
* unambiguous.
|
||||
* <p>This is the only public construction path. It keeps terminal bindings and slot roles in
|
||||
* the same {@link MemberRegistry}, so a configured architect can resolve as an architect.
|
||||
*/
|
||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals, architectTerminals);
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
MemberRegistry members) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
||||
members == null ? null : members::snapshot,
|
||||
members == null ? null : members::roleForSlot,
|
||||
members == null ? null : members::nameForSlot);
|
||||
}
|
||||
|
||||
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
||||
@@ -151,6 +145,21 @@ public final class CallerResolver {
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, memberSlotRoles, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles,
|
||||
Function<String, String> memberSlotNames) {
|
||||
if (tokenMode && (token == null || token.isBlank())) {
|
||||
throw new IllegalArgumentException(
|
||||
"auth.mode=token requires a non-empty token; check that the env var named by "
|
||||
@@ -161,6 +170,8 @@ public final class CallerResolver {
|
||||
this.expectedToken = tokenMode ? token.getBytes(StandardCharsets.UTF_8) : null;
|
||||
this.leadTerminals = leadTerminals == null ? Map::of : leadTerminals;
|
||||
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
||||
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
||||
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -206,11 +217,12 @@ public final class CallerResolver {
|
||||
return Principal.leader(lead, c.terminal(), c.pid());
|
||||
}
|
||||
String slot = architectTerminals.get().get(c.terminal());
|
||||
if (slot != null) {
|
||||
if (slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT) {
|
||||
// The config/live binding names this pane as an architect slot's own. Same
|
||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||
// Checked before the generic worker fallback, per the CB-548 precedence order.
|
||||
return Principal.architect(slot, c.terminal(), c.pid());
|
||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||
// escalating a dev or reviewer into an architect. Checked before the worker fallback.
|
||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||
}
|
||||
@@ -223,7 +235,7 @@ public final class CallerResolver {
|
||||
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (BridgedConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
}
|
||||
+46
-6
@@ -1,12 +1,15 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
@@ -29,7 +32,9 @@ import java.util.Map;
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
*/
|
||||
public final class MemberRegistry {
|
||||
public final class MemberRegistry implements MemberLifecycle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MemberRegistry.class);
|
||||
|
||||
/**
|
||||
* One flattened {@code fleet:} entry.
|
||||
@@ -54,7 +59,7 @@ public final class MemberRegistry {
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(BridgedConfig.Fleet fleet) {
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
@@ -89,7 +94,7 @@ public final class MemberRegistry {
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code bridge_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* {@code fleet_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
@@ -125,6 +130,12 @@ public final class MemberRegistry {
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
@@ -189,4 +200,33 @@ public final class MemberRegistry {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind only architect sessions to a free slot with the resolved profile.
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
String slot = slotForTerminal(terminal);
|
||||
if (slot != null) {
|
||||
unbind(slot, terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
+15
-5
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* A resolved caller: its {@link Role}, and — for a worker — the herdr {@code terminal_id} that
|
||||
@@ -39,16 +39,16 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
*
|
||||
* <p>Carries {@link Role#PRIMARY}: a lead <em>is</em> a primary as far as authorization goes,
|
||||
* so every existing {@code isPrimary()} gate keeps working unchanged and the role table needed
|
||||
* no new entry. The name is reporting only — it lets {@code bridge_whoami} say <em>which</em>
|
||||
* no new entry. The name is reporting only — it lets {@code fleet_whoami} say <em>which</em>
|
||||
* lead is asking once more than one is configured.
|
||||
*
|
||||
* <p><strong>CB-532: a lead now carries the terminal it was matched by.</strong> Under CB-530 it
|
||||
* deliberately did not, because {@code terminal} meant "which worker pane" everywhere and a
|
||||
* non-null one would have enrolled the lead in the worker presence map. That reading was what
|
||||
* made a lead unaddressable: {@link #ownsSession} could never be true for it, so
|
||||
* {@code bridge_reply} was refused and one lead could send to another but never be answered.
|
||||
* {@code fleet_reply} was refused and one lead could send to another but never be answered.
|
||||
* The terminal now means "which pane is this caller", the presence map keys on
|
||||
* {@link #isWorker()} instead, and a lead is a peer that can both send and receive.
|
||||
* {@link #isSpawnedMember()} instead, and a lead is a peer that can both send and receive.
|
||||
*/
|
||||
public static Principal leader(String name, String terminal, long pid) {
|
||||
return new Principal(Role.PRIMARY, terminal, pid, name);
|
||||
@@ -63,7 +63,7 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
* An architect (CB-548), identified by the slot it occupies and the pane bound to it.
|
||||
*
|
||||
* <p>Carries {@link Role#ARCHITECT}. {@code slotName} is reporting only — it lets
|
||||
* {@code bridge_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* {@code fleet_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* (future) spawn lifecycle reads a profile back from. Identity is the {@code terminal}: like a
|
||||
* worker's it comes from the connection and the live terminal→slot binding, so
|
||||
* {@code ownsSession} works exactly as it does for a worker — an architect acts as its own
|
||||
@@ -85,6 +85,16 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
return role == Role.WORKER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller is a spawned member with its own pane.
|
||||
*
|
||||
* <p>Both workers and architects are spawned members. A lead is excluded because recording it
|
||||
* as present would count it as an available member in the roster.
|
||||
*/
|
||||
public boolean isSpawnedMember() {
|
||||
return role == Role.WORKER || role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isAnonymous() {
|
||||
return role == Role.ANONYMOUS;
|
||||
}
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* What a caller is allowed to be on the bus (CB-501).
|
||||
+56
-28
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.config;
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -16,7 +16,7 @@ import java.util.function.Supplier;
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link BridgedConfig}, and read through {@link #get()} at the point
|
||||
* <p>Consumers hold this, not a {@link FleetConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
@@ -27,16 +27,25 @@ import java.util.function.Supplier;
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool and {@code tabLabel}), {@code placement:}, and an existing
|
||||
* profile's {@code weight} / {@code maxLoad}. Those three are read through a supplier on
|
||||
* {@code CompositePeerLauncher}, which is what makes them hot — not the fact that they are
|
||||
* config.</li>
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Fleetd.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code guard:},
|
||||
* {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl} and the rest.
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
@@ -56,30 +65,30 @@ import java.util.function.Supplier;
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<BridgedConfig> current;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
|
||||
public ConfigRef(Path path, BridgedConfig initial) {
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(BridgedConfig cfg) {
|
||||
public static ConfigRef fixed(FleetConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public BridgedConfig get() {
|
||||
public FleetConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
@@ -119,7 +128,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart bridged to apply them.";
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
@@ -140,16 +149,17 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
BridgedConfig old = current.get();
|
||||
BridgedConfig fresh;
|
||||
FleetConfig old = current.get();
|
||||
FleetConfig fresh;
|
||||
try {
|
||||
fresh = BridgedConfig.load(path);
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
@@ -172,7 +182,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
private static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
@@ -180,6 +190,9 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.memberHerdrSocket(), fresh.memberHerdrSocket())) {
|
||||
changed.add("memberHerdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
@@ -192,7 +205,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
private static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
@@ -210,9 +223,15 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
Map<String, BridgedConfig.Profile> before =
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, FleetConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, BridgedConfig.Profile> after =
|
||||
Map<String, FleetConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
@@ -231,7 +250,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
BridgedConfig.Profile now = after.get(name);
|
||||
FleetConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
@@ -245,10 +264,12 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight} and {@code maxLoad} are excluded because those are
|
||||
* read live by the placement policy and really do take effect on the next spawn.
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(BridgedConfig.Profile a, BridgedConfig.Profile b) {
|
||||
private static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
@@ -258,12 +279,19 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
// CB-634: the IDE MCP mount is a launch flag, fixed at spawn like mcpUrl — a
|
||||
// reload changes it only for members spawned after, so a changed value is deferred.
|
||||
&& Objects.equals(a.ideMcpUrl(), b.ideMcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription());
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
}
|
||||
}
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.config;
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -11,7 +11,7 @@ import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Polls {@code bridged.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* Polls {@code fleetd.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* (CB-559). Opt-in through {@code configReload.enabled}.
|
||||
*
|
||||
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
||||
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.guard;
|
||||
package dev.ltms.fleet.guard;
|
||||
|
||||
/** Thrown when the subscription boundary would be violated. Never swallow this. */
|
||||
public class GuardException extends RuntimeException {
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.guard;
|
||||
package dev.ltms.fleet.guard;
|
||||
|
||||
import java.net.URI;
|
||||
import java.net.URISyntaxException;
|
||||
@@ -16,7 +16,7 @@ import java.util.Set;
|
||||
* means its traffic would leave the subscription. That is a hard stop.</li>
|
||||
* </ul>
|
||||
*
|
||||
* Both checks throw {@link GuardException} on violation. {@code bridged} calls
|
||||
* Both checks throw {@link GuardException} on violation. {@code fleetd} calls
|
||||
* {@link #assertWorker} before spawning a worker and {@link #assertPrimaryClean}
|
||||
* against its own environment at startup.
|
||||
*/
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
|
||||
/**
|
||||
* Pure classifier. Collection and repair are deliberately outside this package.
|
||||
* {@link HealthState#ERROR_ON_SCREEN} is not decided yet because it needs a bounded pane detection
|
||||
* read and an adapter-specific fatal signature; status facts alone must not guess it.
|
||||
*/
|
||||
public final class FleetHealth {
|
||||
private FleetHealth() { }
|
||||
|
||||
public static HealthDecision decide(HealthSnapshot s, HealthPrior prior, long nowNanos) {
|
||||
if (s.controlLinkDown()) return result(HealthState.CONTROL_LINK_DOWN, false);
|
||||
if (s.targetNotFound()) return result(HealthState.GONE, false);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING && !s.present() && s.readinessGraceElapsed()) {
|
||||
return result(HealthState.NEVER_READY, false);
|
||||
}
|
||||
if (s.orphanedDelegation()) return result(HealthState.DELEGATION_ORPHANED, false);
|
||||
boolean disagreement = s.sessionState() == MemberSession.State.BUSY && s.acceptedDelivery()
|
||||
&& (s.liveStatus() == AgentStatus.IDLE || s.liveStatus() == AgentStatus.DONE);
|
||||
if (disagreement && prior.busyButDone()) return result(HealthState.TURN_BOUNDARY_LOST, true);
|
||||
if (s.stalled()) return result(HealthState.STALL_SUSPECTED, disagreement);
|
||||
if (s.replyStranded()) return result(HealthState.REPLY_STRANDED, disagreement);
|
||||
if (s.queuedDelivery() || s.inboxMessage()) return result(HealthState.WORK_PENDING, disagreement);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING) return result(HealthState.STARTING, disagreement);
|
||||
if (s.acceptedDelivery() && s.liveStatus() == AgentStatus.BLOCKED) {
|
||||
return result(HealthState.BLOCKED_AMBIGUOUS, disagreement);
|
||||
}
|
||||
// An accepted delivery remains bridge work even when herdr is late, unknown, or has already
|
||||
// reported DONE once. It cannot be IDLE until the delegation has resolved.
|
||||
if (s.acceptedDelivery()) return result(HealthState.WORKING, disagreement);
|
||||
return result(HealthState.IDLE, disagreement);
|
||||
}
|
||||
|
||||
private static HealthDecision result(HealthState state, boolean disagreement) {
|
||||
return new HealthDecision(state, new HealthPrior(disagreement));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long intervalSeconds;
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
* for one tick during an ordinary race — an async ticket exists before its virtual thread has
|
||||
* reached {@code rendezvous.open()}, so for that instant nothing is accepted or queued behind
|
||||
* it. {@code decide} maps the field straight to {@code DELEGATION_ORPHANED} with no cross-tick
|
||||
* smoothing of its own, so a single racy read would log a fault that clears on the next tick.
|
||||
* Requiring two consecutive observations costs one interval of latency on a real orphan and
|
||||
* removes that false positive entirely.
|
||||
*/
|
||||
private final Map<String, Integer> orphanStreaks = new HashMap<>();
|
||||
|
||||
/** How many consecutive ticks a target must look orphaned before health reports it (CB-643). */
|
||||
static final int ORPHAN_CONFIRM_TICKS = 2;
|
||||
|
||||
// CB-643: every HealthSnapshot field now carries real evidence. The NOT_YET_OBSERVED placeholder
|
||||
// that stood in for 7 of the 12 is gone, and with it the reason 8 of the 9 fault states were
|
||||
// unreachable — GONE and NEVER_READY included, which is what kept CB-580's failTarget from ever
|
||||
// firing. Do not reintroduce a constant here: a field with no publisher is a dead state, and the
|
||||
// tests pass either way, so nothing else will tell you.
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code FleetMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
long workingSuspectAfterSeconds, BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.workingSuspectAfterNanos = TimeUnit.SECONDS.toNanos(workingSuspectAfterSeconds);
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
List<Agent> agentsNow;
|
||||
boolean controlLinkDown = false;
|
||||
try {
|
||||
agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
} catch (HerdrException error) {
|
||||
agentsNow = List.of();
|
||||
controlLinkDown = true;
|
||||
log.warn("fleet health control link unavailable; classifying roster", error);
|
||||
}
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
long nowNanos = clock.getAsLong();
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean present = agent != null;
|
||||
boolean targetNotFound = !controlLinkDown && !present
|
||||
&& session.state() != MemberSession.State.SPAWNING;
|
||||
boolean readinessGraceElapsed = nowNanos - session.spawnedAtNanos() >= READINESS_GRACE_NANOS;
|
||||
boolean stalled = session.state() == MemberSession.State.BUSY
|
||||
&& nowNanos - session.lastActivityAtNanos() >= workingSuspectAfterNanos;
|
||||
// CB-643: the three message-layer facts CB-640 published. Read them here rather than
|
||||
// leaving them false — that constant is what made 8 of the 9 fault states dead.
|
||||
boolean queuedDelivery = messages.hasQueuedDelivery(session.terminalId());
|
||||
boolean replyStranded = messages.hasStrandedReply(session.terminalId());
|
||||
boolean orphanedDelegation = confirmOrphan(session.terminalId(),
|
||||
messages.hasOrphanedDelegation(session.terminalId()));
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, queuedDelivery,
|
||||
messages.hasInboxMessage(session.terminalId()), present, targetNotFound, controlLinkDown,
|
||||
readinessGraceElapsed, orphanedDelegation, replyStranded, stalled);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
nowNanos);
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
orphanStreaks.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// Any unclassified collection failure must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Debounce {@link MessageService#hasOrphanedDelegation} across ticks (CB-643). Returns true only
|
||||
* once {@code observed} has held for {@link #ORPHAN_CONFIRM_TICKS} consecutive ticks; a single
|
||||
* false reading resets the streak, so a transient race never reaches the classifier.
|
||||
*/
|
||||
private boolean confirmOrphan(String target, boolean observed) {
|
||||
if (!observed) {
|
||||
orphanStreaks.remove(target);
|
||||
return false;
|
||||
}
|
||||
int streak = orphanStreaks.merge(target, 1, Integer::sum);
|
||||
return streak >= ORPHAN_CONFIRM_TICKS;
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
/** Classification plus the private fact that the next pure decision needs. */
|
||||
public record HealthDecision(HealthState state, HealthPrior prior) { }
|
||||
@@ -0,0 +1,6 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
/** Private cross-tick observation. It is deliberately not a reported health value. */
|
||||
public record HealthPrior(boolean busyButDone) {
|
||||
public static final HealthPrior NONE = new HealthPrior(false);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
|
||||
/** Read-only facts from one fleet collection tick. */
|
||||
public record HealthSnapshot(MemberSession.State sessionState, AgentStatus liveStatus,
|
||||
boolean acceptedDelivery, boolean queuedDelivery, boolean inboxMessage,
|
||||
boolean present, boolean targetNotFound, boolean controlLinkDown,
|
||||
boolean readinessGraceElapsed, boolean orphanedDelegation,
|
||||
boolean replyStranded, boolean stalled) { }
|
||||
@@ -0,0 +1,8 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
/** Health classifications reported for a member. */
|
||||
public enum HealthState {
|
||||
STARTING, IDLE, WORKING, WORK_PENDING, BLOCKED_AMBIGUOUS,
|
||||
NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Counts turns that ended via the completion fallback instead of {@code fleet_reply}.
|
||||
* MUTE is an observation by target and profile, not a classifier state and never suppresses faults.
|
||||
*/
|
||||
public final class MuteCounter {
|
||||
private final Map<String, Integer> byTarget = new ConcurrentHashMap<>();
|
||||
private final Map<String, Integer> byProfile = new ConcurrentHashMap<>();
|
||||
|
||||
/** Record only fallback completion; a structured reply does not make a member mute. */
|
||||
public void observe(String target, String profile, Rendezvous.Kind kind) {
|
||||
if (kind != Rendezvous.Kind.COMPLETION) return;
|
||||
byTarget.merge(target, 1, Integer::sum);
|
||||
byProfile.merge(profile, 1, Integer::sum);
|
||||
}
|
||||
|
||||
public int forTarget(String target) { return byTarget.getOrDefault(target, 0); }
|
||||
public int forProfile(String profile) { return byProfile.getOrDefault(profile, 0); }
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/** Fixed pane-probe limits. Pane content is never retained here. */
|
||||
public final class PaneBudget {
|
||||
public static final long COOLDOWN_NANOS = 60_000_000_000L;
|
||||
public static final int MAX_PER_TICK = 2;
|
||||
private final Map<String, Long> lastProbe = new HashMap<>();
|
||||
private int cursor;
|
||||
|
||||
public List<String> choose(List<String> candidates, long nowNanos, long configuredCooldownNanos) {
|
||||
long cooldown = Math.max(COOLDOWN_NANOS, configuredCooldownNanos);
|
||||
List<String> out = new ArrayList<>();
|
||||
for (int n = 0; n < candidates.size() && out.size() < MAX_PER_TICK; n++) {
|
||||
String target = candidates.get((cursor + n) % candidates.size());
|
||||
Long last = lastProbe.get(target);
|
||||
if (last == null || nowNanos - last >= cooldown) { out.add(target); lastProbe.put(target, nowNanos); }
|
||||
}
|
||||
if (!candidates.isEmpty()) cursor = (cursor + 1) % candidates.size();
|
||||
return List.copyOf(out);
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
+6
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
@@ -38,6 +38,11 @@ public final class AgentControl {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/** The herdr daemon this control object sends its agent calls to. */
|
||||
public HerdrClient herdr() {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
/** One agent-targeted call, translating a terminal id to its pane id (retrying once fresh). */
|
||||
private JsonNode agentCall(String method, String target, Map<String, Object> extra) {
|
||||
String resolved = resolveTarget(target);
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user