Compare commits
428 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| be5ba22c75 | |||
| ba51e0c6cc | |||
| 086c59848e | |||
| 0c10079755 | |||
| ece2091b53 | |||
| b3f917e6f5 | |||
| fa39a5f55e | |||
| d5128a1d35 | |||
| 61097e5cf0 | |||
| ef507bcd12 | |||
| 94f50e507a | |||
| 75b15086b0 | |||
| dab9645906 | |||
| e93b5f6512 | |||
| bfabe13e8f | |||
| 7e49c6eca2 | |||
| 11cbfa79b4 | |||
| c2c2746922 | |||
| 6ed70700a0 | |||
| 0087645da4 | |||
| 19cacf5b62 | |||
| 4dd12083ab | |||
| 30e21adec7 | |||
| 719b79f892 | |||
| 66e5247b6d | |||
| 1e41bd63b4 | |||
| 4887d03d88 | |||
| 7d5434455d | |||
| f71ee4926e | |||
| 282a2fc2b8 | |||
| f04e934b94 | |||
| 3fae35c357 | |||
| 18aecbfe67 | |||
| 5d75f72473 | |||
| e028a0ae54 | |||
| 2fa673d4c0 | |||
| 9020d01b40 | |||
| 1006805027 | |||
| 27aefbf9a0 | |||
| d42c2bc204 | |||
| 3916adc372 | |||
| fa97f598dd | |||
| ea9aa4fd77 | |||
| a2b8caf6b5 | |||
| b5ddbe5757 | |||
| d223a93039 | |||
| 96d8191149 | |||
| f0e7ac73d6 | |||
| e2fe861b4d | |||
| 279d6f5fbd | |||
| 34480cebef | |||
| 9d0bf14c46 | |||
| 5a3ab5764c | |||
| 3bad9f5785 | |||
| 8308c0b68f | |||
| 3b3063eb2b | |||
| df9086263d | |||
| eb568ff451 | |||
| ac790e4cce | |||
| 6b5f3f472f | |||
| 38dec72152 | |||
| 0e8bfb74fc | |||
| 51f7b0a3ca | |||
| 21c539f22e | |||
| bd2774b5f1 | |||
| c50f5b2d61 | |||
| 01a840cc14 | |||
| c4deef08be | |||
| c796eac09c | |||
| e897e5257b | |||
| 2afa3652bb | |||
| 80092ff359 | |||
| 9d37f3aa29 | |||
| eaf89abaf6 | |||
| d895f02bc1 | |||
| 43206cac2f | |||
| 8bba3a8184 | |||
| 5cf3ca9a89 | |||
| ac474981e4 | |||
| ba04b2359b | |||
| 2e5b63f6f6 | |||
| 3437d6313d | |||
| 743377d6cd | |||
| 5952d559c7 | |||
| 205ad823b0 | |||
| d654ccb818 | |||
| a89dcc9b7e | |||
| 0d7b4fb026 | |||
| 321d8dcbb5 | |||
| ef8c97871e | |||
| 26bafe824b | |||
| 838a701109 | |||
| e5eb3534c7 | |||
| 959c83534f | |||
| 31b028e860 | |||
| 7840e9adf6 | |||
| c3672f5472 | |||
| cf54aed451 | |||
| bbf68f3e3c | |||
| 776743cbe2 | |||
| 826e0aeb2a | |||
| c935b181dd | |||
| ee932fd85b | |||
| fe2e5ede34 | |||
| c325054242 | |||
| 7662e2d0c8 | |||
| 4877992a70 | |||
| 1c051c4e47 | |||
| adb7a67880 | |||
| 723fe494e9 | |||
| 0f08b93659 | |||
| 432c1d92d1 | |||
| 60b7e67b42 | |||
| 3fbd43fe3f | |||
| 32ebf065ac | |||
| 1178b3f684 | |||
| 2a434ced2f | |||
| cc919aa2b6 | |||
| 6417b0edd9 | |||
| 39c7ce76f3 | |||
| b40f477210 | |||
| 5c56cb347f | |||
| e694deace3 | |||
| 748367b7d6 | |||
| dcf5fb3be3 | |||
| f9d2ee2a2b | |||
| 866c7f2e9a | |||
| d9168de43e | |||
| c3fa1136d4 | |||
| cabcd87b66 | |||
| bf0e09b1a2 | |||
| e1eb50ce65 | |||
| 049e7d9d54 | |||
| ff3b49cd1e | |||
| 0373b6c41b | |||
| 445a45f6e1 | |||
| 966c58a3b8 | |||
| de026b8f8a | |||
| 97f6c33a45 | |||
| 735c837604 | |||
| 457dc0330d | |||
| a1a9015217 | |||
| 5a811a3695 | |||
| 1fdaa74eb3 | |||
| 7d4a4339c2 | |||
| 847e8bd3fa | |||
| f9fb387427 | |||
| a55079afbd | |||
| e18ad4723b | |||
| 6de8ac8972 | |||
| a052975420 | |||
| 6d82ca95a4 | |||
| 8067ee4ec4 | |||
| 3743789e8d | |||
| 388aba7632 | |||
| ad587eafa3 | |||
| 08ce9aef11 | |||
| ee5f8b932b | |||
| 4ac688b6d9 | |||
| a814d1ef00 | |||
| ea98856130 | |||
| a49e96835a | |||
| 63c19dcba7 | |||
| 2823349c8e | |||
| 6fc301d62c | |||
| bc99d64786 | |||
| 615af4ed0a | |||
| 045d229728 | |||
| d1fd5700f5 | |||
| 2d55b0b9a5 | |||
| d89ae94a2e | |||
| e6193c4098 | |||
| b66f0677ed | |||
| 23ada1981e | |||
| a22480c117 | |||
| 24f404f989 | |||
| a237fbff9d | |||
| 31d5516991 | |||
| 17af61e8dd | |||
| 6af87b6ad6 | |||
| 5ba05d0bdb | |||
| fc655e78c2 | |||
| 25726a5ae7 | |||
| 11c3ff67b6 | |||
| d867c87100 | |||
| b0c4cedfab | |||
| 85417d5215 | |||
| 4accc746bd | |||
| 21c4c8cbef | |||
| 430f5b0dae | |||
| 42731833d0 | |||
| 65acf066ad | |||
| ee8f570fd7 | |||
| fa3f910d44 | |||
| 46ac6e4e38 | |||
| 3bfa82839b | |||
| 65ccf2e4ad | |||
| 7a3b27f76f | |||
| 82e7be564c | |||
| c5e24197bf | |||
| bcb402b688 | |||
| 7c4170ff6d | |||
| a6095743f0 | |||
| 66e776d178 | |||
| 26f64cba45 | |||
| 51047848f1 | |||
| d8c0b657e8 | |||
| 312c0584ce | |||
| da2625acfb | |||
| 97d9cebc59 | |||
| 3ce76a5d69 | |||
| 2e138a199b | |||
| 450a5ed9c5 | |||
| edabccd885 | |||
| 1dbe3a03fc | |||
| 6058b8472b | |||
| f29968c334 | |||
| 7b98cca967 | |||
| 2757bc7185 | |||
| 7655f1b51a | |||
| a5efb7c676 | |||
| 5a8cf4cb4d | |||
| a3fc7e4df8 | |||
| d811b30df3 | |||
| 3f4ac2b24e | |||
| 4644359128 | |||
| 95a8dbcea9 | |||
| 83b50753fe | |||
| 6f968d59a4 | |||
| d432df8e5c | |||
| 4b822731e6 | |||
| e38eac1a33 | |||
| 8101290933 | |||
| 6e7fc12f89 | |||
| 50df14a50f | |||
| ccd882f2fc | |||
| ecc590f344 | |||
| 3b7cdf9365 | |||
| 9b50dd69d8 | |||
| 08f1a79800 | |||
| 51bdec22e7 | |||
| d43f670285 | |||
| c135583402 | |||
| 8d2893b67e | |||
| 555715ced9 | |||
| 446a9d395c | |||
| 8311f2db7f | |||
| f4570ff274 | |||
| 3aea4e1ec6 | |||
| 516366b796 | |||
| d82d0157cb | |||
| 1deb0c90d4 | |||
| 41a3114d03 | |||
| 76e4b577a0 | |||
| 03473a286b | |||
| 5cabd09705 | |||
| cc47672b7c | |||
| 37edd9134b | |||
| 80f167b1f7 | |||
| bd6547fca3 | |||
| 7e97f5bff5 | |||
| 7d41ccccee | |||
| bf616e192a | |||
| f8bd5d0c51 | |||
| ac40de1d30 | |||
| 7930a31b94 | |||
| 850fb12807 | |||
| b14b66ab03 | |||
| 15ff6bcde5 | |||
| 7822772905 | |||
| 0efe1567c0 | |||
| 837fed7690 | |||
| aa4ee64a34 | |||
| bb750cdba3 | |||
| d56c77b368 | |||
| 83e2ff06cf | |||
| f5deaafd06 | |||
| fe46311266 | |||
| 08968bb1b7 | |||
| 27bbd11f06 | |||
| 48d7841fbf | |||
| cc1df11f69 | |||
| fdfd4ac491 | |||
| d5dd5639ae | |||
| 7a120b3256 | |||
| 863d477966 | |||
| cec48832be | |||
| a36b7ccd7c | |||
| 32bf324a1e | |||
| 16de9df000 | |||
| 65c9deb4d1 | |||
| 28ae27b8e1 | |||
| 81a0cf4710 | |||
| 8a837a2830 | |||
| 0a2b3a4a56 | |||
| 613ece92dc | |||
| 8d4206c2b5 | |||
| 03286a589b | |||
| 88b9503c3b | |||
| c553d795d8 | |||
| 3ba6d6784c | |||
| 78ca24dc3f | |||
| 4fa6553db5 | |||
| 2124e043ce | |||
| e4f3620acb | |||
| 032a59a34d | |||
| e689090024 | |||
| 1cc34888fd | |||
| 0331ecd5d3 | |||
| 831a918c30 | |||
| 3db5277ae8 | |||
| 6939e0cbbc | |||
| f0095bf8b2 | |||
| 5206679efd | |||
| ac044e7573 | |||
| 6ebad2a91f | |||
| 3d10ed385c | |||
| 6d0c94dbdb | |||
| e01563a550 | |||
| 30e3225a3c | |||
| ef186a1516 | |||
| 5d5b3bdc76 | |||
| 2f8c98dac9 | |||
| 2db7189067 | |||
| 94476ac109 | |||
| 5fe02b7c98 | |||
| a1052f4fd1 | |||
| 8c9904a7c4 | |||
| 23af5dfdff | |||
| 3a10f6ad17 | |||
| a3842c873d | |||
| c29c3f063d | |||
| 758d62a396 | |||
| 6725642274 | |||
| a1f4dc365a | |||
| 165b62ee20 | |||
| 72d3481de3 | |||
| 29ccb747ad | |||
| 4749e27453 | |||
| e501d39988 | |||
| 6b6cf25862 | |||
| 180de840eb | |||
| 8fd2d7e5e7 | |||
| 129dd4a838 | |||
| e09cac6f1f | |||
| c00a86b32c | |||
| d3ae0350a2 | |||
| 2c2196a1f1 | |||
| 35ade14630 | |||
| 541df87272 | |||
| 337b6ccd6e | |||
| 42f46dfe9a | |||
| 9088d2b2c5 | |||
| 500bfa2c33 | |||
| 9ca9c43dfa | |||
| b525b0f08f | |||
| 1966c69994 | |||
| 976eff8ad1 | |||
| 0af902ec43 | |||
| 9118ce2537 | |||
| 2f48e08f1f | |||
| fec284e7cb | |||
| 8e2e4c5e73 | |||
| 0edc6615fc | |||
| b745e159de | |||
| aac29d604c | |||
| c884802b13 | |||
| 5f5573a24e | |||
| 74b0087ebb | |||
| 927e0151d4 | |||
| 5275922d1d | |||
| e186c7945a | |||
| 33a6e77f0e | |||
| 4e47489d53 | |||
| c76b2f149e | |||
| 826fffe05b | |||
| 75f57cdba7 | |||
| f556af5d4e | |||
| d5f33f0c6e | |||
| 16e17b32ad | |||
| 988494e18b | |||
| c4549a5e20 | |||
| 46fa4f38d5 | |||
| 20e0e68ad7 | |||
| 17468a234a | |||
| 8a53d5bfc6 | |||
| 695da7418e | |||
| abd26c796b | |||
| 24559d81ac | |||
| 6c1c2c3994 | |||
| 01fab15713 | |||
| 0cd00e71c3 | |||
| bf0ff2adbf | |||
| ed4bbc1c56 | |||
| c456402cc5 | |||
| 4f0bf667b1 | |||
| 61af9aa574 | |||
| 3b59b34e76 | |||
| 65b38997f7 | |||
| 7510f7649c | |||
| 619792a81c | |||
| cea1183f75 | |||
| e2af4c5ae4 | |||
| dfd5f82894 | |||
| e81944cef6 | |||
| 7bdd39ab9a | |||
| f4b38f040e | |||
| 799668e129 | |||
| 890190263e | |||
| f8522edacd | |||
| b958747855 | |||
| 078bde2c02 | |||
| 0b10ea987b | |||
| 1553d38182 | |||
| d04b075996 | |||
| 644927636d | |||
| 95e45007aa | |||
| 7a583c4045 | |||
| 65bce058bb | |||
| 48b437083b | |||
| fa0612859b | |||
| 4ffbcd0b7d | |||
| a0cd053fd9 | |||
| f9a5e066b5 | |||
| e9bc192160 | |||
| e0a57988ad | |||
| b414a74c26 | |||
| 2c3796d598 |
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the bridged MCP gateway.",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
@@ -8,7 +8,7 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the bridged MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"description": "Make a project bridge-ready: mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
name: architect
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
name: dev
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,274 @@
|
||||
---
|
||||
name: fleets-status
|
||||
description: Report the status of every fleet that shares one LavinMQ instance. Use for local daemon health, broker-wide fleet presence, and cross-host lead coordination checks.
|
||||
---
|
||||
|
||||
# Status of every fleet on the shared LavinMQ instance
|
||||
|
||||
**The headline: always report what is missing.** This skill starts with the local fleet, then adds
|
||||
broker-wide facts when its read-only credential exists. A missing fleet must appear as `unknown` or
|
||||
`not reachable`, with the reason and the fix. Never leave it out.
|
||||
|
||||
The known topology has one LavinMQ instance on `10.10.20.13` (`fleet01`). AMQP uses port `5672`,
|
||||
and the management API uses port `15672`. The Mac fleet owns vhost `/mac`. The fleet01 fleet owns
|
||||
vhost `/fleet01`.
|
||||
|
||||
## 1. Protect credentials before any probe
|
||||
|
||||
**Hard rule — never print `LAVINMQ_URI`.** It is an AMQP URI with its password inline. It only
|
||||
resolves in a login shell because `${SHARED_ENV}/tools/secrets.sh` supplies it. A non-login shell
|
||||
can make every broker probe look empty.
|
||||
|
||||
- Never run `echo "$LAVINMQ_URI"`.
|
||||
- Never put `${LAVINMQ_URI:-something}` in output. That form expands to the secret value when set.
|
||||
- Parse the user, host, and password into shell or Python variables. Use them without printing them.
|
||||
- Prefer `resolves` or `does not resolve` over any part of the value.
|
||||
- Every command that can read `LAVINMQ_URI` must send all output through this redaction before it
|
||||
reaches the report:
|
||||
|
||||
```bash
|
||||
sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
```
|
||||
|
||||
**The `g` flag is not optional.** Without it `sed` replaces only the first match on each line, so a
|
||||
line carrying two URIs leaks the second one. `scripts/redeploy-fleetd.sh --check` prints lines like
|
||||
that. Checked on 2026-08-27: without `g`, `amqp://u1:p1@h1/mac and http://u2:p2@h2:15672/api`
|
||||
redacts the first pair and prints `u2:p2` in the clear.
|
||||
|
||||
Keep `pipefail` on when applying that filter. Otherwise the filter can hide a failed probe. Apply
|
||||
the same no-print rule to the management password below, even though it is not in an AMQP URI.
|
||||
|
||||
## 2. Tier 1 — this fleet (always run)
|
||||
|
||||
Start here even when the broker tier is blocked. Work from the local fleetd checkout.
|
||||
|
||||
First run the read-only deployment check. It already checks the daemon process, deployed jar versus
|
||||
the checkout `HEAD`, launchd state, and whether each configured token resolves in a login shell.
|
||||
Do not copy those checks into new shell code. The script reads `LAVINMQ_URI`, so redact all output:
|
||||
|
||||
```bash
|
||||
set -o pipefail
|
||||
scripts/redeploy-fleetd.sh --check 2>&1 \
|
||||
| sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
git rev-parse HEAD
|
||||
```
|
||||
|
||||
Treat jar drift as a top-level warning. A merge is not a deployment. State the running jar result
|
||||
as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into a match.
|
||||
|
||||
Report the process identifier (PID) and uptime too:
|
||||
|
||||
```bash
|
||||
PIDS="$(pgrep -f 'target/fleetd.jar' || true)"
|
||||
if [ -z "$PIDS" ]; then
|
||||
printf '%s\n' 'fleetd: not running'
|
||||
else
|
||||
for PID in $PIDS; do
|
||||
ps -p "$PID" -o pid=,etime=,lstart=,command=
|
||||
done
|
||||
fi
|
||||
```
|
||||
|
||||
Read the full health response. Keep the HTTP status because `503` means fleetd is running but herdr
|
||||
is not reachable. Report both `herdr.version` and `herdr.protocol` when present:
|
||||
|
||||
```bash
|
||||
curl -sS --max-time 5 -w '\nHTTP %{http_code}\n' http://127.0.0.1:8765/healthz
|
||||
```
|
||||
|
||||
Call `fleet_whoami`, then call `fleet_list`. Preserve its sections in the report:
|
||||
|
||||
- `leads`, including which row is this lead;
|
||||
- `members`, including state, role, profile, branch, and worktree when present;
|
||||
- every per-profile `capacity` row, including `maxLoad`, `live`, `free`, and quarantine facts;
|
||||
- the exact `healthCoverage` value.
|
||||
|
||||
Do not describe an empty `members` list as an empty fleet. It says only that no members are spawned.
|
||||
Also do not hide a profile with `free: 0`; say whether load or credential quarantine caused it.
|
||||
|
||||
Show only WARN, ERROR, and SEVERE lines after the last `fleetd listening` line. This anchor stops an
|
||||
old incident from looking current:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
path = Path("fleetd/fleetd.out")
|
||||
if not path.exists():
|
||||
print("cannot check current WARN/ERROR: fleetd/fleetd.out does not exist")
|
||||
else:
|
||||
lines = path.read_text(errors="replace").splitlines()
|
||||
starts = [i for i, line in enumerate(lines) if "fleetd listening" in line]
|
||||
if not starts:
|
||||
print("cannot anchor WARN/ERROR: no 'fleetd listening' line exists")
|
||||
else:
|
||||
current = lines[starts[-1]:]
|
||||
alerts = [line for line in current if re.search(r"\b(?:WARN|ERROR|SEVERE)\b", line)]
|
||||
print(f"current WARN/ERROR/SEVERE count: {len(alerts)}")
|
||||
for line in alerts[-50:]:
|
||||
print(line)
|
||||
PY
|
||||
```
|
||||
|
||||
**What this tier cannot see:** it proves facts only about the Mac daemon at `127.0.0.1:8765`.
|
||||
It cannot show the fleet01 daemon, broker queue depth, or broker consumers. The fleet01 REST service
|
||||
at `10.10.20.13:8765` is not reachable from the Mac. Say this in the report rather than omitting
|
||||
fleet01.
|
||||
|
||||
**But fleet01 IS reachable over SSH — checked 2026-08-28.** An older version of this line said SSH
|
||||
was denied. That is true only for the user `dai.ha`. The host alias `fleet01` maps to user `ltms`,
|
||||
and `ssh fleet01` works with key auth:
|
||||
|
||||
```bash
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=6 fleet01 'echo $(id -un)@$(hostname)'
|
||||
```
|
||||
|
||||
So fleet01's daemon PID, uptime, jar and `/healthz` **can** be reported — over SSH, not over REST.
|
||||
Do that rather than writing `not reachable`. `ltms` also has passwordless sudo there.
|
||||
|
||||
## 3. Tier 2 — the shared broker (run when management access exists)
|
||||
|
||||
**This tier is blocked today.** The AMQP user in `LAVINMQ_URI` can connect on port `5672`, but gets
|
||||
HTTP `401` from the management API on port `15672`. An AMQP connection does not grant monitoring
|
||||
access.
|
||||
|
||||
The operator must create a separate, read-only LavinMQ management user with the `monitoring` tag.
|
||||
It needs access to inspect both `/mac` and `/fleet01`. Store its values as
|
||||
`LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD` in
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Do not reuse or print the AMQP URI. Full multi-fleet status stays
|
||||
blocked until this user exists.
|
||||
|
||||
When both variables resolve, run this from a login shell. It calls `GET /api/overview`,
|
||||
`GET /api/vhosts`, `GET /api/queues`, and `GET /api/connections`. It prints selected status fields,
|
||||
but never the user, password, Authorization header, or AMQP URI:
|
||||
|
||||
```bash
|
||||
zsh -lc 'python3 - "$@"' -- <<'PY'
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
base = "http://10.10.20.13:15672"
|
||||
user = os.environ.get("LAVINMQ_MANAGEMENT_USER", "")
|
||||
password = os.environ.get("LAVINMQ_MANAGEMENT_PASSWORD", "")
|
||||
if not user or not password:
|
||||
print("broker tier: BLOCKED — management credential does not resolve in a login shell")
|
||||
sys.exit(0)
|
||||
|
||||
token = base64.b64encode(f"{user}:{password}".encode()).decode()
|
||||
|
||||
def get(path):
|
||||
request = urllib.request.Request(
|
||||
base + path,
|
||||
headers={"Authorization": "Basic " + token, "Accept": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=5) as response:
|
||||
return json.load(response)
|
||||
|
||||
try:
|
||||
overview = get("/api/overview")
|
||||
vhosts = get("/api/vhosts")
|
||||
queues = get("/api/queues")
|
||||
connections = get("/api/connections")
|
||||
except urllib.error.HTTPError as error:
|
||||
print(f"broker tier: BLOCKED — management API returned HTTP {error.code}")
|
||||
sys.exit(0)
|
||||
except Exception as error:
|
||||
print(f"broker tier: BLOCKED — management API is not reachable: {type(error).__name__}")
|
||||
sys.exit(0)
|
||||
|
||||
fleet_names = {"/mac": "Mac fleet", "/fleet01": "fleet01 fleet"}
|
||||
print(json.dumps({
|
||||
"overview": {
|
||||
"lavinmq_version": overview.get("lavinmq_version"),
|
||||
"rabbitmq_version": overview.get("rabbitmq_version"),
|
||||
"queue_totals": overview.get("queue_totals", {}),
|
||||
"object_totals": overview.get("object_totals", {}),
|
||||
},
|
||||
"fleets": [
|
||||
{
|
||||
"fleet": fleet_names.get(vhost.get("name"), "UNKNOWN FLEET"),
|
||||
"vhost": vhost.get("name"),
|
||||
"queues": [
|
||||
{
|
||||
"name": queue.get("name"),
|
||||
"messages": queue.get("messages", 0),
|
||||
"messages_ready": queue.get("messages_ready", 0),
|
||||
"messages_unacknowledged": queue.get("messages_unacknowledged", 0),
|
||||
"consumers": queue.get("consumers", 0),
|
||||
}
|
||||
for queue in queues if queue.get("vhost") == vhost.get("name")
|
||||
],
|
||||
"connections": [
|
||||
{
|
||||
"name": connection.get("name"),
|
||||
"peer_host": connection.get("peer_host"),
|
||||
"state": connection.get("state"),
|
||||
}
|
||||
for connection in connections if connection.get("vhost") == vhost.get("name")
|
||||
],
|
||||
}
|
||||
for vhost in vhosts
|
||||
],
|
||||
}, indent=2, sort_keys=True))
|
||||
PY
|
||||
```
|
||||
|
||||
Map `/mac` to the Mac fleet and `/fleet01` to the fleet01 fleet. Keep any other vhost in the
|
||||
report as `UNKNOWN FLEET`; do not drop it. For each vhost, total the ready, unacknowledged, and all
|
||||
messages. Report every queue's consumer count and each live connection.
|
||||
|
||||
**A vhost with queues but zero consumers means that fleet's daemon is down while its durable state
|
||||
survives. Call this out as a top-level warning.** This is the main reason to use the management API
|
||||
instead of calling each remote daemon.
|
||||
|
||||
**What this tier cannot see:** without the new `monitoring` credential it cannot enumerate any
|
||||
vhost, queue, depth, consumer, or connection. With the credential it still cannot report fleet01's
|
||||
daemon PID, uptime, jar revision, `/healthz`, herdr version, or member capacity. Those need reachable
|
||||
fleet01 REST or SSH access, which the Mac does not have today.
|
||||
|
||||
## 4. Tier 3 — cross-fleet lead coordination
|
||||
|
||||
Use the queue data from Tier 2. Select queues whose names match `lead.<coordId>.inbox`. Report each
|
||||
queue's vhost, depth, consumer count, and the `coordId` between the prefix and suffix.
|
||||
|
||||
- A lead inbox with a consumer shows that a lead mailbox is live on that vhost.
|
||||
- A durable lead inbox with zero consumers shows saved coordination state, but no live receiver.
|
||||
- No lead inbox is not proof that coordination is disabled. The daemon may be down before declaring
|
||||
its queue, or this account may not be allowed to see the vhost.
|
||||
|
||||
This Mac fleet currently sets both `broker.uriEnv` and `coordinator.uriEnv` to the same variable,
|
||||
`LAVINMQ_URI`. Therefore its coordinator connects to `/mac`. Cross-host `fleet_send{coordId}` routes
|
||||
only when both leads share the same coordinator vhost. If the fleet01 lead uses `/fleet01` for its
|
||||
coordinator, the leads cannot see each other and the send will not route.
|
||||
|
||||
**Open question:** the fleet01 coordinator vhost has not been checked. Surface this question in
|
||||
every report until a live `lead.<coordId>.inbox` consumer or fleet01's config proves the answer. Do
|
||||
not claim that fleet01 uses `/fleet01` just because its member queues do.
|
||||
|
||||
Also compare these broker facts with the `leads` rows from local `fleet_list`. A missing remote lead
|
||||
is `not visible from this coordinator`, not `down`, unless the broker consumer facts prove it.
|
||||
|
||||
**What this tier cannot see:** without Tier 2 management access it cannot list lead inboxes or their
|
||||
consumers. Even with that access, a stopped fleet01 daemon leaves only durable queue history. That
|
||||
history cannot prove which coordinator URI its current config would use after restart.
|
||||
|
||||
## 5. Report all fleets
|
||||
|
||||
Use one row per known or discovered fleet. Include blocked rows.
|
||||
|
||||
| Fleet | Daemon | Deployment | Herdr | Members/capacity | Queues/consumers | Lead coordination | Cannot check |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| Mac (`/mac`) | PID + uptime | jar vs `HEAD` | health + version + protocol | `fleet_list` + `healthCoverage` | facts or blocked reason | inbox facts or open question | exact missing facts |
|
||||
| fleet01 (`/fleet01`) | reachable/down/unknown | value or `not reachable` | value or `not reachable` | value or `not reachable` | facts or blocked reason | inbox facts plus coordinator-vhost question | exact missing facts and fix |
|
||||
|
||||
Add rows for unknown vhosts. End with three short sections: `Current warnings`, `Checks that were
|
||||
blocked`, and `Operator action`. Until the management user exists, `Operator action` must say:
|
||||
|
||||
> Create a read-only LavinMQ management user with the `monitoring` tag and access to `/mac` and
|
||||
> `/fleet01`. Put its user and password in `${SHARED_ENV}/tools/secrets.sh` as
|
||||
> `LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD`.
|
||||
@@ -1,11 +1,11 @@
|
||||
---
|
||||
name: implementer
|
||||
description: Implementer-role procedure for a bridged worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over bridged.
|
||||
description: Implementer-role procedure for a fleetd worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over fleetd.
|
||||
---
|
||||
|
||||
# Implementer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge, never commit `.mcp.json` or `wiki/`) is in **`CLAUDE.md` → Bridge communication →
|
||||
Worker** and already applies. This skill is only the *implement-and-hand-off procedure*.
|
||||
|
||||
@@ -39,6 +39,19 @@ a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
**Never run `git stash` (or `git stash pop`/`apply`/`drop`).** Your worktree is isolated, but the
|
||||
stash is **not**: `refs/stash` is one stack shared by the primary's checkout and every other
|
||||
worker's worktree of this repo. Measured on 2026-09-04 — `git stash list` from a worker's worktree
|
||||
and from the primary's tree returned byte-identical output. So a `git stash` you run can be popped
|
||||
into someone else's tree, and a `git stash pop` you run can drop **another worker's** uncommitted
|
||||
edits on top of yours. This has already happened here: two workers were running in parallel and one
|
||||
of them had its in-progress edit silently overwritten by the other's stash.
|
||||
|
||||
The branch is your isolation, so use it instead. To set work aside, commit it on your own branch
|
||||
(`git commit -m "wip: ..."`) and carry on; to try something and back out, use
|
||||
`git diff > /tmp/<your-branch>.patch` then `git checkout -- <file>`. Both stay inside your worktree.
|
||||
If you find a stash entry you did not create, leave it alone and say so in your report.
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
@@ -49,7 +62,7 @@ test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-t
|
||||
your worktree*:
|
||||
|
||||
```bash
|
||||
cd "$(git rev-parse --show-toplevel)/bridged" && mvn clean install
|
||||
cd "$(git rev-parse --show-toplevel)/fleetd" && mvn clean install
|
||||
echo "exit=$?"
|
||||
```
|
||||
|
||||
@@ -85,7 +98,7 @@ host (`GITEA_HOST`) into your env for exactly this — the token can create a PR
|
||||
merge**.
|
||||
|
||||
```bash
|
||||
API="${GITEA_HOST%/}/api/v1/repos/lms/claude-bridge/pulls"
|
||||
API="${GITEA_HOST%/}/api/v1/repos/fleet/fleetd/pulls"
|
||||
BRANCH="$(git branch --show-current)"
|
||||
curl -sS -X POST "$API" \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
@@ -103,7 +116,7 @@ fix it if the cause is yours (e.g. branch not pushed yet), and report the failur
|
||||
inventing a URL. If `GITEA_TOKEN` is unset your profile was not granted PR-create: push the branch
|
||||
and report its name so the lead opens the PR.
|
||||
|
||||
## 6. Hand off — what goes in `bridge_reply`
|
||||
## 6. Hand off — what goes in `fleet_reply`
|
||||
|
||||
The reply is the entire handoff; the lead cannot see your terminal.
|
||||
|
||||
@@ -130,7 +143,7 @@ sequenceDiagram
|
||||
I->>G: git push -u origin HEAD
|
||||
I->>G: POST /pulls (GITEA_TOKEN) — open PR to main
|
||||
G-->>I: html_url
|
||||
I->>L: bridge_reply(PR url, branch, files, tests)
|
||||
I->>L: fleet_reply(PR url, branch, files, tests)
|
||||
Note over L,G: the lead reviews the PR and merges on green — you never merge
|
||||
```
|
||||
|
||||
|
||||
@@ -53,7 +53,7 @@ Map each entry from `.mcp.json`:
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"instructions": ["CLAUDE.md"],
|
||||
"mcp": {
|
||||
"bridged": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"fleetd": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"context7": { "type": "remote", "url": "https://example.dev/mcp", "enabled": true,
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" } },
|
||||
"gitea": { "type": "local", "command": ["gitea-mcp", "-t", "stdio"], "enabled": true,
|
||||
|
||||
@@ -1,11 +1,11 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Reviewer-role procedure for a bridged worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over bridged.
|
||||
description: Reviewer-role procedure for a fleetd worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over fleetd.
|
||||
---
|
||||
|
||||
# Reviewer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies. This
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
@@ -24,13 +24,13 @@ wrong.
|
||||
covers what it was given.
|
||||
- Do **not** edit files or run the build. You review; the owner acts.
|
||||
|
||||
## 3. Reach for `bridge_ask` only for a genuine fork
|
||||
## 3. Reach for `fleet_ask` only for a genuine fork
|
||||
|
||||
Ambiguous requirement, a missing acceptance criterion, "intended or a bug?", or two defensible
|
||||
fixes with different consequences — those are the lead's call, and guessing produces a
|
||||
confident-but-wrong finding. Anything you could settle by reading more code is yours to settle.
|
||||
|
||||
## 4. The finding — what goes in `bridge_reply`
|
||||
## 4. The finding — what goes in `fleet_reply`
|
||||
|
||||
Report the **single most important** real issue in the scope, in these four lines, under
|
||||
~90 words:
|
||||
|
||||
@@ -14,7 +14,7 @@ jobs:
|
||||
# does not depend on the wiki repo being reachable.
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The runner image ships an older default-jdk; bridged sets maven.compiler.release=25, so
|
||||
# The runner image ships an older default-jdk; fleetd sets maven.compiler.release=25, so
|
||||
# provision the JDK explicitly rather than apt-installing whatever "default" means today.
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
@@ -32,7 +32,7 @@ jobs:
|
||||
mvn -version
|
||||
|
||||
- name: Build and test
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
# This IS the mock-socket surface CB-503 asks for: the pom's `default-excludes` profile
|
||||
# already sets excludedGroups=contract, so the @Tag("contract") tests — which need a live
|
||||
# herdr socket and a RabbitMQ container — are excluded without any flag here. Everything
|
||||
@@ -46,7 +46,7 @@ jobs:
|
||||
# the log instead, where they are actually readable.
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||
# straight to it — no Docker, no skipped tests. This separation (build job hermetic and
|
||||
# Docker-free; contract job broker-provided) is deliberate — see the default-excludes/contract
|
||||
# profiles in bridged/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# profiles in fleetd/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# exactly as in the build job above.
|
||||
contract:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -69,8 +69,6 @@ jobs:
|
||||
env:
|
||||
RABBITMQ_DEFAULT_USER: guest
|
||||
RABBITMQ_DEFAULT_PASS: guest
|
||||
ports:
|
||||
- 5672:5672
|
||||
env:
|
||||
# Service containers are reachable from the job by their network alias on their internal port.
|
||||
AMQP_URI: amqp://guest:guest@rabbitmq:5672
|
||||
@@ -93,12 +91,12 @@ jobs:
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
- name: Contract tests
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
|
||||
+12
-4
@@ -6,8 +6,16 @@
|
||||
# Settings backups inherit the env block — and secrets with it.
|
||||
.claude/settings.local.json.bak*
|
||||
|
||||
# Daemon runtime artefacts. bridged appends its log wherever it is launched from, so both the
|
||||
# repo root and bridged/ collect one; neither belongs in git.
|
||||
bridged.out
|
||||
bridged/bridged.out
|
||||
# The default profile parityOverlay copies these primary→worktree, so they appear in EVERY worker
|
||||
# worktree. Two reasons they must be ignored. They hold environment values, which is reason enough.
|
||||
# And since CB-576 a release preserves any worktree that `git status --porcelain` calls dirty —
|
||||
# untracked files included, deliberately. An untracked overlay file would therefore make every
|
||||
# COMPLETED release preserve its worktree, and worktrees would pile up with no error to notice.
|
||||
.env
|
||||
.envrc
|
||||
|
||||
# Daemon runtime artefacts. fleetd appends its log wherever it is launched from, so both the
|
||||
# repo root and fleetd/ collect one; neither belongs in git.
|
||||
fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
[submodule "wiki"]
|
||||
path = wiki
|
||||
url = ssh://git@git.ltms.dev:2224/lms/claude-bridge.wiki.git
|
||||
url = ssh://git@git.ltms.dev:2224/fleet/fleetd.wiki.git
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -4,48 +4,52 @@
|
||||
|
||||
> **Canonical block.** Everything down to §Layering is the portable bridge charter, copied verbatim
|
||||
> into every project that mounts the bridge MCP. Keep it byte-identical with the template in the
|
||||
> wiki ([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
|
||||
If no `bridge_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
`bridged` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **worker**) mount the *same* MCP server and talk only
|
||||
through its `bridge_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
`fleetd` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **member**) mount the *same* MCP server and talk only
|
||||
through its `fleet_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
|
||||
### Which role am I? — settle this before acting
|
||||
|
||||
**Both roles read this file.** A worker runs in a git worktree of this same repo, so it inherits
|
||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||
|
||||
**Call `bridge_whoami`.** It returns `{"role":"primary"}` or `{"role":"worker","sessionId":…,
|
||||
"profile":…,"worktree":…,"branch":…}`, resolved by the daemon from your connection — unforgeable,
|
||||
and the same resolution its authorization gate uses. Don't infer what you can ask.
|
||||
**Call `fleet_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
||||
was bound to. Don't infer what you can ask.
|
||||
|
||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||
fires: the reply charter in your system prompt (*"You are an off-subscription worker in the
|
||||
claude-bridge fleet"*) ⇒ **worker**; bridge tools prefixed `mcp__bridge__*` ⇒ **worker** (the
|
||||
launcher fixes that mount name; a primary's mount is named by whoever wrote its `.mcp.json`, so it
|
||||
varies); `ANTHROPIC_BASE_URL` set ⇒ **worker** (Claude-model workers run on a clean env, so its
|
||||
*absence* proves nothing). **Still unsure ⇒ act as a worker.** The two mistakes are not symmetric: a
|
||||
primary acting as a worker is refused by the authorization gate — loud and self-correcting — while a
|
||||
worker acting as the primary ends its turn with no `bridge_reply`, and the sender silently receives
|
||||
nothing. Fail toward the recoverable error.
|
||||
fires: the reply charter in your system prompt (*"You are a spawned member in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; fleet tools prefixed `mcp__fleet__*` ⇒ **spawned
|
||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||
`.mcp.json`, so it varies — and a member spawned before CB-632 still says `mcp__bridge__*`); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
||||
only `fleet_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
### Invariants — both roles, no exceptions
|
||||
|
||||
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
||||
stays on subscription; only the bridge puts a worker off it, at spawn. Mounting the bridge must
|
||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||
never move a session across that boundary.
|
||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||
side cannot see your screen. An answer that isn't in a `bridge_*` call is silently discarded.
|
||||
side cannot see your screen. An answer that isn't in a `fleet_*` call is silently discarded.
|
||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||
cannot act as another session. Spawn/stop/send/drain are lead-only; reply/ask are
|
||||
only-as-itself — any peer may answer for its own pane, and for no other. A call outside your
|
||||
role is refused, not queued.
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
||||
outside your role is refused, not queued.
|
||||
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`/`blocked`.
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
|
||||
@@ -53,10 +57,10 @@ nothing. Fail toward the recoverable error.
|
||||
|
||||
**Delegate by default — that is the job.** With the bridge mounted you are an orchestrator on a
|
||||
metered subscription, and workers are cheap, parallel, and disposable. The default answer to "who
|
||||
does this?" is **a worker**, not you. Reach for `bridge_send` before you reach for `Edit`. The steps
|
||||
does this?" is **a worker**, not you. Reach for `fleet_send` before you reach for `Edit`. The steps
|
||||
below are the procedure — run them in order, every task, not only the big ones.
|
||||
|
||||
0. **Know your role** — `bridge_whoami`, once per session, before anything else.
|
||||
0. **Know your role** — `fleet_whoami`, once per session, before anything else.
|
||||
1. **Split.** Write the unit list. Every unit carries: scope · the files or PR in question ·
|
||||
acceptance criteria · exactly what to report back. A unit with no acceptance criteria is not
|
||||
ready to delegate — refine it or keep it.
|
||||
@@ -66,20 +70,22 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
delegate. The keep-list is closed: the conversation with the user, decomposition and planning,
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `bridge_spawn{profile, worktree:true, ticket}`, one per
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
4. **Then send them all** — `bridge_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
brief instead. The brief is self-contained — the worker sees your message and the repo, nothing
|
||||
of your context, your plan, or your screen.
|
||||
5. **Collect** — `bridge_poll{ticket}` → `bridge_ack{ticket, msgId}`. Answer a worker's `bridge_ask`
|
||||
with `bridge_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`bridge_status`, never by reading its terminal.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker mounts only the bridge MCP and
|
||||
cannot run your other tooling, and a piped command (`… | tail`) hides failures behind a zero
|
||||
exit — never promote a worker's "clean" to a fact.
|
||||
5. **Collect** — `fleet_poll{ticket}` → `fleet_ack{target, msgId}`. Answer a worker's `fleet_ask`
|
||||
with `fleet_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||
@@ -87,11 +93,11 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
read it yourself.
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `bridge_stop{paneId}`.
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
prefer `wait:false` + `bridge_poll` for anything non-trivial: a blocking `bridge_send` is capped by
|
||||
prefer `wait:false` + `fleet_poll` for anything non-trivial: a blocking `fleet_send` is capped by
|
||||
*your own* MCP client call timeout (~60s), well below the task's real runtime.
|
||||
|
||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||
@@ -99,28 +105,29 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
|
||||
| Intent | Tool |
|
||||
|---|---|
|
||||
| Confirm your own role | `bridge_whoami` |
|
||||
| See backends available | `bridge_profiles` |
|
||||
| Start a worker | `bridge_spawn{profile?, cwd?, worktree?, ticket?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `bridge_list` → `leads` (your peers) + `workers` · one peer's state: `bridge_status{sessionId}` |
|
||||
| Delegate (blocking) | `bridge_send{sessionId, content}` |
|
||||
| Delegate (long task) | `bridge_send{sessionId, content, wait:false}` → ticket → `bridge_poll{ticket}` |
|
||||
| Answer a worker's `bridge_ask` | `bridge_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** | `bridge_send{sessionId: <their terminal>, content}` — `bridge_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `bridge_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `bridge_poll{target}` · then `bridge_ack{target, msgId}` |
|
||||
| Tear down | `bridge_stop{paneId}` |
|
||||
| Confirm your own role | `fleet_whoami` |
|
||||
| See backends available | `fleet_profiles` |
|
||||
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `fleet_status{sessionId}` |
|
||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
`bridge_list` returns `leads` alongside `workers`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own workers, and its own judgment. An empty
|
||||
`workers` array means no workers are spawned; it says nothing about peers.
|
||||
`fleet_list` returns `leads` alongside `members`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own members, and its own judgment. An empty
|
||||
`members` array means no members are spawned; it says nothing about peers.
|
||||
|
||||
**A lead never assigns a task to another lead.** Work goes to workers — only ever downward, never
|
||||
sideways. Sending a peer a brief with acceptance criteria is a category error: a brief is a worker's
|
||||
**A lead never assigns a task to another lead.** Work goes to members — only ever downward, never
|
||||
sideways. Sending a peer a brief with acceptance criteria is a category error: a brief is a member's
|
||||
artefact, and a peer is not yours to task. If a unit needs doing and it falls in your area, spawn a
|
||||
worker and delegate it yourself; if it falls in the peer's area, say so and let the peer assign it.
|
||||
member and delegate it yourself; if it falls in the peer's area, say so and let the peer assign it.
|
||||
The traffic between leads is coordination and nothing else:
|
||||
|
||||
1. **Divide the map, not the work.** Agree who owns which area, then each of you assigns inside your
|
||||
@@ -134,22 +141,30 @@ The traffic between leads is coordination and nothing else:
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `bridge_reply`, and push back on
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
|
||||
### Worker — the turn contract
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
1. **Load the playbook skill the lead named** before doing anything else.
|
||||
2. **Do the assigned scope only.** Note anything you spot outside it in one line; don't go hunt it.
|
||||
3. **`bridge_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||
Don't ask what you could decide yourself.
|
||||
4. **End the turn with exactly one `bridge_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `bridge_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
5. **Report honestly.** State only what you actually ran and its real output, including failures.
|
||||
You mount **only** the bridge MCP — the primary's other servers (IDE, forge, docs) are not yours,
|
||||
so never claim the result of a check you had no way to run.
|
||||
4. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||
lead with its end cut off.
|
||||
5. **Report honestly.** State only what you actually ran and its real output, including failures,
|
||||
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
||||
find there holds a deliberately blocked credential and fails every call, by design.
|
||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
@@ -157,29 +172,115 @@ you.
|
||||
|
||||
| Layer | Scope | Reaches |
|
||||
|---|---|---|
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `bridge_reply`* | every worker, at launch, every peer kind |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every Claude worker — tracked in git, so worktrees inherit it |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a worker told to load one |
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `fleet_reply`* | every spawned member, at launch, every peer kind — never a lead |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every member that reads the repo — tracked in git, so worktrees inherit it |
|
||||
| role agent definition files | role contract and per-job procedure | a member whose launcher binds its role to the matching file in its worktree |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a member told to load one |
|
||||
| the bridge's own docs | design detail, flows, error model | on demand |
|
||||
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. Peers that don't read
|
||||
`CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they* must obey belongs in the
|
||||
charter, not here.
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. A member without a
|
||||
repo checkout still gets the launcher's reply charter, which is why that one rule stays there.
|
||||
Peers that don't read `CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they*
|
||||
must obey belongs in the charter, not here.
|
||||
|
||||
## Project addendum — claude-bridge (not part of the canonical block)
|
||||
|
||||
- **This repo is the bridge.** The daemon is `bridged`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/BridgeMcp` (tools), `auth/Authz` (the role table),
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **`fleet_profiles`/`fleet_list` report two separate outage states, and they are not the same
|
||||
thing.** *Quarantined* (CB-578) means the backend told us it is out of capacity — a long,
|
||||
1800s-default cooldown. *Cooling off* (fleetd #201/#227) means a profile's credential threw two
|
||||
distinct backend errors (a non-exhaustion failure such as an HTTP 5xx) within 60 seconds — a
|
||||
short, fixed 60s cooldown, not configurable per profile. Each check runs independently, so a
|
||||
profile can show both at once. In the JSON: a cooling profile carries `credentialId` and
|
||||
`coolingOffForSeconds`; a quarantined profile carries `quarantinedForSeconds`; a profile hit by
|
||||
both carries all three fields, and either state alone already sets that profile's `free` to `0`.
|
||||
A `fleet_spawn` naming a cooling-off profile is refused before it ever reaches the backend
|
||||
adapter, with a message naming the credential and the remaining seconds ("cooling off after
|
||||
repeated backend errors") — distinct wording from a quarantine refusal, so don't conflate the
|
||||
two when reading a spawn failure.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace).
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace) and
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `bridge_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` §6, kept out of this file because it loads into every
|
||||
session's context.
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
committed copies would otherwise mount the primary's IDE and forge servers (fleetd #134). The
|
||||
worktree's copy of each is a stub, **not** the repo's real file, so a worker that reads one and
|
||||
reports what it found is reporting on the stub. The daemon logs a per-spawn summary, but the
|
||||
worker cannot see that log. From inside its own worktree a worker — or a lead debugging one —
|
||||
reads the list with `git config --worktree --get-all fleet.neutralizedConfig`, and the
|
||||
consequence with `git config --worktree --get fleet.neutralizedConfigNote`. Never brief a worker
|
||||
to edit one of these files: the edit cannot be committed, and it will not tell you so.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, the turn-done
|
||||
fallback and status gating — are diagrammed in `docs/MCP-Contract.md`. That page is now flows
|
||||
only: its pre-build tool catalogue, parameter tables and REST paths were deleted rather than
|
||||
corrected, because a hand-maintained second copy of the tool surface is what drifted for a month
|
||||
while this line pointed every session at it (CB-609 / #114). **The live MCP schema is the tool
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
@@ -192,15 +293,15 @@ Before you call any work done, check the row that matches what you touched:
|
||||
|
||||
| You changed… | Re-read and update… |
|
||||
|---|---|
|
||||
| a `bridge_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| a `fleet_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| `Authz` / the role table | invariant 3, and the primary-only vs worker-only claims |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `bridge_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__bridge__*`), and the layering table's top row |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `fleet_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__fleet__*`), and the layering table's top row |
|
||||
| the injector / status gating | invariant 4 |
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `bridged.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
|
||||
That last row is not bookkeeping. Chapters 1–10 answer *how is this built* and *why this way*;
|
||||
none of them has a home for *what can it do and how do I turn it on*, so for twenty tickets a
|
||||
@@ -211,7 +312,7 @@ is a Roadmap line. A change that touches none of the three earns no entry, and t
|
||||
outcome rather than an omission.
|
||||
|
||||
Then **propagate**: the block in this file and the template in the wiki
|
||||
([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
block*) must stay byte-identical, and other projects carrying the block need the same edit. Verify
|
||||
rather than trust:
|
||||
|
||||
@@ -234,10 +335,10 @@ PY
|
||||
|
||||
Two IDE MCP servers are connected: **intellij-index** (semantic code intelligence) and
|
||||
**jetbrains** (file problems, reformat, debugger). IntelliJ has multiple projects open; our
|
||||
module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
module is **`fleetd`**. Always pass these to IDE MCP tools:
|
||||
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/bridged`
|
||||
- IDE paths are relative to `bridged/` (e.g. `src/main/java/dev/ltms/bridged/...`)
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/fleetd`
|
||||
- IDE paths are relative to `fleetd/` (e.g. `src/main/java/dev/ltms/fleet/...`)
|
||||
|
||||
### After editing any file — mandatory
|
||||
|
||||
@@ -251,7 +352,7 @@ module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
whole-project gate before declaring work done or committing.
|
||||
|
||||
**Whenever dependencies change (or a `pom.xml` edit), validate CVEs with
|
||||
`jetbrains get_file_problems{filePath: "bridged/pom.xml"}`** — its Mend.io check reflects the
|
||||
`jetbrains get_file_problems{filePath: "fleetd/pom.xml"}`** — its Mend.io check reflects the
|
||||
dependencies on disk. (Note: `ide_diagnostics` / intellij-index does NOT re-resolve dependencies
|
||||
after a pom edit without a full Maven reimport, so it reports stale CVE results — don't trust it
|
||||
for this.) Treat a CVE warning like any other: bump to a patched version and confirm
|
||||
|
||||
@@ -9,32 +9,33 @@ Sibling of [`crush-bridge`](https://git.ltms.dev/systems/vms) (which drives a he
|
||||
process*, so it inherits `CLAUDE.md`, hooks, skills, and MCP — just pointed at a
|
||||
cheaper/local model.
|
||||
|
||||
## Leading approach — herdr-centric message server (`bridged`)
|
||||
## Leading approach — herdr-centric message server (`fleetd`)
|
||||
|
||||
A small always-on message server, **`bridged`**, controls
|
||||
A small always-on message server, **`fleetd`**, controls
|
||||
[herdr](https://herdr.dev) (an agent multiplexer) over its Unix-socket API and exposes a
|
||||
clean 2-way messaging API as an **MCP server that both the primary and the workers mount** —
|
||||
one unified Claude setup and the **sole communication gateway** (REST/SSE stays for non-Claude
|
||||
clients; any broker is `bridged`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `bridged` owns
|
||||
clients; any broker is `fleetd`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `fleetd` owns
|
||||
policy (subscription boundary, session lifecycle, status-gated delivery) and the client
|
||||
contract. The worker `claude` launches with `ANTHROPIC_BASE_URL=https://ollama.ltms.dev` + a
|
||||
bearer token; the primary Opus stays env-clean and calls `bridged`'s MCP tools.
|
||||
contract. A Claude member launches with `ANTHROPIC_BASE_URL` pointed at the gateway,
|
||||
`https://llm.ltms.dev/anthropic`, plus a bearer token; the lead stays env-clean and calls
|
||||
`fleetd`'s MCP tools. See the wiki's **[13 User Guide](wiki/13-User-Guide.md)** to run it.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["bridged — standalone daemon (not a claude process)"]
|
||||
subgraph BD["fleetd — standalone daemon (not a claude process)"]
|
||||
SRV["SERVER face<br/>MCP · REST/SSE · policy"]
|
||||
CLI["CLIENT face<br/>status-gated injector · herdr socket"]
|
||||
SRV --> CLI
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
M["ollama.ltms.dev<br/>(worker model)"]
|
||||
M["llm.ltms.dev<br/>(the one gateway)"]
|
||||
|
||||
OPUS -->|"MCP bridge_send (blocks)"| SRV
|
||||
W -.->|"MCP bridge_reply"| SRV
|
||||
OPUS -->|"MCP fleet_send (blocks)"| SRV
|
||||
W -.->|"MCP fleet_reply"| SRV
|
||||
CLI -->|"Unix socket<br/>send_text · events.subscribe"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
W -->|"inference"| M
|
||||
@@ -46,24 +47,26 @@ flowchart LR
|
||||
```
|
||||
|
||||
- **Subscription boundary:** the *primary* never sets `ANTHROPIC_BASE_URL` (stays on
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `bridged` itself is a
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `fleetd` itself is a
|
||||
plain daemon (no Anthropic quota), so it may poll/subscribe freely.
|
||||
- **One gateway (unified MCP setup):** `bridged` is the **sole communication path** for every
|
||||
- **One gateway (unified MCP setup):** `fleetd` is the **sole communication path** for every
|
||||
Claude session. Primary and workers each mount it as an MCP server (one `claude mcp add`
|
||||
line, same on both) and talk over MCP tools — `bridge_send` / `bridge_reply` /
|
||||
`bridge_status` (with `bridge_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `bridged`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`bridge_send`);
|
||||
`bridged` holds it open until the worker calls `bridge_reply` or its turn hits
|
||||
line, same on both) and talk over MCP tools — `fleet_send` / `fleet_reply` /
|
||||
`fleet_status` (with `fleet_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `fleetd`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
**Tool naming:** the tools are `fleet_*` (renamed from `bridge_*` in CB-622). The old
|
||||
`bridge_*` names were removed in CB-634 — only `fleet_*` answers now.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`fleet_send`);
|
||||
`fleetd` holds it open until the worker calls `fleet_reply` or its turn hits
|
||||
`agent_status=done`, then returns the reply as the tool result. No cross-turn busy-poll, so
|
||||
no quota burn. SSE is an optional side-channel for humans/dashboards watching status.
|
||||
- **Worker → primary** rides `bridged`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `bridged` **injects the primary's idle pane** when it's
|
||||
- **Worker → primary** rides `fleetd`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `fleetd` **injects the primary's idle pane** when it's
|
||||
ready), so *no keystroke-into-primary and no broker are involved, even single-host*. The one
|
||||
exception: a split-host primary that isn't a herdr pane wakes via its own `Stop`-hook, which
|
||||
polls **`bridged`** (never a broker). See the wiki for the two topologies.
|
||||
polls **`fleetd`** (never a broker). See the wiki for the two topologies.
|
||||
- **Different model per process** sidesteps Claude Code's lack of per-subagent provider
|
||||
routing — the worker isn't a subagent, it's its own configured process.
|
||||
- **AgentAPI** ([`coder/agentapi`](https://github.com/coder/agentapi)) is retained only as a
|
||||
@@ -72,11 +75,11 @@ flowchart LR
|
||||
|
||||
## Docs
|
||||
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/lms/claude-bridge/wiki)**,
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/fleet/fleetd/wiki)**,
|
||||
vendored here as a submodule under [`wiki/`](./wiki):
|
||||
|
||||
```bash
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/lms/claude-bridge.git
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/fleet/fleetd.git
|
||||
# or, after a plain clone:
|
||||
git submodule update --init
|
||||
```
|
||||
@@ -86,7 +89,7 @@ Gitea wiki.
|
||||
|
||||
## Status
|
||||
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`bridged`** message server is built and in
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`fleetd`** message server is built and in
|
||||
real use: an Opus primary delegates tasks to off-subscription workers that reply through the
|
||||
bridge (code reviews delegated this way have produced committed bug fixes). Selected as the
|
||||
primary approach 2026-07-11, superseding the AgentAPI plan (2026-07-08); AgentAPI retained as a
|
||||
@@ -97,9 +100,9 @@ tests run separately via `mvn test -Pcontract`):
|
||||
|
||||
- **Core gateway** — herdr socket client (contract-tested vs live 0.7.0); guard-checked worker
|
||||
spawn with `ANTHROPIC_BASE_URL` injected only into the worker's env; status-gated injector;
|
||||
blocking `bridge_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `bridge_send` / `bridge_reply` / `bridge_status` (messaging) and `bridge_spawn`
|
||||
/ `bridge_list` / `bridge_stop` / `bridge_profiles` / `bridge_poll` (fleet). Caller identity is
|
||||
blocking `fleet_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `fleet_send` / `fleet_reply` / `fleet_status` (messaging) and `fleet_spawn`
|
||||
/ `fleet_list` / `fleet_stop` / `fleet_profiles` / `fleet_poll` (fleet). Caller identity is
|
||||
connection-based (loopback peer PID → herdr pane), so the same mount serves primary and workers.
|
||||
- **Delivery reliability** — completion fallback (a confirmed `working→idle` turn resolves a
|
||||
send); async fire-and-poll (beats the caller's MCP call timeout for long tasks); and failure
|
||||
@@ -107,7 +110,7 @@ tests run separately via `mvn test -Pcontract`):
|
||||
- **Fleet** — multiple worker profiles, each with an independent base_url guard check; workers
|
||||
inherit the primary's working directory (never `$HOME`); a readiness gate holds delivery until
|
||||
a worker's Claude has connected the bridge MCP (no paste lost into its boot window).
|
||||
- **Blocked-worker path** — `bridge_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
- **Blocked-worker path** — `fleet_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
ask the primary and resumes the *same* turn with the answer (CB-205).
|
||||
- **Session lifecycle** — session manager with spawn/reuse/recycle, `idle_ttl` reaper, `context_cap`,
|
||||
and graceful drain on shutdown (CB-301/CB-303); per-worker git worktrees on their own branch with
|
||||
|
||||
@@ -1,14 +0,0 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from bridged.example.yaml)
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -1,391 +0,0 @@
|
||||
# bridged configuration (example). Copy to bridged.yaml and adjust.
|
||||
#
|
||||
# bridged is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — bridged REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
|
||||
# API authentication (CB-501). Governs how a caller that is NOT an on-host worker pane proves it
|
||||
# is the primary. Worker identity never depends on this: a loopback peer PID that maps to a herdr
|
||||
# pane is unforgeable and is always honoured, so turning auth on cannot lock the fleet out.
|
||||
#
|
||||
# mode: loopback-trust → DEFAULT, and the historical behaviour: any loopback caller that is not
|
||||
# a worker is the primary, no credential needed. Sound ONLY because the
|
||||
# OS refuses remote connections to a loopback socket.
|
||||
# mode: token → such a caller must send `Authorization: Bearer <token>`; without it it
|
||||
# is anonymous and authorized for nothing. REQUIRED for a non-loopback
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# BRIDGED_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
# location / { proxy_pass http://127.0.0.1:8765; proxy_set_header Authorization $http_authorization; }
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: BRIDGED_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from bridge_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
# pushReminders: 5 # max nudges before giving up (default 5)
|
||||
# pushBackoffMs: 15000 # delay between nudges (default 15000)
|
||||
|
||||
# CB-530: MORE THAN ONE LEAD. `primary:` above is singular by construction — every other pane
|
||||
# resolves as a worker — which is right for one lead driving a fleet and wrong the moment two leads
|
||||
# (say a Claude lead and an opencode lead) work as peers: the second is silently demoted and refused
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# terminal → the ONLY field identity depends on; get it from that session's bridge_whoami
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by bridge_whoami
|
||||
#
|
||||
# A lead is never spawned — it pre-exists, which is exactly why it must be named rather than created.
|
||||
# `bridge_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
# destination for its nudges. If both name the same terminal, the `fleet.leaders:` entry wins.
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing bridged places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open bridge_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
# Opt-in on purpose — it SPENDS the operator's subscription on its own initiative (each nudge starts
|
||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||
# block = feature off, exactly as before.
|
||||
#
|
||||
# Three knobs, each with a default that errs on the side of not burning context:
|
||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||
# leadHeartbeat:
|
||||
# idleAfterSeconds: 300
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
# it is for; that is the member's role, and roles live under `fleet:` below. Which profile an
|
||||
# unqualified spawn lands on comes from that role's pool, not from a global default.
|
||||
#
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → bridged mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
# cwd on an MCP spawn, else the daemon's cwd — never $HOME. See
|
||||
# docs/Worker-Startup-and-Trust.md.
|
||||
# configDir → CLAUDE_CONFIG_DIR for the worker, so it inherits that profile's
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.claude/settings.local.json, .env, .envrc].
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
# handed a worker the primary's IDE servers, which are bound to the primary's
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# bridged neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
# Opt-in by design — omit and the worker gets no PR-create grant (push over
|
||||
# SSH is unaffected). The token value itself is never stored in this file.
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. bridged hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# bridged now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.bridged.plist and deploy/bridged.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
# entry cannot repoint a worker past the SubscriptionGuard — which is checked
|
||||
# against `baseUrl` alone.
|
||||
# Put `defaultMode: "auto"` in each ccs profile so the worker runs autonomously.
|
||||
profiles:
|
||||
gx10: # ccs profile name (NOT a hostname)
|
||||
kind: claude-code # which adapter spawns this profile (default; may omit)
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: BRIDGED_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
weight: 0.5 # relative selection weight for placement: weighted
|
||||
maxLoad: 2 # max live workers on this profile (omit for unlimited)
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".claude/settings.local.json", ".env", ".envrc"] # never add .mcp.json — see above
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
# Pin an auto-compact window BELOW the served model's context ceiling. The global
|
||||
# ~/.claude/settings.json value is shared by every ccs instance and the primary, so the
|
||||
# per-profile override belongs here. Equal to the ceiling means auto-compact never fires
|
||||
# before the server rejects the prompt, which kills a worker mid-turn (CB-523).
|
||||
env:
|
||||
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "280000"
|
||||
# CB-402: a second coding-agent kind, proving the PeerLauncher SPI is provider-neutral.
|
||||
# opencode is provider-agnostic and uses NONE of Claude's private seams: no ANTHROPIC_BASE_URL /
|
||||
# SubscriptionGuard (so it needs no `guard` host entry), no --mcp-config / --append-system-prompt.
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → bridge_send →
|
||||
# structured bridge_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
#
|
||||
# CB-508: point an opencode profile at your OWN OpenAI-compatible endpoint (local vLLM, llama.cpp,
|
||||
# LM Studio, TGI…) instead of opencode's gateway. Setting `baseUrl` on a `kind: opencode` profile
|
||||
# makes the bridge emit a custom `provider` block into the generated opencode.json — opencode has
|
||||
# no ANTHROPIC_BASE_URL seam, so this is how the endpoint is pinned.
|
||||
# baseUrl → a bare host:port gets `/v1` appended (where these servers mount the API); a URL that
|
||||
# already has a path is used verbatim, so a custom mount point still works.
|
||||
# model → MUST be "<provider>/<model>". The provider half names the generated block; the model
|
||||
# half must match an id the server reports at /v1/models. One field drives both the
|
||||
# declaration and the `-m` flag, so they cannot drift apart. A bare model name with a
|
||||
# baseUrl set is rejected at spawn rather than silently using the default gateway.
|
||||
# tokenEnv → optional; its value becomes the provider apiKey. Most local servers ignore the key,
|
||||
# so a placeholder is used when unset (the AI SDK still requires a non-empty one).
|
||||
# NOTE: no `guard` entry is needed even with a baseUrl set. The SubscriptionGuard exists to stop a
|
||||
# worker borrowing the primary's Anthropic subscription, and an opencode process has no Anthropic
|
||||
# credential path at all.
|
||||
# opencode-local:
|
||||
# kind: opencode
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
# How an unqualified spawn chooses a profile: fixed (default, reproduces pre-CB-518 behaviour),
|
||||
# round-robin, or weighted. Omitting this key is a strict no-op for existing configs.
|
||||
placement: weighted
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded bridged keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. bridged checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
# Not every key can move under a running daemon, and the difference is about what already exists
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool and
|
||||
# `tabLabel`), `placement:`, and an existing profile's weight / maxLoad. Those are
|
||||
# hot because the placement policy reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# DEFERRED → accepted into the new config, but the wiring built at startup keeps the old value
|
||||
# until you restart: `lifecycle:`, `leadHeartbeat:`, `guard:`, `worktreeRoot:`,
|
||||
# `spawnReadyTimeoutMs` / `spawnReadyPollMs`, ADDING or REMOVING a profile (a new
|
||||
# backend needs its own launcher, and launchers are built once), AND an existing
|
||||
# profile's launch settings — model, baseUrl, argv, env, configDir, mcpUrl, tabLabel.
|
||||
# The launcher takes a copy of `profiles:` at startup and resolves every spawn out of
|
||||
# that copy, so those never reach a launch until you restart. The reload logs them by
|
||||
# name rather than pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
# bound, the broker connection is open, and the auth mode decides who may reach the
|
||||
# port that is already listening.
|
||||
#
|
||||
# A changed COLD key refuses the WHOLE reload — not the hot half applied and the cold half warned
|
||||
# about. A half-applied reload would leave the daemon matching no file on disk, which is the worst
|
||||
# thing a reload can do to an operator debugging one. A file that fails to parse or fails a startup
|
||||
# validator is refused the same way, and the running config stays live.
|
||||
# configReload:
|
||||
# enabled: true
|
||||
# intervalSeconds: 10
|
||||
|
||||
# THE FLEET (CB-557) — who the daemon may run, and under which role. This one block replaced four
|
||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||
#
|
||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
||||
# file, the playbook skill and the authz row.
|
||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||
# which is why the two cannot be one field.
|
||||
#
|
||||
# The ROLE IS THE CONTAINING KEY, not a `role:` field. That is not only tidier: a misspelled role
|
||||
# used to parse into a member with no contract at all, while a misspelled pool name here simply
|
||||
# declares nothing.
|
||||
#
|
||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
||||
# with being listed here; the entry key just names the entry.
|
||||
fleet:
|
||||
# Optional. Template for a member tab's label; {role}, {profile}, {model} and {n} are substituted.
|
||||
# {n} counts per role+profile, so `dev: sonnet #2` really is the second sonnet dev. Because {role}
|
||||
# comes from a closed enum, a generated label can never begin with a lead's tabPrefix.
|
||||
# tabLabel: "{role}: {profile} #{n}"
|
||||
|
||||
# Panes that orchestrate rather than are orchestrated. A lead may now be CREATED as well as
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Give it only a `terminal:` and it is recognise-only, as before.
|
||||
#
|
||||
# `tabPrefix` is the naming convention that finds a lead without pasting a terminal id: label the
|
||||
# tab `lead: <name>` when you open it and the pane is recognised on the next rescan. Reopen the
|
||||
# tab later and the id changes; the label does not.
|
||||
#
|
||||
# A lead the daemon launches is labelled BY the daemon, using the same convention, so it is found
|
||||
# by the same scan. A lead counts as live only when herdr also reports a running agent in that
|
||||
# tab — a label left behind by a session that died does not block the relaunch.
|
||||
#
|
||||
# An auto-launched lead is NOT a member: it gets no worker reply charter, is never registered with
|
||||
# the session lifecycle (the idle reaper would kill your orchestrator), and stays on the
|
||||
# subscription — ANTHROPIC_BASE_URL/AUTH_TOKEN are stripped from its env whatever the profile says.
|
||||
# leaders:
|
||||
# opus-5.0:
|
||||
# profile: opus # omit to never create this lead, only recognise it
|
||||
# instances: 1 # desired live count; only the shortfall is launched. 0 = off
|
||||
# terminal: term_0123456789abcd # optional hand-pin; usually found by tabPrefix instead.
|
||||
# # A running agent on this terminal also counts as live, so a
|
||||
# # lead you opened by hand is not relaunched under you.
|
||||
# tabPrefix: "lead:" # `lead: opus-5.0` ⇒ a lead named opus-5.0 (case-insensitive)
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: bridged's own)
|
||||
# kind: claude # descriptive; reported by bridge_whoami
|
||||
# gpt-sol-5.6:
|
||||
# terminal: term_fedcba9876543
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
# architect-2:
|
||||
# profile: sol # a different vendor on purpose — two architects that share a
|
||||
# # model share its blind spots
|
||||
developers:
|
||||
gx10:
|
||||
profile: gx10
|
||||
# reviewers:
|
||||
# gx10:
|
||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Every profile above must have its host listed here.
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- gx01.gw
|
||||
|
||||
# Spawn-readiness gate (CB-306). The launcher blocks until the worker's herdr status is
|
||||
# injectable (IDLE/BLOCKED/DONE) or the timeout elapses. 0 disables the gate.
|
||||
# NOTE: keys are camelCase — config is bound by plain Jackson with no naming strategy and
|
||||
# unknown keys are ignored, so a snake_case key would be silently dropped (default kept).
|
||||
# spawnReadyTimeoutMs: 20000
|
||||
# spawnReadyPollMs: 300
|
||||
|
||||
# Worktree provisioning root (CB-301-ext). Where per-worker git worktrees are checked out so
|
||||
# each worker owns an isolated branch instead of sharing the primary's tree. Omit to default
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
# contextCap → force-release a session after this many delegated turns
|
||||
# drainTimeoutSeconds → seconds to wait for BUSY sessions on shutdown before forced teardown
|
||||
# lifecycle:
|
||||
# idleTtlSeconds: 300
|
||||
# contextCap: 10
|
||||
# drainTimeoutSeconds: 5
|
||||
|
||||
# Durable reply delivery (CB-307 Stage 2). OMIT this block entirely to keep the default
|
||||
# in-memory, soft-state reply inbox (late worker replies are held only until a daemon bounce).
|
||||
# Set a broker uri to swap in the AMQP-backed inbox: worker replies with no open send are held
|
||||
# on a durable per-target queue (agent.<target>.inbox) and survive a restart — the broker
|
||||
# redelivers anything the primary had not yet drained. Production default is LavinMQ; a stock
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# broker:
|
||||
# uri: amqp://guest:guest@127.0.0.1:5672
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open bridge_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
# terminal → pin the primary's herdr terminal id. Omit to learn it from the connection on
|
||||
# the first orchestration-side MCP call (the normal case). An off-host or
|
||||
# non-herdr primary leaves this unresolved → the loop is a no-op and delivery
|
||||
# degrades to pull; the reply is still never lost.
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just bridged-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off bridge_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
# primary:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
@@ -1,488 +0,0 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.config.ConfigRef;
|
||||
import dev.ltms.bridged.config.ConfigWatcher;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.LeadTabScanner;
|
||||
import dev.ltms.bridged.lead.LeadLauncher;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.CompletionResolver;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.inject.StatusPoller;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.auth.MemberRegistry;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.mcp.BridgeMcp;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.bridged.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.bridged.msg.AmqpReplyInbox;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.ReplyInbox;
|
||||
import dev.ltms.bridged.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.bridged.msg.ReplyPushLoop;
|
||||
import dev.ltms.bridged.rest.BridgedApp;
|
||||
import dev.ltms.bridged.session.GitWorktrees;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.session.SessionReaper;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.member.CompositePeerLauncher;
|
||||
import dev.ltms.bridged.member.HerdrPeerLauncher;
|
||||
import dev.ltms.bridged.member.OpenCodeLauncher;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* {@code bridged} entry point. Wires the real herdr socket client to the REST app and
|
||||
* starts listening. Before anything else it asserts its own environment is clean —
|
||||
* {@code bridged} is not a Claude process and must never carry a base_url.
|
||||
*/
|
||||
public final class Bridged {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Bridged.class);
|
||||
|
||||
/** How often the injector samples a busy worker's status while it has queued work. */
|
||||
private static final long INJECT_POLL_MILLIS = 250;
|
||||
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = Path.of(args.length > 0 ? args[0] : "bridged.yaml");
|
||||
BridgedConfig cfg = BridgedConfig.load(configPath);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched bridged must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, BridgedConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, BridgedConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet().tabLabel()));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(agents, spaces,
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet().tabLabel()));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName));
|
||||
// CB-504: under supervision (launchd/systemd) bridged can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the static registry, discover leads by the tab labels the operator
|
||||
// writes. CB-557 moved the settings onto the lead they describe, so scanning is on whenever
|
||||
// a `fleet.leaders:` entry exists — with no leads configured the supplier is a constant and
|
||||
// never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
// One scanner, so one prefix and one interval. Distinct per-lead prefixes would need a
|
||||
// scanner each; until a config actually wants that, take the first entry's settings and
|
||||
// say so, rather than silently honouring one lead's prefix and dropping another's.
|
||||
var scan = leaders.values().iterator().next();
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(Objects::nonNull)
|
||||
.collect(Collectors.toSet());
|
||||
leads = new LeadTabScanner(herdr, scan.tabPrefix(), memberSpaces, leadTerminals,
|
||||
TimeUnit.SECONDS.toNanos(scan.scanIntervalSeconds()), System::nanoTime);
|
||||
log.info("lead scan: tabs labelled '{}…' host a lead (rescan every {}s, member spaces {} "
|
||||
+ "excluded)",
|
||||
scan.tabPrefix(), scan.scanIntervalSeconds(), memberSpaces);
|
||||
long distinctPrefixes = leaders.values().stream()
|
||||
.map(BridgedConfig.Leader::tabPrefix).distinct().count();
|
||||
if (distinctPrefixes > 1) {
|
||||
log.warn("fleet.leaders declares {} different tabPrefix values; only '{}' is scanned "
|
||||
+ "for. Give every lead the same tabPrefix, or leads under the others "
|
||||
+ "will not be discovered.",
|
||||
distinctPrefixes, scan.tabPrefix());
|
||||
}
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(agents, spaces, cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
completion.onDelivered(target);
|
||||
sessions.onDelivered(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Injector injector = new Injector(agents, turnListener, deliverableTo(presence, leads),
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, INJECT_POLL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block (with a uri) selects the AMQP-backed durable adapter;
|
||||
// absent, bridged stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox;
|
||||
if (cfg.broker() != null && cfg.broker().isConfigured()) {
|
||||
replyInbox = AmqpReplyInbox.open(cfg.broker().uri());
|
||||
log.info("reply inbox: AMQP broker (durable) at {}", cfg.broker().uri());
|
||||
} else {
|
||||
replyInbox = new InMemoryReplyInbox();
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
}
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open bridge_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = BridgedMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, agents, replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, agents, replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(terminal -> {
|
||||
messages.abandon(terminal, "the worker session was released before it replied");
|
||||
replyInbox.release(terminal);
|
||||
primaryRegistry.forgetDelegation(terminal); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): bridge_send/bridge_reply/bridge_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting bridged");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads,
|
||||
members::snapshot);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads,
|
||||
members::snapshot);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
BridgeMcp mcp = new BridgeMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics);
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
herdr.close();
|
||||
}));
|
||||
|
||||
Javalin app = new BridgedApp(herdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("bridged listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a worker whose
|
||||
* agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> worker's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code BridgeMcp} marks presence
|
||||
* only for a worker, deliberately, since that map doubles as the worker roster's availability
|
||||
* signal and a lead counted there would show up as an available worker. So without the second
|
||||
* disjunct a lead is permanently un-deliverable: every lead→lead send sat on the gate for
|
||||
* {@code READINESS_GRACE_POLLS} (~60s) and then failed having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Bridged() {
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,59 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
* identity (CB-105). Given the PID that opened an MCP connection, {@link #terminalForPid} finds
|
||||
* the agent pane whose process tree contains it, so {@code bridged} can tell <em>which worker</em>
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane owns it (e.g. the caller is the primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(paneId, pid)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private boolean paneOwnsPid(String paneId, long pid) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -1,265 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* The CB-106 completion fallback: bridges the {@link Injector}'s turn-completion signal to the
|
||||
* {@link Rendezvous} so a blocking {@code bridge_send} resolves even when the worker finishes its
|
||||
* task without ever calling {@code bridge_reply} — the common case for a real delegated coding task.
|
||||
*
|
||||
* <p>On a confirmed {@code working → idle} boundary it scrapes the worker's recent transcript and
|
||||
* resolves the awaiting send with that tail (a {@link Rendezvous.Kind#COMPLETION} resolution, so the
|
||||
* caller can tell a scrape from a structured reply). It scrapes only when a send is actually waiting
|
||||
* — a fleet worker's own turns, or a send that already timed out, cost no herdr traffic. An explicit
|
||||
* {@code bridge_reply} that raced in first wins; {@link Rendezvous#resolveCompletion} is then a no-op.
|
||||
*
|
||||
* <p>It also handles the CB-109 stall signal ({@link #onTurnFailed}): a worker that ran a turn then
|
||||
* wedged in an {@code unknown} state resolves the send as a failure (with the error screen as
|
||||
* context) rather than leaving it to time out.
|
||||
*
|
||||
* <p>The scrape is cleaned to the last {@code ⏺} assistant block (stripping TUI chrome) and guarded
|
||||
* against misattribution (CB-115): the pane content is baselined on delivery ({@link #onDelivered}),
|
||||
* and a completion whose scrape is unchanged from that baseline — the previous turn's wind-down
|
||||
* sampled as this turn's boundary on a rapid back-to-back send — is suppressed rather than resolving
|
||||
* the send with a stale answer.
|
||||
*
|
||||
* <p><strong>Waiter-specific resolution (CB-116).</strong> On delivery we also capture the exact
|
||||
* {@link Rendezvous} waiter this turn belongs to, and the completion/failure fallbacks resolve
|
||||
* <em>that</em> waiter — never "whatever send is waiting now". A completion fallback runs on a virtual
|
||||
* thread and can land after the worker's {@code bridge_reply} already resolved the turn and the
|
||||
* <em>next</em> send opened its own waiter on the same session; resolving the current waiter would
|
||||
* then deliver turn N's stale scrape as turn N+1's answer. Targeting the captured waiter makes a late
|
||||
* completion a harmless no-op (its waiter is already done) instead of a cross-turn stale reply.
|
||||
*
|
||||
* <p>Wired as the {@link Injector}'s {@link TurnListener}; the handlers hand off to a virtual thread
|
||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||
*/
|
||||
public final class CompletionResolver implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source for the completion scrape. {@code recent} returns the tail of
|
||||
* the transcript (the worker's last output), which is what a delegator wants when the worker
|
||||
* didn't structure a reply.
|
||||
*/
|
||||
static final String SCRAPE_SOURCE = "recent";
|
||||
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
* delivering send opened, plus the assistant block present when it was delivered.
|
||||
*
|
||||
* <p>The {@code waiter} is what makes a late fallback safe (CB-116): we resolve it, not "whoever
|
||||
* is waiting now", so a completion that fires after the next send has opened its own waiter is a
|
||||
* no-op rather than a cross-turn stale reply. The {@code baseline} is the CB-115 staleness
|
||||
* reference: a completion scrape equal to it means the worker produced no new output (the previous
|
||||
* turn's wind-down sampled as this boundary), so it is suppressed. Overwritten on each delivery;
|
||||
* cleared when the turn resolves. Package-private so tests can capture and replay a specific turn.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
}
|
||||
String baseline;
|
||||
try {
|
||||
// Clip to the same cap resolve() applies to the tail (line ~134): the CB-115 misattribution
|
||||
// guard compares baseline.equals(tail), so both sides must be the same capped representation.
|
||||
// An unclipped baseline vs a clipped tail would never match for a >MAX_SCRAPE_CHARS block,
|
||||
// defeating the guard and letting a stale completion resolve the send.
|
||||
baseline = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
InFlight inFlight(String target) {
|
||||
return inFlight.get(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
// Read the in-flight turn on the poller thread — before any next-turn delivery can overwrite
|
||||
// it — then off-load the scrape (a herdr round-trip we must not block polling on) to a vthread.
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("completion-" + target).start(() -> resolve(target, turn));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the completed turn before adapter housekeeping can erase its rendered output. This is
|
||||
* intentionally synchronous and used only when a post-turn context reset is enabled; the normal
|
||||
* path remains off-loaded so polling is not blocked by a scrape.
|
||||
*/
|
||||
public void resolveBeforePostAction(String target) {
|
||||
resolve(target, inFlight.get(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
void resolve(String target, InFlight turn) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = turn == null ? null : turn.waiter();
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
// Nobody is blocked on THIS turn (it had no send, or its bridge_reply already won). Skip
|
||||
// the scrape; resolving the current waiter here would be the CB-116 cross-turn stale reply.
|
||||
inFlight.remove(target, turn);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
tail = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
// The worker finished but we couldn't read its screen — still resolve the send so the
|
||||
// caller unblocks; an empty tail beats hanging until the caller's timeout.
|
||||
log.warn("completion scrape for {} failed; resolving with an empty tail: {}",
|
||||
target, e.getMessage());
|
||||
tail = "";
|
||||
scrapeFailed = true;
|
||||
}
|
||||
// Misattribution guard (CB-115): if the scrape is byte-identical to the pane content at
|
||||
// delivery, this turn produced no new output — the boundary belongs to the previous turn's
|
||||
// wind-down (common on rapid back-to-back sends). Suppress rather than resolve the send with
|
||||
// a stale answer; the real bridge_reply (or a later genuine completion) resolves it instead.
|
||||
// A scrape that failed to read is exempt — an empty tail there is "couldn't see", not "no change".
|
||||
String baseline = turn.baseline();
|
||||
if (!scrapeFailed && baseline != null && baseline.equals(tail)) {
|
||||
log.debug("suppressing misattributed completion for {} (no output change since delivery)",
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
}
|
||||
if (rendezvous.resolveCompletion(waiter, tail)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.debug("resolved send to {} via turn-completion fallback ({} chars scraped)",
|
||||
target, tail.length());
|
||||
}
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
CompletableFuture<Rendezvous.Resolution> waiter =
|
||||
turn != null ? turn.waiter() : rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason;
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.debug("failed send to {} via turn-stall fallback", target);
|
||||
}
|
||||
}
|
||||
|
||||
private static String clip(String s) {
|
||||
if (s == null) return "";
|
||||
String trimmed = s.strip();
|
||||
return trimmed.length() <= MAX_SCRAPE_CHARS
|
||||
? trimmed
|
||||
: trimmed.substring(trimmed.length() - MAX_SCRAPE_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the last assistant message from a raw Claude Code pane scrape (CB-115). Claude Code
|
||||
* prefixes each assistant turn with {@code ⏺}; the delegator wants that answer, not the TUI
|
||||
* chrome around it. Take everything from the final {@code ⏺} onward and stop at the <em>first</em>
|
||||
* hard interface boundary below it — the spinner/status line, input box, {@code ❯} prompt (which
|
||||
* may echo the <em>next</em> turn's text), footer, or tips/warnings. Stopping at the first
|
||||
* boundary (rather than trimming only trailing chrome) is what keeps a following turn's echoed
|
||||
* prompt out of this reply. Blank lines are not boundaries, so a multi-paragraph answer survives;
|
||||
* trailing blanks are trimmed at the end. With no {@code ⏺} marker (an unusual render) the whole
|
||||
* text is scanned the same way, so we never lose the reply.
|
||||
*
|
||||
* <p>Package-private and pure so it is unit-testable without herdr.
|
||||
*/
|
||||
static String lastAssistantBlock(String raw) {
|
||||
if (raw == null || raw.isBlank()) return "";
|
||||
int marker = raw.lastIndexOf('⏺');
|
||||
String block = marker >= 0 ? raw.substring(marker + 1) : raw;
|
||||
StringBuilder out = new StringBuilder();
|
||||
int kept = 0;
|
||||
for (String line : block.split("\n", -1)) {
|
||||
if (isBoundary(line)) break; // first TUI boundary ends the assistant message
|
||||
if (kept++ > 0) out.append('\n');
|
||||
out.append(line);
|
||||
}
|
||||
return out.toString().strip();
|
||||
}
|
||||
|
||||
/**
|
||||
* A hard TUI boundary line that marks the end of an assistant message and the start of interface
|
||||
* chrome (input box, prompt, spinner, footer, tips/warnings). Blank lines are <em>not</em>
|
||||
* boundaries — an answer may contain them — so they are kept and trimmed only if trailing.
|
||||
*/
|
||||
private static boolean isBoundary(String line) {
|
||||
String t = line.strip();
|
||||
if (t.isEmpty()) return false;
|
||||
// A horizontal rule / all box-drawing separators (e.g. "──────").
|
||||
if (t.chars().allMatch(c -> c == '─' || c == '—' || c == '━' || c == '═' || c == '-')) {
|
||||
return true;
|
||||
}
|
||||
String lower = t.toLowerCase();
|
||||
return t.startsWith("╭") || t.startsWith("│") || t.startsWith("╰") || t.startsWith("┌")
|
||||
|| t.startsWith("└") || t.startsWith("❯") || t.startsWith("⏵")
|
||||
|| t.startsWith("⎿") || t.startsWith("⚠")
|
||||
// Status/spinner lines Claude Code renders below a settled or in-flight turn,
|
||||
// e.g. "✻ Baked for 21s", "✶ Forming…".
|
||||
|| t.startsWith("✻") || t.startsWith("✳") || t.startsWith("✽") || t.startsWith("·")
|
||||
|| t.startsWith("●") || t.startsWith("◐") || t.startsWith("✢") || t.startsWith("✶")
|
||||
|| lower.contains("auto mode") || lower.contains("for shortcuts")
|
||||
|| lower.contains("esc to interrupt") || lower.contains("bypass permissions");
|
||||
}
|
||||
}
|
||||
@@ -1,963 +0,0 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.auth.AuditLog;
|
||||
import dev.ltms.bridged.auth.Authz;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.auth.Role;
|
||||
import dev.ltms.bridged.guard.GuardException;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.WorktreeRequest;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import io.modelcontextprotocol.common.McpTransportContext;
|
||||
import io.modelcontextprotocol.json.McpJsonMapper;
|
||||
import io.modelcontextprotocol.json.jackson3.JacksonMcpJsonMapperSupplier;
|
||||
import io.modelcontextprotocol.server.McpServer;
|
||||
import io.modelcontextprotocol.server.McpSyncServer;
|
||||
import io.modelcontextprotocol.server.McpSyncServerExchange;
|
||||
import io.modelcontextprotocol.server.transport.HttpServletStreamableServerTransportProvider;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import jakarta.servlet.http.HttpServlet;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* The MCP SERVER face (CB-105): a Streamable-HTTP MCP server whose tools are <em>thin adapters</em>
|
||||
* over the same {@link MessageService}/{@link Rendezvous} the REST routes use — so the two are
|
||||
* validated by parity, not by re-implementing behaviour. The primary Opus calls {@code bridge_send}
|
||||
* / {@code bridge_status}; the worker calls {@code bridge_reply}.
|
||||
*
|
||||
* <p>Beyond delegation the primary also manages the fleet here (CB-108): {@code bridge_spawn} /
|
||||
* {@code bridge_list} / {@code bridge_stop} drive the {@link PeerLauncher} SPI so a worker's whole
|
||||
* lifecycle is managed through MCP, with each adapter's subscription boundary enforced inside it.
|
||||
*
|
||||
* <p>The tool <em>logic</em> lives in package-private static methods returning a
|
||||
* {@link McpSchema.CallToolResult}, so it is unit-testable without standing up the HTTP transport;
|
||||
* the SDK owns the wire protocol. Mount {@link #servlet()} at {@code /mcp} on the daemon's Jetty.
|
||||
*/
|
||||
public final class BridgeMcp {
|
||||
|
||||
private static final long DEFAULT_TIMEOUT_MS = 25_000;
|
||||
private static final long MAX_TIMEOUT_MS = 120_000;
|
||||
// bridge_ask blocks the WORKER's own MCP call, which its client caps near 60s — default under
|
||||
// that so the bridge returns a clean timeout before the client severs the call (CB-205).
|
||||
private static final long ASK_DEFAULT_TIMEOUT_MS = 55_000;
|
||||
private static final long ASK_MAX_TIMEOUT_MS = 115_000;
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper(); // worker-view JSON projections
|
||||
|
||||
/** Transport-context key under which the extractor stashes the resolved caller identity. */
|
||||
static final String CALLER_TERMINAL = "callerTerminal";
|
||||
/** Transport-context key under which the extractor stashes the caller's PID (for cwd inherit). */
|
||||
static final String CALLER_PID = "callerPid";
|
||||
/** Transport-context key under which the extractor stashes the resolved {@link Role} (CB-501). */
|
||||
static final String CALLER_ROLE = "callerRole";
|
||||
/** Transport-context key for the lead's configured name, when the caller is one (CB-530). */
|
||||
static final String CALLER_NAME = "callerName";
|
||||
|
||||
private final HttpServletStreamableServerTransportProvider transport;
|
||||
private final McpSyncServer server;
|
||||
private final CallerResolver authz; // CB-501: null → authorization not enforced (legacy)
|
||||
private final Metrics metrics; // CB-502: null → auth failures not counted
|
||||
|
||||
/**
|
||||
* Legacy constructor — no authorization. Retained so existing tests exercise tool behaviour
|
||||
* without an auth fixture.
|
||||
*/
|
||||
public BridgeMcp(MessageService messages, PeerLauncher workers,
|
||||
SessionManager sessions, ConnectionIdentity identity, MemberPresence presence,
|
||||
PrimaryRegistry primaryRegistry) {
|
||||
this(messages, workers, sessions, identity, presence, primaryRegistry, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param callers resolves each call's {@link Principal}; {@code null} disables authorization.
|
||||
* This surface needs its own enforcement: {@code /mcp} is a raw servlet on
|
||||
* Jetty's context handler and never passes through Javalin's {@code before}
|
||||
* filter, so the REST guard does not cover it.
|
||||
* @param metrics registry for auth-failure counting; may be {@code null}
|
||||
*/
|
||||
public BridgeMcp(MessageService messages, PeerLauncher workers,
|
||||
SessionManager sessions, ConnectionIdentity identity, MemberPresence presence,
|
||||
PrimaryRegistry primaryRegistry, CallerResolver callers, Metrics metrics) {
|
||||
McpJsonMapper json = new JacksonMcpJsonMapperSupplier().get();
|
||||
this.transport = HttpServletStreamableServerTransportProvider.builder()
|
||||
.jsonMapper(json)
|
||||
.mcpEndpoint("/mcp")
|
||||
// Resolve the caller from the connection (peer PID → herdr pane) in one lookup: the
|
||||
// worker terminal for bridge_reply (no spoofable arg), and the PID so bridge_spawn can
|
||||
// inherit the primary's cwd (CB-112). Any contact from a worker marks it available
|
||||
// (CB-113) — its MCP initialize is the reliable "the agent is up" signal.
|
||||
.contextExtractor(req -> {
|
||||
// One resolution per call, shared with the REST surface via CallerResolver so
|
||||
// the two paths cannot drift on who a caller is.
|
||||
Principal p = callers != null
|
||||
? callers.resolve(req.getRemoteAddr(), req.getRemotePort(),
|
||||
req.getHeader("Authorization"))
|
||||
: legacyPrincipal(identity, req.getRemoteAddr(), req.getRemotePort());
|
||||
// CB-532: guard on the ROLE, not on the terminal being null. A named lead now
|
||||
// carries its pane too, and enrolling a lead in the worker presence map would
|
||||
// have it counted as an available worker.
|
||||
if (p.isWorker()) presence.markPresent(p.terminal());
|
||||
return McpTransportContext.create(Map.of(
|
||||
CALLER_TERMINAL, orEmpty(p.terminal()),
|
||||
CALLER_PID, Long.toString(p.pid()),
|
||||
CALLER_ROLE, p.role().name(),
|
||||
CALLER_NAME, orEmpty(p.name())));
|
||||
})
|
||||
.build();
|
||||
this.server = McpServer.sync(transport)
|
||||
.serverInfo("bridge", "0.1.0")
|
||||
.capabilities(McpSchema.ServerCapabilities.builder().tools(true).build())
|
||||
.toolCall(sendTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SEND,
|
||||
str(req.arguments(), "sessionId"));
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
// CB-548: only a PRIMARY caller may claim the legacy singleton "primary" fallback.
|
||||
// An architect delegates as its own pane but must never become the fallback that
|
||||
// no-delegation inbox nudges target as if it were the primary (the per-target
|
||||
// delegation map does not cure the singleton).
|
||||
recordPrimarySingleton(primaryRegistry, caller, principal(exchange));
|
||||
Map<String, Object> a = req.arguments();
|
||||
String target = str(a, "sessionId");
|
||||
String content = str(a, "content");
|
||||
String turnId = str(a, "turnId");
|
||||
if (turnId != null && !turnId.isBlank()) {
|
||||
// Answering a worker's bridge_ask (CB-205): resolve its blocked question and
|
||||
// block for the worker's reply as it resumes the same turn. This is the same
|
||||
// delegation, so ownership is left untouched (CB-548) — never re-recorded.
|
||||
return answer(messages, turnId, content, timeoutMs(a));
|
||||
}
|
||||
// CB-548: delegator ownership (which lead's reply nudge this worker routes to,
|
||||
// CB-532) is recorded only once the send is ACCEPTED — MessageService has won the
|
||||
// session lock and queued delivery — via the accepted-delivery callback, never at
|
||||
// request time. A concurrent sender that times out BUSY therefore cannot steal a
|
||||
// live turn's reply routing without ever owning the turn.
|
||||
Runnable onAccepted = () -> primaryRegistry.recordDelegation(target, caller);
|
||||
// wait defaults to true (block for the reply); wait:false is fire-and-poll.
|
||||
return Boolean.FALSE.equals(a.get("wait"))
|
||||
? sendAsync(messages, target, content, onAccepted)
|
||||
: send(messages, target, content, timeoutMs(a), onAccepted);
|
||||
})
|
||||
// bridge_reply's identity is the CONNECTION, never an argument — so the authz check
|
||||
// is "is this caller a worker at all", and it can only ever reply as itself.
|
||||
.toolCall(replyTool(), (exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.REPLY, self);
|
||||
if (denied != null) return denied;
|
||||
return reply(messages, self, str(req.arguments(), "content"));
|
||||
})
|
||||
// bridge_ask (CB-205): a worker's mid-turn question — identity from the CONNECTION.
|
||||
.toolCall(askTool(), (exchange, req) -> {
|
||||
String self = callerTerminal(exchange);
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.ASK, self);
|
||||
if (denied != null) return denied;
|
||||
return ask(messages, self, str(req.arguments(), "question"), timeoutMs(req.arguments()));
|
||||
})
|
||||
.toolCall(statusTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return status(messages, str(req.arguments(), "sessionId"));
|
||||
})
|
||||
.toolCall(pollTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
Map<String, Object> a = req.arguments();
|
||||
return poll(messages, str(a, "ticket"), str(a, "target"));
|
||||
})
|
||||
// CB-307 Increment 3: per-msgId ack (not needed in v1 but supported by the inbox).
|
||||
// Acking removes a reply from the inbox, so it is a drain, not a read.
|
||||
.toolCall(ackTool(), (exchange, req) -> {
|
||||
Map<String, Object> a = req.arguments();
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.DRAIN, str(a, "target"));
|
||||
if (denied != null) return denied;
|
||||
return ack(messages, str(a, "target"), str(a, "msgId"));
|
||||
})
|
||||
// Fleet management (CB-108): spawn/list/stop over ClaudeCodeLauncher.
|
||||
.toolCall(spawnTool(), (exchange, req) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.SPAWN, null);
|
||||
if (denied != null) return denied;
|
||||
String caller = callerTerminal(exchange);
|
||||
// SPAWN is already auth-gated to PRIMARY (architects can never call it), but
|
||||
// enforce the same invariant here: only a PRIMARY may claim the legacy singleton.
|
||||
recordPrimarySingleton(primaryRegistry, caller, principal(exchange));
|
||||
Map<String, Object> a = req.arguments();
|
||||
// CB-112: worker inherits the primary's cwd unless the call pins one.
|
||||
// CB-301: carry the caller's identity as the session owner (null for the primary).
|
||||
// CB-301-ext: optional isolated worktree for parallel implementers.
|
||||
String callerCwd = identity.cwdForPid(callerPid(exchange));
|
||||
return spawn(sessions, str(a, "profile"), str(a, "role"), str(a, "cwd"), callerCwd,
|
||||
callerTerminal(exchange), worktreeRequest(a));
|
||||
})
|
||||
.toolCall(listTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return listFleet(workers, sessions,
|
||||
callers == null ? Map.of() : callers.leads(),
|
||||
callerTerminal(exchange));
|
||||
})
|
||||
.toolCall(stopTool(), (exchange, req) -> {
|
||||
String paneId = str(req.arguments(), "paneId");
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.STOP, paneId);
|
||||
if (denied != null) return denied;
|
||||
return stop(sessions, paneId);
|
||||
})
|
||||
.toolCall(profilesTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return profiles(workers);
|
||||
})
|
||||
.toolCall(whoamiTool(), (exchange, _) -> {
|
||||
McpSchema.CallToolResult denied = deny(exchange, Authz.Action.READ, null);
|
||||
if (denied != null) return denied;
|
||||
return whoami(principal(exchange), sessions);
|
||||
})
|
||||
.build();
|
||||
this.authz = callers;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pre-CB-501 identity: worker if the connection maps to a pane, otherwise the primary. Used
|
||||
* only by the legacy constructor, where authorization is not enforced anyway.
|
||||
*/
|
||||
private static Principal legacyPrincipal(ConnectionIdentity identity, String addr, int port) {
|
||||
ConnectionIdentity.Caller c = identity.resolve(addr, port);
|
||||
return c.terminal() != null
|
||||
? Principal.worker(c.terminal(), c.pid())
|
||||
: Principal.primary(c.pid());
|
||||
}
|
||||
|
||||
/** The caller reconstructed from the transport context. */
|
||||
private static Principal principal(McpSyncServerExchange exchange) {
|
||||
return principalFrom(exchange.transportContext().get(CALLER_ROLE),
|
||||
callerTerminal(exchange), callerPid(exchange), callerName(exchange));
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild a {@link Principal} from the three values the context extractor stashed.
|
||||
*
|
||||
* <p>Split out from {@link #principal(McpSyncServerExchange)} so the identity rules are
|
||||
* reachable without an {@code McpSyncServerExchange} — that is an SDK type this project has no
|
||||
* mocking library to fabricate, which is why this logic had no test at all until CB-513.
|
||||
*
|
||||
* @param role the stashed {@link Role} name, or {@code null} on the legacy path
|
||||
* @param terminal the worker terminal, or {@code null} for a non-worker
|
||||
* @param pid the calling pid, or {@code -1}
|
||||
*/
|
||||
static Principal principalFrom(Object role, String terminal, long pid) {
|
||||
return principalFrom(role, terminal, pid, null);
|
||||
}
|
||||
|
||||
/** As {@link #principalFrom(Object, String, long)}, carrying a lead's name (CB-530). */
|
||||
static Principal principalFrom(Object role, String terminal, long pid, String name) {
|
||||
if (role == null) {
|
||||
// No role stashed (legacy path): fall back to the historical interpretation.
|
||||
return terminal != null ? Principal.worker(terminal, pid) : Principal.primary(pid);
|
||||
}
|
||||
return new Principal(Role.valueOf(role.toString()), terminal, pid, name);
|
||||
}
|
||||
|
||||
/**
|
||||
* Gate a tool call on the CB-505 table. Returns {@code null} when the call may proceed, or the
|
||||
* error result to return when it may not.
|
||||
*/
|
||||
private McpSchema.CallToolResult deny(McpSyncServerExchange exchange, Authz.Action action,
|
||||
String target) {
|
||||
return denyFor(principal(exchange), action, target);
|
||||
}
|
||||
|
||||
/**
|
||||
* The policy half of {@link #deny}: everything except pulling the caller out of the MCP
|
||||
* exchange. Kept separate so the authorization decision — the actual control — is unit-testable
|
||||
* without fabricating an SDK {@code McpSyncServerExchange}.
|
||||
*
|
||||
* <p>This surface exists because the enforcement was previously unreachable from a test: no
|
||||
* test constructs a {@code BridgeMcp}, so the whole MCP-side gate ran zero times in the suite
|
||||
* while the REST-side equivalent had ten tests. A security control nothing exercises is a
|
||||
* claim, not a control.
|
||||
*
|
||||
* @return {@code null} when the call may proceed, or the error result to return when it may not
|
||||
*/
|
||||
McpSchema.CallToolResult denyFor(Principal caller, Authz.Action action, String target) {
|
||||
// The enforcement switch lives HERE rather than in the exchange-facing wrapper: any future
|
||||
// tool that calls this directly must not be able to skip the gate by accident.
|
||||
if (authz == null) {
|
||||
return null; // legacy constructor: authorization not enforced
|
||||
}
|
||||
if (Authz.permits(caller, action, target)) {
|
||||
if (action != Authz.Action.READ) {
|
||||
AuditLog.allowed(caller, action, target); // reads would drown the trail
|
||||
}
|
||||
return null;
|
||||
}
|
||||
String reason = Authz.isUnauthenticated(caller) ? "unauthenticated" : "forbidden";
|
||||
AuditLog.denied(caller, action, target, reason);
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.AUTH_FAILURES, "reason", reason);
|
||||
}
|
||||
return error(reason + ": " + caller.describe() + " may not " + action);
|
||||
}
|
||||
|
||||
/**
|
||||
* Update the legacy singleton "primary" fallback used for no-delegation inbox nudges (CB-548).
|
||||
*
|
||||
* <p>Only {@link Role#PRIMARY} callers — the unnamed primary and named leads alike — may claim
|
||||
* it. An architect delegates as its own pane but must never become the fallback: the per-target
|
||||
* delegation map ({@code PrimaryRegistry#recordDelegation}) does not cure the singleton, so an
|
||||
* architect left here would draw nudges that belong to a primary. The decision uses the resolved
|
||||
* role, never name/kind sniffing. A null {@code caller} (legacy/no-auth path) records nothing.
|
||||
*
|
||||
* <p>Split out of the tool handlers so the guard is unit-testable without fabricating an SDK
|
||||
* {@code McpSyncServerExchange} (same pattern as {@link #denyFor}/{@link #principalFrom}).
|
||||
*/
|
||||
static void recordPrimarySingleton(PrimaryRegistry registry, String callerTerminal, Principal caller) {
|
||||
if (caller != null && caller.isPrimary()) {
|
||||
registry.record(callerTerminal);
|
||||
}
|
||||
}
|
||||
|
||||
/** The worker identity resolved from this call's connection, or {@code null} if the primary. */
|
||||
private static String callerTerminal(McpSyncServerExchange exchange) {
|
||||
Object v = exchange.transportContext().get(CALLER_TERMINAL);
|
||||
String s = v == null ? null : v.toString();
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** The lead name resolved from this call's connection, or {@code null} (CB-530). */
|
||||
private static String callerName(McpSyncServerExchange exchange) {
|
||||
Object v = exchange.transportContext().get(CALLER_NAME);
|
||||
String s = v == null ? null : v.toString();
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** The caller's PID resolved from this call's connection, or {@code -1} if unknown. */
|
||||
private static long callerPid(McpSyncServerExchange exchange) {
|
||||
Object v = exchange.transportContext().get(CALLER_PID);
|
||||
try {
|
||||
return v == null ? -1 : Long.parseLong(v.toString());
|
||||
} catch (NumberFormatException e) {
|
||||
return -1;
|
||||
}
|
||||
}
|
||||
|
||||
private static String orEmpty(String s) {
|
||||
return s == null ? "" : s;
|
||||
}
|
||||
|
||||
/** The Streamable-HTTP servlet to mount at {@code /mcp} on the daemon's Jetty. */
|
||||
public HttpServlet servlet() {
|
||||
return transport;
|
||||
}
|
||||
|
||||
/** Graceful shutdown of the MCP server. */
|
||||
public void close() {
|
||||
server.closeGracefully();
|
||||
}
|
||||
|
||||
// --- tool logic (thin adapters over the services; unit-testable) ---------------------------
|
||||
|
||||
/** {@code bridge_send}: delegate {@code content} to a worker session and block for its reply. */
|
||||
static McpSchema.CallToolResult send(MessageService messages, String sessionId, String content, Long timeoutMs) {
|
||||
return send(messages, sessionId, content, timeoutMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(MessageService, String, String, Long)}, wiring an accepted-delivery hook
|
||||
* (CB-548): {@code onAccepted} records delegator ownership the instant the send is accepted, so
|
||||
* a BUSY interloper never claims a turn it did not win. {@code null} disables recording.
|
||||
*/
|
||||
static McpSchema.CallToolResult send(MessageService messages, String sessionId, String content,
|
||||
Long timeoutMs, Runnable onAccepted) {
|
||||
if (isBlank(sessionId) || isBlank(content)) {
|
||||
return error("sessionId and content are required");
|
||||
}
|
||||
long timeout = clamp(timeoutMs == null ? DEFAULT_TIMEOUT_MS : timeoutMs);
|
||||
try {
|
||||
return formatReply(messages.send(sessionId, content, timeout, onAccepted), timeout);
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error contacting session " + sessionId + ": " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_send} carrying a {@code turnId}: the primary's answer to a worker's
|
||||
* {@code bridge_ask} (CB-205). Resolves the worker's blocked question and blocks for its reply as
|
||||
* it resumes the same turn — surfaced to the primary identically to a normal send.
|
||||
*/
|
||||
static McpSchema.CallToolResult answer(MessageService messages, String turnId, String content, Long timeoutMs) {
|
||||
if (isBlank(turnId) || isBlank(content)) {
|
||||
return error("turnId and content are required to answer a worker's question");
|
||||
}
|
||||
long timeout = clamp(timeoutMs == null ? DEFAULT_TIMEOUT_MS : timeoutMs);
|
||||
return formatReply(messages.answer(turnId, content, timeout), timeout);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_ask} (CB-205): a worker pauses its delegated turn to ask the primary, blocking
|
||||
* until the primary answers. The worker is identified by its connection ({@code callerTerminal}),
|
||||
* never an argument — a {@code null} means the caller is not a known worker.
|
||||
*/
|
||||
static McpSchema.CallToolResult ask(MessageService messages, String callerTerminal, String question, Long timeoutMs) {
|
||||
if (callerTerminal == null) {
|
||||
return error("bridge_ask is for workers only — could not identify the calling worker "
|
||||
+ "from the connection");
|
||||
}
|
||||
if (isBlank(question)) {
|
||||
return error("question is required");
|
||||
}
|
||||
long timeout = Math.clamp(timeoutMs == null ? ASK_DEFAULT_TIMEOUT_MS : timeoutMs, 1, ASK_MAX_TIMEOUT_MS);
|
||||
MessageService.AskResult r = messages.ask(callerTerminal, question, timeout);
|
||||
return switch (r.outcome()) {
|
||||
case ANSWERED -> text(r.answer());
|
||||
case NO_WAITER -> error("no primary is awaiting this turn — bridge_ask only works while a "
|
||||
+ "bridge_send delegation is open to answer it");
|
||||
case TIMED_OUT -> text("[no answer within " + timeout + "ms — the primary did not respond; "
|
||||
+ "proceed on your best judgement, then call bridge_reply to end the turn]");
|
||||
};
|
||||
}
|
||||
|
||||
/** Render a {@link MessageService.Reply} as a tool result — shared by {@link #send} and {@link #answer}. */
|
||||
private static McpSchema.CallToolResult formatReply(MessageService.Reply r, long timeout) {
|
||||
return switch (r.outcome()) {
|
||||
case REPLIED -> text(r.text());
|
||||
// The worker's turn finished but it never called bridge_reply — hand back the scraped
|
||||
// transcript tail, flagged so the primary knows it isn't a structured reply.
|
||||
case COMPLETED_UNREPLIED -> text(
|
||||
"[worker finished without a structured bridge_reply — transcript tail follows]\n" + r.text());
|
||||
// The worker ran the turn then wedged (CB-109) — surface the error context.
|
||||
case WORKER_FAILED -> text("[worker failed — turn ended in an unrecoverable state]\n" + r.text());
|
||||
// The worker paused mid-turn to ask (CB-205) — tell the primary how to answer in-turn.
|
||||
case QUESTION -> text("[question] the worker paused to ask before it can finish:\n" + r.text()
|
||||
+ "\n\nAnswer it by calling bridge_send again with turnId=\"" + r.turnId()
|
||||
+ "\" and content set to your answer; the worker resumes the same turn.");
|
||||
case STALE_TURN -> error("that question is no longer open — it timed out or was already "
|
||||
+ "answered (turnId stale)");
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> text("[no reply within " + timeout + "ms — worker "
|
||||
+ r.outcome().name().toLowerCase().replace("timed_out_", "") + "; retry or poll status]");
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_send} with {@code wait:false}: delegate {@code content} and return a ticket
|
||||
* immediately (fire-and-poll), so a long task isn't cut off by the caller's MCP call timeout.
|
||||
*/
|
||||
static McpSchema.CallToolResult sendAsync(MessageService messages, String sessionId, String content) {
|
||||
return sendAsync(messages, sessionId, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(MessageService, String, String)}, wiring the accepted-delivery hook
|
||||
* (CB-548) so an async flooding send records delegator ownership exactly once it is accepted.
|
||||
*/
|
||||
static McpSchema.CallToolResult sendAsync(MessageService messages, String sessionId, String content,
|
||||
Runnable onAccepted) {
|
||||
if (isBlank(sessionId) || isBlank(content)) {
|
||||
return error("sessionId and content are required");
|
||||
}
|
||||
String ticket = messages.sendAsync(sessionId, content, onAccepted);
|
||||
return text("accepted — task delegated. Poll bridge_poll with ticket=" + ticket);
|
||||
}
|
||||
|
||||
/** {@code bridge_poll}: check an async delegation by ticket, or drain a worker's inbox by target. */
|
||||
static McpSchema.CallToolResult poll(MessageService messages, String ticket, String target) {
|
||||
if (!isBlank(target)) {
|
||||
var replies = messages.drainReplies(target);
|
||||
if (replies.isEmpty()) {
|
||||
return text("[]");
|
||||
}
|
||||
return text(json(replies));
|
||||
}
|
||||
if (isBlank(ticket)) {
|
||||
return error("ticket (or target) is required");
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ticket);
|
||||
if (v == null) {
|
||||
return error("unknown ticket: " + ticket + " (never issued, or expired)");
|
||||
}
|
||||
return switch (v.phase()) {
|
||||
case DONE -> text(v.replySource() != null && v.replySource().equals("transcript")
|
||||
? "[done — worker finished without a structured bridge_reply; transcript tail follows]\n" + v.reply()
|
||||
: v.reply());
|
||||
case PENDING -> text("[pending — " + v.detail() + "]");
|
||||
case FAILED -> text("[failed — " + v.detail() + "]");
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_reply}: the worker returns its structured answer, resolving the awaiting send
|
||||
* or — when no send is open — queueing the reply in the inbox for later drain (CB-307).
|
||||
* {@code callerTerminal} is resolved from the connection (never an argument); a {@code null}
|
||||
* means the caller is not a known worker (e.g. the primary called it by mistake).
|
||||
*/
|
||||
static McpSchema.CallToolResult reply(MessageService messages, String callerTerminal, String content) {
|
||||
if (callerTerminal == null) {
|
||||
return error("bridge_reply is for workers only — could not identify the calling worker "
|
||||
+ "from the connection");
|
||||
}
|
||||
if (content == null) {
|
||||
return error("content is required");
|
||||
}
|
||||
messages.reply(callerTerminal, content);
|
||||
return text("delivered");
|
||||
}
|
||||
|
||||
/** {@code bridge_ack}: acknowledge (remove) a specific reply from the inbox. */
|
||||
static McpSchema.CallToolResult ack(MessageService messages, String target, String msgId) {
|
||||
if (isBlank(target) || isBlank(msgId)) {
|
||||
return error("target and msgId are required");
|
||||
}
|
||||
messages.ackReply(target, msgId);
|
||||
return text("acknowledged " + msgId);
|
||||
}
|
||||
|
||||
/** {@code bridge_status}: the live lifecycle status of a worker session. */
|
||||
static McpSchema.CallToolResult status(MessageService messages, String sessionId) {
|
||||
if (isBlank(sessionId)) {
|
||||
return error("sessionId is required");
|
||||
}
|
||||
try {
|
||||
return text(messages.status(sessionId).name().toLowerCase());
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error for session " + sessionId + ": " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_whoami}: the caller's own identity, as the daemon already resolved it.
|
||||
*
|
||||
* <p>Every other tool <em>consumes</em> this identity — the authorization gate, the reply
|
||||
* rendezvous, the cwd inherit — but none reported it, so an agent had to infer its own role
|
||||
* from side channels the daemon does not control: a charter string in its system prompt, the
|
||||
* name its MCP mount happens to carry, or {@code ANTHROPIC_BASE_URL} (which Claude-model
|
||||
* workers do not set). The failure mode of guessing is asymmetric and silent: a primary that
|
||||
* mistakes itself for a worker is refused by {@link Authz} and learns immediately, while a
|
||||
* worker that mistakes itself for the primary ends its turn without {@code bridge_reply} and
|
||||
* the sender simply receives nothing. This tool removes the guess.
|
||||
*
|
||||
* <p>For a worker the session registry adds what it knows about that session. A worker the
|
||||
* registry has no record of — one that outlived a daemon restart — still gets its role and
|
||||
* {@code sessionId}, which is the load-bearing part.
|
||||
*/
|
||||
static McpSchema.CallToolResult whoami(Principal caller, SessionManager sessions) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("role", caller.role().name().toLowerCase());
|
||||
if (caller.isArchitect()) {
|
||||
// CB-548: the role reads "architect"; the name is the gateway-local slot the pane is
|
||||
// bound to, and the pane itself so a peer knows where to reach it.
|
||||
if (caller.name() != null) {
|
||||
m.put("architect", caller.name());
|
||||
}
|
||||
if (caller.terminal() != null) {
|
||||
m.put("sessionId", caller.terminal());
|
||||
}
|
||||
return text(json(m));
|
||||
}
|
||||
if (!caller.isWorker()) {
|
||||
// CB-530: which lead, once more than one pane is configured as one. `role` deliberately
|
||||
// still reads "primary" — the fallback ladder in CLAUDE.md keys on it, and a lead IS a
|
||||
// primary for authorization; the name is additive so no existing reader breaks.
|
||||
if (caller.name() != null) {
|
||||
m.put("leader", caller.name());
|
||||
}
|
||||
// CB-532: a lead's own pane, so it can tell a peer where to reach it — and so an
|
||||
// operator can read off which tab hosts which lead without going to herdr.
|
||||
if (caller.terminal() != null) {
|
||||
m.put("sessionId", caller.terminal());
|
||||
}
|
||||
return text(json(m));
|
||||
}
|
||||
m.put("sessionId", caller.terminal());
|
||||
sessions.roster().stream()
|
||||
.filter(s -> caller.terminal().equals(s.terminalId()))
|
||||
.findFirst()
|
||||
.ifPresent(s -> {
|
||||
m.put("paneId", s.paneId());
|
||||
m.put("profile", s.profile());
|
||||
m.put("state", s.state().name().toLowerCase());
|
||||
if (s.worktree() != null) {
|
||||
m.put("worktree", s.worktree());
|
||||
}
|
||||
if (s.branch() != null) {
|
||||
m.put("branch", s.branch());
|
||||
}
|
||||
if (s.ownerTerminal() != null) {
|
||||
m.put("owner", s.ownerTerminal());
|
||||
}
|
||||
});
|
||||
return text(json(m));
|
||||
}
|
||||
|
||||
// --- fleet management logic (CB-108 / CB-301) --------------------------------------------
|
||||
|
||||
/** {@code bridge_spawn} without cwd/caller context (default resolution). */
|
||||
static McpSchema.CallToolResult spawn(SessionManager sessions, String profile) {
|
||||
return spawn(sessions, profile, null, null, null, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_spawn}: launch a guard-checked member for {@code profile} (blank → the default
|
||||
* profile) under {@code role} (blank → {@code dev}), and return its session id + pane id. The
|
||||
* member's cwd is {@code requestedCwd} if given, else the profile's config, else
|
||||
* {@code callerCwd} (the primary's directory), else the daemon's.
|
||||
* CB-301: the session is registered with {@code ownerTerminal} as its owner.
|
||||
* CB-301-ext: {@code worktreeRequest} non-null provisions an isolated git worktree.
|
||||
*
|
||||
* <p>{@code role} and {@code profile} are independent: the role picks the contract, the profile
|
||||
* picks the backend. A reviewer on the same profile as the dev it reviews is a normal spawn.
|
||||
*/
|
||||
static McpSchema.CallToolResult spawn(SessionManager sessions, String profile, String role,
|
||||
String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest worktreeRequest) {
|
||||
MemberRole memberRole;
|
||||
try {
|
||||
memberRole = isBlank(role) ? MemberRole.DEV : MemberRole.parse(role);
|
||||
} catch (IllegalArgumentException e) {
|
||||
return error(e.getMessage());
|
||||
}
|
||||
try {
|
||||
MemberSession member = sessions.acquire(isBlank(profile) ? null : profile, memberRole,
|
||||
requestedCwd, callerCwd, ownerTerminal, worktreeRequest);
|
||||
return text(json(memberView(member)));
|
||||
} catch (GuardException e) {
|
||||
return error("subscription boundary: " + e.getMessage());
|
||||
} catch (IllegalArgumentException e) {
|
||||
return error(e.getMessage()); // unknown / no-default profile
|
||||
} catch (PeerUnreachableException e) {
|
||||
return error("spawn timed out — worker pane never reached injectable state: " + e.getMessage());
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error spawning worker: " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Build a {@link WorktreeRequest} from {@code bridge_spawn}'s optional {@code worktree}/{@code ticket} args. */
|
||||
private static WorktreeRequest worktreeRequest(Map<String, Object> a) {
|
||||
Object w = a.get("worktree");
|
||||
if (w == null || Boolean.FALSE.equals(w)) {
|
||||
return null;
|
||||
}
|
||||
String ticket = str(a, "ticket");
|
||||
if (w instanceof String s) {
|
||||
if (s.isBlank() || "false".equalsIgnoreCase(s)) {
|
||||
return null;
|
||||
}
|
||||
if ("true".equalsIgnoreCase(s)) {
|
||||
if (isBlank(ticket)) {
|
||||
throw new IllegalArgumentException("worktree=true requires a ticket slug");
|
||||
}
|
||||
return new WorktreeRequest(ticket, null);
|
||||
}
|
||||
return new WorktreeRequest(s, null);
|
||||
}
|
||||
if (w instanceof Boolean b && b) {
|
||||
if (isBlank(ticket)) {
|
||||
throw new IllegalArgumentException("worktree=true requires a ticket slug");
|
||||
}
|
||||
return new WorktreeRequest(ticket, null);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** {@code bridge_profiles}: the configured worker profiles and the default. */
|
||||
static McpSchema.CallToolResult profiles(PeerLauncher workers) {
|
||||
return text(json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile())));
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code bridge_list}: the whole fleet — {@code leads} and {@code workers} — each merged with
|
||||
* live herdr status. CB-519 decoupled the registry key (a host-unique id) from the herdr pane
|
||||
* coordinate, so the join is on the terminal id, which both the session and the live agent carry.
|
||||
*
|
||||
* <p>CB-535 added the {@code leads} half. Until then this listed the worker roster alone, and a
|
||||
* lead asking "who else is here?" got an empty array — which reads as <em>no peers</em> but
|
||||
* actually means <em>no workers spawned</em>. There was no way at all for a lead to learn a
|
||||
* peer's address; it had to be carried across by a human. Both halves are reported even when a
|
||||
* half is empty, so an empty {@code workers} can no longer be mistaken for an empty fleet.
|
||||
*
|
||||
* <p>Leads are drawn from the resolver rather than from a second registry, so an address listed
|
||||
* here is one that would actually resolve as a lead — see {@link CallerResolver#leads()}. The
|
||||
* caller's own row is flagged {@code "self": true}: a peer needs to tell its own pane apart from
|
||||
* a peer's, and the alternative is every lead calling {@code bridge_whoami} to subtract itself.
|
||||
*
|
||||
* @param leads terminal_id → lead name, live from the resolver
|
||||
* @param selfTerm the calling pane's terminal id, or blank for a caller with no pane
|
||||
*/
|
||||
static McpSchema.CallToolResult listFleet(PeerLauncher workers, SessionManager sessions,
|
||||
Map<String, String> leads, String selfTerm) {
|
||||
try {
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
List<Map<String, Object>> leadRows = leads.entrySet().stream()
|
||||
.sorted(Map.Entry.comparingByValue())
|
||||
.map(e -> leadView(e.getKey(), e.getValue(), live.get(e.getKey()), selfTerm))
|
||||
.toList();
|
||||
List<Map<String, Object>> out = sessions.roster().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
return text(json(Map.of("leads", leadRows, "members", out)));
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error listing the fleet: " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* One lead's row: its address, its name, and whether it can be reached right now.
|
||||
*
|
||||
* <p>{@code status} is herdr's live view, and {@code unknown} when herdr is not tracking that
|
||||
* pane as an agent — the honest answer, and the one that matters: a lead whose pane herdr cannot
|
||||
* see is a lead a {@code bridge_send} cannot be typed into. It is reported rather than hidden,
|
||||
* because a peer that has gone unreachable is exactly what the sender needs to know.
|
||||
*/
|
||||
private static Map<String, Object> leadView(String terminal, String name, Agent live,
|
||||
String selfTerm) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("sessionId", terminal);
|
||||
m.put("name", name);
|
||||
m.put("status", live == null || live.status() == null
|
||||
? "unknown" : live.status().name().toLowerCase());
|
||||
if (terminal.equals(selfTerm)) {
|
||||
m.put("self", true);
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
/** {@code bridge_stop}: tear a worker down by its pane id. */
|
||||
static McpSchema.CallToolResult stop(SessionManager sessions, String paneId) {
|
||||
if (isBlank(paneId)) {
|
||||
return error("paneId is required");
|
||||
}
|
||||
try {
|
||||
sessions.release(paneId);
|
||||
return text("stopped " + paneId);
|
||||
} catch (HerdrException e) {
|
||||
return error("herdr error stopping " + paneId + ": " + e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** CB-301 projection from the authoritative session registry. */
|
||||
private static Map<String, Object> memberView(MemberSession s) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("sessionId", s.terminalId());
|
||||
m.put("paneId", s.paneId());
|
||||
// Echo the role back so the caller can see what it actually got, not what it meant to ask
|
||||
// for — a spawn that silently fell back to dev is otherwise invisible.
|
||||
m.put("role", s.role() == null ? "dev" : s.role().wireName());
|
||||
m.put("profile", s.profile());
|
||||
m.put("status", s.state().name().toLowerCase());
|
||||
if (s.worktree() != null) {
|
||||
m.put("worktree", s.worktree());
|
||||
}
|
||||
if (s.branch() != null) {
|
||||
m.put("branch", s.branch());
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
private static String json(Object o) {
|
||||
try {
|
||||
return MAPPER.writeValueAsString(o);
|
||||
} catch (Exception e) {
|
||||
return String.valueOf(o);
|
||||
}
|
||||
}
|
||||
|
||||
// --- tool schemas --------------------------------------------------------------------------
|
||||
|
||||
private static McpSchema.Tool sendTool() {
|
||||
return tool("bridge_send",
|
||||
"Delegate a task to a worker session. By default blocks until the worker replies and "
|
||||
+ "returns its reply (or a 'still working / queued' note on timeout). Pass wait:false "
|
||||
+ "for a long task to return a ticket immediately, then poll it with bridge_poll. To "
|
||||
+ "answer a worker's bridge_ask, pass its turnId (with content) instead of sessionId.",
|
||||
objectSchema(Map.of(
|
||||
"sessionId", stringProp("The worker session id (herdr terminal_id) to delegate to"),
|
||||
"content", stringProp("The task/message to send to the worker (or your answer, with turnId)"),
|
||||
"timeoutMs", Map.of("type", "integer", "description", "Max ms to wait for a reply (blocking mode)"),
|
||||
"wait", Map.of("type", "boolean",
|
||||
"description", "Block for the reply (default true); false returns a ticket to poll"),
|
||||
"turnId", stringProp("When answering a worker's bridge_ask, its question turnId — "
|
||||
+ "routes your answer back into the same turn (omit for a normal delegation)")),
|
||||
List.of("content")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool askTool() {
|
||||
// No target/session arg — the worker's identity is resolved from the connection.
|
||||
return tool("bridge_ask",
|
||||
"Pause your current delegated turn to ask the primary a question, blocking until it "
|
||||
+ "answers — then resume the same turn with the answer. Use this when only the "
|
||||
+ "primary has a decision or detail you need to continue. You do not address the "
|
||||
+ "primary; identity is resolved from your connection.",
|
||||
objectSchema(Map.of(
|
||||
"question", stringProp("The question to put to the primary"),
|
||||
"timeoutMs", Map.of("type", "integer",
|
||||
"description", "Max ms to wait for the primary's answer")),
|
||||
List.of("question")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool pollTool() {
|
||||
return tool("bridge_poll",
|
||||
"Check an async delegation (a bridge_send with wait:false) by its ticket: "
|
||||
+ "pending, done (with the worker's reply), or failed. When target (a worker "
|
||||
+ "session id) is present instead of ticket, drain that worker's inbox of "
|
||||
+ "replies delivered when no send was open.",
|
||||
objectSchema(Map.of(
|
||||
"ticket", stringProp("The ticket returned by bridge_send wait:false"),
|
||||
"target", stringProp("Worker session id to drain pending replies from (optional)")),
|
||||
List.of()));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool ackTool() {
|
||||
return tool("bridge_ack",
|
||||
"Acknowledge (remove) a specific reply from a worker's inbox. Use when the primary "
|
||||
+ "has processed a reply and wants to confirm it, leaving other pending replies "
|
||||
+ "in the inbox for later drain.",
|
||||
objectSchema(Map.of(
|
||||
"target", stringProp("Worker session id whose inbox to ack from"),
|
||||
"msgId", stringProp("The message id to acknowledge")),
|
||||
List.of("target", "msgId")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool spawnTool() {
|
||||
return tool("bridge_spawn",
|
||||
"Spawn a new off-subscription member session. A member has two independent attributes: "
|
||||
+ "role (what it is for) and profile (which backend it runs on). Pass role to pick "
|
||||
+ "the contract — 'dev' implements a unit and opens its own PR, 'reviewer' reviews a "
|
||||
+ "diff it did not write, 'architect' refines a ticket before anyone builds it; omit "
|
||||
+ "it for 'dev'. Pass profile (from bridge_profiles) to pick the backend, or omit it "
|
||||
+ "for the default. The two are independent: a reviewer may run on the same profile "
|
||||
+ "as the dev it reviews. The member opens your current directory by default; pass "
|
||||
+ "cwd to pin a different one. Pass worktree:true (with ticket) or "
|
||||
+ "worktree:<ticket-slug> to provision an isolated git worktree. Returns the member's "
|
||||
+ "sessionId (use with bridge_send) and paneId (use with bridge_stop).",
|
||||
objectSchema(Map.of(
|
||||
"role", stringProp("What the member is for: architect, dev or reviewer (default dev)"),
|
||||
"profile", stringProp("Which backend to run it on (omit for the default profile)"),
|
||||
"cwd", stringProp("Working directory for the member (omit to inherit yours)"),
|
||||
"worktree", Map.of("type", "string", "description", "'true' or a ticket slug — requests an isolated git worktree"),
|
||||
"ticket", stringProp("Ticket slug when worktree:true")),
|
||||
List.of()));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool profilesTool() {
|
||||
return tool("bridge_profiles",
|
||||
"List the configured worker profiles (backends) and which one bridge_spawn uses by default.",
|
||||
objectSchema(Map.of(), List.of()));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool listTool() {
|
||||
return tool("bridge_list",
|
||||
"List the whole fleet the bridge tracks, in two parts. 'leads' are your PEERS — other "
|
||||
+ "orchestrators, each with its sessionId (the address to bridge_send to), "
|
||||
+ "name, live status, and 'self': true on your own row; this is how you "
|
||||
+ "discover a peer lead without being told its address. 'members' are the "
|
||||
+ "sessions delegated to — each with sessionId, paneId, role (architect/dev/"
|
||||
+ "reviewer), profile (the backend it runs on), state, optional "
|
||||
+ "worktree/branch/owner, and live herdr status. An empty 'members' "
|
||||
+ "means no members are spawned; it says nothing about peers.",
|
||||
objectSchema(Map.of(), List.of()));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool stopTool() {
|
||||
return tool("bridge_stop",
|
||||
"Tear down a worker session by its paneId (from bridge_spawn or bridge_list).",
|
||||
objectSchema(Map.of(
|
||||
"paneId", stringProp("The worker's paneId to stop")),
|
||||
List.of("paneId")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool replyTool() {
|
||||
// No session/target arg — the caller's identity is resolved from the connection.
|
||||
return tool("bridge_reply",
|
||||
"Return your structured answer for a message you were sent, resolving the sender's "
|
||||
+ "blocked bridge_send. A worker MUST end every delegated turn with exactly "
|
||||
+ "one of these. A lead uses it only to answer another lead that messaged "
|
||||
+ "it — never to answer a worker, whose turn it is not.",
|
||||
objectSchema(Map.of(
|
||||
"content", stringProp("Your reply/answer")),
|
||||
List.of("content")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool statusTool() {
|
||||
return tool("bridge_status",
|
||||
"Get the live lifecycle status (idle/working/blocked/unknown) of a worker session.",
|
||||
objectSchema(Map.of(
|
||||
"sessionId", stringProp("The worker session id to query")),
|
||||
List.of("sessionId")));
|
||||
}
|
||||
|
||||
private static McpSchema.Tool whoamiTool() {
|
||||
return tool("bridge_whoami",
|
||||
"Report who YOU are on the bridge — your role is resolved from your connection "
|
||||
+ "(unforgeable), never from anything you claim. Returns role 'primary' (you "
|
||||
+ "orchestrate: spawn/send/stop; reply ONLY to answer a peer lead that "
|
||||
+ "messaged you, never to answer a worker), 'architect' (you delegate turns "
|
||||
+ "and reply/ask as your own pane, but cannot spawn/stop/drain), or 'worker' "
|
||||
+ "(you were delegated to: you must end every turn with exactly one "
|
||||
+ "bridge_reply, and cannot spawn or send), plus 'leader'/'architect' naming "
|
||||
+ "which one you are, your own sessionId, and profile/worktree/branch when "
|
||||
+ "you are a worker. Call this first when following role-conditional "
|
||||
+ "instructions rather than guessing.",
|
||||
objectSchema(Map.of(), List.of()));
|
||||
}
|
||||
|
||||
// --- small helpers -------------------------------------------------------------------------
|
||||
|
||||
// The SDK 2.0.0 deprecates its own Tool builders without a stable replacement — isolate it here.
|
||||
@SuppressWarnings("deprecation")
|
||||
private static McpSchema.Tool tool(String name, String description, Map<String, Object> inputSchema) {
|
||||
return McpSchema.Tool.builder(name).description(description).inputSchema(inputSchema).build();
|
||||
}
|
||||
|
||||
private static Map<String, Object> objectSchema(Map<String, Object> properties, List<String> required) {
|
||||
return Map.of("type", "object", "properties", properties, "required", required);
|
||||
}
|
||||
|
||||
private static Map<String, Object> stringProp(String description) {
|
||||
return Map.of("type", "string", "description", description);
|
||||
}
|
||||
|
||||
private static McpSchema.CallToolResult text(String s) {
|
||||
return McpSchema.CallToolResult.builder().addTextContent(s == null ? "" : s).build();
|
||||
}
|
||||
|
||||
private static McpSchema.CallToolResult error(String s) {
|
||||
return McpSchema.CallToolResult.builder().addTextContent(s).isError(true).build();
|
||||
}
|
||||
|
||||
private static String str(Map<String, Object> args, String key) {
|
||||
Object v = args.get(key);
|
||||
return v == null ? null : v.toString();
|
||||
}
|
||||
|
||||
private static Long timeoutMs(Map<String, Object> args) {
|
||||
Object v = args.get("timeoutMs");
|
||||
return v instanceof Number n ? n.longValue() : null;
|
||||
}
|
||||
|
||||
private static long clamp(long ms) {
|
||||
return Math.clamp(ms, 1, MAX_TIMEOUT_MS);
|
||||
}
|
||||
|
||||
private static boolean isBlank(String s) {
|
||||
return s == null || s.isBlank();
|
||||
}
|
||||
}
|
||||
@@ -1,350 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>Claude Code</strong> — the safe path from a
|
||||
* delegation request to a running off-subscription Claude.
|
||||
*
|
||||
* <p>Everything transport-related (tab/pane placement, the CB-306 spawn-readiness gate, unique
|
||||
* naming, CB-117 orphan reap, teardown, listing, cwd resolution) lives in the base. This class
|
||||
* supplies only the two Claude-specific seams:
|
||||
* <ul>
|
||||
* <li>the {@code claude} name prefix (so reap matches {@code claude-*} panes, never another
|
||||
* adapter's), and</li>
|
||||
* <li>{@link #buildLaunch}, which encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em> touching
|
||||
* herdr, and mount the bridge MCP + reply charter as inline launch flags. A worker's base_url
|
||||
* lives in the env map handed to herdr and nowhere else; {@code bridged}'s own environment is
|
||||
* never mutated, and nothing is written to the worker's profile.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "claude";
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
* Standing instruction appended to the worker's system prompt so it returns its result via
|
||||
* {@code bridge_reply}. Injected as a launch flag, so nothing is written to the worker's
|
||||
* profile — it is guidance, and a worker that never replies is caught by the send's timeout.
|
||||
*/
|
||||
static final String REPLY_CHARTER =
|
||||
"You are an off-subscription worker in the claude-bridge fleet. Every message you "
|
||||
+ "receive arrives through the bridge, and the ONLY channel back to the sender is the "
|
||||
+ "bridge_reply MCP tool. Text you write in your terminal is NOT sent anywhere — the "
|
||||
+ "sender cannot see your screen, so an in-terminal answer is silently discarded. "
|
||||
+ "Therefore you MUST end EVERY turn by calling bridge_reply with `content` set to your "
|
||||
+ "complete response. This holds for every message without exception — tasks, questions, "
|
||||
+ "clarifications, acknowledgements, and ordinary back-and-forth conversation. Call "
|
||||
+ "bridge_reply exactly once, as the final action of your turn, with your full answer in "
|
||||
+ "`content`; never wait for confirmation first. If you end a turn without calling "
|
||||
+ "bridge_reply, the sender receives nothing and the exchange stalls.";
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so
|
||||
* existing deployments and tests keep the legacy non-blocking spawn semantics.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with spawn-ready gate enabled. The gate polls {@code agents.status()}
|
||||
* until the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
tabLabelTemplate);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply
|
||||
* fakes for the clock ({@code nowMillis}) and poll-loop wait ({@code sleeper}). The
|
||||
* {@code sleeper} is never called when the gate is disabled ({@code spawnReadyTimeoutMs == 0}).
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param guard subscription-boundary guard (checked before spawning)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (e.g. {@code () -> Thread.sleep(pollMs)}); it
|
||||
* already encodes the poll interval, so the 8th positional argument
|
||||
* (poll ms) is accepted for API symmetry but otherwise unused here
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param tabLabelTemplate {@code fleet.tabLabel}; {@code null}/blank ⇒
|
||||
* {@link BridgedConfig.Fleet#DEFAULT_TAB_LABEL}
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, tabLabelTemplate);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>A legacy spawn with no session identity is a fresh, launcher-derived session — delegate to
|
||||
* the session-aware form with no name and no resume id.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg) {
|
||||
return buildLaunch(cfg, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: assert the profile's base_url is on
|
||||
* the allowlist <em>before</em> any herdr call, then build the worker env with
|
||||
* {@code ANTHROPIC_*}, the parity-neutral git-forge grant, and the bridge MCP + reply charter
|
||||
* mounted as inline launch flags. When the request carries session identity (CB-547a) it is
|
||||
* applied here — see {@link #applySessionIdentity}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, String sessionName, String resumeSessionId) {
|
||||
// CB-539: a profile may deliberately opt into the subscription (subscription: true) when no
|
||||
// off-subscription endpoint exists for it — e.g. `sonnet` on `ccs`. That profile gets no
|
||||
// ANTHROPIC_BASE_URL/AUTH_TOKEN (there is nothing to point them at) and the guard's base_url
|
||||
// requirement is skipped FOR IT ONLY. Every other profile keeps the hard boundary below.
|
||||
boolean onSubscription = cfg.isSubscription();
|
||||
String baseUrl = cfg.baseUrl();
|
||||
|
||||
if (onSubscription) {
|
||||
// NO SILENT CONTRADICTION: subscription:true + a baseUrl state opposite intents; refuse
|
||||
// loudly rather than pick a winner.
|
||||
if (baseUrl != null && !baseUrl.isBlank()) {
|
||||
throw new IllegalStateException("profile '" + cfg.profile()
|
||||
+ "' sets both subscription: true and a baseUrl ('" + baseUrl + "') — the two "
|
||||
+ "are contradictory: a subscription profile must not point at an endpoint. "
|
||||
+ "Drop baseUrl, or drop subscription: true.");
|
||||
}
|
||||
// Visible without anyone going looking for it: this worker bills the subscription.
|
||||
log.warn("spawning profile '{}' on the Claude subscription (subscription: true) — this "
|
||||
+ "worker WILL bill the operator's subscription", cfg.profile());
|
||||
} else {
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
}
|
||||
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
if (onSubscription) {
|
||||
// CB-542 belt-and-braces: on the subscription path no guard vets these two keys, and the
|
||||
// profile's env: is layered in by baseEnv — so strip any that rode in there. Config load
|
||||
// already rejects this (loudly, naming the profile); this makes the boundary hold even
|
||||
// for a profile built in code that never passed through that validation.
|
||||
workerEnv.remove("ANTHROPIC_BASE_URL");
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so bridged chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
// handle is known BEFORE the agent has written anything; a resume spawn adopts its prior
|
||||
// id via -r and passes no --session-id (the two conflict). Both are injected before the
|
||||
// model flag so --model keeps outranking the operator's own argv.
|
||||
// mutableArgv: argvWithBridge may hand back the profile's own (immutable) List.of when it
|
||||
// has no MCP — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithBridge(cfg));
|
||||
String agentSessionId = applySessionIdentity(argv, sessionName, resumeSessionId);
|
||||
return new Launch(workerEnv, argvWithModel(argv, cfg), agentSessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
}
|
||||
if (resuming) {
|
||||
argv.add("-r");
|
||||
argv.add(resumeSessionId);
|
||||
return resumeSessionId;
|
||||
}
|
||||
String minted = UUID.randomUUID().toString();
|
||||
argv.add("--session-id");
|
||||
argv.add(minted);
|
||||
return minted;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus — when {@code worker.mcpUrl} is set — inline {@code --mcp-config} for
|
||||
* the bridge server and {@code --append-system-prompt} for the {@link #REPLY_CHARTER}. Neither
|
||||
* touches the profile's config; both are pure command-line flags. This inline-flag mount is
|
||||
* Claude Code specific — other adapters mount MCP and instructions their own way.
|
||||
*/
|
||||
private List<String> argvWithBridge(BridgedConfig.Profile cfg) {
|
||||
if (!cfg.hasMcp()) {
|
||||
return cfg.argv();
|
||||
}
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(REPLY_CHARTER);
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin the model on the command line as well as in {@code ANTHROPIC_MODEL} (CB-533).
|
||||
*
|
||||
* <p>The env var alone is not a reliable pin for this adapter, because the argv is usually a
|
||||
* launcher rather than {@code claude} itself — {@code ["ccs", "<profile>"]} — and {@code ccs}
|
||||
* exports its profile's own model family ({@code ANTHROPIC_MODEL}, {@code DEFAULT_OPUS/SONNET/
|
||||
* HAIKU}, {@code CLAUDE_CODE_SUBAGENT_MODEL}) over whatever it inherited. A worker profile that
|
||||
* set {@code model:} therefore got silently overruled by its own launcher. Claude Code's
|
||||
* {@code --model} flag outranks the environment, and {@code ccs <profile> [claude-args...]}
|
||||
* passes trailing arguments through, so the flag survives the wrapper.
|
||||
*
|
||||
* <p>Appended last so it also outranks anything in the operator's own {@code argv}. Profiles
|
||||
* that deliberately leave {@code model:} unset (letting {@code ccs} own model selection, as
|
||||
* {@code gx10} does) are untouched — this adds nothing when there is nothing to add. This is
|
||||
* the {@code kind: claude} counterpart of the opencode adapter's {@code -m provider/model}.
|
||||
*/
|
||||
private static List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() == null || cfg.model().isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withModel = mutableArgv(argv);
|
||||
withModel.add("--model");
|
||||
withModel.add(cfg.model());
|
||||
return withModel;
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.CONTEXT_RESET, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_NAME, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
String target = agentTarget(id);
|
||||
if (target == null) {
|
||||
return false;
|
||||
}
|
||||
// This deliberately bypasses Injector: /clear is housekeeping, not a delegated turn.
|
||||
agents().send(target, "/clear");
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (Claude prefix), kept for direct unit testing -------------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a Claude Code bridge worker started by a <em>different</em> process
|
||||
* than {@code currentNonce}. A thin {@code claude}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -1,422 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import dev.ltms.bridged.placement.PlacementCandidate;
|
||||
import dev.ltms.bridged.placement.PlacementContext;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.placement.PlacementPolicies;
|
||||
import dev.ltms.bridged.placement.PlacementPolicy;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link PeerLauncher} the core actually holds when more than one adapter is configured — a thin
|
||||
* router in front of one {@link HerdrPeerLauncher} per peer {@code kind} (Claude Code, opencode, …).
|
||||
* It owns no transport of its own; it dispatches each SPI call to the delegate that owns the profile
|
||||
* involved, and fans the fleet-wide queries (list/reap/caps/profiles) across all delegates.
|
||||
*
|
||||
* <p>Routing rules:
|
||||
* <ul>
|
||||
* <li><strong>By profile</strong> — {@link #spawn}, {@link #effectiveCwd}, {@link #parityOverlay}
|
||||
* resolve the profile (a null/blank name → the global {@link #defaultProfile}) and delegate to
|
||||
* the single adapter that declares it. Profiles partition cleanly across adapters: the
|
||||
* constructor rejects a name claimed by two.</li>
|
||||
* <li><strong>By pane id</strong> — {@link #stop} routes to the adapter that spawned that pane
|
||||
* (recorded at spawn time). A pane the composite never spawned (only real for a caller that
|
||||
* hand-rolls an id) falls back to the first delegate; teardown is pane-id addressed and
|
||||
* tab cleanup is single-occupant guarded, so it is safe either way.</li>
|
||||
* <li><strong>Fleet-wide</strong> — {@link #reapOrphanWorkers} and {@link #capabilities} fan out
|
||||
* and combine. {@link #list} is deduplicated by pane id because every herdr-backed delegate
|
||||
* shares one herdr connection and so reports the same global agent set.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>CB-518: an unqualified spawn is routed through a {@link PlacementPolicy}. The default
|
||||
* {@code fixed} policy reproduces the historical default-profile behaviour; {@code weighted} uses
|
||||
* smooth weighted round-robin with {@code maxLoad} gating. If a chosen profile fails with
|
||||
* {@link PeerUnreachableException}, the composite advances to the next available candidate and
|
||||
* retries, bounded by the number of candidates.
|
||||
*/
|
||||
public final class CompositePeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompositePeerLauncher.class);
|
||||
|
||||
private final List<HerdrPeerLauncher> delegates;
|
||||
private final Map<String, HerdrPeerLauncher> byProfile;
|
||||
private final String defaultProfile;
|
||||
|
||||
/** paneId → the delegate that spawned it, so {@link #stop} tears down through the right adapter. */
|
||||
private final Map<String, HerdrPeerLauncher> spawnedBy = new ConcurrentHashMap<>();
|
||||
|
||||
private final Function<String, Integer> liveCount;
|
||||
|
||||
/**
|
||||
* CB-559: the placement inputs are read <em>per spawn</em>, not captured at construction, so a
|
||||
* config reload changes where the next member lands without a restart. These are the hot keys —
|
||||
* role pools, an existing profile's weight/maxLoad, and the placement policy. What cannot change
|
||||
* this way is the set of adapters ({@link #byProfile}), because a new backend needs a launcher
|
||||
* and launchers are built once; {@code ConfigRef} classifies that as deferred and says so.
|
||||
*/
|
||||
private final Supplier<Map<String, BridgedConfig.Profile>> profileConfigs;
|
||||
private final Supplier<PlacementPolicy> placementPolicy;
|
||||
|
||||
/**
|
||||
* CB-557: the role pools an unqualified spawn draws its candidates from. A supplier that yields
|
||||
* {@code null}, and an empty pool for a role, both fall back to every configured profile — the
|
||||
* pre-CB-557 behaviour.
|
||||
*/
|
||||
private final Supplier<BridgedConfig.Fleet> fleet;
|
||||
|
||||
/**
|
||||
* Backward-compatible constructor: fixed placement, no live-counting. Use this for tests and
|
||||
* simple wiring; it preserves the pre-CB-518 behaviour exactly.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to (may be null)
|
||||
* @throws IllegalArgumentException if {@code delegates} is empty or two adapters claim one profile
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates, String defaultProfile) {
|
||||
this(delegates, defaultProfile, Map.of(), PlacementPolicies.fixed(), _ -> 0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with a placement policy and live-worker counter.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to under {@code fixed} policy
|
||||
* @param profileConfigs all configured worker profiles (used for candidate weights/caps)
|
||||
* @param placementPolicy which policy governs unqualified spawns
|
||||
* @param liveCount live worker count per profile (must never return {@code null})
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, BridgedConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools (CB-557). An unqualified spawn draws its candidates from
|
||||
* {@code fleet.<role>} instead of from every configured profile, so a reviewer is placed on a
|
||||
* reviewer backend and never on, say, the architect-only one.
|
||||
*
|
||||
* @param fleet the configured role pools; {@code null} ⇒ every profile is a candidate for every
|
||||
* role, which is the pre-CB-557 behaviour
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, BridgedConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
BridgedConfig.Fleet fleet) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), so a config
|
||||
* reload retargets the next member without a restart.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<BridgedConfig> config,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile,
|
||||
() -> config.get().profiles(),
|
||||
() -> PlacementPolicies.fromName(config.get().placement()),
|
||||
liveCount,
|
||||
() -> config.get().fleet());
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
private CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<Map<String, BridgedConfig.Profile>> profileConfigs,
|
||||
Supplier<PlacementPolicy> placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
if (delegates.isEmpty()) {
|
||||
throw new IllegalArgumentException("at least one peer adapter must be configured");
|
||||
}
|
||||
this.delegates = List.copyOf(delegates);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.profileConfigs = profileConfigs;
|
||||
this.placementPolicy = placementPolicy;
|
||||
this.liveCount = liveCount;
|
||||
Map<String, HerdrPeerLauncher> index = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : this.delegates) {
|
||||
for (String profile : d.profiles()) {
|
||||
HerdrPeerLauncher prev = index.putIfAbsent(profile, d);
|
||||
if (prev != null) {
|
||||
throw new IllegalArgumentException(
|
||||
"worker profile '" + profile + "' is claimed by two peer adapters");
|
||||
}
|
||||
}
|
||||
}
|
||||
// Order-preserving for the same reason, and because profiles() is user-visible (bridge_profiles).
|
||||
this.byProfile = Collections.unmodifiableMap(index);
|
||||
}
|
||||
|
||||
/** A supplier of a value fixed at construction — how the non-reloading constructors funnel in. */
|
||||
private static <T> Supplier<T> constant(T value) {
|
||||
return () -> value;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-configured profiles, never null.
|
||||
*
|
||||
* <p>Read fresh on every call so a reload is visible; a caller that needs two consistent reads
|
||||
* takes one local, as {@link #poolFor} does.
|
||||
*/
|
||||
private Map<String, BridgedConfig.Profile> profiles0() {
|
||||
Map<String, BridgedConfig.Profile> m = profileConfigs.get();
|
||||
return m == null ? Map.of() : m;
|
||||
}
|
||||
|
||||
/** The adapter owning {@code profileName} (null/blank → the default). Throws on an unknown profile. */
|
||||
private HerdrPeerLauncher route(String profileName) {
|
||||
String resolved = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (resolved == null) {
|
||||
// No profile and no default configured — hand to the first delegate so it raises the
|
||||
// same "no default" error it would on its own; keeps the SPI contract single-sourced.
|
||||
return delegates.getFirst();
|
||||
}
|
||||
HerdrPeerLauncher d = byProfile.get(resolved);
|
||||
if (d == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile: " + resolved);
|
||||
}
|
||||
return d;
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String requestedProfile = req.profileName();
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
// An explicit profile bypasses the placement policy, but not the capacity cap: maxLoad
|
||||
// is documented as an unconditional limit on this profile (BridgedConfig.Profile), and
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
}
|
||||
|
||||
// CB-557: an unqualified spawn is placed inside the pool of the role it asked for, not across
|
||||
// the whole profile list. An EXPLICIT profile (above) is left alone on purpose — it is the
|
||||
// operator overriding, and refusing it would break `bridge_spawn{profile:"opus"}`, which
|
||||
// carries no role and so would be judged against the dev pool it was never meant for.
|
||||
List<PlacementCandidate> candidates = candidates(req.role());
|
||||
String roleDefault = defaultProfileFor(req.role());
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
// Deliberately uncaught: when no candidate is left (all at cap, or all unreachable) the
|
||||
// policy already throws a clear message. Catching it to rethrow a generic
|
||||
// PeerUnreachableException would replace a precise diagnosis with a vague one.
|
||||
PlacementCandidate chosen = placementPolicy.get().select(ctx);
|
||||
|
||||
HerdrPeerLauncher d = byProfile.get(chosen.profile());
|
||||
if (d == null) {
|
||||
// A configured profile with no adapter is a wiring bug; fail fast.
|
||||
unreachable.add(chosen.profile());
|
||||
continue;
|
||||
}
|
||||
|
||||
// CB-547a: route the chosen profile but keep the caller's session identity — dropping it
|
||||
// here would silently sever the resume handle on every policy-routed spawn. CB-557: the
|
||||
// role rides along for the same reason, or a routed spawn would be labelled as a dev.
|
||||
SpawnRequest routedReq = new SpawnRequest(chosen.profile(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
try {
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
} catch (PeerUnreachableException e) {
|
||||
log.warn("spawn on profile {} unreachable, will retry next candidate if any: {}",
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable);
|
||||
}
|
||||
}
|
||||
|
||||
throw new PeerUnreachableException(
|
||||
"no reachable worker profile available after trying " + unreachable.size()
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
|
||||
*
|
||||
* <p>maxLoad is a documented, unconditional capacity limit (see {@code BridgedConfig.Profile#maxLoad}),
|
||||
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
|
||||
* ({@code PlacementPolicyUtil}, package-private, hence not linked) would leave the cap dead config
|
||||
* on every call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
|
||||
*
|
||||
* <p>Deliberately no fallback to another profile: the caller named {@code profile} for a cost/model
|
||||
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
|
||||
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
|
||||
*
|
||||
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
|
||||
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
|
||||
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
|
||||
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
|
||||
* the readiness gate and is a far worse trade.
|
||||
*
|
||||
* @param profile the profile the caller explicitly named
|
||||
* @throws PlacementException when the profile is at capacity
|
||||
*/
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (non-positive ⇒ unlimited at
|
||||
// load), means no cap — never cap what wasn't configured.
|
||||
BridgedConfig.Profile cfg = profiles0().get(profile);
|
||||
Integer cap = (cfg == null) ? null : cfg.maxLoad();
|
||||
if (cap == null) {
|
||||
return;
|
||||
}
|
||||
int live = liveCount.apply(profile);
|
||||
if (live >= cap) {
|
||||
throw new PlacementException("worker profile '" + profile + "' is at maxLoad: " + live
|
||||
+ " live >= " + cap + " cap; refusing spawn — no fallback to another profile");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile names {@code role} may be placed on, in definition order.
|
||||
*
|
||||
* <p>An empty or absent pool means "unconstrained", not "nothing allowed": a config that declares
|
||||
* no pool for a role must keep spawning, so it falls back to every configured profile. Names in a
|
||||
* pool that no adapter declares are dropped here rather than thrown — config load already rejects
|
||||
* a pool entry with no profile, so a survivor is a profile this particular composite does not own.
|
||||
*/
|
||||
private List<String> poolFor(MemberRole role) {
|
||||
Map<String, BridgedConfig.Profile> configured = profiles0();
|
||||
BridgedConfig.Fleet f = fleet.get();
|
||||
List<String> pool = (f == null) ? List.of() : f.profilesFor(role);
|
||||
List<String> known = pool.stream().filter(configured::containsKey).toList();
|
||||
return known.isEmpty() ? List.copyOf(configured.keySet()) : known;
|
||||
}
|
||||
|
||||
/** The profile an unqualified spawn for {@code role} falls back to under {@code fixed} placement. */
|
||||
private String defaultProfileFor(MemberRole role) {
|
||||
List<String> pool = poolFor(role);
|
||||
return pool.isEmpty() ? defaultProfile : pool.getFirst();
|
||||
}
|
||||
|
||||
/** Build the candidate list from {@code role}'s pool, in definition order. */
|
||||
private List<PlacementCandidate> candidates(MemberRole role) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (String name : poolFor(role)) {
|
||||
BridgedConfig.Profile w = profiles0().get(name);
|
||||
if (w != null) {
|
||||
out.add(new PlacementCandidate(name, null, w.weight(), w.maxLoad()));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return route(req.profileName()).effectiveCwd(req);
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return route(profileName).parityOverlay(profileName);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.remove(id);
|
||||
if (d == null) {
|
||||
log.debug("stop({}) — no recorded owner, routing to the first adapter (pane-addressed)", id);
|
||||
d = delegates.getFirst();
|
||||
}
|
||||
d.stop(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
HerdrPeerLauncher delegate = spawnedBy.get(id);
|
||||
if (delegate == null) {
|
||||
log.debug("clearContext({}) ignored — no recorded owning adapter", id);
|
||||
return false;
|
||||
}
|
||||
return delegate.clearContext(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return byProfile.keySet();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/** Every herdr agent, deduplicated by pane id (all delegates share one herdr and list globally). */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
Map<String, Agent> byPane = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
for (Agent a : d.list()) {
|
||||
if (a.paneId() != null) {
|
||||
byPane.putIfAbsent(a.paneId(), a);
|
||||
}
|
||||
}
|
||||
}
|
||||
return List.copyOf(byPane.values());
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
int reaped = 0;
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
reaped += d.reapOrphanWorkers();
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The union of every adapter's capabilities — a capability any adapter offers, the fleet offers. */
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
EnumSet<Capability> caps = EnumSet.noneOf(Capability.class);
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
caps.addAll(d.capabilities());
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
}
|
||||
@@ -1,738 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ConcurrentMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Abstract base for {@link PeerLauncher} adapters that materialize a peer as a <em>herdr</em>
|
||||
* agent (a CLI coding agent running in a herdr tab/pane). It owns everything that is the same
|
||||
* regardless of <em>which</em> coding agent runs: tab/pane placement, the CB-306 spawn-readiness
|
||||
* gate, unique naming, CB-117 orphan reap, teardown, {@link #list() listing}, and cwd resolution.
|
||||
*
|
||||
* <p>Two seams are peer-specific and supplied by the concrete adapter:
|
||||
* <ul>
|
||||
* <li>{@code namePrefix} (constructor arg) — the label prefix ({@code claude}, {@code opencode})
|
||||
* that drives both unique naming and the orphan-reap pattern, so each adapter reaps only its
|
||||
* own kind of pane and never another's.</li>
|
||||
* <li>{@link #buildLaunch(BridgedConfig.Profile)} — the peer-specific env map + argv, including any
|
||||
* subscription/guard check, MCP mount, and instruction injection. The base never sees how the
|
||||
* peer is configured; it only places and starts the returned {@link Launch}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Placement: in the default {@code tab} policy a peer lands in its own tab inside a dedicated
|
||||
* worker space (found-or-created once, then shared), so peers never split or clutter the user's
|
||||
* real work spaces. Teardown removes the peer's pane <em>and</em> its now-empty tab, tolerating an
|
||||
* already-gone peer so a repeated DELETE is harmless.
|
||||
*/
|
||||
public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
/**
|
||||
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
||||
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
||||
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
||||
*/
|
||||
private static final int SHELL_READY_RETRIES = 20;
|
||||
|
||||
private final String namePrefix; // label prefix: naming + reap scheme
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final Map<String, BridgedConfig.Profile> profiles; // profile name → spawn settings
|
||||
private final String defaultProfile; // profile a no-arg spawn uses (nullable)
|
||||
|
||||
/** Host env lookup (injectable for tests); adapters read it in {@link #buildLaunch}. */
|
||||
protected final Function<String, String> env;
|
||||
|
||||
private final AtomicLong nameSeq = new AtomicLong(); // per-peer counter (herdr agent names only)
|
||||
|
||||
/**
|
||||
* The {@code fleet.tabLabel} template; a {@code null} supplier or a {@code null}/blank value ⇒
|
||||
* {@link BridgedConfig.Fleet#DEFAULT_TAB_LABEL}. A profile's own {@code tabLabel} still
|
||||
* overrides it.
|
||||
*
|
||||
* <p>CB-559: a supplier rather than a String, so a config reload renames the <em>next</em> tab
|
||||
* without a restart. Existing tabs keep the label they were given — bridged does not rewrite a
|
||||
* label it already wrote.
|
||||
*/
|
||||
private final Supplier<String> tabLabelTemplate;
|
||||
|
||||
/**
|
||||
* Tab numbers, counted per {@code role/profile} pair (CB-557).
|
||||
*
|
||||
* <p>Deliberately not {@link #nameSeq}. That counter is shared by every profile this launcher
|
||||
* serves, because its job is to make herdr <em>agent names</em> unique. Reusing it for the tab
|
||||
* label made the numbers global, so sibling tabs read {@code #4}, {@code #9}, {@code #17} — gaps
|
||||
* that look like a member died. Counting per role+profile makes {@code dev: sonnet #2} mean the
|
||||
* second sonnet dev, which is what a reader assumes it means.
|
||||
*
|
||||
* <p>Resets when the daemon restarts, and that is fine: the label is a human-facing hint, not an
|
||||
* identity. Identity is {@link PeerHandle#id()}.
|
||||
*/
|
||||
private final ConcurrentMap<String, AtomicLong> labelSeq = new ConcurrentHashMap<>();
|
||||
|
||||
private final long spawnReadyTimeoutMs; // 0 = disable gate (legacy non-blocking spawn)
|
||||
private final LongSupplier nowMillis; // monotonic clock (injectable for tests)
|
||||
private final Runnable sleeper; // sleep/wait hook (injectable for tests; never real-sleep in unit tests)
|
||||
|
||||
// Per-process token mixed into each peer name so a fresh process (nameSeq back at 0) cannot
|
||||
// collide with same-profile peers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
|
||||
// CB-519: PeerHandle.id() is a host-unique opaque UUID, decoupled from the herdr pane id. The
|
||||
// routing/registry key is the UUID; the herdr pane id is a launcher-private placement/teardown
|
||||
// coordinate. This map bridges the two so stop(id) can resolve a host-unique key back to the
|
||||
// exact pane it must tear down. The pane id is launcher-private (never the routing key) — see
|
||||
// PeerHandle.id().
|
||||
private final ConcurrentMap<String, String> paneByAgentId = new ConcurrentHashMap<>();
|
||||
private final AtomicBoolean resetUnsupportedLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* @param namePrefix label prefix for this peer kind (drives naming and reap)
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured peer profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (never called when the gate is disabled); the poll
|
||||
* interval is baked into this hook, so the base needs no poll field
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(namePrefix, agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus the {@code fleet.tabLabel} template (CB-557).
|
||||
*
|
||||
* @param tabLabelTemplate fleet-wide tab-label template, read per spawn (CB-559); {@code null},
|
||||
* or a supplier yielding {@code null}/blank ⇒
|
||||
* {@link BridgedConfig.Fleet#DEFAULT_TAB_LABEL}. A separate constructor
|
||||
* rather than a new parameter on the one above, so every existing call
|
||||
* site keeps the default without an edit.
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
this.tabLabelTemplate = tabLabelTemplate;
|
||||
this.namePrefix = namePrefix;
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.profiles = Map.copyOf(profiles);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.env = env;
|
||||
this.spawnReadyTimeoutMs = spawnReadyTimeoutMs;
|
||||
this.nowMillis = nowMillis;
|
||||
this.sleeper = sleeper;
|
||||
}
|
||||
|
||||
// --- adapter seams -------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Build the peer-specific launch for {@code cfg}: the environment map and argv handed to herdr.
|
||||
* Any subscription/guard check, MCP mount, and instruction injection happen here. The env map
|
||||
* and argv are adapter-private; the base only places and starts what is returned.
|
||||
*/
|
||||
protected abstract Launch buildLaunch(BridgedConfig.Profile cfg);
|
||||
|
||||
/**
|
||||
* Session-aware variant of {@link #buildLaunch(BridgedConfig.Profile)} (CB-547a). Default
|
||||
* discards the session identity and delegates to the profile-only form, so an adapter that
|
||||
* carries no durable peer session (opencode, say) inherits byte-identical behaviour and needs
|
||||
* no change. An adapter that does (Claude Code) overrides this to mint/resume the id and to
|
||||
* surface it on the returned {@link Launch#agentSessionId()}.
|
||||
*
|
||||
* @param cfg the resolved profile to spawn
|
||||
* @param sessionName the bridge's logical session name, or null/blank for launcher-derived
|
||||
* @param resumeSessionId the peer's own prior session id to resume, or null/blank for fresh
|
||||
*/
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, String sessionName, String resumeSessionId) {
|
||||
return buildLaunch(cfg);
|
||||
}
|
||||
|
||||
/** Direct transport access for peer-specific, non-turn control operations. */
|
||||
protected final AgentControl agents() {
|
||||
return agents;
|
||||
}
|
||||
|
||||
/** Resolve the public peer id to the launcher's private herdr target. */
|
||||
protected final String agentTarget(String id) {
|
||||
return paneByAgentId.get(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
if (resetUnsupportedLogged.compareAndSet(false, true)) {
|
||||
log.warn("context reset is unsupported for peer kind {}; clearAfterTurn is a no-op",
|
||||
namePrefix);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* A peer-specific launch: the herdr {@code env} map and {@code argv}, plus — for an adapter
|
||||
* that carries durable session identity (CB-547a) — the peer's OWN session id
|
||||
* ({@link PeerHandle#agentSessionId()}), known before the peer has written anything. Null for
|
||||
* a launch that carries no identity.
|
||||
*/
|
||||
protected record Launch(Map<String, String> env, List<String> argv, String agentSessionId) {
|
||||
|
||||
/** A launch without a discoverable agent session id (an adapter that carries none). */
|
||||
Launch(Map<String, String> env, List<String> argv) {
|
||||
this(env, argv, null);
|
||||
}
|
||||
}
|
||||
|
||||
// --- profile surface -----------------------------------------------------------------------
|
||||
|
||||
/** The configured peer profile names (what {@code spawn(profile)} accepts). */
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return profiles.keySet();
|
||||
}
|
||||
|
||||
/** The parity-overlay file list for {@code profileName} (default list when unset). */
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
return cfg == null ? List.of() : cfg.parityOverlay();
|
||||
}
|
||||
|
||||
/** The profile a no-argument spawn uses, or {@code null} if none is configured. */
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/** The configured profiles, for adapter capability decisions (e.g. any git-token grant). */
|
||||
protected Collection<BridgedConfig.Profile> profileConfigs() {
|
||||
return profiles.values();
|
||||
}
|
||||
|
||||
/** Resolve {@code profileName} (null/blank → default) to its config, or throw with the options. */
|
||||
protected BridgedConfig.Profile requireProfile(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
throw new IllegalArgumentException("no default worker profile is configured — "
|
||||
+ "pass a profile; configured: " + profiles.keySet());
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
if (cfg == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile '" + name
|
||||
+ "' — configured: " + profiles.keySet());
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
// --- spawn ---------------------------------------------------------------------------------
|
||||
|
||||
/** A started peer plus the launch's agent-session id (the resume handle, or null). */
|
||||
private record Spawned(Agent agent, String agentSessionId) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer. {@code profileName} null/blank → the default profile. The working directory
|
||||
* (CB-112) is resolved by {@link #resolveCwd}: an explicit {@code requestedCwd}, else the
|
||||
* profile's configured {@code cwd}, else {@code callerCwd} (the primary's cwd, when the spawn
|
||||
* came from the primary over MCP), else the daemon's cwd — never assumed to be {@code $HOME}.
|
||||
* The adapter's {@link #buildLaunch} runs before any herdr call.
|
||||
*/
|
||||
protected Agent spawnInternal(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, null, null).agent();
|
||||
}
|
||||
|
||||
/** Pre-CB-557 shape: no explicit role, so the tab is labelled as a {@code dev}. */
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId,
|
||||
MemberRole.DEV);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer with session identity (CB-547a). {@code sessionName} and {@code resumeSessionId}
|
||||
* are threaded from the {@link SpawnRequest} into {@link #buildLaunch(BridgedConfig.Profile,
|
||||
* String, String)}, and the launch's resolved agent-session id is returned alongside the agent
|
||||
* so the caller can put it on the {@link PeerHandle}.
|
||||
*/
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId, MemberRole role) {
|
||||
BridgedConfig.Profile cfg = requireProfile(profileName);
|
||||
Launch launch = buildLaunch(cfg, sessionName, resumeSessionId);
|
||||
String cwd = resolveCwd(requestedCwd, cfg, callerCwd);
|
||||
Agent agent = cfg.tabPlacement()
|
||||
? spawnInTab(cfg, launch.env(), launch.argv(), cwd, role)
|
||||
: spawnAsPane(cfg, launch.env(), launch.argv(), cwd);
|
||||
return new Spawned(agent, launch.agentSessionId());
|
||||
}
|
||||
|
||||
/**
|
||||
* The next tab number for {@code role} on {@code profile}, starting at 1.
|
||||
*
|
||||
* <p>Starts at 1 rather than 0 because the number is read by a person: {@code "dev: sonnet #1"}
|
||||
* is the first one, and {@code #0} invites the question of where {@code #1} went.
|
||||
*/
|
||||
private long nextLabelSeq(MemberRole role, String profile) {
|
||||
String key = (role == null ? "" : role.wireName()) + "/" + profile;
|
||||
return labelSeq.computeIfAbsent(key, _ -> new AtomicLong()).incrementAndGet();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to {@link #spawnInternal} and wraps the resulting herdr {@link Agent} in a
|
||||
* {@link WorkerHandle} whose {@link PeerHandle#id()} is a fresh <em>host-unique</em> opaque
|
||||
* UUID (CB-519), deliberately decoupled from the herdr pane id: the id is the registry/routing
|
||||
* key and must never collide across daemon processes on the same host, while the herdr pane id
|
||||
* stays a launcher-private placement/teardown coordinate, remembered here so {@link #stop}
|
||||
* can resolve the host-unique key back to its pane. When {@code spawnReadyTimeoutMs > 0},
|
||||
* blocks until the peer's herdr status is injectable or the timeout elapses; on timeout the
|
||||
* pane is closed (no orphan) and a {@link PeerUnreachableException} is thrown.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
Spawned spawned = spawnInternal(req.profileName(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
Agent agent = spawned.agent();
|
||||
String paneId = agent.paneId();
|
||||
if (spawnReadyTimeoutMs > 0) {
|
||||
waitUntilInjectableOrThrow(paneId);
|
||||
}
|
||||
// CB-519: the handle id is a host-unique UUID; the herdr pane it maps to stays internal.
|
||||
String id = UUID.randomUUID().toString();
|
||||
paneByAgentId.put(id, paneId);
|
||||
return new WorkerHandle(id, agent.terminalId(), requireProfile(req.profileName()).profile(),
|
||||
req.sessionName(), spawned.agentSessionId());
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return effectiveCwd(req.profileName(), req.requestedCwd(), req.callerCwd());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-301: the effective working directory a spawn for {@code profileName} would use, without
|
||||
* actually spawning.
|
||||
*/
|
||||
private String effectiveCwd(String profileName, String requestedCwd, String callerCwd) {
|
||||
return resolveCwd(requestedCwd, requireProfile(profileName), callerCwd);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-112 cwd resolution: spawn arg → profile config → the primary's cwd → the daemon's cwd.
|
||||
* Never returns {@code null}/blank: {@code "."} (the daemon's own working directory) is the
|
||||
* guaranteed last resort so a pathological environment with an unset {@code user.dir} still
|
||||
* honours the "never assume {@code $HOME}" contract rather than letting herdr default the pane.
|
||||
*/
|
||||
private static String resolveCwd(String requestedCwd, BridgedConfig.Profile cfg, String callerCwd) {
|
||||
return firstNonBlank(requestedCwd, cfg.cwd(), callerCwd, System.getProperty("user.dir"), ".");
|
||||
}
|
||||
|
||||
private static String firstNonBlank(String... values) {
|
||||
for (String v : values) {
|
||||
if (v != null && !v.isBlank()) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Dedicated worker space → own tab (carrying cwd+env) → start the peer into the seed pane. */
|
||||
private Agent spawnInTab(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd, MemberRole role) {
|
||||
Workspace space = spaces.ensureWorkspace(cfg.workspace());
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId(), cwd, workerEnv);
|
||||
log.info("spawning {} profile={} space={} tab={} cwd={}",
|
||||
namePrefix, cfg.profile(), space.workspaceId(), tab.tab().tabId(), cwd);
|
||||
|
||||
Started started;
|
||||
try {
|
||||
if (tab.rootPaneId() == null) {
|
||||
// Protocol 19 starts the agent INTO the seed pane — without one there is nowhere
|
||||
// to start, and a partial tab would be left behind.
|
||||
throw new IllegalStateException("tab " + tab.tab().tabId()
|
||||
+ " had no seed pane in the create response — cannot start a peer in it");
|
||||
}
|
||||
started = startUniquelyNamed(cfg, argv, tab.rootPaneId());
|
||||
} catch (RuntimeException e) {
|
||||
// The peer never started — don't leave the tab we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned tab {} after spawn error: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
|
||||
// The peer is LIVE now, in the seed pane itself (no shell pane to drop — protocol 19).
|
||||
// Labelling is cosmetic: it must not fail the spawn or orphan the running peer — on error
|
||||
// we log and still return it so the caller gets its paneId and can tear it down.
|
||||
tidy("label tab " + tab.tab().tabId(),
|
||||
() -> spaces.renameTab(tab.tab().tabId(),
|
||||
cfg.renderTabLabel(
|
||||
tabLabelTemplate == null ? null : tabLabelTemplate.get(),
|
||||
role, nextLabelSeq(role, cfg.profile()))));
|
||||
log.info("{} started pane={} tab={} terminal={}",
|
||||
namePrefix, started.agent().paneId(), started.agent().tabId(), started.agent().terminalId());
|
||||
return started.agent();
|
||||
}
|
||||
|
||||
/** Run a best-effort post-start cleanup step, logging (not throwing) on failure. */
|
||||
private void tidy(String what, Runnable step) {
|
||||
try {
|
||||
step.run();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("post-start step failed ({}) — peer is running regardless: {}", what, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Legacy placement: split the currently-focused tab; the peer still starts in {@code cwd}. */
|
||||
private Agent spawnAsPane(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd) {
|
||||
log.info("spawning {} (pane placement) profile={} cwd={} argv={}",
|
||||
namePrefix, cfg.profile(), cwd, argv);
|
||||
String paneId = spaces.splitPane(cwd, workerEnv);
|
||||
if (paneId == null) {
|
||||
throw new IllegalStateException("pane.split returned no pane — cannot start a peer");
|
||||
}
|
||||
Agent peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
log.info("{} started pane={} terminal={}", namePrefix, peer.paneId(), peer.terminalId());
|
||||
return peer;
|
||||
}
|
||||
|
||||
/** A started peer together with the sequence its unique name/label used. */
|
||||
private record Started(Agent agent, long seq) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the peer under a unique herdr agent name. herdr requires each running agent's
|
||||
* {@code name} to be distinct (a 2nd identical {@code name} fails {@code agent_name_taken}) —
|
||||
* the exact case that makes multiple peers useful. The name is
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>}: {@code seq} distinguishes peers within this process,
|
||||
* and the per-process {@code nonce} keeps a fresh process (whose {@code seq} restarts at 0) from
|
||||
* colliding with same-profile peers that outlived a restart. The retry is a belt-and-braces
|
||||
* backstop for the astronomically unlikely nonce+seq clash; the name is a label only — herdr
|
||||
* detects kind and status from terminal output, not from it.
|
||||
*/
|
||||
private Started startUniquelyNamed(BridgedConfig.Profile cfg, List<String> argv, String paneId) {
|
||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(startAwaitingShellPrompt(name, args, paneId), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("peer name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
||||
HerdrException busy = null;
|
||||
for (int attempt = 0; attempt < SHELL_READY_RETRIES; attempt++) {
|
||||
try {
|
||||
return agents.start(name, namePrefix, args, paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_pane_busy".equals(e.code())) throw e;
|
||||
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
||||
busy = e;
|
||||
sleeper.run();
|
||||
}
|
||||
}
|
||||
throw busy;
|
||||
}
|
||||
|
||||
// --- discovery + reap ----------------------------------------------------------------------
|
||||
|
||||
/** All herdr-tracked agents — discovery for "what peers exist". */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
return agents.list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Reap peer panes left behind by an earlier daemon process (CB-117). herdr keeps a peer's pane
|
||||
* alive across a daemon restart <em>by design</em>, and that pane's id is held only by its
|
||||
* spawner — so a peer whose owning process exited before issuing the matching teardown leaks
|
||||
* with nothing tracking it. On boot we scan herdr for agents whose name matches our
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>} scheme with a nonce <em>other</em> than this
|
||||
* process's {@link #nameNonce}, and tear each one down (its pane and, via {@link #stop}, its
|
||||
* now-empty dedicated tab). A current-nonce peer is ours and live, so it is left running; a
|
||||
* user's own session carries no such name and is never touched. A peer from a <em>different</em>
|
||||
* adapter (different prefix) is likewise never touched. Best-effort: a failed listing, or a
|
||||
* failure to stop any one peer, is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
List<Agent> all;
|
||||
try {
|
||||
all = agents.list();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("orphan-peer reap skipped — agent.list failed: {}", e.getMessage());
|
||||
return 0;
|
||||
}
|
||||
int reaped = 0;
|
||||
for (Agent a : all) {
|
||||
if (!isForeignWorker(namePrefix, a.name(), nameNonce)) continue;
|
||||
try {
|
||||
stop(a.paneId());
|
||||
reaped++;
|
||||
log.info("reaped orphan {} {} (pane={} tab={}) left by a prior daemon",
|
||||
namePrefix, a.name(), a.paneId(), a.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("could not reap orphan {} {} (pane={}): {}",
|
||||
namePrefix, a.name(), a.paneId(), e.getMessage());
|
||||
}
|
||||
}
|
||||
if (reaped > 0) {
|
||||
log.info("orphan-peer reap complete — {} stale {} peer(s) removed at startup", reaped, namePrefix);
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The {@code <prefix>-<profile>-<nonce>-<seq>} name pattern; group 1 captures the 6-hex nonce. */
|
||||
static Pattern workerNamePattern(String prefix) {
|
||||
return Pattern.compile(prefix + "-.*-([0-9a-f]{6})-\\d+");
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a peer of kind {@code prefix} started by a <em>different</em> process
|
||||
* than {@code currentNonce} — the reap predicate (CB-117). True only for the prefix's naming
|
||||
* scheme with a foreign nonce: a non-peer name, a different adapter's name, or our own live
|
||||
* nonce is excluded. Pure and package-private so the decision is unit-testable without herdr.
|
||||
*/
|
||||
static boolean isForeignWorker(String prefix, String name, String currentNonce) {
|
||||
String nonce = workerNonce(prefix, name);
|
||||
return nonce != null && !nonce.equals(currentNonce);
|
||||
}
|
||||
|
||||
/** The 6-hex nonce embedded in a {@code prefix} peer name, or {@code null} if not one. */
|
||||
static String workerNonce(String prefix, String name) {
|
||||
if (name == null) return null;
|
||||
Matcher m = workerNamePattern(prefix).matcher(name);
|
||||
return m.matches() ? m.group(1) : null;
|
||||
}
|
||||
|
||||
/** This process's peer-name nonce (a label component only; exposed for reaper tests). */
|
||||
String nameNonce() {
|
||||
return nameNonce;
|
||||
}
|
||||
|
||||
// --- teardown ------------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Tear a peer down: close the pane, and close its tab <em>only</em> when the peer is that tab's
|
||||
* sole occupant. The single-pane check is what makes this safe regardless of how the peer was
|
||||
* placed (or a placement-config change across a restart): a pane-placement peer sitting in one
|
||||
* of the user's shared tabs has siblings, so its tab is never closed — we only ever remove a
|
||||
* tab we created to hold one peer.
|
||||
*
|
||||
* <p>{@code idOrPane} is the {@link PeerHandle#id()} of a peer this launcher spawned (CB-519's
|
||||
* host-unique opaque UUID), resolved through {@link #paneByAgentId} to the pane it must tear
|
||||
* down. An argument that is not one of our ids is treated as a raw herdr pane id — the
|
||||
* {@link #reapOrphanWorkers() orphan-reap} and spawn-gate-timeout paths, plus any caller that
|
||||
* passes a pane directly, keep working without an owning id.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed peer) is treated as success; any other failure propagates so a
|
||||
* genuinely failed teardown is not reported as done.
|
||||
*/
|
||||
@Override
|
||||
public void stop(String idOrPane) {
|
||||
// Teardown knows only the pane, not which profile spawned it. Attempt tab cleanup when any
|
||||
// profile uses tab placement (so the bridge may have created a dedicated peer tab); the
|
||||
// single-occupant check below is what actually protects the user's shared tabs.
|
||||
String paneId = paneByAgentId.remove(idOrPane);
|
||||
if (paneId == null) {
|
||||
paneId = idOrPane; // raw-pane fallback (reap, gate timeout, pane-addressed callers)
|
||||
}
|
||||
WorkspaceControl.PaneLocation loc = usesTabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) throw e;
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated peer tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether any configured profile places peers in their own tab (so tabs may need cleanup). */
|
||||
private boolean usesTabPlacement() {
|
||||
return profiles.values().stream().anyMatch(BridgedConfig.Profile::tabPlacement);
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
// --- spawn-readiness gate (CB-306) ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until the pane reports an injectable state or the configured
|
||||
* timeout elapses. On timeout, close the pane (self-reap) and throw.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long deadline = nowMillis.getAsLong() + spawnReadyTimeoutMs;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (agents.status(paneId).injectable()) {
|
||||
log.debug("peer pane={} reached injectable state", paneId);
|
||||
return;
|
||||
}
|
||||
sleeper.run();
|
||||
}
|
||||
log.warn("peer pane={} did not become injectable within {}ms — closing", paneId, spawnReadyTimeoutMs);
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " did not reach injectable state within "
|
||||
+ spawnReadyTimeoutMs + "ms");
|
||||
}
|
||||
|
||||
/**
|
||||
* A concrete {@link PeerHandle} wrapping herdr agent coordinates, the profile that spawned it,
|
||||
* and the session identity the launch resolved (CB-547a): the bridge's logical name and the
|
||||
* peer's own session id, both null when the spawn carried no identity.
|
||||
*/
|
||||
private record WorkerHandle(String id, String terminalId, String profile,
|
||||
String sessionName, String agentSessionId) implements PeerHandle {
|
||||
}
|
||||
|
||||
// --- shared helpers ------------------------------------------------------------------------
|
||||
|
||||
/** Put {@code k → v} only when {@code v} is present (non-null, non-blank). */
|
||||
protected static void putIfPresent(Map<String, String> m, String k, String v) {
|
||||
if (v != null && !v.isBlank()) {
|
||||
m.put(k, v);
|
||||
}
|
||||
}
|
||||
|
||||
/** Host env lookup that tolerates an unconfigured (null/blank) var name — returns null then. */
|
||||
protected String resolveEnv(String name) {
|
||||
return (name == null || name.isBlank()) ? null : env.apply(name);
|
||||
}
|
||||
|
||||
/**
|
||||
* The parity-neutral git-forge token grant (CB-302): when {@code cfg} opts in via
|
||||
* {@code gitTokenEnv} and the token resolves, inject {@code GITEA_TOKEN} plus its paired
|
||||
* {@code GITEA_HOST}. Push over SSH is unaffected; the only incremental grant is PR-create.
|
||||
* Peer-neutral, so every herdr adapter reuses it unchanged.
|
||||
*/
|
||||
protected void applyGitToken(Map<String, String> workerEnv, BridgedConfig.Profile cfg) {
|
||||
if (!cfg.hasGitToken()) {
|
||||
return;
|
||||
}
|
||||
String gitToken = resolveEnv(cfg.gitTokenEnv());
|
||||
if (gitToken != null) {
|
||||
workerEnv.put("GITEA_TOKEN", gitToken);
|
||||
putIfPresent(workerEnv, "GITEA_HOST", resolveEnv(cfg.gitHostEnv()));
|
||||
}
|
||||
}
|
||||
|
||||
/** A fresh mutable env map — the conventional starting point for {@link #buildLaunch}. */
|
||||
/**
|
||||
* Seed a worker's environment (CB-511): the daemon's own {@code PATH}, then the profile's
|
||||
* {@code env:} entries.
|
||||
*
|
||||
* <p>Why this exists: bridged passes herdr an explicit env map, and herdr merges it into
|
||||
* <em>its own</em> process environment. So before this, a worker inherited whatever PATH the
|
||||
* herdr server happened to be started with — on this host, one from weeks earlier with no JDK
|
||||
* and no Maven, which left workers unable to run the build they were being asked to run. The
|
||||
* worker's toolchain must follow from configuration, not from how a long-lived daemon was
|
||||
* launched.
|
||||
*
|
||||
* <p>Adapter-specific variables are layered on top of this by {@code buildLaunch} and therefore
|
||||
* win. That ordering is deliberate and load-bearing: it stops a profile's {@code env:} from
|
||||
* overriding {@code ANTHROPIC_BASE_URL} and slipping past {@link
|
||||
* dev.ltms.bridged.guard.SubscriptionGuard}, which is checked against the profile's
|
||||
* {@code baseUrl} and nothing else.
|
||||
*/
|
||||
protected Map<String, String> baseEnv(BridgedConfig.Profile cfg) {
|
||||
Map<String, String> workerEnv = new LinkedHashMap<>();
|
||||
String path = env.apply("PATH");
|
||||
if (path != null && !path.isBlank()) {
|
||||
workerEnv.put("PATH", path);
|
||||
}
|
||||
if (cfg != null && cfg.env() != null) {
|
||||
workerEnv.putAll(cfg.env());
|
||||
}
|
||||
return workerEnv;
|
||||
}
|
||||
|
||||
/** Defensive copy of {@code argv} plus room to append launch flags. */
|
||||
protected static List<String> mutableArgv(List<String> argv) {
|
||||
return new ArrayList<>(argv);
|
||||
}
|
||||
|
||||
/**
|
||||
* Uninterruptible sleep — the production {@link #sleeper}. Tests supply their own no-op /
|
||||
* fast-faking sleeper so they never real-sleep.
|
||||
*/
|
||||
protected static void sleepUninterruptibly(long ms) {
|
||||
try {
|
||||
Thread.sleep(ms);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
// preserve the interrupt flag but continue — poll loops should not be aborted by an
|
||||
// interrupt that was not meant for them.
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,507 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>opencode</strong> — an open-source,
|
||||
* provider-agnostic terminal coding agent. Its whole reason for existing is to prove the
|
||||
* {@code PeerLauncher} SPI is genuinely provider-neutral: opencode shares none of Claude Code's
|
||||
* private launch seams, yet reuses every line of shared transport in the base (tab/pane placement,
|
||||
* the CB-306 readiness gate, unique naming + CB-117 reap, teardown, listing, cwd).
|
||||
*
|
||||
* <p>The divergences from {@link ClaudeCodeLauncher}, all confined to {@link #buildLaunch}:
|
||||
* <ul>
|
||||
* <li><strong>No subscription boundary.</strong> opencode carries no {@code ANTHROPIC_BASE_URL}
|
||||
* and there is no {@link dev.ltms.bridged.guard.SubscriptionGuard} — the guard is a
|
||||
* Claude-private concern, not part of the SPI. opencode reads the operator's own provider
|
||||
* credentials from its global {@code auth.json}; the bridge injects none.</li>
|
||||
* <li><strong>File-based MCP mount + instructions.</strong> opencode has no inline
|
||||
* {@code --mcp-config}/{@code --append-system-prompt}. Instead the bridge writes an ephemeral
|
||||
* {@code opencode.json} that declares the bridge as a {@code remote} MCP server and lists a
|
||||
* reply-charter file under {@code instructions}, then points the worker at it with
|
||||
* {@code OPENCODE_CONFIG}. This is the one place the launcher touches disk — Claude never did.</li>
|
||||
* <li><strong>Model as a flag.</strong> the {@code provider/model} selector is passed as
|
||||
* {@code -m}, not an env var.</li>
|
||||
* <li><strong>{@code opencode} name prefix</strong> so reap matches {@code opencode-*} panes and
|
||||
* never another adapter's.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "opencode";
|
||||
|
||||
/** Writer for the generated {@code opencode.json}. */
|
||||
private static final ObjectMapper JSON = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* Standing instruction written to the charter file and mounted via the config's
|
||||
* {@code instructions} so the worker returns its result through {@code bridge_reply}. Kept on
|
||||
* disk (not a launch flag) because opencode's {@code instructions} takes file paths, not inline
|
||||
* text — the file is regenerated per spawn and never touches the worker's own profile.
|
||||
*/
|
||||
static final String REPLY_CHARTER =
|
||||
"You are an off-subscription worker in the claude-bridge fleet, running under opencode. "
|
||||
+ "Every message you receive arrives through the bridge, and the ONLY channel back to the "
|
||||
+ "sender is the bridge_reply MCP tool. Text you write in your terminal is NOT sent "
|
||||
+ "anywhere — the sender cannot see your screen, so an in-terminal answer is silently "
|
||||
+ "discarded. Therefore you MUST end EVERY turn by calling bridge_reply with `content` set "
|
||||
+ "to your complete response. This holds for every message without exception — tasks, "
|
||||
+ "questions, clarifications, acknowledgements, and ordinary back-and-forth conversation. "
|
||||
+ "Call bridge_reply exactly once, as the final action of your turn, with your full answer "
|
||||
+ "in `content`; never wait for confirmation first. If you end a turn without calling "
|
||||
+ "bridge_reply, the sender receives nothing and the exchange stalls.";
|
||||
|
||||
/** Root under which per-spawn opencode config dirs are created (injectable for tests). */
|
||||
private final Path configRoot;
|
||||
|
||||
/**
|
||||
* The current spawn's resume-target session id, threaded from {@link #spawn(SpawnRequest)} to
|
||||
* {@link #buildLaunch} across the base's {@code spawn -> spawnInternal -> buildLaunch} chain,
|
||||
* which carries no request. A plain field would race under concurrent spawns (the base supports
|
||||
* them), so it is thread-local: each spawn captures its own request's id on its own thread, and
|
||||
* {@code buildLaunch}, synchronous and same-thread, reads exactly that one. Set only around the
|
||||
* {@code super.spawn} call and cleared in {@code finally}, so a paused/leftover value can never
|
||||
* bleed into the next spawn.
|
||||
*/
|
||||
private final ThreadLocal<String> resumeSessionId = new ThreadLocal<>();
|
||||
|
||||
/**
|
||||
* Session discovery against opencode's on-disk storage ({@link OpenCodeSessionDiscovery}) —
|
||||
* the one seam that knows opencode's private session-file layout. Its root is injectable for
|
||||
* tests so they never touch the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with the spawn-ready gate enabled. Polls {@code agents.status()} until
|
||||
* the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), tabLabelTemplate);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply a
|
||||
* fake clock ({@code nowMillis}), poll-loop wait ({@code sleeper}), and a temp {@code configRoot}
|
||||
* they can inspect the generated {@code opencode.json}/charter under.
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (encodes the poll interval; never called when the
|
||||
* gate is disabled)
|
||||
* @param configRoot existing directory under which per-spawn config dirs are created
|
||||
* @param discoveryRoot opencode's on-disk storage root to scan for session records
|
||||
* (injectable for tests; opencode's layout is matched at
|
||||
* {@link OpenCodeSessionDiscovery})
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, configRoot, discoveryRoot, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param tabLabelTemplate {@code fleet.tabLabel}; {@code null}/blank ⇒
|
||||
* {@link BridgedConfig.Fleet#DEFAULT_TAB_LABEL}
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<String> tabLabelTemplate) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, tabLabelTemplate);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Builds the opencode launch: no {@code ANTHROPIC_*} and no guard (opencode reads its own
|
||||
* provider credentials); when the profile mounts the bridge MCP, generate an ephemeral
|
||||
* {@code opencode.json} (remote MCP server + reply-charter instructions) and point the worker at
|
||||
* it via {@code OPENCODE_CONFIG}; carry the parity-neutral git-forge grant; and select the model
|
||||
* with {@code -m}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg) {
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
// A config file is needed for the bridge MCP mount, for a pinned endpoint (CB-508), or both.
|
||||
if (cfg.hasMcp() || hasCustomProvider(cfg)) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
return new Launch(workerEnv, argvWithResume(argvWithModel(argvWithAuto(cfg), cfg)));
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
*
|
||||
* <p>Note this reuses {@code baseUrl}, the same field the Claude adapter injects as
|
||||
* {@code ANTHROPIC_BASE_URL} — but it does <em>not</em> go through {@code SubscriptionGuard}.
|
||||
* That asymmetry is deliberate and safe: the guard exists to stop a worker borrowing the
|
||||
* primary's Anthropic subscription, and an opencode process has no Anthropic credential path
|
||||
* at all. Pointing it at a local vLLM cannot leak the subscription.
|
||||
*/
|
||||
private static boolean hasCustomProvider(BridgedConfig.Profile cfg) {
|
||||
return cfg.baseUrl() != null && !cfg.baseUrl().isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus the unconditional {@code --auto} flag, which auto-approves the
|
||||
* permissions opencode does not explicitly deny. It is unconditional, not a preference: a
|
||||
* spawned peer has no human at its pane — the bridge spawned it — so one that stops at an
|
||||
* approval prompt is a wedged agent, indistinguishable from a legitimate mid-turn wait and
|
||||
* unable to end its turn with {@code bridge_reply}. opencode's own help calls this
|
||||
* "dangerous!", but the blast radius here is already bounded by design: a worker runs in its
|
||||
* own git worktree on its own branch, is off-subscription, and cannot merge — the lead is the
|
||||
* gate.
|
||||
*/
|
||||
private List<String> argvWithAuto(BridgedConfig.Profile cfg) {
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--auto");
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus, on a resumed spawn, opencode's {@code -s <id>} flag to continue a prior
|
||||
* conversation by its session id. {@code -s, --session <id>} resumes an existing session; on a
|
||||
* fresh spawn (no resume target) no flag is added, letting opencode start a brand-new session.
|
||||
* The id comes from the current spawn request's {@code resumeSessionId}, threaded per-thread by
|
||||
* {@link #spawn(SpawnRequest)}.
|
||||
*/
|
||||
private List<String> argvWithResume(List<String> argv) {
|
||||
String id = resumeSessionId.get();
|
||||
if (id == null || id.isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withResume = mutableArgv(argv);
|
||||
withResume.add("-s");
|
||||
withResume.add(id);
|
||||
return withResume;
|
||||
}
|
||||
|
||||
/** The launch argv plus, when a model is configured, the opencode {@code -m provider/model} flag. */
|
||||
private List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()) {
|
||||
argv.add("-m");
|
||||
argv.add(cfg.model());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write an ephemeral {@code opencode.json} (and the reply-charter file it references) into a
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*/
|
||||
private Path writeConfig(BridgedConfig.Profile cfg) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "bridged-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
root.put("$schema", "https://opencode.ai/config.json");
|
||||
// CB-523, opencode side: a worker that runs out of context dies mid-turn, and its reply
|
||||
// — the entire point of the turn — is lost with it. Auto-compaction is therefore not an
|
||||
// operator preference for a bridged worker, it is a condition of the turn contract.
|
||||
//
|
||||
// Stated deliberately even though it is redundant today: OPENCODE_CONFIG is MERGED over
|
||||
// ~/.config/opencode/config.json rather than replacing it, so a worker already inherits
|
||||
// an `auto: true` set at home. We do not want that inheritance to be what the guarantee
|
||||
// rests on — the home file is outside this repo, differs per machine, and is not ours.
|
||||
//
|
||||
// Know the cost before removing it: this key WINS over the home config (verified — an
|
||||
// OPENCODE_CONFIG value overrides the home value, it does not defer to it), so an
|
||||
// operator who sets `compaction.auto: false` at home cannot turn it off for bridged
|
||||
// workers. That is the intended trade for peers we spawn and whose turns we must land;
|
||||
// if per-profile control is ever wanted, add a profile knob rather than dropping this.
|
||||
root.putObject("compaction").put("auto", true);
|
||||
|
||||
if (cfg.hasMcp()) {
|
||||
Path charter = dir.resolve("reply-charter.md");
|
||||
Files.writeString(charter, REPLY_CHARTER);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode bridge = root.putObject("mcp").putObject("bridge");
|
||||
bridge.put("type", "remote");
|
||||
bridge.put("url", cfg.mcpUrl());
|
||||
bridge.put("enabled", true);
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
if (hasCustomProvider(cfg)) {
|
||||
addCustomProvider(root, cfg);
|
||||
}
|
||||
|
||||
Path cfgFile = dir.resolve("opencode.json");
|
||||
// Built with Jackson rather than string concatenation: the provider block is nested and
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
"cannot write opencode config for profile " + cfg.profile(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
*
|
||||
* <p>The provider id comes from the {@code provider/model} selector in {@code model:}, so one
|
||||
* field drives both the declaration and the {@code -m} flag and they cannot drift apart.
|
||||
*/
|
||||
private void addCustomProvider(ObjectNode root, BridgedConfig.Profile cfg) {
|
||||
String[] parts = splitModelSelector(cfg);
|
||||
String providerId = parts[0];
|
||||
String modelId = parts[1];
|
||||
|
||||
ObjectNode provider = root.putObject("provider").putObject(providerId);
|
||||
provider.put("npm", "@ai-sdk/openai-compatible");
|
||||
provider.put("name", providerId + " (bridged)");
|
||||
|
||||
ObjectNode options = provider.putObject("options");
|
||||
options.put("baseURL", openAiBaseUrl(cfg.baseUrl()));
|
||||
// vLLM and friends usually ignore the key, but the AI SDK still requires a non-empty one.
|
||||
String token = resolveEnv(cfg.tokenEnv());
|
||||
options.put("apiKey", (token == null || token.isBlank()) ? "bridged-local-noauth" : token);
|
||||
|
||||
provider.putObject("models").putObject(modelId).put("name", modelId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Split {@code model:} into its {@code provider} and {@code model} halves. A pinned endpoint
|
||||
* needs both, so a bare model name is rejected loudly rather than silently falling back to the
|
||||
* default gateway — a worker quietly talking to the wrong endpoint is the failure this avoids.
|
||||
*/
|
||||
private static String[] splitModelSelector(BridgedConfig.Profile cfg) {
|
||||
String model = cfg.model();
|
||||
int slash = model == null ? -1 : model.indexOf('/');
|
||||
if (model == null || model.isBlank() || slash <= 0 || slash == model.length() - 1) {
|
||||
throw new IllegalArgumentException(
|
||||
"profile " + cfg.profile() + " sets baseUrl (a pinned opencode endpoint) so"
|
||||
+ " model: must be \"<provider>/<model>\", e.g."
|
||||
+ " \"local-vllm/deepseek-v4-flash\"; got "
|
||||
+ (model == null ? "null" : '"' + model + '"'));
|
||||
}
|
||||
return new String[]{model.substring(0, slash), model.substring(slash + 1)};
|
||||
}
|
||||
|
||||
/**
|
||||
* The OpenAI-compatible base URL for {@code baseUrl}. A bare {@code host:port} gets {@code /v1}
|
||||
* appended (where these servers put the API); a URL that already carries a path is taken as-is,
|
||||
* so an endpoint mounted somewhere unusual is still reachable.
|
||||
*/
|
||||
private static String openAiBaseUrl(String baseUrl) {
|
||||
String trimmed = baseUrl.trim();
|
||||
while (trimmed.endsWith("/")) {
|
||||
trimmed = trimmed.substring(0, trimmed.length() - 1);
|
||||
}
|
||||
int schemeEnd = trimmed.indexOf("://");
|
||||
String afterScheme = schemeEnd < 0 ? trimmed : trimmed.substring(schemeEnd + 3);
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>adds this adapter's session-identity work around the base's spawn — as opencode cannot be
|
||||
* told its session id at spawn (see {@link Capability#SESSION_RESUME} vs
|
||||
* {@link Capability#SESSION_NAME}), identity is only ever adopted after the fact:
|
||||
* <ul>
|
||||
* <li>the request's {@code resumeSessionId} is remembered for {@link #buildLaunch} to turn
|
||||
* into {@code -s <id>}; and</li>
|
||||
* <li>the returned handle is wrapped so its
|
||||
* {@link dev.ltms.bridged.peer.PeerHandle#agentSessionId()} performs lazy session
|
||||
* discovery against opencode's storage (see {@link OpenCodeSessionDiscovery}) — always
|
||||
* non-blocking, {@code null} until opencode has persisted the session record.</li>
|
||||
* </ul>
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
resumeSessionId.set(req.resumeSessionId());
|
||||
try {
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
} finally {
|
||||
// Never let a paused/leftover resume id bleed into the next spawn on this thread.
|
||||
resumeSessionId.remove();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A {@link PeerHandle} that delegates everything to the base's worker handle but resolves
|
||||
* {@link #agentSessionId()} lazily through opencode session discovery. Delegate-only, so the
|
||||
* base's id/terminalId/profile semantics (CB-519's host-unique routing key, herdr coordinates)
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String id() {
|
||||
return delegate.id();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String terminalId() {
|
||||
return delegate.terminalId();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String profile() {
|
||||
return delegate.profile();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String sessionName() {
|
||||
return delegate.sessionName();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
}
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.ORPHAN_REAP, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
// Deliberately NOT SESSION_NAME: opencode has no display-name flag, so the bridge's logical
|
||||
// name can't surface in the peer's own UI — declaring the capability would hide that
|
||||
// asymmetry rather than make it honest. For opencode the name lives only in the bridge's
|
||||
// roster (see PeerHandle.sessionName() returning null).
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (opencode prefix), kept for direct unit testing -----------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is an opencode bridge worker started by a <em>different</em> process than
|
||||
* {@code currentNonce}. A thin {@code opencode}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -1,122 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a bridged worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every bridged worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A bridged worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
*/
|
||||
String sessionIdForDirectory(String directory) {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
} catch (IOException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,242 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import com.rabbitmq.client.AMQP;
|
||||
import com.rabbitmq.client.Channel;
|
||||
import com.rabbitmq.client.Connection;
|
||||
import com.rabbitmq.client.ConnectionFactory;
|
||||
import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
* port {@link InMemoryReplyInbox} implements as soft state.
|
||||
*
|
||||
* <p><strong>Mapping — consume-and-hold with deferred manual ack.</strong> Each target has a durable
|
||||
* queue {@code agent.<target>.inbox}. The gateway that owns the target starts a manual-ack consumer
|
||||
* ({@link #own}) that pulls persistent messages off that queue into an in-memory <em>held</em> map
|
||||
* (keyed by {@code msgId}) but does <em>not</em> ack them. {@link #peek} returns that snapshot;
|
||||
* {@link #ack} acks the broker delivery-tag and drops the entry. Because messages stay unacked until
|
||||
* the owning gateway actually drains them, a crash (or a {@code java -jar} bounce) before caller-ack
|
||||
* leaves them on the broker — it redelivers on reconnect. That is the durability the in-memory
|
||||
* adapter cannot give, with the port contract preserved.
|
||||
*
|
||||
* <p><strong>Ownership is explicit.</strong> {@link #own} declares the queue and starts the consumer;
|
||||
* {@link #release} cancels it. {@link #publish} sends to the queue but does <em>not</em> imply ownership
|
||||
* and does not attach a consumer. This split is required by CB-308 federation, where one gateway may
|
||||
* publish to an agent owned by another gateway; in that case the publisher must not compete for
|
||||
* deliveries.
|
||||
*
|
||||
* <p><strong>Dedup.</strong> The consumer keys the held map by {@code msgId}; a redelivered duplicate
|
||||
* (at-least-once, or a producer double-publish) is acked-and-dropped on arrival, so it never
|
||||
* double-queues.
|
||||
*
|
||||
* <p><strong>Visibility.</strong> Unlike the in-memory adapter, publish → broker → consumer is
|
||||
* asynchronous, so a {@link #peek} immediately after {@link #publish} may not yet see the message
|
||||
* (broker delivery latency). Callers that need the reply drained poll (as the primary already does);
|
||||
* the contract test waits for visibility. This is inherent to broker-backed delivery, not a defect.
|
||||
*
|
||||
* <p>The default deploy targets LavinMQ; a stock RabbitMQ speaks the same AMQP 0-9-1 (URI-only swap),
|
||||
* so the {@code @Tag("contract")} integration test runs against a RabbitMQ container.
|
||||
*/
|
||||
public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(AmqpReplyInbox.class);
|
||||
|
||||
private static final String QUEUE_PREFIX = "agent.";
|
||||
private static final String QUEUE_SUFFIX = ".inbox";
|
||||
|
||||
private final Connection connection;
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
/** A message pulled off the broker but not yet acked: its delivery-tag plus the port payload. */
|
||||
private record Held(long deliveryTag, InboxMessage message) {}
|
||||
|
||||
/** Connect to {@code uri} (e.g. {@code amqp://guest:guest@127.0.0.1:5672/}) and open the inbox. */
|
||||
public static AmqpReplyInbox open(String uri) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new AmqpReplyInbox(factory.newConnection("bridged-reply-inbox"));
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection) {
|
||||
this.connection = connection;
|
||||
try {
|
||||
this.channel = connection.createChannel();
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot open AMQP channel", e);
|
||||
}
|
||||
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
|
||||
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue).
|
||||
if (connection instanceof Recoverable recoverable) {
|
||||
recoverable.addRecoveryListener(new RecoveryListener() {
|
||||
@Override
|
||||
public void handleRecovery(Recoverable recoverable) {
|
||||
held.clear();
|
||||
log.info("AMQP connection recovered; cleared held replies for fresh redelivery");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void handleRecoveryStarted(Recoverable recoverable) {
|
||||
// no-op: we act once recovery completes
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void own(String target) {
|
||||
synchronized (channelLock) {
|
||||
if (consumerTags.containsKey(target)) {
|
||||
return; // already owning this target
|
||||
}
|
||||
String queue = queueName(target);
|
||||
try {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void publish(String target, String msgId, String content) {
|
||||
AMQP.BasicProperties props = new AMQP.BasicProperties.Builder()
|
||||
.messageId(msgId)
|
||||
.deliveryMode(2) // persistent — survives a broker restart
|
||||
.contentType("text/plain")
|
||||
.build();
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
channel.basicPublish("", queueName(target), props, content.getBytes(StandardCharsets.UTF_8));
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot publish reply to " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
return perTarget.values().stream().map(Held::message).toList();
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return;
|
||||
}
|
||||
Held h;
|
||||
synchronized (perTarget) {
|
||||
h = perTarget.remove(msgId);
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(h.deliveryTag(), false);
|
||||
}
|
||||
} catch (IOException e) {
|
||||
// Ack didn't reach the broker: restore the entry so a later ack (or a redelivery after
|
||||
// reconnect) can retry. Keeps the at-least-once contract — a reply is never silently lost.
|
||||
synchronized (perTarget) {
|
||||
perTarget.putIfAbsent(msgId, h);
|
||||
}
|
||||
throw new IllegalStateException("cannot ack reply " + msgId + " on " + queueName(target), e);
|
||||
}
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback(String target) {
|
||||
return (_, delivery) -> {
|
||||
String msgId = delivery.getProperties().getMessageId();
|
||||
long tag = delivery.getEnvelope().getDeliveryTag();
|
||||
if (msgId == null || msgId.isBlank()) {
|
||||
msgId = Long.toHexString(tag); // synthesize an id so dedup still has a key
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
duplicate = true;
|
||||
} else {
|
||||
perTarget.put(msgId, new Held(tag, new InboxMessage(msgId, target, content)));
|
||||
duplicate = false;
|
||||
}
|
||||
}
|
||||
if (duplicate) {
|
||||
// Redelivered duplicate: ack the new tag and drop it so the broker stops resending.
|
||||
synchronized (channelLock) {
|
||||
channel.basicAck(tag, false);
|
||||
}
|
||||
}
|
||||
};
|
||||
}
|
||||
|
||||
private static String queueName(String target) {
|
||||
return QUEUE_PREFIX + target + QUEUE_SUFFIX;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
try {
|
||||
channel.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP channel close: {}", e.toString());
|
||||
}
|
||||
try {
|
||||
connection.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("AMQP connection close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,569 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionException;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.locks.ReentrantLock;
|
||||
|
||||
/**
|
||||
* The blocking delegation feature (CB-104): deliver {@code content} into a worker and block until
|
||||
* the worker returns a <em>structured reply</em> via {@code bridge_reply} (the {@link Rendezvous}),
|
||||
* then hand that reply back. Delivery is the {@link Injector}'s job (the background poller sends it
|
||||
* when the worker is injectable); this service never drives the injector or scrapes the terminal —
|
||||
* completion is the worker's explicit reply, not a guess about {@code agent_status}.
|
||||
*
|
||||
* <p>Sends are serialized per session so exactly one reply can be outstanding per worker, which is
|
||||
* what lets a reply map unambiguously to its send (no cross-talk between concurrent callers).
|
||||
*
|
||||
* <p>If the worker never replies within the timeout, the caller gets a typed "still working" /
|
||||
* "queued" outcome — the message may still be mid-flight. A finished-but-unreplied turn is caught
|
||||
* by the CB-106 completion fallback (see {@link Rendezvous#resolveCompletion}).
|
||||
*
|
||||
* <p><strong>Async fire-and-poll (CB-107).</strong> A caller's MCP client caps a blocking call at
|
||||
* ~60s, but a real delegated task runs for minutes. {@link #sendAsync} therefore runs the same
|
||||
* blocking {@link #send} on a background virtual thread and hands back a <em>ticket</em> the caller
|
||||
* polls with {@link #poll}. The blocking and async paths share one code path (and the same per-target
|
||||
* serialization), so async inherits the reply + completion resolution behaviour for free.
|
||||
*/
|
||||
public final class MessageService {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MessageService.class);
|
||||
|
||||
/**
|
||||
* The window a fire-and-poll send waits for resolution — generous, since no caller is blocked on
|
||||
* it; a real delegated task resolves (reply or completion) well within this, and only a genuinely
|
||||
* hung worker rides it out.
|
||||
*/
|
||||
private static final long ASYNC_TIMEOUT_MS = 30 * 60 * 1_000L;
|
||||
|
||||
/** How long a finished (terminal) ticket is retained for polling before it is pruned. */
|
||||
private static final long TICKET_TTL_NANOS = 10 * 60 * 1_000_000_000L;
|
||||
|
||||
/** Outcome of a blocking send. */
|
||||
public enum Outcome {
|
||||
/** The worker called {@code bridge_reply}; {@code text} holds the structured answer. */
|
||||
REPLIED,
|
||||
/**
|
||||
* The worker's delegated turn finished without a {@code bridge_reply} (CB-106 fallback);
|
||||
* {@code text} is the scraped transcript tail rather than a structured answer.
|
||||
*/
|
||||
COMPLETED_UNREPLIED,
|
||||
/**
|
||||
* The worker ran the turn then wedged in an unrecoverable state (CB-109); {@code text} is the
|
||||
* failure context (e.g. the error screen). Terminal, but not a successful completion.
|
||||
*/
|
||||
WORKER_FAILED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||
* {@link #answer(String, String, long)} and the turn resumes.
|
||||
*/
|
||||
QUESTION,
|
||||
/** Timed out after the message was delivered — the worker is still working. */
|
||||
TIMED_OUT_WORKING,
|
||||
/** Timed out before delivery — the message is still queued for the worker. */
|
||||
TIMED_OUT_QUEUED,
|
||||
/** Another send to this session was in flight for the whole window. */
|
||||
BUSY,
|
||||
/**
|
||||
* An answer ({@link #answer(String, String, long)}) referenced a {@code turnId} that is no
|
||||
* longer open — the worker's {@code bridge_ask} already timed out or was answered.
|
||||
*/
|
||||
STALE_TURN
|
||||
}
|
||||
|
||||
/**
|
||||
* @param outcome how the send ended (or paused)
|
||||
* @param text the worker's answer when {@link #completed()} (a structured {@code bridge_reply}
|
||||
* for {@link Outcome#REPLIED}, a scraped transcript tail for
|
||||
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
||||
* else {@code null}
|
||||
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
||||
* {@link #answer(String, String, long)}), else {@code null}
|
||||
*/
|
||||
public record Reply(Outcome outcome, String text, String turnId) {
|
||||
/** A reply with no correlation id (the common terminal outcomes). */
|
||||
public Reply(Outcome outcome, String text) {
|
||||
this(outcome, text, null);
|
||||
}
|
||||
|
||||
/** Whether the worker's turn actually finished with an answer (replied or scraped). */
|
||||
public boolean completed() {
|
||||
return outcome == Outcome.REPLIED || outcome == Outcome.COMPLETED_UNREPLIED;
|
||||
}
|
||||
}
|
||||
|
||||
/** How a worker's {@code bridge_ask} (CB-205) resolved. */
|
||||
public enum AskOutcome {
|
||||
/** The primary answered; {@link AskResult#answer} carries it. */
|
||||
ANSWERED,
|
||||
/** No delegation was open to surface the question to — the worker has no one to ask. */
|
||||
NO_WAITER,
|
||||
/** The primary did not answer within the window. */
|
||||
TIMED_OUT
|
||||
}
|
||||
|
||||
/** The outcome of a worker's {@code bridge_ask}: how it resolved and (if answered) the answer. */
|
||||
public record AskResult(AskOutcome outcome, String answer) {
|
||||
}
|
||||
|
||||
/** Lifecycle phase of an async delegation ticket. */
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
PENDING,
|
||||
/** The worker's turn finished; {@link TaskView#reply} holds the answer. */
|
||||
DONE,
|
||||
/** The delegation could not complete (timed out, worker gone, or busy). */
|
||||
FAILED
|
||||
}
|
||||
|
||||
/**
|
||||
* A poll snapshot of an async delegation.
|
||||
*
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, else {@code null}
|
||||
* @param replySource {@code "reply"} (structured {@code bridge_reply}) or {@code "transcript"}
|
||||
* (completion scrape) when {@link Phase#DONE}, else {@code null}
|
||||
* @param detail a human note (live worker status while pending, or the failure reason)
|
||||
*/
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail) {
|
||||
}
|
||||
|
||||
/** An in-flight or finished async delegation, keyed by its ticket. */
|
||||
private record Task(String target, CompletableFuture<Reply> future, long createdNanos) {
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Injector injector;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ReplyInbox inbox;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
|
||||
/**
|
||||
* Create with an explicit {@link ReplyInbox} and optional {@link ReplyPushLoop}.
|
||||
*
|
||||
* @param pushLoop nullable — when non-null, the push loop is notified on the no-waiter reply
|
||||
* branch ({@link #reply}) so it can nudge the primary to drain the inbox
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with a metric registry (CB-502). Instrumenting here rather than at the REST and MCP
|
||||
* edges means both surfaces are counted by one piece of code and cannot drift.
|
||||
*
|
||||
* @param metrics nullable — when null, nothing is recorded
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop, Metrics metrics) {
|
||||
this.agents = agents;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
this.pushLoop = pushLoop;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Create with an explicit {@link ReplyInbox} and no push loop. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox) {
|
||||
this(agents, injector, rendezvous, inbox, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor that uses a default {@link InMemoryReplyInbox}. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous) {
|
||||
this(agents, injector, rendezvous, new InMemoryReplyInbox());
|
||||
}
|
||||
|
||||
/** Current lifecycle status of a worker (the {@code GET /sessions/{id}/status} surface). */
|
||||
public AgentStatus status(String target) {
|
||||
return agents.status(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code bridge_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
* result is <em>not</em> a failure — the reply is held for later drain.
|
||||
*
|
||||
* <p><strong>Do NOT use this for mid-turn questions.</strong> {@code bridge_ask} /
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* @return always {@code true} — the reply either resolved a live send or was queued
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(BridgedMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content);
|
||||
// A rising inbox share is the signal CB-307 exists to make visible: the worker finished but
|
||||
// nobody was waiting, so delivery now depends on the push loop and a drain.
|
||||
count(BridgedMetrics.REPLIES, "path", "inbox");
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
}
|
||||
|
||||
/** Record a counter sample when a registry is wired; a no-op in unit tests. */
|
||||
private void count(String name, String... labels) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(name, labels);
|
||||
}
|
||||
}
|
||||
|
||||
/** Count a send's terminal outcome and pass the reply through unchanged. */
|
||||
private Reply recorded(Reply r) {
|
||||
String label = sendOutcomeLabel(r.outcome());
|
||||
if (label != null) {
|
||||
count(BridgedMetrics.SENDS, "outcome", label);
|
||||
}
|
||||
return r;
|
||||
}
|
||||
|
||||
/** Map a terminal send outcome to its metric label, or {@code null} for non-terminal ones. */
|
||||
private static String sendOutcomeLabel(Outcome o) {
|
||||
return switch (o) {
|
||||
case REPLIED -> "replied";
|
||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Abandon any send still waiting on {@code target} because its session has gone away (CB-516).
|
||||
*
|
||||
* <p>Without this, tearing a worker down left its rendezvous waiter open: a blocking
|
||||
* {@code bridge_send} kept blocking, and an async one kept reporting {@code PENDING} until
|
||||
* {@link #ASYNC_TIMEOUT_MS} — thirty minutes — even though the worker provably no longer
|
||||
* existed and the delegation could never complete. Worse, {@code poll} already had the evidence
|
||||
* (it calls {@code liveStatus} to build its detail string and gets back {@code "unknown"}) and
|
||||
* reported {@code PENDING} anyway.
|
||||
*
|
||||
* <p>Resolving the waiter as a failure — rather than letting it time out — also means the
|
||||
* outcome is counted, so a torn-down delegation stops being invisible to {@code /metrics}.
|
||||
*
|
||||
* @return true if a live waiter was failed
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
return false; // nobody is blocked on this worker — nothing to abandon
|
||||
}
|
||||
boolean failed = rendezvous.resolveFailure(waiter, reason);
|
||||
if (failed) {
|
||||
log.debug("abandoned send to {}: {}", target, reason);
|
||||
}
|
||||
return failed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain (peek + ack) all pending inbox replies for {@code target}. At-least-once: returns the
|
||||
* messages and acknowledges them; an in-flight failure between returning and the caller
|
||||
* processing them re-surfaces them on a subsequent drain (the ack is local).
|
||||
*
|
||||
* @return the drained messages, newest last (FIFO); empty list if none
|
||||
*/
|
||||
public List<ReplyInbox.InboxMessage> drainReplies(String target) {
|
||||
var messages = inbox.peek(target);
|
||||
for (var msg : messages) {
|
||||
inbox.ack(target, msg.msgId());
|
||||
}
|
||||
return messages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis) {
|
||||
return send(target, content, timeoutMillis, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
||||
*
|
||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||
* queued, so a throwing hook fails the send cleanly (the waiter it already opened is closed and
|
||||
* nothing is left queued). It is <em>not</em> invoked when the send is {@link Outcome#BUSY}
|
||||
* (lock never taken). A caller uses this to record that <em>it</em> now owns the delegation's
|
||||
* reply routing (CB-548: {@code PrimaryRegistry} delegator ownership) — recording only on
|
||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
// Open the waiter BEFORE queueing delivery (CB-548). A fast reply — the worker already
|
||||
// injectable the instant we enqueue — otherwise arrives before the waiter is registered
|
||||
// and orphans into the inbox while this send blocks to the timeout (the enqueue-before-
|
||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
try {
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
return recorded(new Reply(
|
||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
}
|
||||
} finally {
|
||||
rendezvous.close(target, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker's mid-turn question (CB-205 reverse rendezvous): surface {@code question} to the
|
||||
* primary by resolving its open blocking {@code bridge_send}, then block this (worker) call until
|
||||
* the primary answers via {@link #answer} or {@code timeoutMillis} elapses. Identity is the
|
||||
* worker's own session — it does not address the primary.
|
||||
*
|
||||
* <p>Returns {@link AskOutcome#NO_WAITER} when no delegation is open to surface the question to
|
||||
* (nothing to answer it), {@link AskOutcome#ANSWERED} with the primary's answer, or
|
||||
* {@link AskOutcome#TIMED_OUT} if the primary stayed silent. The worker resumes its turn either
|
||||
* way — an answered ask hands back the answer; an unanswered one leaves it to proceed alone.
|
||||
*/
|
||||
public AskResult ask(String workerSession, String question, long timeoutMillis) {
|
||||
Rendezvous.AskTicket ticket = rendezvous.openAsk(workerSession);
|
||||
// Only the freshly-opening caller surfaces the question; a coalesced duplicate simply blocks on
|
||||
// the shared answer future that the fresh owner is already responsible for.
|
||||
if (ticket.fresh()) {
|
||||
// Register the reverse waiter first, then surface the question — so the answer, which can
|
||||
// arrive the instant the primary reacts, always finds an open waiter to resolve.
|
||||
if (!rendezvous.resolveQuestion(workerSession, question, ticket.turnId())) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("bridge_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the primary's answer for " + workerSession, e);
|
||||
} finally {
|
||||
// Only the fresh owner tears down the shared turn; a duplicate must leave it open.
|
||||
if (ticket.fresh()) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The primary's answer to a worker's {@code bridge_ask} (CB-205): resolve the worker's blocked
|
||||
* question identified by {@code turnId}, then — like a fresh {@link #send} — block for the worker's
|
||||
* eventual {@code bridge_reply} as it finishes the resumed turn. The worker session is derived from
|
||||
* {@code turnId}, never a caller argument.
|
||||
*
|
||||
* <p>Unlike {@link #send} this does not re-inject through the {@link Injector}: the worker is
|
||||
* mid-turn (already picked up), so the answer flows back through its own open {@code bridge_ask}
|
||||
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
||||
* is unblocked so a reply that lands the instant it resumes is not lost.
|
||||
*/
|
||||
public Reply answer(String turnId, String content, long timeoutMillis) {
|
||||
String workerSession = rendezvous.askSession(turnId);
|
||||
if (workerSession == null) {
|
||||
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
||||
}
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null);
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
// turn (it never re-entered the injector), so a silent worker rides out the window.
|
||||
return new Reply(Outcome.TIMED_OUT_WORKING, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fire-and-poll variant of {@link #send}: deliver {@code content} to {@code target} on a
|
||||
* background virtual thread and return immediately with a ticket to {@link #poll}. This is how a
|
||||
* long task is delegated without tripping the caller's MCP client call timeout.
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content) {
|
||||
return sendAsync(target, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(String, String)}, with the accepted-delivery hook of
|
||||
* {@link #send(String, String, long, Runnable)} — the running {@code send} invokes {@code onAccepted}
|
||||
* the moment it becomes the accepted target turn, so async flooding records delegator ownership
|
||||
* exactly as the blocking path does (CB-548).
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
CompletableFuture<Reply> future = CompletableFuture.supplyAsync(
|
||||
() -> send(target, content, ASYNC_TIMEOUT_MS, onAccepted), asyncExecutor);
|
||||
tasks.put(ticket, new Task(target, future, System.nanoTime()));
|
||||
pruneTerminalTickets();
|
||||
log.debug("async send {} -> {}", ticket, target);
|
||||
return ticket;
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket;
|
||||
* otherwise a {@link Phase#PENDING} view (with the live worker status as detail), a
|
||||
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
||||
*/
|
||||
public TaskView poll(String ticket) {
|
||||
Task task = tasks.get(ticket);
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future();
|
||||
if (!f.isDone()) {
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target()));
|
||||
}
|
||||
Reply r;
|
||||
try {
|
||||
r = f.getNow(null);
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
Throwable cause = (e instanceof CompletionException ce && ce.getCause() != null) ? ce.getCause() : e;
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage());
|
||||
}
|
||||
if (r.completed()) {
|
||||
String source = r.outcome() == Outcome.REPLIED ? "reply" : "transcript";
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null);
|
||||
}
|
||||
// A wedged worker (CB-109) carries the error context as its reason; the timeout/busy
|
||||
// outcomes carry none, so fall back to the outcome name.
|
||||
String detail = r.outcome() == Outcome.WORKER_FAILED && r.text() != null
|
||||
? r.text()
|
||||
: "no reply — " + r.outcome().name().toLowerCase();
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
return agents.status(target).name().toLowerCase();
|
||||
} catch (RuntimeException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/** Drop finished tickets older than the TTL so the registry cannot grow without bound. */
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = System.nanoTime() - TICKET_TTL_NANOS;
|
||||
tasks.values().removeIf(t -> t.future().isDone() && t.createdNanos() < cutoff);
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
public void close() {
|
||||
asyncExecutor.shutdown();
|
||||
}
|
||||
|
||||
/** Map a rendezvous {@link Rendezvous.Kind} onto its send {@link Outcome} (shared by send/answer). */
|
||||
private static Outcome outcomeOf(Rendezvous.Kind kind) {
|
||||
return switch (kind) {
|
||||
case REPLY -> Outcome.REPLIED;
|
||||
case COMPLETION -> Outcome.COMPLETED_UNREPLIED;
|
||||
case FAILED -> Outcome.WORKER_FAILED;
|
||||
case QUESTION -> Outcome.QUESTION;
|
||||
};
|
||||
}
|
||||
|
||||
private static boolean tryLock(ReentrantLock lock, long millis) {
|
||||
try {
|
||||
return lock.tryLock(Math.max(0, millis), TimeUnit.MILLISECONDS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the session send lock", e);
|
||||
}
|
||||
}
|
||||
|
||||
private static long remainingMillis(long deadlineNanos) {
|
||||
return (deadlineNanos - System.nanoTime()) / 1_000_000L;
|
||||
}
|
||||
}
|
||||
@@ -1,199 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Mechanism (b) of CB-307: a dedicated, status-gated push loop that nudges the primary's own
|
||||
* herdr pane when a worker reply lands with no live {@code bridge_send} to resolve it.
|
||||
*
|
||||
* <p>The loop is triggered by {@link #onReplyQueued(String)} (called from
|
||||
* {@link MessageService#reply} after the durable inbox publish). It checks four conditions
|
||||
* at each tick via {@link #decide(String, int)}, then either injects a drain nudge,
|
||||
* waits for the primary to become injectable, or stops reminding.
|
||||
*
|
||||
* <p>Bounded: at most {@link #maxReminders} nudges per target, with a configurable backoff
|
||||
* between them. The reply is never lost — the durable inbox is the backstop.
|
||||
*/
|
||||
public final class ReplyPushLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ReplyPushLoop.class);
|
||||
static final String NUDGE_FORMAT = "Worker %s returned a reply — run bridge_poll(target=%s) to collect it";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
private final ReplyInbox inbox;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final int maxReminders;
|
||||
private final long backoffMs;
|
||||
private final Metrics metrics; // CB-512: nullable — no registry in unit tests
|
||||
|
||||
/** Track targets that have an active schedule. */
|
||||
private final ConcurrentHashMap<String, Boolean> activeTargets = new ConcurrentHashMap<>();
|
||||
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs) {
|
||||
this(primaryRegistry, agents, inbox, scheduler, maxReminders, backoffMs, null);
|
||||
}
|
||||
|
||||
/** As above, with a metric registry (CB-512) so push outcomes are counted. */
|
||||
public ReplyPushLoop(PrimaryRegistry primaryRegistry, AgentControl agents, ReplyInbox inbox,
|
||||
ScheduledExecutorService scheduler,
|
||||
int maxReminders, long backoffMs, Metrics metrics) {
|
||||
this.primaryRegistry = primaryRegistry;
|
||||
this.agents = agents;
|
||||
this.inbox = inbox;
|
||||
this.scheduler = scheduler;
|
||||
this.maxReminders = maxReminders;
|
||||
this.backoffMs = backoffMs;
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(BridgedMetrics.PUSH_NUDGES, "outcome", outcome);
|
||||
}
|
||||
}
|
||||
|
||||
// --- decision logic (package-private for unit-testing) -------------------------------------
|
||||
|
||||
/** The action the loop should take for a target at the given reminder count. */
|
||||
enum Action { INJECT, WAIT_BUSY, STOP }
|
||||
|
||||
/**
|
||||
* Pure decision function: examine the current state and return what the loop should do.
|
||||
*
|
||||
* @param target the worker session (target terminal id)
|
||||
* @param reminderCount how many nudges have been sent so far for this target
|
||||
* @return the action the caller should take
|
||||
*/
|
||||
Action decide(String target, int reminderCount) {
|
||||
// CB-532: the destination is per-delegation — the lead that sent this worker its work, not
|
||||
// "the primary". With two leads orchestrating one fleet the singular question has no right
|
||||
// answer, and answering it anyway interrupted whichever lead happened to call bridge_send
|
||||
// first with results it never asked for.
|
||||
var nudgeTarget = primaryRegistry.nudgeTargetFor(target);
|
||||
if (nudgeTarget.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}, stopping reminder", target);
|
||||
return Action.STOP;
|
||||
}
|
||||
if (inbox.peek(target).isEmpty()) {
|
||||
log.debug("push: inbox empty for {}, stopping reminder", target);
|
||||
return Action.STOP;
|
||||
}
|
||||
if (reminderCount >= maxReminders) {
|
||||
log.debug("push: reminder cap ({}) reached for {}, stopping", maxReminders, target);
|
||||
countNudge("exhausted");
|
||||
return Action.STOP;
|
||||
}
|
||||
String leadTerminal = nudgeTarget.get();
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(leadTerminal);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("push: status check failed for lead {}, will retry", leadTerminal, e);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
if (status.injectable()) {
|
||||
return Action.INJECT;
|
||||
}
|
||||
log.debug("push: lead {} is {} (not injectable), waiting", leadTerminal, status);
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
|
||||
// --- public entrypoint ---------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Called when a reply is queued for {@code target}. Idempotent per target: a second call while
|
||||
* a schedule is active is a no-op. The schedule nudges the primary, then schedules a follow-up
|
||||
* check (reminder on backoff, or re-check on WAIT_BUSY), until the inbox is empty or the cap
|
||||
* is reached.
|
||||
*/
|
||||
public void onReplyQueued(String target) {
|
||||
if (activeTargets.putIfAbsent(target, Boolean.TRUE) != null) {
|
||||
log.debug("push: already active for {}, ignoring duplicate trigger", target);
|
||||
return; // already scheduled
|
||||
}
|
||||
log.debug("push: starting reminder loop for {}", target);
|
||||
scheduleNext(target, 0);
|
||||
}
|
||||
|
||||
/** Execute one loop tick — called on the scheduler thread. */
|
||||
private void tick(String target, int reminderCount) {
|
||||
var action = decide(target, reminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(target, reminderCount);
|
||||
scheduleNext(target, reminderCount + 1);
|
||||
}
|
||||
// Re-check after the configured backoff; the primary may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(target, reminderCount);
|
||||
case STOP -> {
|
||||
activeTargets.remove(target);
|
||||
log.debug("push: reminder loop ended for {}", target);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Send the nudge and log the event. */
|
||||
private void injectNudge(String target, int reminderCount) {
|
||||
// Re-read rather than threading it down from decide(): the delegating lead can change
|
||||
// between the decision and the injection, and the nudge should follow the current one.
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: lead for {} disappeared before the nudge could be sent", target);
|
||||
return;
|
||||
}
|
||||
String leadTerminal = lead.get();
|
||||
String nudge = NUDGE_FORMAT.formatted(target, target);
|
||||
try {
|
||||
agents.send(leadTerminal, nudge);
|
||||
log.debug("push: nudge {}/{} sent to lead {} for target {}",
|
||||
reminderCount + 1, maxReminders, leadTerminal, target);
|
||||
countNudge("delivered");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} for target {} (reminder {}/{}): {}",
|
||||
leadTerminal, target, reminderCount + 1, maxReminders, e.toString());
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
private void scheduleNext(String target, int nextReminderCount) {
|
||||
scheduler.schedule(() -> tick(target, nextReminderCount), backoffMs, TimeUnit.MILLISECONDS);
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Whether any reminder loop is currently active for some target (CB-551). The idle-lead heartbeat
|
||||
* uses this to stand aside: while the push loop is actively nudging the lead, a concurrent
|
||||
* heartbeat injection would start a second competing turn in the same pane — racing loops multiply
|
||||
* turns and context burn. "Active" means a schedule exists in {@link #activeTargets}; the set is
|
||||
* bounded by what has been triggered, not by any persistent state.
|
||||
*/
|
||||
public boolean isActive() {
|
||||
return !activeTargets.isEmpty();
|
||||
}
|
||||
|
||||
/** Shut down the scheduler. Outstanding reminders are cancelled. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
activeTargets.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
public void close() {
|
||||
stop();
|
||||
}
|
||||
}
|
||||
@@ -1,94 +0,0 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* SPI for materializing a connected peer — the only way the bridge core creates or tears down
|
||||
* a peer process. Every launcher is a first-party, in-tree adapter selected by (future) profile
|
||||
* config; today's single adapter is the {@code ClaudeCodeLauncher} / Claude Code over herdr.
|
||||
*
|
||||
* <p>The core delegates spawn and teardown to this interface without knowing how the peer is set
|
||||
* up. Environment variables, CLI flags, subscription guards, transport (herdr tab/pane) layout,
|
||||
* and naming conventions are all adapter-private — the core sees only the returned
|
||||
* {@link PeerHandle} whose {@code id()} is the registry/routing key.
|
||||
*
|
||||
* <p>The interface is a superset of what {@code SessionManager} and {@code Bridged.main} call
|
||||
* on the concrete launcher today.
|
||||
*/
|
||||
public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The set of {@link Capability capabilities} this launcher declares. A peer whose profile
|
||||
* opts into a git-forge token should include {@link Capability#SELF_PR}; the base set for
|
||||
* the Claude Code herdr adapter is always {@code MID_TURN_ASK, WORKTREE, ORPHAN_REAP}.
|
||||
*/
|
||||
Set<Capability> capabilities();
|
||||
|
||||
/**
|
||||
* {@code profileName}/requestedCwd null/blank → default resolution. Returns after the peer
|
||||
* process is live (env + argv + placement complete). Never returns {@code null}.
|
||||
*
|
||||
* @param req the spawn parameters (profile, requested cwd, caller cwd)
|
||||
* @return a handle whose {@link PeerHandle#id()} is the registry/routing key
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
PeerHandle spawn(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The configured worker profile names — the set of names {@code spawn(profileName)} accepts.
|
||||
*/
|
||||
Set<String> profiles();
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
*
|
||||
* @return the resolved absolute path, never null/blank
|
||||
*/
|
||||
String effectiveCwd(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The parity-overlay file list for {@code profileName} (default list when unset). Used by
|
||||
* worktree provisioning to copy config files into the isolated checkout before spawning.
|
||||
*/
|
||||
List<String> parityOverlay(String profileName);
|
||||
|
||||
/**
|
||||
* The set of all agents this launcher currently tracks, transport-specific. Each element
|
||||
* exposes at minimum a pane-like {@code id()} matching this launcher's {@link PeerHandle}
|
||||
* scheme, plus transport-level status. Callers merge this set with the session registry to
|
||||
* build a live roster view.
|
||||
*/
|
||||
List<?> list();
|
||||
|
||||
/**
|
||||
* Reap orphaned peers left behind by a prior daemon process. Only peers whose naming scheme
|
||||
* matches this launcher's and whose nonce differs from the current process are eligible.
|
||||
* Best-effort: a failure to list or to stop any one peer is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
int reapOrphanWorkers();
|
||||
|
||||
/**
|
||||
* Tear a peer down by its registry/routing key ({@link PeerHandle#id()}). Tolerates an
|
||||
* already-gone peer. Also cleans up launcher-private resources (e.g. empty dedicated tabs)
|
||||
* when safe to do so.
|
||||
*/
|
||||
void stop(String id);
|
||||
|
||||
/**
|
||||
* Discard the context of the peer identified by {@code id}. Implementations must bypass normal
|
||||
* bridge delivery/turn accounting. Unsupported peer kinds return {@code false} without sending
|
||||
* a guessed command.
|
||||
*
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
}
|
||||
@@ -1,22 +0,0 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank()) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
PlacementCandidate first = ctx.candidates().getFirst();
|
||||
return new PlacementCandidate(first.profile(), null, first.weight(), first.maxLoad());
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
}
|
||||
@@ -1,19 +0,0 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
|
||||
/**
|
||||
* Everything a {@link PlacementPolicy} needs to make one selection.
|
||||
*
|
||||
* @param defaultProfile profile a {@code fixed} policy should return (may be {@code null})
|
||||
* @param candidates every configured candidate; the policy filters out those at cap or unreachable
|
||||
* @param liveCount current live worker count per profile (from the session registry)
|
||||
* @param unreachable profiles already known to have failed in this spawn attempt
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable) {
|
||||
}
|
||||
@@ -1,65 +0,0 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Shared filtering and empty-set reporting used by the built-in placement policies.
|
||||
*/
|
||||
final class PlacementPolicyUtil {
|
||||
|
||||
private PlacementPolicyUtil() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Candidates that are not known-unreachable and have not reached their maxLoad.
|
||||
* A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (ctx.unreachable().contains(c.profile())) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
if (cap != null) {
|
||||
int live = ctx.liveCount().apply(c.profile());
|
||||
if (live >= cap) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.add(c);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all at capacity,
|
||||
* all unreachable, or a mix.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int atCap = 0;
|
||||
int unreachable = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
atCap++;
|
||||
}
|
||||
}
|
||||
|
||||
int total = ctx.candidates().size();
|
||||
if (total == 0) {
|
||||
return new PlacementException("no worker profiles configured");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
if (unreachable == total) {
|
||||
return new PlacementException("all worker profiles are unreachable");
|
||||
}
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + (total - atCap - unreachable) + " remaining");
|
||||
}
|
||||
}
|
||||
@@ -1,277 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStreamReader;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* Production {@link Worktrees} implementation that shells {@code git} via {@link ProcessBuilder}.
|
||||
* Non-zero exits become {@link WorktreeException}. Worktree directories live under a configurable
|
||||
* root (default: a sibling {@code .bridged-worktrees} of the repo root) so they are never nested
|
||||
* inside the primary working tree.
|
||||
*/
|
||||
public final class GitWorktrees implements Worktrees {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(GitWorktrees.class);
|
||||
|
||||
/** Project-level MCP config. Present in the repo, so every worktree would otherwise inherit the
|
||||
* primary's IDE server mounts (a CB-523 worker edited the primary checkout; see the isolation
|
||||
* javadoc). Neutralized unconditionally. */
|
||||
private static final String MCP_CONFIG = ".mcp.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .mcp.json}: a valid, explicitly empty server map. */
|
||||
private static final String NEUTRAL_MCP_CONFIG = "{\n \"mcpServers\": {}\n}\n";
|
||||
|
||||
/** OpenCode's repo-level config. Tracked here, so it lands in every worktree, and it mounts the
|
||||
* primary's gitea and context7 servers with the primary's credentials. Neutralized so the worker
|
||||
* gets only the config its launcher writes via {@code OPENCODE_CONFIG}.
|
||||
*
|
||||
* <p>The reason has changed shape and is now stronger. It used to be a crash: the file carried
|
||||
* {@code {file:.secrets/...}} references to gitignored files that never reached a worktree, and
|
||||
* opencode refuses to start on a dangling reference (CB-543). Those credentials now live in one
|
||||
* shell-level store and the file reads them as {@code {env:...}}, so in a worktree the reference
|
||||
* resolves instead of failing. That is worse, not better: a member would silently inherit the
|
||||
* primary's admin-scoped {@code GITEA_ACCESS_TOKEN}. A loud crash became a quiet privilege leak,
|
||||
* so this entry protects a boundary now rather than papering over a startup error. */
|
||||
private static final String OPENCODE_CONFIG = "opencode.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code opencode.json}: a valid, empty JSON object. */
|
||||
private static final String NEUTRAL_OPENCODE_CONFIG = "{}\n";
|
||||
|
||||
/** Autoenv's repo-level config. Not tracked today, but re-landing it must stay safe: autoenv
|
||||
* authorizes by path, so a fresh worktree path is always unauthorized and its interactive prompt
|
||||
* would block every spawn — neutralize it so it can never be committed. */
|
||||
private static final String AUTOENV_CONFIG = ".autoenv";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .autoenv}: a valid, empty env file. */
|
||||
private static final String NEUTRAL_AUTOENV_CONFIG = "";
|
||||
|
||||
/**
|
||||
* A tracked project config that is hostile in a provisioned worktree, and what to replace it
|
||||
* with. {@link #file} is the repo-relative path; {@link #stub} is a neutral but VALID payload for
|
||||
* that file's format — a malformed stub would only trade one crash for another;
|
||||
* {@link #createIfAbsent} keeps {@code .mcp.json}'s long-standing behaviour of writing its stub
|
||||
* even when the repo carries no such file, whereas the others are only touched when present.
|
||||
*/
|
||||
private record WorktreeHostileConfig(String file, String stub, boolean createIfAbsent) {}
|
||||
|
||||
/** The worktree-hostile configs neutralized in every provisioned worktree, in order. */
|
||||
private static final List<WorktreeHostileConfig> WORKTREE_HOSTILE_CONFIGS = List.of(
|
||||
new WorktreeHostileConfig(MCP_CONFIG, NEUTRAL_MCP_CONFIG, true),
|
||||
new WorktreeHostileConfig(OPENCODE_CONFIG, NEUTRAL_OPENCODE_CONFIG, false),
|
||||
new WorktreeHostileConfig(AUTOENV_CONFIG, NEUTRAL_AUTOENV_CONFIG, false)
|
||||
);
|
||||
|
||||
private final String configuredRoot;
|
||||
private final SecureRandom random = new SecureRandom();
|
||||
private final AtomicLong seq = new AtomicLong();
|
||||
|
||||
/** Default constructor: worktree root is derived per-repo as {@code <repoRoot>/../.bridged-worktrees}. */
|
||||
public GitWorktrees() {
|
||||
this(null);
|
||||
}
|
||||
|
||||
/** @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of the repo root. */
|
||||
public GitWorktrees(String configuredRoot) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
String base = (baseRef == null || baseRef.isBlank()) ? "HEAD" : baseRef;
|
||||
String nonce = nonce();
|
||||
Path root = resolveRoot(repoRoot);
|
||||
Path path = root.resolve(nonce);
|
||||
try {
|
||||
Files.createDirectories(root);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
isolateToolSurface(wt);
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Neutralize the worktree's worktree-hostile project configs so a worker inherits only the tools
|
||||
* and environment its launcher mounts (the bridge via {@code --mcp-config}, the opencode config
|
||||
* via {@code OPENCODE_CONFIG}) — never the primary's.
|
||||
*
|
||||
* <p>This is unconditional, and it is not the same job as the parity overlay. The repo's own
|
||||
* committed {@code .mcp.json} declares the primary's IDE servers, so a fresh checkout mounts them
|
||||
* whether or not the overlay copies anything; a worker that inherits them navigates and edits
|
||||
* through tools bound to the <em>primary's</em> IntelliJ project, which silently hands it absolute
|
||||
* paths outside its own worktree. That is not hypothetical: a CB-523 worker made all 59 of its
|
||||
* edits in the primary checkout while compiling its worktree, so every build it ran was of code
|
||||
* that did not contain its changes. {@code opencode.json} is the same trap one tool over — tracked,
|
||||
* so it lands in every worktree, and it mounts gitea and context7 with the primary's own
|
||||
* credentials, which a member must never hold. {@code .autoenv} extends the principle to a
|
||||
* config that is not tracked today: autoenv authorizes by path, so a fresh worktree path is always
|
||||
* unauthorized and its interactive prompt would block every spawn, so re-landing one must be safe.
|
||||
*
|
||||
* <p>Where a config exists it is replaced by a valid neutral stub (an explicitly empty
|
||||
* map/object, or an empty env file — never a deletion, which would still let a later
|
||||
* {@code git checkout} restore the hostile copy). The {@code --skip-worktree} bit keeps the
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
}
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot neutralize " + cfg.file() + " in the worktree: "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing to remove", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.info("removing worktree {}", worktreePath);
|
||||
exec("git", "-C", repoRoot, "worktree", "remove", "--force", worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlay == null || overlay.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Path srcRoot = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
Path dstRoot = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (String rel : overlay) {
|
||||
Path src = srcRoot.resolve(rel).normalize();
|
||||
if (!Files.exists(src)) {
|
||||
log.debug("parity overlay source missing — skipping {}", rel);
|
||||
continue;
|
||||
}
|
||||
Path dst = dstRoot.resolve(rel).normalize();
|
||||
try {
|
||||
Files.createDirectories(dst.getParent());
|
||||
Files.copy(src, dst, StandardCopyOption.REPLACE_EXISTING, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
log.debug("copied parity overlay {}", rel);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy overlay " + rel + ": " + e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(dstRoot, rel)) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", rel);
|
||||
log.debug("marked overlay --skip-worktree {}", rel);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
String out = exec("git", "-C", cwd, "rev-parse", "--show-toplevel");
|
||||
return Path.of(out.trim()).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/** Resolve the directory that will hold per-session worktree checkouts. */
|
||||
private Path resolveRoot(String repoRoot) {
|
||||
if (configuredRoot != null && !configuredRoot.isBlank()) {
|
||||
return Path.of(configuredRoot).toAbsolutePath().normalize();
|
||||
}
|
||||
Path repo = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
return repo.resolveSibling(".bridged-worktrees");
|
||||
}
|
||||
|
||||
private String nonce() {
|
||||
return String.format("%06x", random.nextInt(1 << 24)) + "-" + seq.incrementAndGet();
|
||||
}
|
||||
|
||||
private boolean isTracked(Path worktreeRoot, String rel) {
|
||||
return exitCode("git", "-C", worktreeRoot.toString(), "ls-files", "--error-unmatch", rel) == 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a command and return its stdout. Non-zero exit → {@link WorktreeException} with both
|
||||
* stdout and stderr (merged by redirectErrorStream).
|
||||
*/
|
||||
private String exec(String... command) {
|
||||
String out;
|
||||
int code;
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try (BufferedReader r = new BufferedReader(new InputStreamReader(p.getInputStream(), StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(Collectors.joining("\n"));
|
||||
} catch (IOException e) {
|
||||
p.destroyForcibly();
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command) + "\n" + out);
|
||||
}
|
||||
code = p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
if (code != 0) {
|
||||
throw new WorktreeException("exit " + code + " for: " + String.join(" ", command)
|
||||
+ (out.isBlank() ? "" : "\n" + out));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
private int exitCode(String... command) {
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command));
|
||||
}
|
||||
return p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,68 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
/**
|
||||
* A bridge-owned worker session — the authoritative in-daemon record of a worker this
|
||||
* process spawned. Immutable; state transitions are performed by replacing the record in
|
||||
* {@link SessionManager}'s registry.
|
||||
*
|
||||
* @param paneId the host-unique opaque id (CB-519) — the registry key and the argument to
|
||||
* teardown. Despite the historical name this is the {@link
|
||||
* dev.ltms.bridged.peer.PeerHandle#id()}, a UUID, and is distinct from the
|
||||
* launcher-private herdr pane coordinate.
|
||||
* @param terminalId herdr terminal handle — the {@code target} for send/read/status
|
||||
* @param profile the profile name that spawned this session — WHICH BACKEND
|
||||
* @param role the contract this member runs under — WHAT IT IS FOR. Independent of
|
||||
* {@code profile}: a reviewer may run on the same profile as the dev
|
||||
* whose diff it reads. {@code null} only for a session recorded before
|
||||
* the role was known
|
||||
* @param cwd the resolved working directory the worker started in
|
||||
* @param ownerTerminal the caller that requested this worker ({@code null} = daemon/anon)
|
||||
* @param spawnedAtNanos {@link System#nanoTime()} when the session was registered
|
||||
* @param lastActivityAtNanos {@link System#nanoTime()} of the most recent lifecycle event
|
||||
* @param turnCount number of delegated turns that have been delivered to this session
|
||||
* @param state current lifecycle state in the one-shot FSM
|
||||
*/
|
||||
public record MemberSession(
|
||||
String paneId,
|
||||
String terminalId,
|
||||
String profile,
|
||||
MemberRole role,
|
||||
String cwd,
|
||||
String ownerTerminal,
|
||||
long spawnedAtNanos,
|
||||
long lastActivityAtNanos,
|
||||
int turnCount,
|
||||
State state,
|
||||
String worktree,
|
||||
String branch) {
|
||||
|
||||
/** One-shot worker lifecycle states. */
|
||||
public enum State {
|
||||
SPAWNING,
|
||||
READY,
|
||||
BUSY,
|
||||
DONE,
|
||||
FAILED,
|
||||
RELEASED
|
||||
}
|
||||
|
||||
/** Return a copy of this session in {@code state}. */
|
||||
public MemberSession withState(State state) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch);
|
||||
}
|
||||
|
||||
/** Return a copy with {@code lastActivityAtNanos} updated to {@code nowNanos}. */
|
||||
public MemberSession withActivity(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount, state, worktree, branch);
|
||||
}
|
||||
|
||||
/** Return a copy with the turn count incremented and activity timestamped at {@code nowNanos}. */
|
||||
public MemberSession bumpTurn(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount + 1, state, worktree, branch);
|
||||
}
|
||||
}
|
||||
@@ -1,619 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Authoritative in-daemon registry of the worker sessions this {@code bridged} process spawned.
|
||||
* Delegates spawn/teardown to a {@link PeerLauncher} (which performs subscription-guarded env
|
||||
* setup and process/materialization) and adds lifecycle tracking, ownership, and deterministic
|
||||
* teardown on top.
|
||||
*
|
||||
* <p>The state machine is intentionally one-shot / no-reuse: every acquired worker is fresh,
|
||||
* and a finished or released worker is torn down, never pooled. {@link #recycle} is a convenience
|
||||
* for {@code release + acquire} with a new distinct pane id.
|
||||
*
|
||||
* <p>The manager implements {@link TurnListener} so the injector's turn boundaries drive
|
||||
* {@code READY → BUSY → DONE} (or {@code FAILED}). It exposes a {@link MemberPresence} view via
|
||||
* {@link #asPresence()}: any MCP contact from a worker marks it present and simultaneously
|
||||
* transitions the session {@code SPAWNING → READY}.
|
||||
*/
|
||||
public final class SessionManager implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(SessionManager.class);
|
||||
|
||||
private final PeerLauncher launcher;
|
||||
private final Worktrees worktrees;
|
||||
private final ConcurrentHashMap<String /*paneId*/, MemberSession> registry = new ConcurrentHashMap<>();
|
||||
private final MemberPresence presence;
|
||||
private final SecureRandom nonceRandom = new SecureRandom();
|
||||
private final AtomicLong nonceSeq = new AtomicLong();
|
||||
private final LongSupplier nowNanos;
|
||||
private final int contextCap;
|
||||
private final boolean clearAfterTurn;
|
||||
|
||||
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
|
||||
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
/** CB-516: notified with a terminalId on every release; no-op until wired. */
|
||||
private final List<Consumer<String>> releaseListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
|
||||
/** Backward-compatible constructor: shared-tree sessions, production git seam. */
|
||||
public SessionManager(PeerLauncher launcher) {
|
||||
this(launcher, new GitWorktrees(), System::nanoTime, 0, false);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor with an injectable worktree seam. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees) {
|
||||
this(launcher, worktrees, System::nanoTime, 0, false);
|
||||
}
|
||||
|
||||
/** Test constructor with an injectable clock. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos) {
|
||||
this(launcher, worktrees, nowNanos, 0, false);
|
||||
}
|
||||
|
||||
/** Production constructor with a configured context turn cap. */
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, int contextCap) {
|
||||
this(launcher, worktrees, System::nanoTime, contextCap, false);
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap) {
|
||||
this(launcher, worktrees, nowNanos, contextCap, false);
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
this.launcher = launcher;
|
||||
this.worktrees = worktrees;
|
||||
this.presence = new PresenceBridge(this);
|
||||
this.nowNanos = nowNanos;
|
||||
this.contextCap = contextCap;
|
||||
this.clearAfterTurn = clearAfterTurn;
|
||||
}
|
||||
|
||||
/**
|
||||
* The single {@link MemberPresence} view of this manager: it records availability and forwards
|
||||
* the signal to the {@code SPAWNING → READY} transition. Pass this to the {@code Injector} and
|
||||
* {@code BridgeMcp} where they previously accepted a plain {@link MemberPresence}. The same
|
||||
* instance is returned every call — presence is shared state, so a fresh bridge per call would
|
||||
* fragment the {@code present} set and lose signals across callers.
|
||||
*/
|
||||
public MemberPresence asPresence() {
|
||||
return presence;
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a worker and register it as {@link MemberSession.State#SPAWNING}. The caller's
|
||||
* identity is recorded as {@code ownerTerminal} ({@code null} for daemon/anon callers).
|
||||
*/
|
||||
public MemberSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal) {
|
||||
return acquire(profile, requestedCwd, callerCwd, ownerTerminal, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a member, optionally inside a fresh git worktree, defaulting the role to
|
||||
* {@link MemberRole#DEV}.
|
||||
*
|
||||
* <p>{@code DEV} is the right default because it is exactly what the old "worker" meant: an
|
||||
* unqualified spawn is a unit of implementation work. An architect or a reviewer is always
|
||||
* asked for on purpose, so neither is ever what a caller silently gets.
|
||||
*/
|
||||
public MemberSession acquire(String profile, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
return acquire(profile, MemberRole.DEV, requestedCwd, callerCwd, ownerTerminal, wt);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a member, optionally inside a fresh git worktree. When {@code wt} is non-null the
|
||||
* worktree is provisioned, parity-overlaid, and its path becomes the member's cwd. On any
|
||||
* failure before registration the worktree is removed so no dangling checkout is left.
|
||||
*
|
||||
* @param profile which backend to run on — a {@code profiles:} key
|
||||
* @param role which contract the member runs under; never {@code null}
|
||||
*/
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
if (wt == null) {
|
||||
// CB-557: the role must ride on the SpawnRequest, not stay a local. The launcher needs it
|
||||
// to pick the profile out of that role's pool and to label the tab; a role kept only on
|
||||
// the MemberSession is recorded after the spawn it was supposed to steer.
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd, null, null, memberRole);
|
||||
PeerHandle handle = launcher.spawn(req);
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
null,
|
||||
null);
|
||||
registry.put(handle.id(), session);
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
return acquireWithWorktree(profile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt);
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id and remove it from the registry. Idempotent. */
|
||||
public void release(String paneId) {
|
||||
release(paneId, ReleaseCause.COMPLETED);
|
||||
}
|
||||
|
||||
/**
|
||||
* Core teardown: always stops the worker pane and deregisters the session; whether the worker's
|
||||
* git worktree is also removed depends on {@code cause}.
|
||||
*
|
||||
* <p>CB-544: these are two concerns that used to be fused. Stopping the pane is correct on every
|
||||
* teardown — the worker process must end. Removing the worktree is a destructive act that is only
|
||||
* correct for a deliberately-finished teardown (an explicit stop of a completed session, the
|
||||
* reaper releasing a genuinely idle one, a context-capped or recycled session). A shutdown drain
|
||||
* must stop panes but preserve worktrees: a worker's uncommitted work exists in exactly one
|
||||
* place — its worktree — so deleting it while the daemon simply goes down is silent data loss,
|
||||
* with no copy and no error. Do NOT fuse these back together; the cost of an orphaned worktree
|
||||
* is a logged path an operator can reclaim, the cost of a deleted one is unrecoverable work.
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
boolean preserveWorktree = cause == ReleaseCause.SHUTDOWN;
|
||||
if (removed != null) {
|
||||
log.debug("releasing session pane={} terminal={} state={} cause={}",
|
||||
removed.paneId(), removed.terminalId(), removed.state(), cause);
|
||||
if (preserveWorktree && removed.worktree() != null) {
|
||||
logPreservedForShutdown(removed);
|
||||
}
|
||||
// CB-516: a send still waiting on this worker can never be answered now. Tell the
|
||||
// listener BEFORE the pane is torn down, so a blocked caller fails fast with a real
|
||||
// reason instead of sitting on a rendezvous nothing will ever resolve.
|
||||
notifyReleased(removed.terminalId());
|
||||
}
|
||||
launcher.stop(paneId);
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Why a session is being released — governs whether its worktree is preserved or removed.
|
||||
* Worktree removal is reserved for the one case that is genuinely finished; everything else
|
||||
* must keep the worker's only copy of its work.
|
||||
*/
|
||||
public enum ReleaseCause {
|
||||
/** Deliberate teardown of a finished session. Stops the pane and removes the worktree. */
|
||||
COMPLETED,
|
||||
/** Daemon shutdown drain. Stops the pane but PRESERVES the worktree. */
|
||||
SHUTDOWN
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-544 shutdown drain log for a worktree we deliberately kept. A session still {@code BUSY}
|
||||
* when the drain timeout expired was abandoned mid-turn — that work may be uncommitted and is
|
||||
* the only copy — so the message is loud and points at the path an operator needs to reclaim.
|
||||
*/
|
||||
private void logPreservedForShutdown(MemberSession session) {
|
||||
if (session.state() == MemberSession.State.BUSY) {
|
||||
log.warn("shutdown drain abandoned BUSY session pane={} terminal={} mid-turn; "
|
||||
+ "worktree preserved at {}", session.paneId(), session.terminalId(),
|
||||
session.worktree());
|
||||
} else {
|
||||
log.info("shutdown drain preserved worktree at {} for pane={}",
|
||||
session.worktree(), session.paneId());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a callback invoked with a session's {@code terminalId} whenever it is acquired
|
||||
* (CB-520). This is the hook that lets the reply inbox {@code own} a target's queue.
|
||||
*/
|
||||
public void onAcquire(Consumer<String> listener) {
|
||||
if (listener != null) {
|
||||
acquireListeners.add(listener);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a callback invoked with a session's {@code terminalId} whenever it is released
|
||||
* (CB-516). Every teardown path funnels through {@link #release}, so one hook covers the REST
|
||||
* and MCP stop tools, the idle-TTL reaper, {@code recycle}, and shutdown drain alike.
|
||||
*
|
||||
* <p>Added rather than injected because {@code MessageService} — one intended listener — is
|
||||
* constructed after this manager (it needs the injector and rendezvous, which need the session
|
||||
* presence view this manager exposes). Wiring it at construction would require breaking that
|
||||
* cycle for one callback.
|
||||
*/
|
||||
public void onRelease(Consumer<String> listener) {
|
||||
if (listener != null) {
|
||||
releaseListeners.add(listener);
|
||||
}
|
||||
}
|
||||
|
||||
/** A listener failure must never prevent the acquisition it is reacting to. */
|
||||
private void notifyAcquired(String terminalId) {
|
||||
if (terminalId == null) {
|
||||
return;
|
||||
}
|
||||
for (Consumer<String> listener : acquireListeners) {
|
||||
try {
|
||||
listener.accept(terminalId);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("acquire listener failed for terminal {}: {}", terminalId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** A listener failure must never prevent the teardown it is reacting to. */
|
||||
private void notifyReleased(String terminalId) {
|
||||
if (terminalId == null) {
|
||||
return;
|
||||
}
|
||||
for (Consumer<String> listener : releaseListeners) {
|
||||
try {
|
||||
listener.accept(terminalId);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("release listener failed for terminal {}: {}", terminalId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private MemberSession acquireWithWorktree(String profile, MemberRole memberRole, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
// daemon cwd → "."), never the raw args. A plain REST spawn supplies neither a requested
|
||||
// nor a caller cwd, so taking the first non-blank of those two yielded null and put
|
||||
// `git -C null` on the command line — an NPE out of ProcessBuilder, surfacing as HTTP 500.
|
||||
// The non-worktree path always used this chain; only this branch was missed.
|
||||
String repoRoot = worktrees.repoRoot(
|
||||
launcher.effectiveCwd(new SpawnRequest(preResolvedProfile, requestedCwd, callerCwd)));
|
||||
String branch = "worker/" + slug(wt.ticketSlug()) + "-" + nonce();
|
||||
String path = null;
|
||||
PeerHandle handle;
|
||||
try {
|
||||
path = worktrees.add(repoRoot, branch, wt.baseRef());
|
||||
worktrees.overlayParity(repoRoot, path, launcher.parityOverlay(preResolvedProfile));
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd, null, null, memberRole));
|
||||
} catch (RuntimeException e) {
|
||||
if (path != null) {
|
||||
try {
|
||||
worktrees.remove(repoRoot, path);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up worktree {} after spawn error: {}", path, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
path,
|
||||
branch);
|
||||
registry.put(handle.id(), session);
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
|
||||
private String slug(String raw) {
|
||||
return raw == null ? "ticket" : raw.toLowerCase().replaceAll("[^a-z0-9]+", "-").replaceAll("^-+|-+$", "");
|
||||
}
|
||||
|
||||
private String nonce() {
|
||||
return String.format("%06x", nonceRandom.nextInt(1 << 24)) + "-" + nonceSeq.incrementAndGet();
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile to record for a session. A launcher that performed dynamic selection tells us
|
||||
* the actual profile via {@link PeerHandle#profile()}; otherwise fall back to what the caller
|
||||
* requested (or the launcher's default for a no-profile spawn).
|
||||
*/
|
||||
private String resolveProfile(PeerHandle handle, String requestedProfile) {
|
||||
String fromHandle = handle.profile();
|
||||
if (fromHandle != null && !fromHandle.isBlank()) {
|
||||
return fromHandle;
|
||||
}
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
return requestedProfile;
|
||||
}
|
||||
return launcher.defaultProfile();
|
||||
}
|
||||
|
||||
/**
|
||||
* Release the old session and acquire a fresh one with the same profile and working directory.
|
||||
* The new session is guaranteed to have a pane id distinct from the old one (no-reuse invariant).
|
||||
*/
|
||||
public MemberSession recycle(String paneId) {
|
||||
MemberSession old = registry.get(paneId);
|
||||
if (old == null) {
|
||||
throw new IllegalArgumentException("no session for paneId " + paneId);
|
||||
}
|
||||
release(paneId);
|
||||
return acquire(old.profile(), old.cwd(), old.cwd(), old.ownerTerminal());
|
||||
}
|
||||
|
||||
/** The session for {@code paneId}, if it is still registered and not released. */
|
||||
public Optional<MemberSession> get(String paneId) {
|
||||
return Optional.ofNullable(registry.get(paneId));
|
||||
}
|
||||
|
||||
/** Bridge-owned roster: all registered sessions (acquired minus released). */
|
||||
public List<MemberSession> roster() {
|
||||
return List.copyOf(registry.values());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-304 merged roster+live view. The registry is authoritative for worktree, branch,
|
||||
* profile, owner, and state; the optional live agent supplies the herdr-reported status.
|
||||
*/
|
||||
public static Map<String, Object> rosterView(MemberSession session, Agent live) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("sessionId", session.terminalId());
|
||||
m.put("paneId", session.paneId());
|
||||
// Both axes, always: profile says which backend this member runs on, role says what it is
|
||||
// for. A lead reading the roster needs both — two rows may share a profile and still be
|
||||
// allowed to do entirely different things.
|
||||
m.put("profile", session.profile());
|
||||
m.put("role", session.role() == null ? "dev" : session.role().wireName());
|
||||
m.put("state", session.state().name().toLowerCase());
|
||||
if (session.worktree() != null) {
|
||||
m.put("worktree", session.worktree());
|
||||
}
|
||||
if (session.branch() != null) {
|
||||
m.put("branch", session.branch());
|
||||
}
|
||||
if (session.ownerTerminal() != null) {
|
||||
m.put("owner", session.ownerTerminal());
|
||||
}
|
||||
m.put("liveStatus", live == null ? "unknown" : live.status().name().toLowerCase());
|
||||
return m;
|
||||
}
|
||||
|
||||
/** Lifecycle hook: worker became available on the bridge MCP. */
|
||||
void onReady(String terminalId) {
|
||||
transitionByTerminal(terminalId, MemberSession.State.SPAWNING, MemberSession.State.READY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Lifecycle hook: a message was delivered into the worker — it is now busy on a turn.
|
||||
* The turn count is bumped and the activity timestamp is refreshed. A {@code DONE} session
|
||||
* can be re-delivered for multi-turn reuse until it is released.
|
||||
*/
|
||||
@Override
|
||||
public void onDelivered(String target) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) return;
|
||||
if (current.state() != MemberSession.State.READY && current.state() != MemberSession.State.DONE) {
|
||||
return;
|
||||
}
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession updated = current.withState(MemberSession.State.BUSY).bumpTurn(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} {} -> BUSY turn={}",
|
||||
target, current.paneId(), current.state(), updated.turnCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** Lifecycle hook: the worker's delegated turn completed successfully. */
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completeTurn(target, false);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
if (!clearAfterTurn) return false;
|
||||
MemberSession current = findByTerminal(target);
|
||||
return current != null && current.state() == MemberSession.State.BUSY
|
||||
&& (contextCap <= 0 || current.turnCount() < contextCap);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
return completeTurn(target, true);
|
||||
}
|
||||
|
||||
private boolean completeTurn(String target, boolean startContextReset) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) return false;
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession updated = current.withState(MemberSession.State.DONE).withActivity(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
}
|
||||
if (contextCap > 0 && updated.turnCount() >= contextCap) {
|
||||
release(current.paneId());
|
||||
return false;
|
||||
}
|
||||
if (!startContextReset || !clearAfterTurn) return false;
|
||||
try {
|
||||
return launcher.clearContext(current.paneId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("context reset failed for terminal={} pane={}; continuing without reset: {}",
|
||||
target, current.paneId(), e.getMessage());
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Lifecycle hook: the worker's delegated turn failed. */
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
onFailed(target);
|
||||
}
|
||||
|
||||
/** Lifecycle hook: the worker vanished or was dropped mid-life. */
|
||||
void onFailed(String target) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) return;
|
||||
if (current.state() == MemberSession.State.RELEASED) return;
|
||||
if (replace(current, current.withState(MemberSession.State.FAILED))) {
|
||||
log.debug("session marked failed terminal={} pane={}", target, current.paneId());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort reap of sessions that have been idle longer than {@code idleTtlNanos}. Only
|
||||
* {@code READY} and {@code DONE} sessions are eligible — never a {@code SPAWNING} or
|
||||
* {@code BUSY} worker. Returns the number of sessions released.
|
||||
*/
|
||||
int reapIdle(long idleTtlNanos) {
|
||||
long now = nowNanos.getAsLong();
|
||||
int reaped = 0;
|
||||
for (MemberSession s : roster()) {
|
||||
if (s.state() != MemberSession.State.READY && s.state() != MemberSession.State.DONE) {
|
||||
continue;
|
||||
}
|
||||
if (now - s.lastActivityAtNanos() > idleTtlNanos) {
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
}
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/**
|
||||
* Gracefully drain all registered sessions on daemon shutdown. For each session that is
|
||||
* {@code BUSY}, poll up to {@code timeoutNanos} for it to leave {@code BUSY}, then release it
|
||||
* regardless. Non-busy sessions are released immediately. A failure releasing one session is
|
||||
* logged and does not abort the rest.
|
||||
*
|
||||
* <p>CB-544: this is a {@link ReleaseCause#SHUTDOWN} release — the worker's pane is stopped
|
||||
* (the process must end) but its worktree is preserved and its path logged. Shutdown is never
|
||||
* a reason to delete a worker's only copy of its uncommitted work. A session still {@code BUSY}
|
||||
* when the timeout expired is abandoned mid-turn and logged loudly so an operator can find its
|
||||
* kept worktree.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
for (MemberSession s : roster()) {
|
||||
try {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
while (System.nanoTime() < deadline) {
|
||||
MemberSession current = registry.get(s.paneId());
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) {
|
||||
break;
|
||||
}
|
||||
try {
|
||||
long remaining = deadline - System.nanoTime();
|
||||
Thread.sleep(Math.min(TimeUnit.NANOSECONDS.toMillis(remaining), 50));
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
release(s.paneId(), ReleaseCause.SHUTDOWN);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("drain failed for pane={}; continuing with remaining sessions", s.paneId(), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Close this manager by draining all sessions. The timeout comes from configuration when set,
|
||||
* otherwise a sensible default.
|
||||
*/
|
||||
public void close(Integer drainTimeoutSeconds) {
|
||||
int seconds = (drainTimeoutSeconds != null && drainTimeoutSeconds > 0) ? drainTimeoutSeconds : 5;
|
||||
drainAll(TimeUnit.SECONDS.toNanos(seconds));
|
||||
}
|
||||
|
||||
/** Number of sessions currently registered. */
|
||||
public int size() {
|
||||
return registry.size();
|
||||
}
|
||||
|
||||
/**
|
||||
* The registered session owning {@code terminalId}, or {@code null} if none does.
|
||||
*
|
||||
* <p>A null {@code terminalId} is a normal input, not a caller bug: every lifecycle hook here is
|
||||
* fed from the MCP transport, where the <em>primary</em> resolves to a {@link
|
||||
* dev.ltms.bridged.auth.Principal} with no terminal. {@code BridgeMcp} documents that contact as
|
||||
* a no-op, and {@link dev.ltms.bridged.inject.MemberPresence#markPresent} honours it — but
|
||||
* {@code PresenceBridge} then forwards the same null here. Matching on a null id can never
|
||||
* succeed anyway (a registered session always has a terminal), so answer "no match" rather than
|
||||
* throwing: an NPE on this path takes down an unrelated tool call for the primary.
|
||||
*/
|
||||
private MemberSession findByTerminal(String terminalId) {
|
||||
if (terminalId == null) return null;
|
||||
for (MemberSession s : registry.values()) {
|
||||
if (terminalId.equals(s.terminalId())) return s;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private void transitionByTerminal(String terminalId, MemberSession.State from,
|
||||
MemberSession.State to) {
|
||||
MemberSession current = findByTerminal(terminalId);
|
||||
if (current == null || current.state() != from) return;
|
||||
long now = nowNanos.getAsLong();
|
||||
if (replace(current, current.withState(to).withActivity(now))) {
|
||||
log.debug("session transitioned terminal={} pane={} {} -> {}",
|
||||
terminalId, current.paneId(), from, to);
|
||||
}
|
||||
}
|
||||
|
||||
private boolean replace(MemberSession expected, MemberSession updated) {
|
||||
return registry.replace(expected.paneId(), expected, updated);
|
||||
}
|
||||
|
||||
/** MemberPresence bridge that also drives the manager's READY transition. */
|
||||
private static final class PresenceBridge extends MemberPresence {
|
||||
private final SessionManager sessions;
|
||||
|
||||
PresenceBridge(SessionManager sessions) {
|
||||
this.sessions = sessions;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void markPresent(String terminal) {
|
||||
if (terminal == null || terminal.isBlank()) {
|
||||
return; // the primary's contact carries no worker terminal — not a readiness signal
|
||||
}
|
||||
super.markPresent(terminal);
|
||||
sessions.onReady(terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,70 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Periodic virtual-thread reaper that tears down {@code READY}/{@code DONE} sessions which have
|
||||
* exceeded their idle TTL. Modeled on {@link dev.ltms.bridged.inject.StatusPoller}: a single
|
||||
* virtual-thread loop, idempotent start/stop, and no {@code ScheduledExecutorService}.
|
||||
*/
|
||||
public final class SessionReaper {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(SessionReaper.class);
|
||||
private static final long DEFAULT_INTERVAL_MILLIS = 5000;
|
||||
|
||||
private final SessionManager sessions;
|
||||
private final long idleTtlNanos;
|
||||
private final long intervalMillis;
|
||||
private volatile boolean running;
|
||||
private Thread thread;
|
||||
|
||||
/** Construct a reaper with the default 5-second polling interval. */
|
||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds) {
|
||||
this(sessions, idleTtlSeconds, DEFAULT_INTERVAL_MILLIS);
|
||||
}
|
||||
|
||||
/** Construct a reaper with an explicit polling interval (useful for tests). */
|
||||
public SessionReaper(SessionManager sessions, long idleTtlSeconds, long intervalMillis) {
|
||||
this.sessions = sessions;
|
||||
this.idleTtlNanos = TimeUnit.SECONDS.toNanos(idleTtlSeconds);
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
/** Start the reaper loop on a virtual thread. Idempotent. */
|
||||
public synchronized void start() {
|
||||
if (running) return;
|
||||
running = true;
|
||||
thread = Thread.ofVirtual().name("session-reaper").start(this::loop);
|
||||
log.info("session reaper started (idle ttl {}s, interval {}ms)",
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos), intervalMillis);
|
||||
}
|
||||
|
||||
private void loop() {
|
||||
while (running) {
|
||||
try {
|
||||
sessions.reapIdle(idleTtlNanos);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("session reaper iteration failed; continuing", e);
|
||||
}
|
||||
sleep();
|
||||
}
|
||||
}
|
||||
|
||||
private void sleep() {
|
||||
try {
|
||||
Thread.sleep(intervalMillis);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
running = false;
|
||||
}
|
||||
}
|
||||
|
||||
/** Stop the reaper loop. Idempotent. */
|
||||
public synchronized void stop() {
|
||||
running = false;
|
||||
if (thread != null) thread.interrupt();
|
||||
}
|
||||
}
|
||||
@@ -1,18 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/** Seam between {@link SessionManager} and git worktree operations. Tests use a recording fake. */
|
||||
public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree add <path> -b <branch> <baseRef|HEAD>. Returns the worktree path. */
|
||||
String add(String repoRoot, String branch, String baseRef);
|
||||
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/** Copy each existing overlay path repoRoot→worktree; mark tracked ones --skip-worktree. */
|
||||
void overlayParity(String repoRoot, String worktreePath, List<String> overlay);
|
||||
|
||||
/** git -C <cwd> rev-parse --show-toplevel — the repo root that owns cwd. */
|
||||
String repoRoot(String cwd);
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,27 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** Unit tests for PID → pane resolution (the herdr half of connection-based MCP identity). */
|
||||
class PaneLocatorTest {
|
||||
|
||||
private final PaneLocator loc = new PaneLocator(new FakeHerdr());
|
||||
|
||||
@Test
|
||||
void resolvesTerminalForAForegroundPid() {
|
||||
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForAPidInNoPane() {
|
||||
assertNull(loc.terminalForPid(999_999));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForNonPositivePid() {
|
||||
assertNull(loc.terminalForPid(0));
|
||||
assertNull(loc.terminalForPid(-1));
|
||||
}
|
||||
}
|
||||
@@ -1,304 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
||||
class CompletionResolverTest {
|
||||
|
||||
@Test
|
||||
void skipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
resolver.resolve("term_a", null); // no in-flight turn captured for this target
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a turn nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failSkipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
resolver.fail("term_a", null); // no in-flight turn, and no registered waiter to fall back to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a wedge nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void captureBaselineSkipsTheReadWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ X\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
resolver.captureBaseline("term_a"); // no send to attribute a later completion to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"with no waiting send there is no turn to baseline — skip the scrape");
|
||||
}
|
||||
|
||||
// --- CB-115 clean scrape: extract the last assistant block ----------------
|
||||
|
||||
@Test
|
||||
void extractsTheLastAssistantBlockStrippingChrome() {
|
||||
String raw = """
|
||||
⏺ Reading the file…
|
||||
|
||||
⏺ Done. The bug was an off-by-one in the loop bound.
|
||||
|
||||
╭──────────────────────────────────────╮
|
||||
│ > │
|
||||
╰──────────────────────────────────────╯
|
||||
⏵⏵ auto mode on · ? for shortcuts
|
||||
""";
|
||||
assertEquals("Done. The bug was an off-by-one in the loop bound.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void keepsMultiLineAssistantContent() {
|
||||
String raw = "⏺ Line one.\nLine two.\n❯ ";
|
||||
assertEquals("Line one.\nLine two.", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void fallsBackToRawTextWhenThereIsNoMarker() {
|
||||
String raw = "plain worker output with no glyph";
|
||||
assertEquals("plain worker output with no glyph", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void blankScrapeYieldsEmpty() {
|
||||
assertTrue(CompletionResolver.lastAssistantBlock("").isEmpty());
|
||||
assertTrue(CompletionResolver.lastAssistantBlock(null).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void stripsSpinnerAndRuleChrome() {
|
||||
String raw = """
|
||||
⏺ Channel check confirmed — your message got through.
|
||||
|
||||
✻ Brewed for 11s
|
||||
|
||||
─────────────────────────────────────
|
||||
""";
|
||||
assertEquals("Channel check confirmed — your message got through.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void cutsANextTurnPromptEchoAndTrailingTipsFromTheBlock() {
|
||||
// The exact turn-2 leak: the scrape captured the settled answer, then a "✻ Cooked" spinner,
|
||||
// then the NEXT turn's echoed prompt, then a "✶ Forming…" spinner and trailing tips/warnings
|
||||
// whose lines (⎿, ⚠) are not themselves chrome-terminated. Stopping at the first boundary
|
||||
// (the ✻ spinner) is what keeps every one of those interface lines out of the reply.
|
||||
String raw = """
|
||||
⏺ Channel confirmed — the bridge reply delivered successfully.
|
||||
|
||||
✻ Cooked for 9s
|
||||
|
||||
❯ Thanks. Now a small task: what is 17 * 23? Show just the number.
|
||||
|
||||
|
||||
|
||||
✶ Forming…
|
||||
⎿ Tip: Name your conversations with /rename
|
||||
⚠ claude.ai connectors are disabled because ANTHROPIC_API_KEY is set
|
||||
""";
|
||||
assertEquals("Channel confirmed — the bridge reply delivered successfully.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
// --- CB-115 misattribution guard: suppress a stale (unchanged) completion -------
|
||||
|
||||
@Test
|
||||
void suppressesACompletionWhoseScrapeIsUnchangedFromDelivery() {
|
||||
// Rapid back-to-back turn: the pane still shows the PREVIOUS turn's answer when this turn's
|
||||
// (misattributed) completion boundary fires. The scrape == the delivery baseline, so the
|
||||
// send must NOT be resolved with the stale answer.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ 391\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
// The turn as captured at delivery: its waiter, and the previous turn's answer still on screen.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn); // scrape still "391" == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(), "a completion with no output change must not resolve the send");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesACompletionWhoseScrapeChangedSinceDelivery() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ No, 391 = 17 × 23.\n❯ "); // the worker's real answer
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
// Delivery baseline was the previous turn's "391"; the scrape now differs → resolve.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(), "a completion with new output must resolve the send");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("No, 391 = 17 × 23.", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesSynchronouslyBeforePostTurnContextClearing() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ previous answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.captureBaseline("term_a");
|
||||
herdr.readText("⏺ answer that /clear would erase\n❯ ");
|
||||
|
||||
resolver.resolveBeforePostAction("term_a");
|
||||
|
||||
assertTrue(waiter.isDone(), "the answer is captured before the adapter sends /clear");
|
||||
assertEquals("answer that /clear would erase", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void suppressesAnUnchangedCompletionEvenWhenTheBlockExceedsTheScrapeCap() {
|
||||
// The fan-out issue-hunt finding: captureBaseline once stored the RAW (unclipped) assistant
|
||||
// block while resolve compares against a clip()'d tail. For a block longer than MAX_SCRAPE_CHARS
|
||||
// the two capped representations differ even when the pane never changed, so the CB-115
|
||||
// byte-identical guard failed to fire and a stale completion could resolve the send. Both sides
|
||||
// must clip identically; here an unchanged >cap block on rapid back-to-back turns stays suppressed.
|
||||
String longBlock = "⏺ " + "x".repeat(CompletionResolver.MAX_SCRAPE_CHARS + 500) + "\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(longBlock);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
resolver.captureBaseline("term_a"); // baseline is the clipped >cap block
|
||||
var turn = resolver.inFlight("term_a");
|
||||
assertEquals(CompletionResolver.MAX_SCRAPE_CHARS, turn.baseline().length(),
|
||||
"the delivery baseline is clipped to the same cap resolve() applies to the tail");
|
||||
|
||||
resolver.resolve("term_a", turn); // scrape unchanged → clipped tail == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(),
|
||||
"an unchanged >cap block must still be recognised as stale and suppressed");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenThereIsNoBaseline() {
|
||||
// No delivery baseline (e.g. the pre-turn read failed) ⇒ never suppress; the completion resolves.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ hello\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "with no baseline a completion resolves as before");
|
||||
assertEquals("hello", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenTheScrapeItselfFailsEvenWithABaselinePresent() {
|
||||
// The most important branch of the CB-115 guard: a failed read means the resolver could not
|
||||
// SEE the screen — "couldn't see", not "no change". It must still resolve the send (an empty
|
||||
// tail beats hanging until the caller's timeout), even though a baseline was captured. The
|
||||
// baseline here is "" (an empty pane at delivery), so without the !scrapeFailed clause the
|
||||
// byte-identical guard would wrongly match the empty tail and suppress.
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false); // agent.read throws HerdrException
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, ""); // empty pane baselined at delivery
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(),
|
||||
"a failed scrape must still resolve the send, not hang until the caller's timeout");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("", waiter.getNow(null).text(), "the tail is empty because the screen was unreadable");
|
||||
}
|
||||
|
||||
// --- CB-115/CB-116 fail guard: an already-done or absent waiter is left alone ---------
|
||||
|
||||
@Test
|
||||
void failLeavesAnAlreadyResolvedWaiterUntouchedAndSkipsTheScrape() {
|
||||
// The send was already resolved (e.g. by the worker's explicit reply) before fail fired.
|
||||
// fail must not overwrite that value, and must not even scrape the worker — nobody needs it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied"));
|
||||
|
||||
resolver.fail("term_a", turn);
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"fail must not scrape a waiter that is already done");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"fail must not overwrite the existing resolution");
|
||||
assertEquals("already replied", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void failFallsBackToTheRegisteredWaiterWhenThereIsNoInFlightTurn() {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fail falls back to the waiter currently registered on the Rendezvous and fails it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // send registered, but no captureBaseline ever ran
|
||||
resolver.fail("term_a", null); // no in-flight turn → fall back to the registered waiter
|
||||
|
||||
assertTrue(waiter.isDone(), "fail falls back to the registered waiter when no turn is in flight");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals("stuck on an error screen", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
// --- CB-116 waiter identity: a late completion never crosses into the next turn ---------
|
||||
|
||||
@Test
|
||||
void aLateCompletionForOneTurnNeverResolvesTheNextTurnsWaiter() {
|
||||
// The cross-turn stale reply the conversation test surfaced: turn N's completion fallback
|
||||
// fires AFTER turn N was resolved by an explicit bridge_reply and turn N+1 has opened its own
|
||||
// waiter on the same session. Resolving "whatever is waiting now" would hand turn N's stale
|
||||
// scrape to turn N+1; targeting turn N's captured waiter makes the late completion a no-op.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ turn N answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous);
|
||||
|
||||
var waiterN = rendezvous.open("term_a"); // turn N's send
|
||||
// The turn as the injector captured it at delivery (waiter + pre-turn baseline).
|
||||
var turnN = new CompletionResolver.InFlight(waiterN, "an earlier answer");
|
||||
|
||||
// Turn N is resolved by the worker's explicit reply, and its send deregisters the waiter.
|
||||
assertTrue(rendezvous.resolve("term_a", "N replied"));
|
||||
rendezvous.close("term_a", waiterN); // the sender's finally, before the next turn opens
|
||||
|
||||
// Turn N+1's send opens its own waiter on the same session (CB-548: open fails if the
|
||||
// previous waiter is still registered, so a clean turn deregisters it first as above).
|
||||
var waiterN1 = rendezvous.open("term_a");
|
||||
|
||||
resolver.resolve("term_a", turnN); // turn N's completion fallback finally fires
|
||||
|
||||
assertFalse(waiterN1.isDone(), "turn N's late completion must not resolve turn N+1's waiter");
|
||||
assertEquals(Rendezvous.Kind.REPLY, waiterN.getNow(null).kind(),
|
||||
"turn N stays resolved by its own reply");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "turn N+1 is still awaiting its own resolution");
|
||||
}
|
||||
}
|
||||
@@ -1,201 +0,0 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.auth.Authz;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.auth.Role;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.session.FakeWorktrees;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-513 — the CB-505 authorization gate on the <strong>MCP</strong> entry path.
|
||||
*
|
||||
* <p>Why this file exists: CB-505 claimed authorization is "enforced on both entry paths", and it
|
||||
* is — but only REST was ever tested ({@code BridgedAppAuthTest}). Coverage showed
|
||||
* {@code BridgeMcp.deny()}, {@code principal()} and every tool-registration lambda at <em>zero</em>
|
||||
* executed lines, because no test had ever constructed a {@code BridgeMcp} — the existing
|
||||
* {@code BridgeMcpTest} calls only the static handler methods. An unexercised security control is
|
||||
* a claim, not a control.
|
||||
*
|
||||
* <p>These tests construct a real {@code BridgeMcp} (which also exercises the constructor and the
|
||||
* tool wiring) and drive the policy half of the gate directly.
|
||||
*/
|
||||
class BridgeMcpAuthzTest {
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private Metrics metrics;
|
||||
private BridgeMcp mcp;
|
||||
|
||||
@AfterEach
|
||||
void close() {
|
||||
if (mcp != null) mcp.close();
|
||||
}
|
||||
|
||||
/** A fully wired BridgeMcp on fakes — constructing it is itself part of what is under test. */
|
||||
private BridgeMcp mcp(boolean enforce) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN", null,
|
||||
"tab", "bridged-workers", "worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(agents, new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "tok");
|
||||
SessionManager sessions = new SessionManager(workers, new FakeWorktrees());
|
||||
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(),
|
||||
new InMemoryReplyInbox());
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> 999_999);
|
||||
metrics = BridgedMetrics.create(sessions, new InMemoryReplyInbox());
|
||||
|
||||
mcp = new BridgeMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||
new PrimaryRegistry(null),
|
||||
enforce ? new CallerResolver(identity) : null,
|
||||
metrics);
|
||||
return mcp;
|
||||
}
|
||||
|
||||
private static final Principal PRIMARY = Principal.primary(100);
|
||||
private static final Principal WORKER_A = Principal.worker("term_a", 200);
|
||||
private static final Principal ANON = Principal.anonymous();
|
||||
private static final Principal ARCH_DESIGN = Principal.architect("lead-designer", "term_design", 400);
|
||||
|
||||
// --- the table, enforced on THIS path too ---------------------------------------------------
|
||||
|
||||
@Test
|
||||
void primaryMayOrchestrate() {
|
||||
BridgeMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.SEND, Authz.Action.DRAIN, Authz.Action.READ}) {
|
||||
assertNull(m.denyFor(PRIMARY, a, "term_a"), a + " is the primary's to perform");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayNotOrchestrateOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.SEND, Authz.Action.DRAIN}) {
|
||||
McpSchema.CallToolResult denied = m.denyFor(WORKER_A, a, "term_a");
|
||||
assertNotNull(denied, a + " must be refused to a worker");
|
||||
assertTrue(denied.isError(), "a refusal is returned as an MCP tool error");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayReplyAndAskOnlyAsItself() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.REPLY, "term_a"), "its own session is allowed");
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.ASK, "term_a"));
|
||||
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.REPLY, "term_b"),
|
||||
"worker A must not reply on worker B's session");
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.ASK, "term_b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void thePrimaryMayNotForgeAWorkerReplyOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
// A forged reply would resolve the very rendezvous the primary is blocked on.
|
||||
assertNotNull(m.denyFor(PRIMARY, Authz.Action.REPLY, "term_a"));
|
||||
assertNotNull(m.denyFor(PRIMARY, Authz.Action.ASK, "term_a"));
|
||||
}
|
||||
|
||||
// --- CB-548: the architect on this path ------------------------------------------------
|
||||
|
||||
@Test
|
||||
void anArchitectMaySendAndReadButNotOrchestrateOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.SEND, "term_a"),
|
||||
"delegating a turn is the architect's job");
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.READ, null));
|
||||
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.DRAIN}) {
|
||||
McpSchema.CallToolResult denied = m.denyFor(ARCH_DESIGN, a, null);
|
||||
assertNotNull(denied, a + " must be refused to an architect");
|
||||
assertTrue(denied.isError(), "a refusal is returned as an MCP tool error");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReplyAndAskOnlyAsItsOwnPaneOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.REPLY, "term_design"));
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.ASK, "term_design"));
|
||||
|
||||
assertNotNull(m.denyFor(ARCH_DESIGN, Authz.Action.REPLY, "term_a"),
|
||||
"architect 'lead-designer' must not reply on worker term_a's session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anonymousIsRefusedEverythingAndCountedAsUnauthenticated() {
|
||||
BridgeMcp m = mcp(true);
|
||||
McpSchema.CallToolResult denied = m.denyFor(ANON, Authz.Action.READ, null);
|
||||
|
||||
assertNotNull(denied, "authenticated as nothing ⇒ authorized for nothing");
|
||||
assertEquals(1, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "unauthenticated"));
|
||||
assertEquals(0, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "forbidden"),
|
||||
"a missing credential is 401-shaped, not 403-shaped");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWrongRoleIsCountedAsForbiddenNotUnauthenticated() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.SPAWN, null));
|
||||
|
||||
assertEquals(1, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "forbidden"));
|
||||
assertEquals(0, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "unauthenticated"),
|
||||
"the caller IS authenticated — it is just not the right role");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theLegacyConstructorLeavesTheGateOpen() {
|
||||
// The 22 pre-existing BridgeMcpTest cases rely on no authorization being enforced.
|
||||
BridgeMcp m = mcp(false);
|
||||
assertNull(m.denyFor(ANON, Authz.Action.SPAWN, null),
|
||||
"no CallerResolver supplied ⇒ authorization not enforced (legacy behaviour)");
|
||||
}
|
||||
|
||||
// --- identity reconstruction from the transport context ------------------------------------
|
||||
|
||||
@Test
|
||||
void principalIsRebuiltFromTheStashedRole() {
|
||||
assertEquals(Role.WORKER, BridgeMcp.principalFrom("WORKER", "term_a", 7).role());
|
||||
assertEquals("term_a", BridgeMcp.principalFrom("WORKER", "term_a", 7).terminal());
|
||||
assertEquals(Role.PRIMARY, BridgeMcp.principalFrom("PRIMARY", null, 7).role());
|
||||
assertEquals(Role.ANONYMOUS, BridgeMcp.principalFrom("ANONYMOUS", null, -1).role());
|
||||
// CB-548: an architect round-trips through the same stash, carrying its slot name.
|
||||
Principal arch = BridgeMcp.principalFrom("ARCHITECT", "term_design", 7, "lead-designer");
|
||||
assertEquals(Role.ARCHITECT, arch.role());
|
||||
assertEquals("lead-designer", arch.name());
|
||||
assertEquals("term_design", arch.terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingRoleFallsBackToTheHistoricalInterpretation() {
|
||||
// Legacy path: no role stashed. A terminal means worker; its absence meant "the primary",
|
||||
// which is exactly the pre-CB-501 default CB-501 inverted — preserved only here.
|
||||
assertEquals(Role.WORKER, BridgeMcp.principalFrom(null, "term_a", 7).role());
|
||||
assertEquals(Role.PRIMARY, BridgeMcp.principalFrom(null, null, 7).role());
|
||||
}
|
||||
}
|
||||
@@ -1,590 +0,0 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.session.FakeWorktrees;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.WorktreeRequest;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Parity tests for the MCP tool adapters — they must produce the same outcomes as the REST routes,
|
||||
* since both drive the same {@link MessageService}/{@link Rendezvous}. The MCP wire protocol itself
|
||||
* is the SDK's concern; here we test the thin adapter logic directly.
|
||||
*/
|
||||
class BridgeMcpTest {
|
||||
|
||||
private static final String T = "term_a";
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private final Rendezvous rendezvous = new Rendezvous();
|
||||
private final InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
private final MessageService messages = new MessageService(agents, new Injector(agents), rendezvous, inbox);
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
// CB-520: the inbox only peeks/acks targets it owns.
|
||||
inbox.own(T);
|
||||
}
|
||||
|
||||
private static String textOf(McpSchema.CallToolResult r) {
|
||||
return ((McpSchema.TextContent) r.content().getFirst()).text();
|
||||
}
|
||||
|
||||
private static ClaudeCodeLauncher workerService(FakeHerdr h, String baseUrl, Set<String> allow) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", baseUrl, "coder", null, "BRIDGED_WORKER_TOKEN", null,
|
||||
"tab", "bridged-workers", "worker: {profile} #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(allow), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> "tok");
|
||||
}
|
||||
|
||||
private static SessionManager sessionManager(FakeHerdr h, String baseUrl, Set<String> allow) {
|
||||
return new SessionManager(workerService(h, baseUrl, allow));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendThenReplyRoundTrips() throws Exception {
|
||||
// bridge_send blocks; bridge_reply resolves it with the worker's structured answer.
|
||||
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.send(messages, "term_a", "review this", 4000L));
|
||||
|
||||
// Wait until the send has opened its waiter so the reply resolves it (CB-307: reply now
|
||||
// queues in the inbox if no waiter is open, which would break the round-trip).
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "send should have opened its waiter");
|
||||
|
||||
McpSchema.CallToolResult reply = BridgeMcp.reply(messages, "term_a", "LGTM");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
|
||||
McpSchema.CallToolResult res = send.get(6, TimeUnit.SECONDS);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertEquals("LGTM", textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void asyncSendReturnsATicketThenPollReportsTheReply() throws Exception {
|
||||
// wait:false parity — a ticket is issued, resolved by a reply, and surfaced by bridge_poll.
|
||||
McpSchema.CallToolResult accepted = BridgeMcp.sendAsync(messages, "term_a", "do it");
|
||||
assertNotEquals(Boolean.TRUE, accepted.isError());
|
||||
String out = textOf(accepted);
|
||||
assertTrue(out.contains("ticket="), out);
|
||||
String ticket = out.substring(out.indexOf("ticket=") + "ticket=".length()).trim();
|
||||
|
||||
// Wait until the send has opened its waiter before replying (CB-307: reply never errors,
|
||||
// so the old retry-on-error pattern no longer works — it would queue instead of resolve).
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "send should have opened its waiter");
|
||||
|
||||
McpSchema.CallToolResult reply = BridgeMcp.reply(messages, "term_a", "async LGTM");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
|
||||
// Poll until the async send completes and reports the reply.
|
||||
McpSchema.CallToolResult polled = BridgeMcp.poll(messages, ticket, null);
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (!textOf(polled).contains("async LGTM") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(10);
|
||||
polled = BridgeMcp.poll(messages, ticket, null);
|
||||
}
|
||||
assertEquals("async LGTM", textOf(polled));
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollUnknownTicketIsAnError() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.poll(messages, "task-999", null);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("unknown ticket"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendTimesOutWithAWorkingNote() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.send(messages, "term_a", "hi", 120L);
|
||||
assertNotEquals(Boolean.TRUE, res.isError(), "a timeout is informational, not a tool error");
|
||||
assertTrue(textOf(res).contains("no reply"), "got: " + textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendRejectsMissingArgs() {
|
||||
assertTrue(BridgeMcp.send(messages, null, "hi", null).isError());
|
||||
assertTrue(BridgeMcp.send(messages, "term_a", " ", null).isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyWithNoPendingSendIsQueuedNotError() {
|
||||
// CB-307: a reply with no open send is now queued in the inbox, not an error.
|
||||
McpSchema.CallToolResult res = BridgeMcp.reply(messages, "term_a", "orphan");
|
||||
assertNotEquals(Boolean.TRUE, res.isError(), "a queued reply is not an error");
|
||||
assertEquals("delivered", textOf(res));
|
||||
|
||||
// The reply is drainable by target.
|
||||
var drained = messages.drainReplies("term_a");
|
||||
assertEquals(1, drained.size());
|
||||
assertEquals("orphan", drained.getFirst().content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgePollWithTargetDrainsReplies() {
|
||||
// A reply with no open send queues it in the inbox.
|
||||
BridgeMcp.reply(messages, "term_a", "queued-msg");
|
||||
|
||||
// bridge_poll with target drains the inbox.
|
||||
McpSchema.CallToolResult res = BridgeMcp.poll(messages, null, "term_a");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String text = textOf(res);
|
||||
assertTrue(text.contains("queued-msg"), "the drained reply should appear in the result");
|
||||
|
||||
// Second drain returns empty.
|
||||
McpSchema.CallToolResult empty = BridgeMcp.poll(messages, null, "term_a");
|
||||
assertEquals("[]", textOf(empty));
|
||||
}
|
||||
|
||||
@Test
|
||||
void askThenAnswerRoundTrips() throws Exception {
|
||||
// The primary delegates and blocks; wait until its waiter is open before the worker asks.
|
||||
CompletableFuture<McpSchema.CallToolResult> send = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.send(messages, "term_a", "do X", 5000L));
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send must be waiting for the ask to surface to");
|
||||
|
||||
// The worker asks mid-turn; the call blocks for the primary's answer.
|
||||
CompletableFuture<McpSchema.CallToolResult> ask = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.ask(messages, "term_a", "which config?", 5000L));
|
||||
|
||||
// The primary's send unblocks with the question and a turnId to answer on.
|
||||
McpSchema.CallToolResult q = send.get(6, TimeUnit.SECONDS);
|
||||
assertNotEquals(Boolean.TRUE, q.isError());
|
||||
String qt = textOf(q);
|
||||
assertTrue(qt.contains("[question]"), qt);
|
||||
String afterMarker = qt.substring(qt.indexOf("turnId=\"") + "turnId=\"".length());
|
||||
String turnId = afterMarker.substring(0, afterMarker.indexOf('"'));
|
||||
|
||||
// The primary answers via bridge_send(turnId); this blocks again for the worker's reply.
|
||||
CompletableFuture<McpSchema.CallToolResult> answer = CompletableFuture.supplyAsync(
|
||||
() -> BridgeMcp.answer(messages, turnId, "config.yaml", 5000L));
|
||||
|
||||
// The worker's ask returns the answer — it resumes the same turn.
|
||||
assertEquals("config.yaml", textOf(ask.get(6, TimeUnit.SECONDS)));
|
||||
|
||||
// The resumed worker replies, resolving the answering send (wait for the reopened waiter).
|
||||
deadline = System.currentTimeMillis() + 3000;
|
||||
while (!rendezvous.isWaiting("term_a") && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the answer should have reopened a waiter");
|
||||
McpSchema.CallToolResult reply = BridgeMcp.reply(messages, "term_a", "done");
|
||||
assertEquals("delivered", textOf(reply));
|
||||
assertEquals("done", textOf(answer.get(6, TimeUnit.SECONDS)));
|
||||
}
|
||||
|
||||
@Test
|
||||
void askFromANonWorkerConnectionIsAnError() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.ask(messages, null, "which config?", 500L);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("workers only"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void answerToAStaleTurnIsAnError() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.answer(messages, "term_a#999", "too late", 500L);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("no longer open"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsTheNewWorkersSessionAndPane() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sm = sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(sm, null);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_new_1\""), out);
|
||||
// CB-519: the "paneId" wire field now carries the host-unique opaque id, not the herdr pane.
|
||||
MemberSession s = sm.roster().getFirst();
|
||||
assertTrue(out.contains("\"paneId\":\"" + s.paneId() + "\""), out);
|
||||
assertNotEquals("w9:pRoot_1", s.paneId(), "the id is decoupled from the herdr pane coordinate");
|
||||
assertTrue(out.contains("\"status\":\"spawning\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnOffAllowlistProfileWithoutTouchingHerdr() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res =
|
||||
BridgeMcp.spawn(sessionManager(h, "https://api.anthropic.com", Set.of("gx00.gw")), null);
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("subscription boundary"));
|
||||
assertFalse(h.called("agent.start"), "the guard must block before any spawn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnUnknownProfileAsAnError() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res =
|
||||
BridgeMcp.spawn(sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), "nope");
|
||||
assertTrue(res.isError());
|
||||
assertTrue(textOf(res).contains("unknown worker profile"), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnPassesTheRequestedCwdToTheWorker() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, null, "/req/dir", null, null, null);
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
// Protocol 19: the requested cwd roots the worker's pane at creation (tab.create).
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, Object> create = (Map<String, Object>) h.lastCall("tab.create").params();
|
||||
assertEquals("/req/dir", create.get("cwd"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesListsConfiguredProfilesAndDefault() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.profiles(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("ltms-local"), out);
|
||||
assertTrue(out.contains("\"default\":\"ltms-local\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsTrackedWorkers() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary",
|
||||
new WorktreeRequest("cb-304", null));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), "");
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"sessionId\":\"" + s.terminalId() + "\""), out);
|
||||
assertTrue(out.contains("\"paneId\":\"" + s.paneId() + "\""), out);
|
||||
assertTrue(out.contains("\"profile\":\"ltms-local\""), out);
|
||||
assertTrue(out.contains("\"state\":\"spawning\""), out);
|
||||
assertTrue(out.contains("\"worktree\":\"" + s.worktree() + "\""), out);
|
||||
assertTrue(out.contains("\"branch\":\"" + s.branch() + "\""), out);
|
||||
assertTrue(out.contains("\"owner\":\"term_primary\""), out);
|
||||
assertTrue(out.contains("\"liveStatus\":\"unknown\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsLeadsAndFlagsTheCallersOwnRow() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions,
|
||||
Map.of("term_me", "opus-5.0", "term_peer", "gpt-sol-5.6"), "term_me");
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"name\":\"opus-5.0\""), out);
|
||||
assertTrue(out.contains("\"name\":\"gpt-sol-5.6\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_peer\""), out);
|
||||
// The caller's own row is flagged, and only the caller's — a peer must be distinguishable
|
||||
// from self without a second bridge_whoami call.
|
||||
assertEquals(1, out.split("\"self\":true", -1).length - 1, out);
|
||||
assertTrue(out.indexOf("term_me") < out.indexOf("\"self\":true"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsBothHalvesEvenWhenEmpty() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), "");
|
||||
|
||||
// An absent "leads" key is what made an empty member roster read as "no peers" (CB-535).
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"leads\":[]"), out);
|
||||
assertTrue(out.contains("\"members\":[]"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void listReportsALeadHerdrCannotSeeAsUnknown() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions,
|
||||
Map.of("term_ghost", "gone-away"), "term_me");
|
||||
|
||||
// Reported, not hidden: an unreachable peer is exactly what a would-be sender needs to see.
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"name\":\"gone-away\""), out);
|
||||
assertTrue(out.contains("\"status\":\"unknown\""), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopTearsDownAWorkerByPane() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.stop(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), "w9:pW");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertEquals("stopped w9:pW", textOf(res));
|
||||
assertTrue(h.called("pane.close"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopRequiresAPaneId() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
assertTrue(BridgeMcp.stop(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), " ").isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckReturnsConfirmationForValidArgs() {
|
||||
McpSchema.CallToolResult res = BridgeMcp.ack(messages, "term_a", "msg-1");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("msg-1"), "response should mention the msgId");
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckRejectsMissingArgs() {
|
||||
assertTrue(BridgeMcp.ack(messages, null, "msg-1").isError());
|
||||
assertTrue(BridgeMcp.ack(messages, "term_a", null).isError());
|
||||
assertTrue(BridgeMcp.ack(messages, " ", "msg-1").isError());
|
||||
}
|
||||
|
||||
@Test
|
||||
void bridgeAckRemovesSpecificReply() {
|
||||
// Queue a reply and capture its msgId.
|
||||
BridgeMcp.reply(messages, "term_a", "orphan");
|
||||
var before = messages.drainReplies("term_a");
|
||||
assertEquals(1, before.size(), "one reply in the inbox");
|
||||
String msgId = before.getFirst().msgId();
|
||||
|
||||
// Publish the same reply again and ack it via bridge_ack surface.
|
||||
BridgeMcp.reply(messages, "term_a", "orphan-again");
|
||||
var peeked = messages.drainReplies("term_a");
|
||||
assertEquals(1, peeked.size(), "one fresh reply in the inbox");
|
||||
|
||||
// ackReply works (no-op since published with a different UUID, but callable).
|
||||
assertDoesNotThrow(() -> messages.ackReply("term_a", msgId));
|
||||
}
|
||||
|
||||
@Test
|
||||
void statusReportsLiveAgentStatus() {
|
||||
FakeHerdr blocked = new FakeHerdr().agentStatus("blocked");
|
||||
AgentControl blockedAgents = new AgentControl(blocked);
|
||||
McpSchema.CallToolResult res = BridgeMcp.status(
|
||||
new MessageService(blockedAgents, new Injector(blockedAgents), rendezvous), "term_a");
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertEquals("blocked", textOf(res));
|
||||
}
|
||||
|
||||
// --- bridge_whoami: the caller's own identity, so an agent never has to guess its role -------
|
||||
|
||||
@Test
|
||||
void whoamiReportsThePrimaryAsPrimaryAndNothingElse() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(
|
||||
Principal.primary(100), sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"primary\""), out);
|
||||
// The primary owns no session — leaking a sessionId here would invite it to reply as one.
|
||||
assertFalse(out.contains("sessionId"), out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void whoamiReportsAWorkerWithItsRegisteredSession() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary",
|
||||
new WorktreeRequest("cb-517", null));
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(Principal.worker(s.terminalId(), 200), sessions);
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"worker\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"" + s.terminalId() + "\""), out);
|
||||
assertTrue(out.contains("\"profile\":\"ltms-local\""), out);
|
||||
assertTrue(out.contains("\"worktree\":\"" + s.worktree() + "\""), out);
|
||||
assertTrue(out.contains("\"branch\":\"" + s.branch() + "\""), out);
|
||||
assertTrue(out.contains("\"owner\":\"term_primary\""), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker the registry has no record of — it outlived a daemon restart — must still learn the
|
||||
* load-bearing fact. Degrading to "I don't know who you are" would put it back to guessing,
|
||||
* which is the failure this tool exists to remove.
|
||||
*/
|
||||
@Test
|
||||
void whoamiStillReportsWorkerRoleWhenTheSessionIsUnregistered() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(Principal.worker("term_orphan", 200),
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"worker\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_orphan\""), out);
|
||||
assertFalse(out.contains("profile"), out); // nothing invented for a session we don't track
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548: an architect reports its role and which gateway-local slot its pane is bound to —
|
||||
* the same shape as a lead, under the architect key, so it can tell a peer where to reach it.
|
||||
*/
|
||||
@Test
|
||||
void whoamiReportsAnArchitectWithItsSlotAndPane() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.whoami(
|
||||
Principal.architect("lead-designer", "term_design", 400),
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"role\":\"architect\""), out);
|
||||
assertTrue(out.contains("\"architect\":\"lead-designer\""), out);
|
||||
assertTrue(out.contains("\"sessionId\":\"term_design\""), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548: an architect SEND delegates as its own pane (recording the per-target delegation) but
|
||||
* must NEVER become the legacy singleton "primary" fallback — the per-target map does not cure
|
||||
* the singleton, so an architect left there would draw no-delegation inbox nudges meant for a
|
||||
* primary. Only PRIMARY callers (the unnamed primary and named leads alike) may claim it, and
|
||||
* the decision keys on the resolved role, not name/kind sniffing.
|
||||
*/
|
||||
@Test
|
||||
void architectSendDoesNotClaimThePrimarySingletonButALeadSendStillCan() {
|
||||
// Architect SEND: does not change the legacy primary fallback.
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(reg, "term_design", Principal.architect("design", "term_design", 400));
|
||||
assertTrue(reg.primaryTerminal().isEmpty(),
|
||||
"an architect must never become the legacy primary fallback");
|
||||
|
||||
// Lead SEND (a named PRIMARY) still claims it — preserved from CB-530/CB-532.
|
||||
PrimaryRegistry leadReg = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(leadReg, "term_lead_opus", Principal.leader("opus", "term_lead_opus", 100));
|
||||
assertEquals("term_lead_opus", leadReg.primaryTerminal().orElseThrow(),
|
||||
"a named lead is a primary and may claim the fallback");
|
||||
|
||||
// Unnamed primary likewise.
|
||||
PrimaryRegistry primaryReg = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(primaryReg, "term_p", Principal.primary(50));
|
||||
assertEquals("term_p", primaryReg.primaryTerminal().orElseThrow(),
|
||||
"an unnamed primary may claim the fallback");
|
||||
|
||||
// A null caller (legacy/no-auth path) records nothing.
|
||||
PrimaryRegistry legacy = new PrimaryRegistry(null);
|
||||
BridgeMcp.recordPrimarySingleton(legacy, "term_x", null);
|
||||
assertTrue(legacy.primaryTerminal().isEmpty(), "no caller means nothing is recorded");
|
||||
}
|
||||
|
||||
// --- member taxonomy (CB-557) ---------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void bridgeListReportsMembersNotWorkersAndCarriesEachRole() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
sessions.acquire("ltms-local", MemberRole.REVIEWER, null, "/caller/proj", "term_primary", null);
|
||||
|
||||
McpSchema.CallToolResult res = BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), "");
|
||||
|
||||
String out = textOf(res);
|
||||
assertTrue(out.contains("\"members\":"), "the roster half is named members: " + out);
|
||||
assertFalse(out.contains("\"workers\":"), "the old key must be gone: " + out);
|
||||
assertTrue(out.contains("\"role\":\"reviewer\""), out);
|
||||
}
|
||||
|
||||
/**
|
||||
* Role and profile are separate axes, so the roster has to report both. Two members on one
|
||||
* backend may still be allowed to do entirely different things.
|
||||
*/
|
||||
@Test
|
||||
void aRosterRowCarriesBothItsRoleAndItsProfile() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
SessionManager sessions = new SessionManager(workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")));
|
||||
sessions.acquire("ltms-local", MemberRole.DEV, null, "/caller/proj", "term_primary", null);
|
||||
sessions.acquire("ltms-local", MemberRole.REVIEWER, null, "/caller/proj", "term_primary", null);
|
||||
|
||||
String out = textOf(BridgeMcp.listFleet(
|
||||
workerService(h, "http://gx00.gw:8000", Set.of("gx00.gw")), sessions, Map.of(), ""));
|
||||
|
||||
assertTrue(out.contains("\"role\":\"dev\""), out);
|
||||
assertTrue(out.contains("\"role\":\"reviewer\""), out);
|
||||
assertEquals(2, out.split("\"profile\":\"ltms-local\"", -1).length - 1,
|
||||
"both members share one profile — that is the point: " + out);
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnDefaultsTheRoleToDev() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, null, null, null, null, null);
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("\"role\":\"dev\""), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnAcceptsAnExplicitRole() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, "architect",
|
||||
null, null, null, null);
|
||||
|
||||
assertNotEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("\"role\":\"architect\""), textOf(res));
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnUnknownRoleAndNamesTheValidOnes() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
McpSchema.CallToolResult res = BridgeMcp.spawn(
|
||||
sessionManager(h, "http://gx00.gw:8000", Set.of("gx00.gw")), null, "worker",
|
||||
null, null, null, null);
|
||||
|
||||
assertEquals(Boolean.TRUE, res.isError());
|
||||
assertTrue(textOf(res).contains("architect, dev, reviewer"), textOf(res));
|
||||
}
|
||||
}
|
||||
@@ -1,881 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.GuardException;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** The step-4 launch-flag injection: the bridge MCP + reply charter are appended to the argv. */
|
||||
class ClaudeCodeLauncherTest {
|
||||
|
||||
private ClaudeCodeLauncher service(FakeHerdr herdr, List<String> argv, String mcpUrl) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
argv, "tab", "bridged-workers", "worker: {profile} #{n}", mcpUrl, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
}
|
||||
|
||||
/** The {@code args} of the last agent.start — protocol 19: everything after the executable. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private List<String> spawnedArgs(FakeHerdr herdr) {
|
||||
return (List<String>) ((Map<String, Object>) herdr.lastCall("agent.start").params()).get("args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void appendsBridgeMcpAndReplyCharterWhenMcpUrlSet() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("claude"), "http://127.0.0.1:8765/mcp").spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertTrue(args.contains("--mcp-config"));
|
||||
assertTrue(args.stream().anyMatch(a -> a.contains("\"bridge\"") && a.contains("http://127.0.0.1:8765/mcp")),
|
||||
"inline bridge MCP config present");
|
||||
assertTrue(args.contains("--append-system-prompt"));
|
||||
assertTrue(args.stream().anyMatch(a -> a.contains("bridge_reply")), "reply charter present");
|
||||
}
|
||||
|
||||
@Test
|
||||
void startRetriesWhileTheSeedShellBoots() {
|
||||
// tab.create returns before the seed shell reaches its prompt; herdr refuses agent.start
|
||||
// into a not-ready pane with agent_pane_busy. The launcher must wait it out, not fail.
|
||||
FakeHerdr herdr = new FakeHerdr().agentPaneBusyTimes(2);
|
||||
long[] clock = {0};
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
Map.of("ltms-local", new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null)),
|
||||
"ltms-local", _ -> null,
|
||||
0, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn succeeds once the shell is ready");
|
||||
assertEquals(3, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"two busy rejections, then the successful start");
|
||||
}
|
||||
|
||||
@Test
|
||||
void startResolvesTheExecutableFromKindAndDropsArgvZero() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("claude"), null).spawn();
|
||||
|
||||
Map<?, ?> start = (Map<?, ?>) herdr.lastCall("agent.start").params();
|
||||
assertEquals("claude", start.get("kind"), "herdr launches the canonical executable by kind");
|
||||
// CB-533: the shared fixture pins model "coder", so the model flag is the whole args list.
|
||||
// What this test guards is that argv[0] is NOT repeated — herdr supplies it from `kind`.
|
||||
assertEquals(List.of("--model", "coder"), start.get("args"),
|
||||
"the configured executable is not repeated in args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noBridgeFlagsWhenMcpUrlAbsent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("claude", "--verbose"), null).spawn();
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("--mcp-config"), "no bridge mount without mcpUrl");
|
||||
assertFalse(args.contains("--append-system-prompt"), "no reply charter without mcpUrl");
|
||||
// CB-533: the model flag is independent of the MCP mount — pinning the model is not part of
|
||||
// "mount the bridge", so an unmounted worker still runs the model its profile names.
|
||||
assertEquals(List.of("--verbose", "--model", "coder"), args,
|
||||
"the operator's own args are preserved, in order, ahead of the model flag");
|
||||
}
|
||||
|
||||
private ClaudeCodeLauncher multiProfile(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile gx10 = new BridgedConfig.Profile("gx10", "http://gx10.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
BridgedConfig.Profile ollama = new BridgedConfig.Profile("ollama", "http://ollama.ltms.dev", null,
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx10.gw", "ollama.ltms.dev")),
|
||||
Map.of("gx10", gx10, "ollama", ollama), "gx10", _ -> "tok");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnPicksTheNamedProfilesBaseUrl() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
multiProfile(herdr).spawn("ollama");
|
||||
|
||||
assertEquals("http://ollama.ltms.dev", startEnv(herdr).get("ANTHROPIC_BASE_URL"),
|
||||
"the named profile's base_url");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRejectsAnUnknownProfile() {
|
||||
try (FakeHerdr herdr = new FakeHerdr()) {
|
||||
assertThrows(IllegalArgumentException.class, () -> multiProfile(herdr).spawn("nope"));
|
||||
}
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static String startCwd(FakeHerdr herdr) {
|
||||
// Protocol 19: the worker's cwd is set at pane creation (tab.create), where the seed
|
||||
// shell — which the agent starts into — is rooted.
|
||||
Object v = ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("cwd");
|
||||
return v == null ? null : v.toString();
|
||||
}
|
||||
|
||||
@Test
|
||||
void requestedCwdRootsTheWorker() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("ccs", "ltms-local"), null).spawn("ltms-local", "/work/proj", "/caller/home");
|
||||
assertEquals("/work/proj", startCwd(herdr), "an explicit spawn cwd wins over everything");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profileConfigCwdBeatsTheCallerCwd() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile("ltms-local", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"w #{n}", null, "/pinned/dir", null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("ltms-local", cfg), "ltms-local", _ -> null);
|
||||
svc.spawn("ltms-local", null, "/caller/home");
|
||||
assertEquals("/pinned/dir", startCwd(herdr), "a profile-pinned cwd overrides the caller's");
|
||||
}
|
||||
|
||||
@Test
|
||||
void inheritsTheCallerCwdWhenNothingElseIsSet() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, List.of("ccs", "ltms-local"), null).spawn("ltms-local", null, "/primary/project");
|
||||
assertEquals("/primary/project", startCwd(herdr), "no explicit/config cwd → inherit the primary's");
|
||||
}
|
||||
|
||||
// --- CB-302 git-forge token injection (worker checkpoint grant) ------------
|
||||
|
||||
/** Protocol 19: the worker's env is injected at pane creation (tab.create), not agent.start. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, String> startEnv(FakeHerdr herdr) {
|
||||
return (Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||
}
|
||||
|
||||
@Test
|
||||
void injectsForgeTokenAndHostWhenProfileGrantsIt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"impl", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "impl"), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
"GITEA_ACCESS_TOKEN", null); // parityOverlay null; gitHostEnv null → defaults to GITEA_HOST
|
||||
Function<String, String> host = name -> switch (name) {
|
||||
case "GITEA_ACCESS_TOKEN" -> "gt-secret";
|
||||
case "GITEA_HOST" -> "git.ltms.dev";
|
||||
default -> null;
|
||||
};
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("impl", cfg), "impl", host).spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertEquals("gt-secret", env.get("GITEA_TOKEN"), "the forge token is injected for a granting profile");
|
||||
assertEquals("git.ltms.dev", env.get("GITEA_HOST"), "the paired forge host rides along with the token");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noForgeTokenWhenProfileDoesNotGrantIt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// gitTokenEnv unset (12-arg ctor); the env would resolve a token if asked, proving the gate
|
||||
// is the profile config, not a missing env var.
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile("ltms-local", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("ccs"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("ltms-local", cfg), "ltms-local",
|
||||
_ -> "would-be-secret").spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("GITEA_TOKEN"), "no forge token when the profile does not opt in");
|
||||
assertNull(env.get("GITEA_HOST"), "no forge host without a granted token");
|
||||
}
|
||||
|
||||
// --- CB-117 orphan reap: the pure predicate --------------------------------
|
||||
|
||||
@Test
|
||||
void isForeignWorkerMatchesOurSchemeWithANonSelfNonce() {
|
||||
assertTrue(ClaudeCodeLauncher.isForeignWorker("claude-ollama-be09c2-2", "aaaaaa"),
|
||||
"a bridge worker name with a different nonce is a prior daemon's orphan");
|
||||
assertTrue(ClaudeCodeLauncher.isForeignWorker("claude-gx10-4127af-11", "aaaaaa"),
|
||||
"profile and multi-digit seq are still parsed; foreign nonce ⇒ reap");
|
||||
}
|
||||
|
||||
@Test
|
||||
void isForeignWorkerSparesOurOwnLiveWorkersAndNonWorkers() {
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("claude-ollama-abcdef-3", "abcdef"),
|
||||
"a worker with THIS process's nonce is ours and live — never reap it");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker(null, "abcdef"), "an unnamed agent is not a worker");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("claude", "abcdef"), "a bare kind name is not a worker");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("my-repl", "abcdef"), "a user's own label is not a worker");
|
||||
assertFalse(ClaudeCodeLauncher.isForeignWorker("claude-ollama-XYZ123-2", "abcdef"),
|
||||
"a non-hex nonce does not match our scheme");
|
||||
}
|
||||
|
||||
// --- CB-117 orphan reap: the wiring through stop() -------------------------
|
||||
|
||||
private static long paneCloseCount(FakeHerdr herdr, String paneId) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("pane.close"))
|
||||
.filter(c -> paneId.equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapsAForeignOrphanButSparesOurOwnWorkerAndUserSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
herdr.withAgent("claude-ollama-be09c2-2", "term_orphan", "wQ:pF", "wQ:t8") // prior daemon's leak
|
||||
.withAgent("claude-gx10-" + svc.nameNonce() + "-1", "term_mine", "wQ:pMine", "wQ:tMine"); // ours, live
|
||||
// (the fake's default unnamed term_a stands in for a user's own Claude session)
|
||||
|
||||
int reaped = svc.reapOrphanWorkers();
|
||||
|
||||
assertEquals(1, reaped, "exactly the one foreign-nonce orphan is reaped");
|
||||
assertEquals(1, paneCloseCount(herdr, "wQ:pF"), "the orphan's pane is closed");
|
||||
assertEquals(0, paneCloseCount(herdr, "wQ:pMine"), "our own live worker's pane is left running");
|
||||
assertEquals(0, paneCloseCount(herdr, "w2:p7"), "a user's own session is never touched");
|
||||
assertTrue(herdr.called("tab.close"), "the orphan's now-empty dedicated tab is closed too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapCountsAnAlreadyGoneOrphanAsReaped() {
|
||||
FakeHerdr herdr = new FakeHerdr().paneCloseFailsWith("pane_not_found");
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
herdr.withAgent("claude-ollama-0d856d-3", "term_gone", "wQ:pS", "wQ:tD");
|
||||
|
||||
assertEquals(1, svc.reapOrphanWorkers(),
|
||||
"a pane that vanished between list and close is a successful reap, not a failure");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIsSkippedWhenHerdrCannotBeListed() {
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false); // agent.list throws
|
||||
assertEquals(0, multiProfile(herdr).reapOrphanWorkers(), "a listing failure reaps nothing and does not throw");
|
||||
}
|
||||
|
||||
// --- PeerHandle indirection ----------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void spawnReturnsPeerHandleWithHostUniqueOpaqueId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn must return a non-null handle");
|
||||
// CB-519: id() is a host-unique opaque UUID, decoupled from the herdr pane coordinate.
|
||||
assertNotEquals("w9:pRoot_1", handle.id(),
|
||||
"handle.id() must NOT be the herdr pane id");
|
||||
assertDoesNotThrow(() -> UUID.fromString(handle.id()),
|
||||
"handle.id() must be a UUID: " + handle.id());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsPeerHandleWithCorrectTerminalId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, "/caller"));
|
||||
|
||||
assertEquals("term_new_1", handle.terminalId(), "handle.terminalId() must equal the agent's terminalId");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesIncludeMidTurnAskWorktreeOrphanReap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
Set<Capability> caps = svc.capabilities();
|
||||
|
||||
assertTrue(caps.contains(Capability.MID_TURN_ASK), "every Claude Code peer supports mid-turn ask");
|
||||
assertTrue(caps.contains(Capability.WORKTREE), "every CLI peer supports worktree cwd");
|
||||
assertTrue(caps.contains(Capability.ORPHAN_REAP), "every herdr launcher supports orphan reap");
|
||||
assertTrue(caps.contains(Capability.CONTEXT_RESET), "Claude Code supports /clear");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearContextUsesTheClaudeCommandThroughTheOwningHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("claude"), null);
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertTrue(svc.clearContext(handle.id()));
|
||||
|
||||
Map<?, ?> prompt = (Map<?, ?>) herdr.lastCall("agent.prompt").params();
|
||||
assertEquals("/clear", prompt.get("text"));
|
||||
assertEquals("w9:pRoot_1", prompt.get("target"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesIncludeSelfPrWhenProfileHasGitToken() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"impl", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "impl"), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
"GITEA_ACCESS_TOKEN", null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("impl", cfg), "impl",
|
||||
_ -> "tok");
|
||||
|
||||
assertTrue(svc.capabilities().contains(Capability.SELF_PR),
|
||||
"a profile with a git token grants SELF_PR");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesExcludeSelfPrWhenNoGitToken() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
assertFalse(svc.capabilities().contains(Capability.SELF_PR),
|
||||
"no git token profile → no SELF_PR capability");
|
||||
}
|
||||
|
||||
@Test
|
||||
void effectiveCwdViaSpawnRequestMatchesExistingResolution() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
String cwd = svc.effectiveCwd(new SpawnRequest("ltms-local", "/work/proj", "/caller/home"));
|
||||
|
||||
assertEquals("/work/proj", cwd, "effectiveCwd via SpawnRequest must match the three-arg resolution");
|
||||
}
|
||||
|
||||
// --- CB-547a: durable session identity (mint / resume / no-identity legacy) -----------------
|
||||
|
||||
@Test
|
||||
void freshSpawnMintsASessionIdAndPassesTheName() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, null, "my-session", null));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int flag = args.indexOf("--session-id");
|
||||
assertTrue(flag >= 0 && flag + 1 < args.size(), "--session-id present: " + args);
|
||||
String minted = args.get(flag + 1);
|
||||
assertDoesNotThrow(() -> UUID.fromString(minted), "--session-id is a valid UUID: " + minted);
|
||||
assertEquals("my-session", args.get(args.indexOf("-n") + 1), "the logical name rides as -n");
|
||||
assertEquals(minted, handle.agentSessionId(),
|
||||
"the resume handle is the minted id, known before the agent has written anything");
|
||||
assertEquals("my-session", handle.sessionName(), "the handle carries the logical name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSpawnPassesDashRAndNeverASessionId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, null, "my-session", "cb-resume-1"));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("--session-id"), "--session-id must NOT be passed on a resume (conflicts with -r)");
|
||||
assertEquals("cb-resume-1", args.get(args.indexOf("-r") + 1), "-r carries the prior session id");
|
||||
assertEquals("cb-resume-1", handle.agentSessionId(), "a resume adopts the prior id as its own");
|
||||
assertEquals("my-session", handle.sessionName(), "the logical name survives a resume");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noIdentitySpawnKeepsTheLegacyArgvAndCarriesNoSessionHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest("ltms-local", null, null));
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertFalse(args.contains("--session-id"), "no identity → no --session-id");
|
||||
assertFalse(args.contains("-n"), "no identity → no -n");
|
||||
assertFalse(args.contains("-r"), "no identity → no -r");
|
||||
assertNull(handle.agentSessionId(), "no identity → no resume handle");
|
||||
assertNull(handle.sessionName(), "no identity → no logical name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesIncludeSessionNameAndSessionResume() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
Set<Capability> caps = svc.capabilities();
|
||||
assertTrue(caps.contains(Capability.SESSION_NAME), "Claude Code surfaces the bridge's logical name (-n)");
|
||||
assertTrue(caps.contains(Capability.SESSION_RESUME), "Claude Code can relaunch onto a prior conversation (-r)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesViaPeerLauncherMatchesExistingApi() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
|
||||
assertEquals(Set.of("gx10", "ollama"), svc.profiles(), "profiles() via PeerLauncher must match");
|
||||
}
|
||||
|
||||
@Test
|
||||
void defaultProfileViaPeerLauncherMatches() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = multiProfile(herdr);
|
||||
|
||||
assertEquals("gx10", svc.defaultProfile(), "defaultProfile() via PeerLauncher must match");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopViaPeerLauncherTearsDownByHandleId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
svc.stop(handle.id());
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "stop via handle.id() must close the pane");
|
||||
}
|
||||
|
||||
// --- CB-519: host-unique id, decoupled from the pane coordinate ------------------------------
|
||||
|
||||
@Test
|
||||
void twoSpawnsOnTheSamePaneNeverCollideOnHostUniqueId() {
|
||||
// Two spawns may be placed on the same herdr pane coordinate (e.g. a pane that was reused
|
||||
// or re-reported after a restart); the host-unique id must not collide even then.
|
||||
FakeHerdr herdr = new FakeHerdr().pinNextStarts(2, "term_shared", "w9:pShared");
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle a = svc.spawn(new SpawnRequest(null, null, null));
|
||||
PeerHandle b = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotEquals(a.id(), b.id(),
|
||||
"two spawns on the same pane coordinate get distinct host-unique ids");
|
||||
assertNotEquals("w9:pShared", a.id(), "id is not the pane coordinate");
|
||||
assertNotEquals("w9:pShared", b.id(), "id is not the pane coordinate");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopResolvesTheHostUniqueIdToThePaneThatSpawnedIt() {
|
||||
// CB-519: id() != paneId, so stop(id) must tear down the exact pane the id names — and no
|
||||
// other live peer's pane.
|
||||
FakeHerdr herdr = new FakeHerdr(); // deterministic panes w9:pRoot_1, w9:pRoot_2 per spawn
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle a = svc.spawn(new SpawnRequest(null, null, null));
|
||||
PeerHandle b = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
svc.stop(b.id());
|
||||
|
||||
assertEquals(1, paneCloseCount(herdr, "w9:pRoot_2"), "stop(b.id()) closes only b's pane");
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"), "a's pane is untouched");
|
||||
}
|
||||
|
||||
// --- CB-306 spawn-readiness gate -----------------------------------------------------------
|
||||
|
||||
private static Map<String, BridgedConfig.Profile> workerConfigMap(String profile, String mcpUrl) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
profile, "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", profile), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", mcpUrl, null, null);
|
||||
return Map.of(cfg.profile(), cfg);
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnWaitsUntilInjectableThenReturnsHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // first status call sees UNKNOWN
|
||||
long[] clock = {0};
|
||||
boolean[] firstSleep = {true};
|
||||
// The sleeper: advance the fake clock, and on the first call flip the
|
||||
// agent status to IDLE so the next poll succeeds.
|
||||
Runnable sleeper = () -> {
|
||||
clock[0] += 300;
|
||||
if (firstSleep[0]) {
|
||||
herdr.agentStatus("idle");
|
||||
firstSleep[0] = false;
|
||||
}
|
||||
};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
5000, () -> clock[0], sleeper);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn returns a handle when worker becomes injectable");
|
||||
assertNotEquals("w9:pRoot_1", handle.id(),
|
||||
"handle id is a host-unique opaque id, not the started pane");
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"no pane.close when worker becomes injectable before timeout");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnThrowsPeerUnreachableWhenNeverInjectableAndReapsPane() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // always UNKNOWN
|
||||
long[] clock = {0};
|
||||
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
1000, () -> clock[0], () -> clock[0] += 50);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(
|
||||
PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
|
||||
assertTrue(ex.getMessage().contains("w9:pRoot_1"),
|
||||
"exception message references the paneId: " + ex.getMessage());
|
||||
assertTrue(ex.getMessage().contains("1000"),
|
||||
"exception message references the timeout: " + ex.getMessage());
|
||||
assertTrue(clock[0] >= 1000, "fake clock advanced past the timeout: " + clock[0]);
|
||||
assertEquals(1, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"pane was closed on timeout (no orphan left behind)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsImmediatelyWhenGateIsDisabled() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// The default 6-arg constructor has spawnReadyTimeoutMs=0 (gate disabled).
|
||||
ClaudeCodeLauncher svc = service(herdr, List.of("ccs", "ltms-local"), null);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn returns a handle when the gate is disabled");
|
||||
assertFalse(herdr.called("agent.get"),
|
||||
"agent.get is never called when the gate is disabled (no polling)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnGateRespectsZeroTimeoutEvenWithFullConstructor() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
long[] clock = {0};
|
||||
|
||||
// Explicit zero timeout with the full testability constructor — should
|
||||
// skip polling entirely, just like the legacy default path.
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
workerConfigMap("ltms-local", null), "ltms-local", _ -> null,
|
||||
0, () -> clock[0], () -> clock[0] += 1);
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle, "spawn still succeeds with zero timeout");
|
||||
assertEquals(0, paneCloseCount(herdr, "w9:pRoot_1"),
|
||||
"no orphan pane close from the gate path");
|
||||
assertDoesNotThrow(() -> UUID.fromString(handle.id()));
|
||||
}
|
||||
|
||||
// --- CB-511: worker environment seeding -----------------------------------------------------
|
||||
|
||||
@Test
|
||||
void workerInheritsTheDaemonPath() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
k -> "PATH".equals(k) ? "/opt/tools/bin:/usr/bin" : null).spawn();
|
||||
|
||||
assertEquals("/opt/tools/bin:/usr/bin", startEnv(herdr).get("PATH"),
|
||||
"a worker with no PATH cannot run the build it is asked to run");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profileEnvIsInjectedIntoTheWorker() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null, null, null,
|
||||
null, Map.of("JAVA_HOME", "/opt/jdk", "PATH", "/profile/bin"), null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
k -> "PATH".equals(k) ? "/daemon/bin" : null).spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertEquals("/opt/jdk", env.get("JAVA_HOME"), "profile env: is passed through");
|
||||
assertEquals("/profile/bin", env.get("PATH"), "an explicit profile PATH overrides the daemon's");
|
||||
}
|
||||
|
||||
/**
|
||||
* The security-relevant ordering. {@code SubscriptionGuard} is checked against the profile's
|
||||
* {@code baseUrl} only, so if a profile's {@code env:} could overwrite ANTHROPIC_BASE_URL a
|
||||
* worker could be pointed at an unguarded host while the guard passed on a benign one.
|
||||
*/
|
||||
@Test
|
||||
void profileEnvCannotOverrideGuardCheckedAnthropicVars() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "w #{n}", null, null, null, null, null,
|
||||
null, Map.of("ANTHROPIC_BASE_URL", "http://evil.example.com"), null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null).spawn();
|
||||
|
||||
assertEquals("http://gx00.gw:8000", startEnv(herdr).get("ANTHROPIC_BASE_URL"),
|
||||
"the guard-checked baseUrl must win over any env: entry, or the boundary is bypassable");
|
||||
}
|
||||
|
||||
// ── CB-533: the model is pinned on the command line, not only in the environment ────────────
|
||||
|
||||
/** A launcher for a profile identical but for its {@code model:} — the only variable here. */
|
||||
private ClaudeCodeLauncher serviceWithModel(FakeHerdr herdr, String model) {
|
||||
BridgedConfig.Profile cfg = profileWithModel(model);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null);
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile profileWithModel(String model) {
|
||||
return new BridgedConfig.Profile("sonnet", "http://gx00.gw:8000", model, null,
|
||||
"BRIDGED_WORKER_TOKEN", List.of("ccs", "sonnet"), "tab", "bridged-workers",
|
||||
"w #{n}", "http://127.0.0.1:8765/mcp", null, null);
|
||||
}
|
||||
|
||||
@Test
|
||||
void aConfiguredModelIsPassedAsAModelFlagAsWellAsTheEnvVar() {
|
||||
// ANTHROPIC_MODEL alone loses to `ccs`, which exports its own model family over whatever it
|
||||
// inherited — so a profile that set model: was silently overruled by its own launcher.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithModel(herdr, "claude-sonnet-5").spawn("sonnet", null, null);
|
||||
|
||||
assertEquals("claude-sonnet-5", startEnv(herdr).get("ANTHROPIC_MODEL"));
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
int flag = args.indexOf("--model");
|
||||
assertTrue(flag >= 0, "the flag is what survives a wrapper argv like [ccs, sonnet]");
|
||||
assertEquals("claude-sonnet-5", args.get(flag + 1));
|
||||
}
|
||||
|
||||
@Test
|
||||
void theModelFlagComesLastSoItOutranksTheOperatorsOwnArgv() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithModel(herdr, "claude-sonnet-5").spawn("sonnet", null, null);
|
||||
|
||||
List<String> args = spawnedArgs(herdr);
|
||||
assertEquals(args.size() - 2, args.indexOf("--model"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithNoModelGetsNoModelFlag() {
|
||||
// gx10 deliberately leaves model: unset so ccs owns selection; adding a flag would make
|
||||
// this file a second source of truth for exactly the thing it declines to decide.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
serviceWithModel(herdr, null).spawn("sonnet", null, null);
|
||||
|
||||
assertFalse(spawnedArgs(herdr).contains("--model"));
|
||||
assertNull(startEnv(herdr).get("ANTHROPIC_MODEL"));
|
||||
}
|
||||
|
||||
// --- CB-539: subscription-profile opt-in ----------------------------------------------------
|
||||
|
||||
/** A claude-code profile on the subscription: no baseUrl (by design), no off-sub endpoint. */
|
||||
private static BridgedConfig.Profile subscriptionCfg(String profile, String baseUrl) {
|
||||
return new BridgedConfig.Profile(
|
||||
profile, baseUrl, "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", profile), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
null, null, null, Map.of(), null, null, true);
|
||||
}
|
||||
|
||||
@Test
|
||||
void defaultRefusalIsPreservedForClaudeProfileWithNoBaseUrl() {
|
||||
// Requirement 1: absent subscription:true ⇒ byte-identical refusal to today. A claude-code
|
||||
// profile with no baseUrl and no subscription must still be refused (it would bill the sub).
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", null, "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
|
||||
GuardException ex = assertThrows(GuardException.class, () -> svc.spawn("ltms-local", null, null));
|
||||
assertTrue(ex.getMessage().contains("no ANTHROPIC_BASE_URL"),
|
||||
"the refusal names the missing baseUrl: " + ex.getMessage());
|
||||
assertEquals(0, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"nothing was spawned before the refusal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void subscriptionProfileSpawnsWithoutInjectedAnthropicVars() {
|
||||
// Requirement on subscription:true: no baseUrl is required (or injected), and neither
|
||||
// ANTHROPIC_BASE_URL nor ANTHROPIC_AUTH_TOKEN is injected even though the token env would
|
||||
// resolve one if asked.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = subscriptionCfg("sonnet", null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "would-be-token").spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"), "no baseUrl injected for a subscription profile");
|
||||
assertNull(env.get("ANTHROPIC_AUTH_TOKEN"), "no auth token injected for a subscription profile");
|
||||
assertEquals("sonnet", env.get("ANTHROPIC_MODEL"),
|
||||
"the model alias is still injected; only the subscription-boundary vars are dropped");
|
||||
}
|
||||
|
||||
@Test
|
||||
void subscriptionPlusBaseUrlIsRefused() {
|
||||
// Requirement 2: subscription:true + a baseUrl state opposite intents — refuse at spawn,
|
||||
// naming the profile, rather than silently picking a winner.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = subscriptionCfg("sonnet", "http://gx00.gw:8000");
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class,
|
||||
() -> svc.spawn("sonnet", null, null));
|
||||
assertTrue(ex.getMessage().contains("sonnet"), "refusal names the profile: " + ex.getMessage());
|
||||
assertTrue(ex.getMessage().contains("subscription"), "refusal explains the contradiction: " + ex.getMessage());
|
||||
assertEquals(0, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"nothing was spawned before the contradiction was refused");
|
||||
}
|
||||
|
||||
@Test
|
||||
void nonSubscriptionProfilesAreStillAllowlistChecked() {
|
||||
// Requirement 3: the guard keeps its teeth for every other profile — a base_url whose host is
|
||||
// not on the allowlist is still refused, whether or not any subscription profile exists.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile rogue = new BridgedConfig.Profile(
|
||||
"rogue", "http://evil.example.com:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "rogue"), "tab", "bridged-workers", "w #{n}", null, null, null);
|
||||
ClaudeCodeLauncher svc = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(rogue.profile(), rogue), rogue.profile(), _ -> null);
|
||||
|
||||
GuardException ex = assertThrows(GuardException.class, () -> svc.spawn("rogue", null, null));
|
||||
assertTrue(ex.getMessage().contains("not on the"), "refusal cites the allowlist: " + ex.getMessage());
|
||||
assertEquals(0, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count(),
|
||||
"nothing was spawned before the allowlist refusal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSubscriptionProfileHasAnEnvSuppliedAnthropicBindingStripped() {
|
||||
// CB-542: even a subscription profile whose env: carries ANTHROPIC_BASE_URL (or AUTH_TOKEN)
|
||||
// must not hand them to the worker — on the subscription path no guard would vet them. Config
|
||||
// load refuses this loudly; this launcher-side strip is the belt-and-braces that makes the
|
||||
// invariant hold for a profile built in code that never passed through that validation.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"sonnet", null, "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "sonnet"), "tab", "bridged-workers", "w #{n}", null, null, null,
|
||||
null, null, null,
|
||||
Map.of("ANTHROPIC_BASE_URL", "http://evil.example.com",
|
||||
"ANTHROPIC_AUTH_TOKEN", "sk-ant-bad", "JAVA_HOME", "/opt/jdk"),
|
||||
null, null, true);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "would-be-token").spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"),
|
||||
"the unguarded endpoint must not survive into the worker");
|
||||
assertNull(env.get("ANTHROPIC_AUTH_TOKEN"),
|
||||
"the unguarded token must not survive into the worker");
|
||||
assertEquals("/opt/jdk", env.get("JAVA_HOME"),
|
||||
"only the Anthropic binding keys are stripped; the rest of env: still applies");
|
||||
}
|
||||
|
||||
// ── CB-557: role-aware tab labels ─────────────────────────────────────────────────────────
|
||||
|
||||
/** The {@code label} of every {@code tab.rename}, in call order. */
|
||||
private List<String> tabLabels(FakeHerdr herdr) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("tab.rename"))
|
||||
.map(c -> (String) ((Map<?, ?>) c.params()).get("label"))
|
||||
.toList();
|
||||
}
|
||||
|
||||
/** A profile with no {@code tabLabel:} of its own — the fleet template decides. */
|
||||
private ClaudeCodeLauncher labelService(FakeHerdr herdr, Supplier<String> fleetTemplate) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", null, null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null, 0, 0L, fleetTemplate);
|
||||
}
|
||||
|
||||
/**
|
||||
* The knob must reach the rename call. It was inert once — {@code HerdrPeerLauncher} accepted a
|
||||
* template while {@code Bridged} passed none, and the label stayed right only because the
|
||||
* fallback happened to match. Pin the wiring, not the coincidence.
|
||||
*/
|
||||
@Test
|
||||
void theFleetTemplateNamesTheRoleTheMemberWasSpawnedFor() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = labelService(herdr, () -> "{role}: {profile} #{n}");
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.REVIEWER));
|
||||
|
||||
assertEquals(List.of("reviewer: sonnet #1"), tabLabels(herdr));
|
||||
}
|
||||
|
||||
/** The counter is per role+profile, so a dev and a reviewer on one profile both start at #1. */
|
||||
@Test
|
||||
void theCounterRunsPerRoleAndProfileNotPerFleet() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher svc = labelService(herdr, () -> "{role}: {profile} #{n}");
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.REVIEWER));
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertEquals(List.of("dev: sonnet #1", "reviewer: sonnet #1", "dev: sonnet #2"),
|
||||
tabLabels(herdr));
|
||||
}
|
||||
|
||||
/** No fleet template configured ⇒ the built-in default, still role-first. */
|
||||
@Test
|
||||
void aBlankFleetTemplateFallsBackToTheRoleFirstDefault() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
labelService(herdr, () -> null).spawn(
|
||||
new SpawnRequest("sonnet", null, null, null, null, MemberRole.ARCHITECT));
|
||||
|
||||
assertEquals(List.of("architect: sonnet #1"), tabLabels(herdr));
|
||||
assertEquals("{role}: {profile} #{n}", BridgedConfig.Fleet.DEFAULT_TAB_LABEL);
|
||||
}
|
||||
|
||||
/** A profile that wants its own label still outranks the fleet template. */
|
||||
@Test
|
||||
void aProfileTabLabelOverridesTheFleetTemplate() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"sonnet", "http://gx00.gw:8000", "sonnet", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("claude"), "tab", "bridged-workers", "pinned {profile}", null, null, null);
|
||||
new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> null, 0, 0L, () -> "{role}: {profile} #{n}")
|
||||
.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.REVIEWER));
|
||||
|
||||
assertEquals(List.of("pinned sonnet"), tabLabels(herdr));
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-559: the template is read per spawn, not captured at construction. This is what makes
|
||||
* {@code fleet.tabLabel} a hot key — a launcher built at boot must see an edit made an hour later
|
||||
* without being rebuilt.
|
||||
*/
|
||||
@Test
|
||||
void theTemplateIsReadOnEverySpawnSoAnEditTakesEffect() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AtomicReference<String> template = new AtomicReference<>("{role}: {profile} #{n}");
|
||||
ClaudeCodeLauncher svc = labelService(herdr, template::get);
|
||||
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
template.set("[{profile}] {role} {n}");
|
||||
svc.spawn(new SpawnRequest("sonnet", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertEquals(List.of("dev: sonnet #1", "[sonnet] dev 2"), tabLabels(herdr));
|
||||
}
|
||||
}
|
||||
@@ -1,555 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The composite router: profile → owning adapter for spawn/cwd/parity, pane id → owner for stop,
|
||||
* and fleet-wide union/dedup for list/reap/caps/profiles. Exercised through two real adapters —
|
||||
* claude-code + opencode — over one FakeHerdr, so each call is observed reaching the right adapter
|
||||
* (the started herdr agent name carries that adapter's {@code claude-}/{@code opencode-} prefix).
|
||||
*/
|
||||
class CompositePeerLauncherTest {
|
||||
|
||||
private ClaudeCodeLauncher claudeAdapter(FakeHerdr herdr) {
|
||||
// 12-arg back-compat Worker ctor → kind defaults to claude-code.
|
||||
BridgedConfig.Profile claude = new BridgedConfig.Profile("claude", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("claude", claude), "claude", _ -> null);
|
||||
}
|
||||
|
||||
private OpenCodeLauncher opencodeAdapter(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile gemini = new BridgedConfig.Profile("gemini", null, "google/gemini-2.5-pro",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("opencode"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null, "GITEA_ACCESS_TOKEN", null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", gemini), "gemini", _ -> "tok");
|
||||
}
|
||||
|
||||
private CompositePeerLauncher composite(FakeHerdr herdr) {
|
||||
return new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(herdr), opencodeAdapter(herdr)), "claude");
|
||||
}
|
||||
|
||||
/**
|
||||
* A minimal concrete HerdrPeerLauncher for policy tests. It either returns a fake handle for the
|
||||
* requested profile or throws, depending on {@code failProfiles}. buildLaunch is a stub; only
|
||||
* spawn/stop/list/caps/reap are exercised by the composite.
|
||||
*/
|
||||
private static final class StubLauncher extends HerdrPeerLauncher {
|
||||
private final Set<String> failProfiles;
|
||||
private final Map<String, Integer> spawnCounts = new HashMap<>();
|
||||
|
||||
StubLauncher(String prefix, FakeHerdr herdr,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Set<String> failProfiles) {
|
||||
super(prefix, new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
profiles, defaultProfile, _ -> null, 0L, System::currentTimeMillis, () -> { });
|
||||
this.failProfiles = Set.copyOf(failProfiles);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg) {
|
||||
return new Launch(Map.of(), List.of());
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String p = (req.profileName() == null || req.profileName().isBlank())
|
||||
? defaultProfile() : req.profileName();
|
||||
spawnCounts.merge(p, 1, Integer::sum);
|
||||
if (failProfiles.contains(p)) {
|
||||
throw new PeerUnreachableException(p + " is down");
|
||||
}
|
||||
return new PeerHandle() {
|
||||
@Override public String id() { return "pane-" + p; }
|
||||
@Override public String terminalId() { return "term-" + p; }
|
||||
@Override public String profile() { return p; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) { }
|
||||
|
||||
@Override
|
||||
public List<Agent> list() { return List.of(); }
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() { return EnumSet.noneOf(Capability.class); }
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() { return 0; }
|
||||
|
||||
int spawnCount(String profile) {
|
||||
return spawnCounts.getOrDefault(profile, 0);
|
||||
}
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile stubWorker(String profile) {
|
||||
return new BridgedConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null);
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile stubWorker(String profile, float weight, Integer maxLoad) {
|
||||
return new BridgedConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null,
|
||||
weight, maxLoad);
|
||||
}
|
||||
|
||||
/**
|
||||
* An <em>order-preserving</em> profile map. Never {@code Map.of} here: its iteration order is
|
||||
* salted per JVM run, and the weighted policy breaks an exact-weight tie on candidate order —
|
||||
* so a {@code Map.of} would make "which profile is tried first" a coin flip per run and any
|
||||
* assertion about the first attempt intermittently false.
|
||||
*/
|
||||
private static Map<String, BridgedConfig.Profile> ordered(String first, BridgedConfig.Profile a,
|
||||
String second, BridgedConfig.Profile b) {
|
||||
Map<String, BridgedConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put(first, a);
|
||||
m.put(second, b);
|
||||
return m;
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static String startedName(FakeHerdr herdr) {
|
||||
return (String) ((Map<String, Object>) herdr.lastCall("agent.start").params()).get("name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRoutesEachProfileToItsOwningAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
|
||||
composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("opencode-"),
|
||||
"the gemini profile is spawned by the opencode adapter: " + startedName(herdr));
|
||||
|
||||
composite.spawn(new SpawnRequest("claude", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"the claude profile is spawned by the claude-code adapter: " + startedName(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullProfileResolvesTheDefaultAndRoutesToItsOwner() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
composite(herdr).spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"a no-profile spawn resolves the default (claude) and routes to its adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unknownProfileIsRejected() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> composite.spawn(new SpawnRequest("nope", null, null)),
|
||||
"a profile no adapter declares is an error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesAndDefaultAreExposedAcrossAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(Set.of("claude", "gemini"), composite.profiles(),
|
||||
"profiles are the union of every adapter's profiles");
|
||||
assertEquals("claude", composite.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesAreTheUnionOfEveryAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher claude = claudeAdapter(herdr);
|
||||
OpenCodeLauncher opencode = opencodeAdapter(herdr);
|
||||
PeerLauncher composite = new CompositePeerLauncher(List.of(claude, opencode), "claude");
|
||||
|
||||
assertTrue(composite.capabilities().containsAll(claude.capabilities()),
|
||||
"the fleet offers every claude-code capability");
|
||||
assertTrue(composite.capabilities().containsAll(opencode.capabilities()),
|
||||
"the fleet offers every opencode capability (incl. SELF_PR from its git-token profile)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void listIsDeduplicatedByPaneIdAcrossAdaptersSharingHerdr() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
// Both adapters wrap the same herdr, so each list() returns the same global agent set;
|
||||
// the composite must return each pane once, not once per adapter.
|
||||
assertEquals(1, composite.list().size(),
|
||||
"the single herdr-tracked pane appears once, not duplicated per adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapSumsAcrossAdaptersAndEachAdapterReapsOnlyItsOwnPrefix() {
|
||||
// One foreign opencode orphan + one foreign claude orphan, from a prior daemon (different nonce).
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withAgent("opencode-gemini-ffffff-1", "term_o", "wQ:pO", "wQ:tO")
|
||||
.withAgent("claude-claude-eeeeee-1", "term_c", "wQ:pC", "wQ:tC");
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(2, composite.reapOrphanWorkers(),
|
||||
"both orphans are reaped — one by each adapter, summed by the composite");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopTearsDownAPaneSpawnedThroughTheComposite() {
|
||||
// CB-519: handle.id() is a host-unique opaque UUID, not the herdr pane — stop(id) must
|
||||
// resolve it through the owning adapter down to the actual pane coordinate it spawned.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertNotEquals("w9:pRoot_1", handle.id(), "the id is decoupled from the pane coordinate");
|
||||
|
||||
composite.stop(handle.id());
|
||||
assertTrue(herdr.calls.stream()
|
||||
.anyMatch(c -> c.method().equals("pane.close")
|
||||
&& "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id"))),
|
||||
"stop routes to the spawning adapter and closes exactly that worker's pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void opencodeContextResetIsANoOpAndWarnsOnlyOnce() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertFalse(opencodeAdapter(herdr).capabilities().contains(Capability.CONTEXT_RESET));
|
||||
assertTrue(herdr.calls.stream().noneMatch(c -> "agent.prompt".equals(c.method())),
|
||||
"never type Claude's /clear into an opencode prompt");
|
||||
assertEquals(1, appender.list.stream()
|
||||
.filter(e -> e.getFormattedMessage().contains("context reset is unsupported"))
|
||||
.count(), "unsupported reset is logged once per adapter, not once per turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAProfileClaimedByTwoAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// Two opencode adapters both declaring "gemini" — a profile-name collision.
|
||||
OpenCodeLauncher a = opencodeAdapter(herdr);
|
||||
OpenCodeLauncher b = opencodeAdapter(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(a, b), "gemini"),
|
||||
"a profile two adapters both claim is a configuration error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAnEmptyAdapterList() {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(), "claude"),
|
||||
"at least one adapter must be configured");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedDefaultIsNoOpForUnqualifiedSpawns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"fixed placement still routes an unqualified spawn to the default profile");
|
||||
assertEquals("claude", h.profile(), "the returned handle carries the resolved default profile");
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyGatesProfileAtMaxLoad() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
for (int i = 0; i < 5; i++) {
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "profile a is at maxLoad, so every spawn must land on b");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyDistributesAccordingToWeightRatio() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.75f, null),
|
||||
"b", stubWorker("b", 0.25f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 40; i++) {
|
||||
String p = composite.spawn(new SpawnRequest(null, null, null)).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(30, a, "weighted distribution should hold the 3:1 ratio");
|
||||
assertEquals(10, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverRetriesNextCandidateWhenProfileIsUnreachable() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "the spawn must fail over from unreachable a to b");
|
||||
assertEquals(1, adapter.spawnCount("a"), "a was tried once and failed");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b was tried once and succeeded");
|
||||
}
|
||||
|
||||
/**
|
||||
* Definition order — not hash order — decides an exact-weight tie. Paired with the test above
|
||||
* (same two profiles, opposite declaration order, opposite expected first attempt) this pins the
|
||||
* ordering contract from both sides: under a salted map one of the two must fail on every run.
|
||||
*/
|
||||
@Test
|
||||
void reversingDefinitionOrderReversesWhichProfileIsTriedFirst() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"b", stubWorker("b"),
|
||||
"a", stubWorker("a"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("a", h.profile(), "b is declared first and unreachable, so the spawn lands on a");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b, declared first, is the one tried first");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverBoundedByCandidateCount() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a", "b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerUnreachableException e = assertThrows(PeerUnreachableException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("no reachable worker profile"), e.getMessage());
|
||||
assertEquals(1, adapter.spawnCount("a"));
|
||||
assertEquals(1, adapter.spawnCount("b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnAtMaxLoadThrowsPlacementExceptionNamingProfileLiveAndCap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 2 : 0);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("a", null, null)));
|
||||
assertTrue(e.getMessage().contains("'a'"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 live"), "message names the live count: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 cap"), "message names the cap: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("a"), "at cap, the spawn is refused before any delegation");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnUnderMaxLoadStillSucceeds() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile under its cap accepts an explicit spawn");
|
||||
assertEquals(1, adapter.spawnCount("a"), "the under-cap spawn is delegated");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnWithNullMaxLoadIsNeverCapped() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, null),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
// A deliberately absurd live count: an unset maxLoad means unlimited, so it must never refuse.
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 1000);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile with no maxLoad is never capped, however many live workers");
|
||||
}
|
||||
|
||||
@Test
|
||||
void emptyCandidateSetThrowsClearException() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, 1));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 1);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
// ── CB-557: an unqualified spawn is placed inside its role's pool ─────────────────────────
|
||||
|
||||
/** Three profiles in definition order — pools are carved out of this set. */
|
||||
private static Map<String, BridgedConfig.Profile> threeProfiles() {
|
||||
Map<String, BridgedConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("opus", stubWorker("opus", 1.0f, null));
|
||||
m.put("sonnet", stubWorker("sonnet", 1.0f, null));
|
||||
m.put("terra", stubWorker("terra", 1.0f, null));
|
||||
return m;
|
||||
}
|
||||
|
||||
private static Map<String, BridgedConfig.Slot> pool(String... names) {
|
||||
Map<String, BridgedConfig.Slot> m = new LinkedHashMap<>();
|
||||
for (String n : names) {
|
||||
m.put(n, new BridgedConfig.Slot(n));
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
private static CompositePeerLauncher withPools(FakeHerdr herdr, BridgedConfig.Fleet fleet) {
|
||||
Map<String, BridgedConfig.Profile> profiles = threeProfiles();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
return new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* The point of the pools: a role is placed only on a backend its pool names. Before CB-557 an
|
||||
* unqualified spawn ranged over every configured profile, so a reviewer could land on the
|
||||
* architect-only one.
|
||||
*/
|
||||
@Test
|
||||
void anUnqualifiedSpawnIsPlacedInsideItsRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), pool("sonnet"), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.ARCHITECT)).profile());
|
||||
assertEquals("terra", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile());
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.REVIEWER)).profile());
|
||||
}
|
||||
|
||||
/** Under `fixed`, the pool's first entry wins — not the global defaultProfile. */
|
||||
@Test
|
||||
void theRolePoolOutranksTheGlobalDefaultProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), Map.of(), pool("sonnet", "terra"), Map.of(), null));
|
||||
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"the dev pool starts at sonnet, so the global default 'opus' must not win");
|
||||
}
|
||||
|
||||
/**
|
||||
* A role with no pool is unconstrained, not blocked. A config that declares pools for some roles
|
||||
* and not others must keep spawning the rest.
|
||||
*/
|
||||
@Test
|
||||
void aRoleWithNoPoolFallsBackToEveryProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("sonnet"), Map.of(), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"no dev pool ⇒ all profiles are candidates, so `fixed` takes the first one");
|
||||
}
|
||||
|
||||
/** No fleet at all is the pre-CB-557 wiring, and must behave exactly as it did. */
|
||||
@Test
|
||||
void noFleetConfiguredKeepsTheOldWholeProfileListBehaviour() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, null);
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest(null, null, null)).profile());
|
||||
}
|
||||
|
||||
/**
|
||||
* An explicit profile is the operator overriding and is NOT judged against the pool. It must
|
||||
* stay that way: an unrolled `bridge_spawn{profile:"opus"}` carries no role, so it defaults to
|
||||
* DEV, and enforcing the pool here would refuse a spawn the operator asked for by name.
|
||||
*/
|
||||
@Test
|
||||
void anExplicitProfileIsNotConfinedToTheRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest("opus", null, null)).profile(),
|
||||
"naming opus explicitly must work even though the dev pool holds only terra");
|
||||
}
|
||||
|
||||
/** Placement still respects maxLoad, but only across the pool — never by escaping it. */
|
||||
@Test
|
||||
void aFullPoolIsRefusedRatherThanSpilledOntoAnotherRolesProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("opus", stubWorker("opus", 1.0f, null)); // architect-only, uncapped
|
||||
profiles.put("terra", stubWorker("terra", 1.0f, 1)); // the sole dev, capped at 1
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.weighted(), name -> "terra".equals(name) ? 1 : 0,
|
||||
new BridgedConfig.Fleet(Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
}
|
||||
@@ -1,328 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The opencode adapter's launch build: a file-based MCP mount + reply-charter instructions (no
|
||||
* inline flags, no {@code ANTHROPIC_*}, no guard), the {@code -m} model flag, and the shared base
|
||||
* transport (naming, reap, readiness gate) proving the {@link HerdrPeerLauncher} SPI is neutral.
|
||||
*/
|
||||
class OpenCodeLauncherTest {
|
||||
|
||||
private static BridgedConfig.Profile opencodeCfg(String model, String mcpUrl, String gitTokenEnv) {
|
||||
return new BridgedConfig.Profile("gemini", null, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, gitTokenEnv, null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
/** Gate-disabled launcher whose per-spawn config dirs land under an inspectable temp root. */
|
||||
private OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, BridgedConfig.Profile cfg) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), k -> "GITEA_ACCESS_TOKEN".equals(k) ? "tok" : null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, Object> lastStart(FakeHerdr herdr) {
|
||||
return (Map<String, Object>) herdr.lastCall("agent.start").params();
|
||||
}
|
||||
|
||||
/** Protocol 19: the worker's env is injected at pane creation (tab.create), not agent.start. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, String> startEnv(FakeHerdr herdr) {
|
||||
Map<String, String> env =
|
||||
(Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||
return env == null ? Map.of() : env;
|
||||
}
|
||||
|
||||
/** Protocol 19: agent.start carries only the args after the kind-resolved executable. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static List<String> startArgs(FakeHerdr herdr) {
|
||||
return (List<String>) lastStart(herdr).get("args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void writesRemoteMcpConfigAndCharterInstructionsWhenMcpUrlSet(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null))
|
||||
.spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"), "opencode carries no ANTHROPIC_* / subscription boundary");
|
||||
String cfgPath = env.get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "OPENCODE_CONFIG points the worker at the generated config file");
|
||||
assertTrue(Path.of(cfgPath).startsWith(root), "config file is generated under the injected root");
|
||||
|
||||
// Assert on parsed structure, not substrings: the generated config is real JSON and its
|
||||
// whitespace is the formatter's business, not the contract's.
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
assertTrue(json.path("compaction").path("auto").asBoolean(),
|
||||
"spawned opencode peers explicitly enable automatic compaction");
|
||||
JsonNode bridge = json.path("mcp").path("bridge");
|
||||
assertEquals("remote", bridge.path("type").asText(), "bridge is mounted as a remote MCP server");
|
||||
assertEquals("http://127.0.0.1:8765/mcp", bridge.path("url").asText(),
|
||||
"the profile's bridge MCP url is present");
|
||||
assertTrue(bridge.path("enabled").asBoolean(), "the bridge server is enabled");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the reply charter is mounted via instructions");
|
||||
|
||||
// The instructions entry is a real file path holding the reply charter.
|
||||
Path charter = Path.of(cfgPath).resolveSibling("reply-charter.md");
|
||||
assertTrue(Files.exists(charter), "the charter file the config references was written");
|
||||
assertTrue(Files.readString(charter).contains("bridge_reply"),
|
||||
"the charter instructs the worker to answer via bridge_reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noConfigFileWhenMcpUrlAbsent(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
assertNull(startEnv(herdr).get("OPENCODE_CONFIG"),
|
||||
"no bridge MCP url → no config file and no OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void passesTheModelAsDashMFlagAlongsideAutoApprove(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
assertTrue(args.contains("--auto"),
|
||||
"--auto is present alongside -m so a spawned peer never blocks on approval");
|
||||
int m = args.indexOf("-m");
|
||||
assertTrue(m >= 0, "model is selected with -m");
|
||||
assertEquals("google/gemini-2.5-pro", args.get(m + 1), "the provider/model selector follows -m");
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoApproveIsUnconditionalWhenModelBlank(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null)).spawn();
|
||||
assertEquals(List.of("--auto"), startArgs(herdr),
|
||||
"--auto is unconditional: a model-less worker still must never block on approval");
|
||||
}
|
||||
|
||||
@Test
|
||||
void injectsForgeTokenWhenProfileGrantsIt(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN")).spawn();
|
||||
assertEquals("tok", startEnv(herdr).get("GITEA_TOKEN"),
|
||||
"a git-token profile gets the peer-neutral GITEA_TOKEN grant, same as Claude");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesDeclareOrphanReapAndMcpAskAndConditionalSelfPr(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
assertEquals(java.util.Set.of(Capability.MID_TURN_ASK, Capability.WORKTREE, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_RESUME),
|
||||
service(herdr, root, opencodeCfg(null, null, null)).capabilities(),
|
||||
"opencode can be resumed by its own session id, so SESSION_RESUME is always declared");
|
||||
assertFalse(service(herdr, root, opencodeCfg(null, null, null))
|
||||
.capabilities().contains(Capability.SESSION_NAME),
|
||||
"opencode has no display-name flag, so SESSION_NAME must NOT be declared");
|
||||
assertTrue(service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN"))
|
||||
.capabilities().contains(Capability.SELF_PR),
|
||||
"a git-token profile adds SELF_PR");
|
||||
}
|
||||
|
||||
// --- CB-547: resume + post-hoc session discovery --------------------------------------------
|
||||
|
||||
@Test
|
||||
void aResumeSpawnPassesTheSessionIdAsDashS(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, "ses_41b79fc90ffeI9E8uZv6VprUn2"));
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
int s = args.indexOf("-s");
|
||||
assertTrue(s >= 0, "a resumed spawn carries opencode's -s flag");
|
||||
assertEquals("ses_41b79fc90ffeI9E8uZv6VprUn2", args.get(s + 1),
|
||||
"the resume target id follows -s");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFreshSpawnCarriesNoSessionFlag(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, null));
|
||||
|
||||
assertFalse(startArgs(herdr).contains("-s"),
|
||||
"no resume target → a fresh session with no -s flag");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theHandleDiscoversTheSessionIdForTheWorkersCwdOnlyAfterItAppears(@TempDir Path root,
|
||||
@TempDir Path discRoot)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = new OpenCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), Map.of("gemini", opencodeCfg(null, null, null)),
|
||||
"gemini", _ -> null, 0, System::currentTimeMillis, () -> { }, root, discRoot);
|
||||
|
||||
PeerHandle handle = launcher.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
|
||||
// opencode writes the record only when the session is first persisted — the instant the
|
||||
// pane is ready it does not exist, so agentSessionId() is null (never a spawn failure).
|
||||
assertNull(handle.agentSessionId(), "no record yet → null, not a spawn-time block");
|
||||
// Once the record appears (here: same cwd), lazy discovery resolves it — the handle's
|
||||
// session id matches its own worktree, not another's.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "p1", "ses_a.json",
|
||||
"ses_resolved", "/work/dir", 1000L);
|
||||
assertEquals("ses_resolved", handle.agentSessionId(),
|
||||
"agentSessionId() re-scans and picks up a record that has since been written");
|
||||
}
|
||||
|
||||
@Test
|
||||
void foreignWorkerMatchesOpencodePrefixButNotClaude() {
|
||||
String nonce = "abc123";
|
||||
assertTrue(OpenCodeLauncher.isForeignWorker("opencode-gemini-def456-1", nonce),
|
||||
"an opencode pane from another process is foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("opencode-gemini-" + nonce + "-1", nonce),
|
||||
"our own opencode pane (same nonce) is not foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("claude-ltms-local-def456-1", nonce),
|
||||
"a claude pane is never reaped by the opencode adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionConstructorsWireThroughToTheBase() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = opencodeCfg(null, null, null);
|
||||
// 5-arg (gate disabled) and 7-arg (gate enabled) production constructors both expose the profile.
|
||||
OpenCodeLauncher disabled = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
OpenCodeLauncher gated = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null, 5000, 100);
|
||||
assertEquals(java.util.Set.of("gemini"), disabled.profiles());
|
||||
assertEquals("gemini", gated.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnGateThrowsPeerUnreachableWhenNeverInjectable(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never injectable
|
||||
long[] clock = {0};
|
||||
OpenCodeLauncher svc = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", opencodeCfg(null, null, null)), "gemini", _ -> null,
|
||||
1000, () -> clock[0], () -> clock[0] += 50, root, root);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(clock[0] >= 1000, "the fake clock advanced past the timeout: " + clock[0]);
|
||||
long closes = herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("pane.close"))
|
||||
.filter(c -> "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
assertEquals(1, closes, "the worker pane was reaped on timeout (no orphan)");
|
||||
assertNotNull(ex.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsHandleWhenGateDisabled(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerHandle handle = service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null));
|
||||
assertNotNull(handle, "spawn returns a handle when the gate is disabled");
|
||||
assertFalse(herdr.called("agent.get"), "no polling when the gate is disabled");
|
||||
}
|
||||
|
||||
// --- CB-508: pinned OpenAI-compatible endpoint (e.g. a local vLLM) ---------------------------
|
||||
|
||||
/** A profile with a baseUrl but no model provider prefix cannot be resolved — fail loudly. */
|
||||
private static BridgedConfig.Profile pinnedCfg(String model, String baseUrl, String mcpUrl) {
|
||||
return new BridgedConfig.Profile("local", baseUrl, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, null, null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
@Test
|
||||
void baseUrlDeclaresACustomOpenAiCompatibleProvider(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash", "http://127.0.0.1:8000", null))
|
||||
.spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a pinned endpoint needs a config file even with no bridge MCP url");
|
||||
JsonNode provider = new ObjectMapper().readTree(Path.of(cfgPath).toFile())
|
||||
.path("provider").path("local-vllm");
|
||||
|
||||
assertFalse(provider.isMissingNode(), "the provider id comes from the model selector");
|
||||
assertEquals("@ai-sdk/openai-compatible", provider.path("npm").asText());
|
||||
assertEquals("http://127.0.0.1:8000/v1", provider.path("options").path("baseURL").asText(),
|
||||
"a bare host:port gets /v1 appended — that is where these servers mount the API");
|
||||
assertFalse(provider.path("options").path("apiKey").asText().isBlank(),
|
||||
"the AI SDK requires a non-empty key even when the server ignores it");
|
||||
assertFalse(provider.path("models").path("deepseek-v4-flash").isMissingNode(),
|
||||
"the model half of the selector is declared under the provider");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBaseUrlThatAlreadyCarriesAPathIsUsedVerbatim(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/m", "http://127.0.0.1:8000/openai/v1", null)).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("http://127.0.0.1:8000/openai/v1",
|
||||
json.path("provider").path("local-vllm").path("options").path("baseURL").asText(),
|
||||
"an endpoint mounted on a custom path must not have /v1 bolted on");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointRejectsAModelWithNoProviderPrefix(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher =
|
||||
service(herdr, root, pinnedCfg("deepseek-v4-flash", "http://127.0.0.1:8000", null));
|
||||
|
||||
// Silently falling back to the default gateway would point the worker at the wrong LLM
|
||||
// while looking healthy — the one failure mode worth being loud about.
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, launcher::spawn);
|
||||
assertTrue(e.getMessage().contains("<provider>/<model>"), "the error says how to fix it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointAndTheBridgeMcpCoexistInOneConfig(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash",
|
||||
"http://127.0.0.1:8000", "http://127.0.0.1:8766/mcp")).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("remote", json.path("mcp").path("bridge").path("type").asText(),
|
||||
"pinning an endpoint must not drop the bridge MCP mount");
|
||||
assertFalse(json.path("provider").path("local-vllm").isMissingNode(),
|
||||
"and the provider block is still declared alongside it");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the reply charter survives too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noBaseUrlDeclaresNoProviderSoTheDefaultGatewayIsUsed(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("opencode/some-free-model", "http://127.0.0.1:8766/mcp", null))
|
||||
.spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertTrue(json.path("provider").isMissingNode(),
|
||||
"without a baseUrl opencode resolves its own provider as before");
|
||||
}
|
||||
}
|
||||
@@ -1,90 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.FileTime;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* {@link OpenCodeSessionDiscovery} matches an opencode session record by the worker's cwd (its
|
||||
* {@code directory}) against opencode's on-disk storage. These tests populate a TEMP storage root
|
||||
* themselves — never the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
class OpenCodeSessionDiscoveryTest {
|
||||
|
||||
/**
|
||||
* Write a session record {@code {"id":..., "directory":...}} under
|
||||
* {@code <root>/session/<projectID>/<fileName>} and stamp it with a known last-modified time,
|
||||
* so "most recently modified wins" is deterministic. Static so the launcher test can reuse it.
|
||||
*/
|
||||
static void writeRecord(Path root, String projectId, String fileName, String id,
|
||||
String directory, long lastModifiedEpochMillis) throws Exception {
|
||||
Path dir = root.resolve("session").resolve(projectId);
|
||||
Files.createDirectories(dir);
|
||||
Path file = dir.resolve(fileName);
|
||||
Files.writeString(file, "{\"id\":\"" + id + "\",\"directory\":\"" + directory
|
||||
+ "\",\"projectID\":\"" + projectId + "\",\"version\":\"1.1.31\"}");
|
||||
Files.setLastModifiedTime(file, FileTime.fromMillis(lastModifiedEpochMillis));
|
||||
}
|
||||
|
||||
@Test
|
||||
void findsTheRecordWhoseDirectoryEqualsTheCwd(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "ses_b.json", "ses_bbb", "/w/b", 2000L);
|
||||
|
||||
assertEquals("ses_bbb", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/b"),
|
||||
"the record whose directory equals the cwd is the one found");
|
||||
assertEquals("ses_aaa", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingDirectoryYieldsNullRatherThanAMismatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/other"),
|
||||
"no record for this cwd yet → null, not a wrong session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void prefersTheMostRecentlyModifiedRecordWhenSeveralMatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "old.json", "ses_old", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "new.json", "ses_new", "/w/a", 5000L);
|
||||
|
||||
assertEquals("ses_new", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"the freshest record for the cwd wins");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingOrEmptyStorageRootYieldsNullWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
// Missing: no session dir at all under the root.
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
|
||||
// Present but empty: a session dir with nothing in it produces no match, not a throw.
|
||||
Path emptyRoot = root.resolve("empty");
|
||||
Files.createDirectories(emptyRoot.resolve("session"));
|
||||
assertNull(new OpenCodeSessionDiscovery(emptyRoot).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankOrNullDirectoryYieldsNull(@TempDir Path root) {
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertNull(discovery.sessionIdForDirectory(null));
|
||||
assertNull(discovery.sessionIdForDirectory(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMalformedRecordIsSkippedRatherThanFatal(@TempDir Path root) throws Exception {
|
||||
// A record that fails to parse must not abort the scan of its siblings.
|
||||
Path dir = root.resolve("session").resolve("p1");
|
||||
Files.createDirectories(dir);
|
||||
Files.writeString(dir.resolve("broken.json"), "{not valid json");
|
||||
writeRecord(root, "p1", "good.json", "ses_good", "/w/a", 1000L);
|
||||
|
||||
assertEquals("ses_good", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"an unreadable record is skipped; a later valid one still matches");
|
||||
}
|
||||
}
|
||||
@@ -1,612 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.inject.CompletionResolver;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* The message layer's resolution paths (CB-104 reply + CB-106 completion fallback). The turn is
|
||||
* driven deterministically by feeding {@code onStatus} rather than running a real poller.
|
||||
*/
|
||||
class MessageServiceTest {
|
||||
|
||||
private static final String T = "term_a";
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr().readText("BUILD GREEN: 391 files");
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private final Rendezvous rendezvous = new Rendezvous();
|
||||
private final CompletionResolver completion = new CompletionResolver(agents, rendezvous);
|
||||
private final Injector injector = new Injector(agents, completion);
|
||||
private final InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
private final MessageService messages = new MessageService(agents, injector, rendezvous, inbox);
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
// CB-520: the inbox only peeks/acks targets it owns.
|
||||
inbox.own(T);
|
||||
}
|
||||
|
||||
/** Run {@code send} on a background thread; the current thread drives the worker's turn. */
|
||||
private CompletableFuture<MessageService.Reply> sendAsync() {
|
||||
return CompletableFuture.supplyAsync(() -> messages.send(T, "do the task", 5000));
|
||||
}
|
||||
|
||||
private void awaitWaiting() throws InterruptedException {
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while (!rendezvous.isWaiting(T) && System.currentTimeMillis() < deadline) {
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertTrue(rendezvous.isWaiting(T), "send should have opened its rendezvous waiter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackResolvesATurnThatNeverCalledBridgeReply() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
herdr.readText("$ prompt"); // pre-turn pane: no answer yet (baseline reference)
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver the task (baselines the pre-turn content)
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up and works
|
||||
herdr.readText("BUILD GREEN: 391 files"); // the worker's turn produced new output
|
||||
injector.onStatus(T, AgentStatus.IDLE); // working → idle: turn complete, no bridge_reply
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.COMPLETED_UNREPLIED, reply.outcome(),
|
||||
"an unreplied but finished turn resolves via the completion fallback");
|
||||
assertEquals("BUILD GREEN: 391 files", reply.text(), "the scraped transcript tail is returned");
|
||||
assertTrue(reply.completed(), "a scraped completion still counts as completed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitBridgeReplyResolvesAsReplied() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker working
|
||||
assertTrue(rendezvous.resolve(T, "LGTM ship it"), "an explicit reply resolves the send");
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, reply.outcome());
|
||||
assertEquals("LGTM ship it", reply.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWedgedWorkerResolvesTheSendAsFailedWithTheErrorContext() throws Exception {
|
||||
herdr.readText("API Error: Unable to connect to API (ENOTFOUND)");
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker starts the turn
|
||||
for (int i = 0; i < 130; i++) injector.onStatus(T, AgentStatus.UNKNOWN); // then wedges (CB-109)
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.WORKER_FAILED, reply.outcome());
|
||||
assertFalse(reply.completed(), "a wedge is terminal but not a successful completion");
|
||||
assertTrue(reply.text().contains("ENOTFOUND"), "the error screen is carried as the failure reason");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerThatVanishesMidTurnResolvesTheSendAsFailed() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker starts the turn
|
||||
// The worker's pane crashes — the poller sees a *_not_found and drops it (CB-110).
|
||||
injector.drop(T, new HerdrException("worker gone", "pane_not_found", null));
|
||||
|
||||
MessageService.Reply reply = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.WORKER_FAILED, reply.outcome(),
|
||||
"a delivered send whose worker vanishes fails instead of hanging to the timeout");
|
||||
assertFalse(reply.completed());
|
||||
}
|
||||
|
||||
// --- bridge_ask reverse rendezvous (CB-205) ------------------------------------------------
|
||||
|
||||
@Test
|
||||
void askSurfacesAsAQuestionAndTheAnswerResumesTheSameTurn() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
// The worker asks mid-turn on its own thread; the call blocks for the primary's answer.
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
|
||||
// The primary's blocking send unblocks with the question and a turnId to answer on.
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertEquals("which config file?", q.text());
|
||||
assertNotNull(q.turnId(), "a question carries a turnId to answer on");
|
||||
|
||||
// The primary answers via bridge_send(turnId); this blocks again for the worker's reply.
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||
|
||||
// The worker's ask returns the answer — it resumes the same turn.
|
||||
MessageService.AskResult a = ask.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a.outcome());
|
||||
assertEquals("config.yaml", a.answer());
|
||||
|
||||
// The resumed worker finishes with a structured reply, resolving the answering send.
|
||||
awaitWaiting(); // the answering send has (re)opened its forward waiter
|
||||
assertTrue(rendezvous.resolve(T, "done"), "the worker's final reply resolves the answering send");
|
||||
MessageService.Reply done = answer.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, done.outcome());
|
||||
assertEquals("done", done.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void duplicateAsksFromTheSameSessionCoalesceToOneTurn() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
// A transport retry: two concurrent bridge_ask calls from the same worker session.
|
||||
CompletableFuture<MessageService.AskResult> ask1 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
CompletableFuture<MessageService.AskResult> ask2 =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
|
||||
// The primary's single blocked send surfaces exactly ONE question (one turnId).
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertEquals("which config file?", q.text());
|
||||
assertNotNull(q.turnId(), "only one turnId should be minted");
|
||||
|
||||
// The primary answers that one turnId; both asks unblock with the same answer.
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||
|
||||
MessageService.AskResult a1 = ask1.get(5, TimeUnit.SECONDS);
|
||||
MessageService.AskResult a2 = ask2.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a1.outcome());
|
||||
assertEquals("config.yaml", a1.answer());
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a2.outcome());
|
||||
assertEquals("config.yaml", a2.answer());
|
||||
|
||||
// The resumed worker finishes with a structured reply, resolving the answering send.
|
||||
awaitWaiting();
|
||||
assertTrue(rendezvous.resolve(T, "done"), "the worker's final reply resolves the answering send");
|
||||
MessageService.Reply done = answer.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, done.outcome());
|
||||
assertEquals("done", done.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void askWithNoOpenDelegationReturnsNoWaiter() {
|
||||
MessageService.AskResult r = messages.ask(T, "anyone listening?", 500);
|
||||
assertEquals(MessageService.AskOutcome.NO_WAITER, r.outcome(),
|
||||
"a question with no blocked send has no primary to answer it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void askTimesOutWhenThePrimaryNeverAnswers() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
MessageService.AskResult r = messages.ask(T, "still there?", 200); // primary never answers
|
||||
assertEquals(MessageService.AskOutcome.TIMED_OUT, r.outcome());
|
||||
|
||||
// The send itself already unblocked with the question the instant the ask surfaced.
|
||||
MessageService.Reply q = send.get(2, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
}
|
||||
|
||||
@Test
|
||||
void answeringAnUnknownTurnIsStale() {
|
||||
MessageService.Reply r = messages.answer(T + "#999", "too late", 500);
|
||||
assertEquals(MessageService.Outcome.STALE_TURN, r.outcome(),
|
||||
"an answer to a turn that never existed (or already lapsed) is stale, not a hang");
|
||||
}
|
||||
|
||||
// --- timeout, answer, poll, and lock-contention edges ----------------------------------
|
||||
|
||||
@Test
|
||||
void sendTimesOutBeforeDeliveryIsQueuedNotWorking() {
|
||||
// Nothing ever delivers the message and nothing resolves the send, so the reply future
|
||||
// times out with delivery still incomplete — the message is still queued for the worker.
|
||||
MessageService.Reply r = messages.send(T, "never delivered", 50);
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_QUEUED, r.outcome(),
|
||||
"an undelivered send that times out is still queued, not working");
|
||||
assertNull(r.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendTimesOutAfterDeliveryIsStillWorking() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "do the task", 300));
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver — the delivered future now completes
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker starts but never replies
|
||||
// No rendezvous.resolve(T, ...) — the reply future rides out its short timeout.
|
||||
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, r.outcome(),
|
||||
"a delivered send whose worker never replies times out as still working");
|
||||
}
|
||||
|
||||
@Test
|
||||
void answerTimesOutWhenTheResumedWorkerNeverReplies() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config file?", 5000));
|
||||
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertNotNull(q.turnId());
|
||||
|
||||
// The primary answers, unblocking the worker; but the worker never sends the follow-up
|
||||
// bridge_reply, so the answering send rides out its short window as still-working.
|
||||
MessageService.Reply answer = messages.answer(q.turnId(), "config.yaml", 200);
|
||||
assertEquals(MessageService.Outcome.TIMED_OUT_WORKING, answer.outcome(),
|
||||
"an answered worker that never replies times out as still working");
|
||||
|
||||
MessageService.AskResult a = ask.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.AskOutcome.ANSWERED, a.outcome());
|
||||
assertEquals("config.yaml", a.answer());
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollReturnsNullForAnUnknownTicket() {
|
||||
assertNull(messages.poll("task-999999"), "a ticket that was never minted is unknown");
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollReportsACompletedTicket() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker works
|
||||
assertTrue(rendezvous.resolve(T, "async result"), "a reply resolves the async send");
|
||||
|
||||
// Wait for the background send to finish and publish a DONE view.
|
||||
MessageService.TaskView view = null;
|
||||
long deadline = System.currentTimeMillis() + 2000;
|
||||
while (view == null || view.phase() != MessageService.Phase.DONE) {
|
||||
if (System.currentTimeMillis() >= deadline) break;
|
||||
view = messages.poll(ticket);
|
||||
//noinspection BusyWait
|
||||
Thread.sleep(5);
|
||||
}
|
||||
assertNotNull(view, "a resolved async send must become DONE");
|
||||
assertEquals(MessageService.Phase.DONE, view.phase());
|
||||
assertEquals("async result", view.reply(), "the completed ticket reports the reply");
|
||||
assertEquals("reply", view.replySource(), "a structured bridge_reply is sourced from 'reply'");
|
||||
}
|
||||
|
||||
@Test
|
||||
void concurrentSendToSameSessionWhileFirstHoldsItIsBusy() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> first =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "first", 5000));
|
||||
awaitWaiting(); // the first send now holds the session lock, blocked on its reply
|
||||
|
||||
// A second send to the SAME session cannot take the lock within its short window.
|
||||
MessageService.Reply busy = messages.send(T, "second", 100);
|
||||
assertEquals(MessageService.Outcome.BUSY, busy.outcome(),
|
||||
"a second send while another holds the session is busy, not a hang");
|
||||
assertNull(busy.text());
|
||||
|
||||
// Release the first send so it resolves cleanly and the test thread is not left pinned.
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver the first message
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up
|
||||
assertTrue(rendezvous.resolve(T, "first done"), "the first send resolves with a reply");
|
||||
MessageService.Reply firstReply = first.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, firstReply.outcome());
|
||||
assertEquals("first done", firstReply.text());
|
||||
}
|
||||
|
||||
// --- CB-548: delegator ownership is recorded only on an ACCEPTED send ----------------------
|
||||
|
||||
private static final String LEAD_L = "term_lead_l";
|
||||
private static final String LEAD_A = "term_lead_a";
|
||||
|
||||
/**
|
||||
* The bug CB-548 fixes: L holds worker W, then architect A attempts W and times out BUSY. With
|
||||
* delegator ownership recorded at {@code bridge_send} <em>request</em> time, A's rejected call
|
||||
* would overwrite L — and W's late no-waiter reply would be pushed to A, who never owned the
|
||||
* turn. The accepted-delivery hook must not fire for a BUSY send, so L stays the delegator.
|
||||
*/
|
||||
@Test
|
||||
void busySenderDoesNotBecomeTheDelegatingOwner() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
// L accepts a delegation to W: the send wins the lock and queues delivery → L is recorded.
|
||||
CompletableFuture<MessageService.Reply> first = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "first", 5000, () -> reg.recordDelegation(T, LEAD_L)));
|
||||
awaitWaiting();
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"an accepted send owns the delegation");
|
||||
|
||||
// A attempts W while L holds it → BUSY (lock never taken) → its hook never fires.
|
||||
MessageService.Reply busy = messages.send(T, "second", 100, () -> reg.recordDelegation(T, LEAD_A));
|
||||
assertEquals(MessageService.Outcome.BUSY, busy.outcome());
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"a BUSY send must not steal the delegator ownership it never earned");
|
||||
|
||||
// L completes so the test thread is not left pinned.
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
assertTrue(rendezvous.resolve(T, "first done"));
|
||||
MessageService.Reply firstReply = first.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, firstReply.outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* Once L's accepted delegation is fully done, a later <em>accepted</em> send from A may
|
||||
* legitimately become the new delegator — ownership follows the turn, not the first caller.
|
||||
*/
|
||||
@Test
|
||||
void anAcceptedSendAfterThePriorOwnerFinishesBecomesTheNewOwner() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
CompletableFuture<MessageService.Reply> first = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "first", 5000, () -> reg.recordDelegation(T, LEAD_L)));
|
||||
awaitWaiting();
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
assertTrue(rendezvous.resolve(T, "first done"));
|
||||
MessageService.Reply firstReply = first.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, firstReply.outcome());
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(), "L owned the first turn");
|
||||
|
||||
// L finished; A's later accepted send takes the delegation over.
|
||||
CompletableFuture<MessageService.Reply> second = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "second", 5000, () -> reg.recordDelegation(T, LEAD_A)));
|
||||
awaitWaiting();
|
||||
assertEquals(LEAD_A, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"an accepted send after the owner finished becomes the new delegator");
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
assertTrue(rendezvous.resolve(T, "second done"));
|
||||
MessageService.Reply secondReply = second.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, secondReply.outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548 requirement: answering an existing {@code bridge_ask} is the SAME delegation, so it must
|
||||
* not rewrite ownership. L accepted the send (owned), the worker paused to ask, and L answers via
|
||||
* turnId — ownership stays L throughout; the answer path never touches the registry.
|
||||
*/
|
||||
@Test
|
||||
void answeringAnAskDoesNotRewriteDelegatorOwnership() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
CompletableFuture<MessageService.Reply> send = CompletableFuture.supplyAsync(
|
||||
() -> messages.send(T, "do X", 5000, () -> reg.recordDelegation(T, LEAD_L)));
|
||||
awaitWaiting();
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(), "L owns the delegation");
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up, then pauses to ask
|
||||
|
||||
CompletableFuture<MessageService.AskResult> ask =
|
||||
CompletableFuture.supplyAsync(() -> messages.ask(T, "which config?", 5000));
|
||||
MessageService.Reply q = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.QUESTION, q.outcome());
|
||||
assertNotNull(q.turnId());
|
||||
|
||||
// L answers the ask on the same turn; the answer path must not touch ownership.
|
||||
CompletableFuture<MessageService.Reply> answer =
|
||||
CompletableFuture.supplyAsync(() -> messages.answer(q.turnId(), "config.yaml", 5000));
|
||||
assertEquals("config.yaml", ask.get(5, TimeUnit.SECONDS).answer());
|
||||
awaitWaiting(); // the answering send reopened its forward waiter
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"answering an ask keeps L as the delegator — ownership is not rewritten");
|
||||
|
||||
assertTrue(rendezvous.resolve(T, "done"));
|
||||
assertEquals(MessageService.Outcome.REPLIED, answer.get(5, TimeUnit.SECONDS).outcome());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-548: {@code onAccepted} is a public callback, so a throwing one must not orphan the turn.
|
||||
* The waiter is opened first, then the hook runs BEFORE delivery is queued — so a throw fails
|
||||
* the send loudly, closes its waiter, and never enqueues a message the worker would pick up and
|
||||
* reply into the void.
|
||||
*/
|
||||
@Test
|
||||
void aThrowingAcceptedHookLeavesNoStaleWaiterOrQueuedOrphan() {
|
||||
assertThrows(IllegalStateException.class,
|
||||
() -> messages.send(T, "doomed", 500,
|
||||
() -> { throw new IllegalStateException("ownership hook failed"); }),
|
||||
"a throwing ownership hook fails the send loudly");
|
||||
|
||||
assertFalse(rendezvous.isWaiting(T), "the failed send must not leave a stale rendezvous waiter");
|
||||
// Give the injector a delivery window: with nothing enqueued, nothing may reach the worker.
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
boolean doomedQueued = herdr.calls.stream()
|
||||
.anyMatch(c -> c.method().equals("agent.prompt")
|
||||
&& String.valueOf(c.params()).contains("doomed"));
|
||||
assertFalse(doomedQueued, "a throwing ownership hook must not leave a queued, orphanable message");
|
||||
}
|
||||
|
||||
/**
|
||||
* The async (fire-and-poll) path runs the same {@code send} on a background thread, so the
|
||||
* accepted-delivery hook must thread through it — ownership is recorded exactly as blocking sends.
|
||||
*/
|
||||
@Test
|
||||
void asyncSendRecordsOwnershipOnAcceptance() throws Exception {
|
||||
PrimaryRegistry reg = new PrimaryRegistry(null);
|
||||
messages.sendAsync(T, "async task", () -> reg.recordDelegation(T, LEAD_L));
|
||||
awaitWaiting(); // the background send won the lock, queued, and opened its waiter
|
||||
assertEquals(LEAD_L, reg.nudgeTargetFor(T).orElseThrow(),
|
||||
"the async path records delegator ownership on acceptance, like the blocking path");
|
||||
}
|
||||
|
||||
// --- CB-307 reply inbox ----------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void replyQueuesInInboxWhenNoSendIsOpen() {
|
||||
// No send is open for this session — reply should queue in the inbox.
|
||||
assertTrue(messages.reply(T, "queued-text"), "reply should succeed (queued)");
|
||||
|
||||
var drained = messages.drainReplies(T);
|
||||
assertEquals(1, drained.size());
|
||||
assertEquals("queued-text", drained.getFirst().content());
|
||||
}
|
||||
|
||||
@Test
|
||||
void replyResolvesOpenSendDoesNotQueue() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitUninterruptibly(T);
|
||||
|
||||
// An explicit reply resolves the open send.
|
||||
assertTrue(messages.reply(T, "send-resolved"), "reply should succeed (resolved live send)");
|
||||
|
||||
// The inbox should be empty — the reply went to the send, not the inbox.
|
||||
assertTrue(messages.drainReplies(T).isEmpty(), "no reply in the inbox");
|
||||
|
||||
MessageService.Reply r = send.get(3, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.REPLIED, r.outcome());
|
||||
assertEquals("send-resolved", r.text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainRepliesReturnsAllPendingThenEmptyOnNextCall() {
|
||||
messages.reply(T, "msg-1");
|
||||
messages.reply(T, "msg-2");
|
||||
|
||||
var first = messages.drainReplies(T);
|
||||
assertEquals(2, first.size());
|
||||
|
||||
var second = messages.drainReplies(T);
|
||||
assertTrue(second.isEmpty(), "second drain should be empty (acked)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aQuestionIsNeverQueuedInTheInbox() {
|
||||
// No send is open — bridge_ask with no delegation returns NO_WAITER,
|
||||
// and the question text MUST NOT appear in the reply inbox.
|
||||
// The inbox is only fed by MessageService.reply(), not by bridge_ask.
|
||||
MessageService.AskResult r = messages.ask(T, "anyone there?", 500);
|
||||
assertEquals(MessageService.AskOutcome.NO_WAITER, r.outcome(),
|
||||
"bridge_ask with no open delegation must return NO_WAITER, never queued");
|
||||
|
||||
assertTrue(messages.drainReplies(T).isEmpty(), "questions must never be queued");
|
||||
}
|
||||
|
||||
@Test
|
||||
void completionFallbackIsNeverQueued() throws Exception {
|
||||
// The fallback resolves a captured waiter, never the inbox.
|
||||
CompletableFuture<MessageService.Reply> send = sendAsync();
|
||||
awaitUninterruptibly(T);
|
||||
injectDelivery();
|
||||
|
||||
// The worker never sends bridge_reply, but the turn completes.
|
||||
herdr.readText("done-scraped");
|
||||
completion.onTurnComplete(T); // The fallback arms and resolves the captured waiter.
|
||||
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.COMPLETED_UNREPLIED, r.outcome());
|
||||
|
||||
// The inbox should be empty — the reply went to the captured waiter.
|
||||
assertTrue(messages.drainReplies(T).isEmpty(), "completion fallback must not queue");
|
||||
}
|
||||
|
||||
// --- helpers ---------------------------------------------------------------------------
|
||||
|
||||
/** Like {@link #awaitWaiting()} but rethrows as unchecked. */
|
||||
private void awaitUninterruptibly(String session) {
|
||||
try {
|
||||
awaitWaiting();
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException(e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Set up a delivered turn so the worker is working, ready for an ask or completion. */
|
||||
private void injectDelivery() {
|
||||
herdr.readText("$ prompt"); // pre-turn content baseline
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver the task
|
||||
injector.onStatus(T, AgentStatus.WORKING); // worker picks it up
|
||||
}
|
||||
|
||||
// --- CB-516: a released session must not leave a send hanging ------------------------------
|
||||
|
||||
/**
|
||||
* The bug this fixes: tearing a worker down left its rendezvous waiter open, so a blocking send
|
||||
* kept blocking and an async one kept reporting PENDING until the 30-minute async timeout —
|
||||
* even though the worker provably no longer existed.
|
||||
*/
|
||||
@Test
|
||||
void abandonFailsASendThatIsStillWaitingOnAReleasedSession() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "work", 30_000));
|
||||
awaitWaiting();
|
||||
|
||||
assertTrue(messages.abandon(T, "session released"), "a live waiter is abandoned");
|
||||
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals(MessageService.Outcome.WORKER_FAILED, r.outcome(),
|
||||
"an abandoned send fails rather than riding out its timeout");
|
||||
assertEquals("session released", r.text(), "the caller is told why");
|
||||
}
|
||||
|
||||
@Test
|
||||
void abandonIsANoOpWhenNobodyIsWaiting() {
|
||||
assertFalse(messages.abandon(T, "session released"),
|
||||
"no open send ⇒ nothing to abandon");
|
||||
}
|
||||
|
||||
@Test
|
||||
void abandonDoesNotOverwriteAnAlreadyResolvedSend() throws Exception {
|
||||
CompletableFuture<MessageService.Reply> send =
|
||||
CompletableFuture.supplyAsync(() -> messages.send(T, "work", 30_000));
|
||||
awaitWaiting();
|
||||
assertTrue(rendezvous.resolve(T, "the real answer"));
|
||||
|
||||
assertFalse(messages.abandon(T, "session released"),
|
||||
"a send already answered by the worker must not be clobbered");
|
||||
MessageService.Reply r = send.get(5, TimeUnit.SECONDS);
|
||||
assertEquals("the real answer", r.text());
|
||||
}
|
||||
|
||||
/** The async path is the one that hung: poll must report FAILED, not PENDING forever. */
|
||||
@Test
|
||||
void anAbandonedAsyncTaskPollsAsFailedNotPending() throws Exception {
|
||||
String ticket = messages.sendAsync(T, "long task");
|
||||
awaitWaiting();
|
||||
assertEquals(MessageService.Phase.PENDING, messages.poll(ticket).phase());
|
||||
|
||||
messages.abandon(T, "session released");
|
||||
|
||||
MessageService.TaskView view = null;
|
||||
long deadline = System.currentTimeMillis() + 3000;
|
||||
while (System.currentTimeMillis() < deadline) {
|
||||
view = messages.poll(ticket);
|
||||
if (view.phase() != MessageService.Phase.PENDING) break;
|
||||
Thread.sleep(10);
|
||||
}
|
||||
assertNotNull(view);
|
||||
assertEquals(MessageService.Phase.FAILED, view.phase(),
|
||||
"a delegation whose worker is gone must not keep reporting PENDING");
|
||||
assertTrue(view.detail() != null && view.detail().contains("released"),
|
||||
"and the detail says why, rather than 'worker unknown'");
|
||||
}
|
||||
}
|
||||
@@ -1,318 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.BeforeEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Unit tests for {@link ReplyPushLoop}: decision logic, nudge injection, idempotency,
|
||||
* bounded reminders, and stop conditions.
|
||||
*
|
||||
* <p>Uses a {@link RecordingHerdrClient} that synchronizes access to its call list so the
|
||||
* scheduler thread and test thread never have memory ordering issues. The {@code decide()}
|
||||
* tests use a simple client with no concurrency concern.
|
||||
*/
|
||||
class ReplyPushLoopTest {
|
||||
|
||||
private static final String PRIMARY = "term_primary";
|
||||
private static final String WORKER = "term_worker";
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
private PrimaryRegistry registry;
|
||||
private AgentControl agents;
|
||||
private InMemoryReplyInbox inbox;
|
||||
private ScheduledExecutorService scheduler;
|
||||
|
||||
@BeforeEach
|
||||
void setUp() {
|
||||
registry = new PrimaryRegistry(PRIMARY);
|
||||
inbox = new InMemoryReplyInbox();
|
||||
inbox.own(WORKER); // CB-520: the inbox only peeks/acks targets it owns
|
||||
scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
}
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
scheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// --- decide() logic ------------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void decideWithoutPrimaryIsStop() {
|
||||
agents = agentWithStatus("idle");
|
||||
var loop = new ReplyPushLoop(
|
||||
new PrimaryRegistry(null), agents, inbox, scheduler, 5, 100);
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop.decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideWithEmptyInboxIsStop() {
|
||||
agents = agentWithStatus("idle");
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideAtCapIsStop() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop(2, 100).decide(WORKER, 2));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithInjectablePrimaryIsInject() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithBlockedPrimaryIsInject() {
|
||||
agents = agentWithStatus("blocked");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0),
|
||||
"BLOCKED is injectable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithDonePrimaryIsInject() {
|
||||
agents = agentWithStatus("done");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0),
|
||||
"DONE is injectable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithBusyPrimaryIsWaitBusy() {
|
||||
agents = agentWithStatus("working");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.WAIT_BUSY, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideUnderCapWithUnknownPrimaryIsWaitBusy() {
|
||||
agents = agentWithStatus("unknown");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.WAIT_BUSY, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void decideStopsAfterInboxIsEmptied() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
assertEquals(ReplyPushLoop.Action.INJECT, loop().decide(WORKER, 0));
|
||||
inbox.ack(WORKER, "m1");
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop().decide(WORKER, 0));
|
||||
}
|
||||
|
||||
// --- onReplyQueued integration -------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void injectablePrimaryCausesExactlyOneNudge() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
|
||||
loop(1, 50).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"one nudge (1 agent.prompt call) should have been sent");
|
||||
|
||||
// Exactly one nudge = exactly 1 agent.prompt call (it submits itself)
|
||||
assertEquals(1, rec.sendCount());
|
||||
assertTrue(rec.sentParams().stream()
|
||||
.anyMatch(e -> e.getValue().toString().contains("bridge_poll")),
|
||||
"nudge text should contain bridge_poll");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onReplyQueuedIsIdempotentPerTarget() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
|
||||
var loop = loop(1, 100);
|
||||
loop.onReplyQueued(WORKER);
|
||||
loop.onReplyQueued(WORKER); // second call — should be a no-op
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"expected exactly one nudge (1 prompt)");
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, rec.sendCount(),
|
||||
"second onReplyQueued must not trigger another nudge");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendsUpToCapThenStops() throws Exception {
|
||||
int cap = 2;
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
rec.sendLatch = new CountDownLatch(cap);
|
||||
|
||||
loop(cap, 50).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(5, TimeUnit.SECONDS),
|
||||
cap + " nudges (" + cap + " prompts) should have fired");
|
||||
Thread.sleep(300);
|
||||
assertEquals(cap, rec.sendCount(),
|
||||
"exactly " + cap + " agent.prompt calls (cap=" + cap + ")");
|
||||
}
|
||||
|
||||
// --- nudge format --------------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void isActiveReflectsALiveScheduleForTheHeartbeatStandDown() {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
ReplyPushLoop loop = loop(1, 100_000); // long backoff so the tick cannot fire mid-test
|
||||
|
||||
loop.onReplyQueued(WORKER);
|
||||
assertTrue(loop.isActive(), "CB-551: the heartbeat must stand aside while a reminder is live");
|
||||
|
||||
loop.stop();
|
||||
assertFalse(loop.isActive(), "stopping clears the active schedule");
|
||||
}
|
||||
|
||||
@Test
|
||||
void nudgeFormatIsCorrect() {
|
||||
String nudge = ReplyPushLoop.NUDGE_FORMAT.formatted(WORKER, WORKER);
|
||||
assertTrue(nudge.contains("Worker term_worker"));
|
||||
assertTrue(nudge.contains("bridge_poll(target=term_worker)"));
|
||||
}
|
||||
|
||||
// --- metrics (CB-512) ----------------------------------------------------------------------
|
||||
|
||||
@Test
|
||||
void successfulNudgeIncrementsDelivered() throws Exception {
|
||||
var rec = recordingClient();
|
||||
agents = new AgentControl(rec);
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
Metrics metrics = new Metrics();
|
||||
|
||||
loop(1, 50, metrics).onReplyQueued(WORKER);
|
||||
|
||||
assertTrue(rec.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"one nudge (1 agent.prompt call) should have been sent");
|
||||
// The delivered count is bumped on the scheduler thread right after the send that releases
|
||||
// the latch — settle briefly so the counter is published before we read it.
|
||||
Thread.sleep(200);
|
||||
assertEquals(1, metrics.count(BridgedMetrics.PUSH_NUDGES, "outcome", "delivered"),
|
||||
"a successfully sent nudge must count as delivered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reminderCapIncrementsExhausted() {
|
||||
agents = agentWithStatus("idle");
|
||||
inbox.publish(WORKER, "m1", "hello");
|
||||
Metrics metrics = new Metrics();
|
||||
|
||||
assertEquals(ReplyPushLoop.Action.STOP, loop(2, 100, metrics).decide(WORKER, 2));
|
||||
|
||||
assertEquals(1, metrics.count(BridgedMetrics.PUSH_NUDGES, "outcome", "exhausted"),
|
||||
"hitting the reminder cap must count as exhausted");
|
||||
assertEquals(0, metrics.count(BridgedMetrics.PUSH_NUDGES, "outcome", "delivered"));
|
||||
}
|
||||
|
||||
// --- helpers -------------------------------------------------------------------------------
|
||||
|
||||
private ReplyPushLoop loop() {
|
||||
return loop(5, 100);
|
||||
}
|
||||
|
||||
private ReplyPushLoop loop(int maxReminders, long backoffMs) {
|
||||
return new ReplyPushLoop(registry, agents, inbox, scheduler, maxReminders, backoffMs);
|
||||
}
|
||||
|
||||
private ReplyPushLoop loop(int maxReminders, long backoffMs, Metrics metrics) {
|
||||
return new ReplyPushLoop(registry, agents, inbox, scheduler, maxReminders, backoffMs, metrics);
|
||||
}
|
||||
|
||||
private static AgentControl agentWithStatus(String status) {
|
||||
return new AgentControl(new FakeHerdrClient(status));
|
||||
}
|
||||
|
||||
/** Non-recording (single-threaded) fake — safe for decide() tests. */
|
||||
private static final class FakeHerdrClient implements HerdrClient {
|
||||
private final String agentStatus;
|
||||
|
||||
FakeHerdrClient(String agentStatus) {
|
||||
this.agentStatus = agentStatus;
|
||||
}
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode()
|
||||
.set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", PRIMARY)
|
||||
.put("agent_status", agentStatus));
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Thread-safe recording fake that counts agent.prompt calls (protocol 19: one nudge = one
|
||||
* prompt). Uses synchronized access so the scheduler thread and test thread never race.
|
||||
*/
|
||||
private static final class RecordingHerdrClient implements HerdrClient {
|
||||
private final List<Map.Entry<String, Object>> calls =
|
||||
Collections.synchronizedList(new ArrayList<>());
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode()
|
||||
.set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", PRIMARY)
|
||||
.put("agent_status", "idle")); // recording double is always injectable
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
calls.add(Map.entry(method, params));
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
long sendCount() {
|
||||
return calls.size();
|
||||
}
|
||||
|
||||
List<Map.Entry<String, Object>> sentParams() {
|
||||
return List.copyOf(calls);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
|
||||
private static RecordingHerdrClient recordingClient() {
|
||||
return new RecordingHerdrClient();
|
||||
}
|
||||
}
|
||||
@@ -1,183 +0,0 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Unit tests for the placement policies. They run with no herdr and no launcher — pure selection
|
||||
* logic exercised through the descriptor type so CB-308 host expansion will not need to rewrite
|
||||
* these assertions.
|
||||
*/
|
||||
class PlacementPolicyTest {
|
||||
|
||||
private static Function<String, Integer> noSessions() {
|
||||
return name -> 0;
|
||||
}
|
||||
|
||||
private static PlacementContext ctx(List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable) {
|
||||
return new PlacementContext("b", candidates, liveCount, unreachable);
|
||||
}
|
||||
|
||||
private static PlacementContext ctx(List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount) {
|
||||
return ctx(candidates, liveCount, Set.of());
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedReturnsDefaultEvenIfOtherProfilesExist() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = ctx(List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b")), noSessions());
|
||||
assertEquals("b", policy.select(ctx).profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedFallsBackToFirstCandidateWhenNoDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null,
|
||||
List.of(PlacementCandidate.profile("a"), PlacementCandidate.profile("b")),
|
||||
noSessions(), Set.of());
|
||||
assertEquals("a", policy.select(ctx).profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedThrowsWhenNoProfilesAndNoDefault() {
|
||||
PlacementPolicy policy = PlacementPolicies.fixed();
|
||||
PlacementContext ctx = new PlacementContext(null, List.of(), noSessions(), Set.of());
|
||||
assertThrows(PlacementException.class, () -> policy.select(ctx));
|
||||
}
|
||||
|
||||
@Test
|
||||
void roundRobinCyclesThroughAvailableProfiles() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b"),
|
||||
PlacementCandidate.profile("c"));
|
||||
assertEquals("a", policy.select(ctx(candidates, noSessions())).profile());
|
||||
assertEquals("b", policy.select(ctx(candidates, noSessions())).profile());
|
||||
assertEquals("c", policy.select(ctx(candidates, noSessions())).profile());
|
||||
assertEquals("a", policy.select(ctx(candidates, noSessions())).profile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void roundRobinSkipsProfilesAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 2),
|
||||
PlacementCandidate.profile("b", 1.0f, null));
|
||||
Function<String, Integer> liveCount = Map.of("a", 2)::get;
|
||||
for (int i = 0; i < 5; i++) {
|
||||
assertEquals("b", policy.select(ctx(candidates, liveCount)).profile());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void roundRobinThrowsWhenAllAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.roundRobin();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, 1));
|
||||
Function<String, Integer> liveCount = Map.of("a", 1, "b", 1)::get;
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, liveCount)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedAlternatesEvenlyWithEqualWeights() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 0.5f, null),
|
||||
PlacementCandidate.profile("b", 0.5f, null));
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 100; i++) {
|
||||
String p = policy.select(ctx(candidates, noSessions())).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(50, a, "equal weights should split 50/50");
|
||||
assertEquals(50, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedHoldsThreeToOneRatio() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 0.75f, null),
|
||||
PlacementCandidate.profile("b", 0.25f, null));
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 40; i++) {
|
||||
String p = policy.select(ctx(candidates, noSessions())).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(30, a, "0.75/0.25 should yield a 3:1 ratio over a multiple of 4");
|
||||
assertEquals(10, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedSkipsProfileAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, null));
|
||||
Function<String, Integer> liveCount = name -> "a".equals(name) ? 1 : 0;
|
||||
for (int i = 0; i < 5; i++) {
|
||||
assertEquals("b", policy.select(ctx(candidates, liveCount)).profile());
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedThrowsWhenAllAtMaxLoad() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, 1));
|
||||
Function<String, Integer> liveCount = name -> 1;
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, liveCount)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedThrowsWhenAllUnreachable() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a"),
|
||||
PlacementCandidate.profile("b"));
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, noSessions(), Set.of("a", "b"))));
|
||||
assertTrue(e.getMessage().contains("unreachable"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void mixedExclusionMessageNamesBothReasons() {
|
||||
PlacementPolicy policy = PlacementPolicies.weighted();
|
||||
List<PlacementCandidate> candidates = List.of(
|
||||
PlacementCandidate.profile("a", 1.0f, 1),
|
||||
PlacementCandidate.profile("b", 1.0f, null));
|
||||
Function<String, Integer> liveCount = name -> "a".equals(name) ? 1 : 0;
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
unreachable.add("b");
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> policy.select(ctx(candidates, liveCount, unreachable)));
|
||||
assertTrue(e.getMessage().contains("1 at maxLoad"), e.getMessage());
|
||||
assertTrue(e.getMessage().contains("1 unreachable"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void unknownPolicyNameThrows() {
|
||||
assertThrows(IllegalArgumentException.class, () -> PlacementPolicies.fromName("random"));
|
||||
}
|
||||
}
|
||||
@@ -1,130 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
|
||||
/** Recording fake {@link Worktrees} for CB-301-ext acceptance tests (no live git). */
|
||||
public final class FakeWorktrees implements Worktrees {
|
||||
|
||||
public record AddCall(String repoRoot, String branch, String baseRef) {
|
||||
}
|
||||
|
||||
public record RemoveCall(String repoRoot, String worktreePath) {
|
||||
}
|
||||
|
||||
public record OverlayCall(String repoRoot, String worktreePath,
|
||||
List<String> requested, List<String> copied, List<String> skipWorktree) {
|
||||
}
|
||||
|
||||
public record RepoRootCall(String cwd) {
|
||||
}
|
||||
|
||||
private final List<AddCall> addCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<RemoveCall> removeCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<OverlayCall> overlayCalls = new CopyOnWriteArrayList<>();
|
||||
private final List<RepoRootCall> repoRootCalls = new CopyOnWriteArrayList<>();
|
||||
private final Set<String> existingPaths = ConcurrentHashMap.newKeySet();
|
||||
private final Set<String> trackedPaths = ConcurrentHashMap.newKeySet();
|
||||
private volatile RuntimeException addFailure;
|
||||
private volatile String repoRoot = "/repo";
|
||||
private volatile String prefix = "/worktrees";
|
||||
|
||||
public FakeWorktrees withRepoRoot(String root) {
|
||||
this.repoRoot = root;
|
||||
return this;
|
||||
}
|
||||
|
||||
public FakeWorktrees withPrefix(String prefix) {
|
||||
this.prefix = prefix;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Paths that exist in the primary repo and will be copied to the worktree. */
|
||||
public FakeWorktrees exists(String... paths) {
|
||||
Collections.addAll(existingPaths, paths);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Paths that exist AND are tracked, so overlayParity should --skip-worktree them. */
|
||||
public FakeWorktrees track(String... paths) {
|
||||
exists(paths);
|
||||
Collections.addAll(trackedPaths, paths);
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make subsequent {@link #add} calls throw (simulates git worktree add failure). */
|
||||
public FakeWorktrees failAdd(String message) {
|
||||
this.addFailure = new WorktreeException(message);
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
addCalls.add(new AddCall(repoRoot, branch, baseRef));
|
||||
if (addFailure != null) {
|
||||
throw addFailure;
|
||||
}
|
||||
// The branch already carries a unique nonce, so the derived path is distinct per acquire
|
||||
// without an extra counter — keep it a pure function of the branch the test can predict.
|
||||
return prefix + "/" + branch.replace('/', '_');
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
removeCalls.add(new RemoveCall(repoRoot, worktreePath));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
List<String> copied = new java.util.ArrayList<>();
|
||||
List<String> skipped = new java.util.ArrayList<>();
|
||||
for (String rel : overlay) {
|
||||
if (!existingPaths.contains(rel)) {
|
||||
continue; // missing source is silently skipped
|
||||
}
|
||||
copied.add(rel);
|
||||
if (trackedPaths.contains(rel)) {
|
||||
skipped.add(rel);
|
||||
}
|
||||
}
|
||||
overlayCalls.add(new OverlayCall(repoRoot, worktreePath, List.copyOf(overlay),
|
||||
List.copyOf(copied), List.copyOf(skipped)));
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
repoRootCalls.add(new RepoRootCall(cwd));
|
||||
return repoRoot;
|
||||
}
|
||||
|
||||
public List<AddCall> addCalls() {
|
||||
return List.copyOf(addCalls);
|
||||
}
|
||||
|
||||
public List<RemoveCall> removeCalls() {
|
||||
return List.copyOf(removeCalls);
|
||||
}
|
||||
|
||||
public List<OverlayCall> overlayCalls() {
|
||||
return List.copyOf(overlayCalls);
|
||||
}
|
||||
|
||||
public List<RepoRootCall> repoRootCalls() {
|
||||
return List.copyOf(repoRootCalls);
|
||||
}
|
||||
|
||||
public AddCall lastAdd() {
|
||||
return addCalls.isEmpty() ? null : addCalls.getLast();
|
||||
}
|
||||
|
||||
public RemoveCall lastRemove() {
|
||||
return removeCalls.isEmpty() ? null : removeCalls.getLast();
|
||||
}
|
||||
|
||||
public OverlayCall lastOverlay() {
|
||||
return overlayCalls.isEmpty() ? null : overlayCalls.getLast();
|
||||
}
|
||||
}
|
||||
@@ -1,208 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-525 acceptance test for tool-surface isolation. This is one of the few tests that drives real
|
||||
* {@code git} — the behaviour under test is precisely what {@link GitWorktrees} does to a checkout,
|
||||
* so a fake would assert nothing. Everything happens inside a {@link TempDir} throwaway repo.
|
||||
*/
|
||||
class GitWorktreesTest {
|
||||
|
||||
/** A project MCP config with servers in it — what this repo actually commits. */
|
||||
private static final String WITH_SERVERS = """
|
||||
{
|
||||
"mcpServers": {
|
||||
"jetbrains": { "type": "sse", "url": "http://localhost:64342/sse" }
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** An opencode config carrying a {@code {file:.secrets/...}} reference — CB-543's crash repro. */
|
||||
private static final String OPENCODE_WITH_FILE_REF = """
|
||||
{
|
||||
"env": {
|
||||
"CONTEXT7_TOKEN": "{file:.secrets/context7-token}"
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** A non-empty autoenv file — the form that would prompt for authorization in a worktree. */
|
||||
private static final String AUTOENV_WITH_DIRECTIVE = "export HELLO=world\n";
|
||||
|
||||
private static Path initRepo(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
git(dir, "init", "-q", "-b", "main");
|
||||
git(dir, "config", "user.email", "test@example.invalid");
|
||||
git(dir, "config", "user.name", "Test");
|
||||
Files.writeString(dir.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||
git(dir, "add", ".mcp.json", "README.md");
|
||||
git(dir, "commit", "-q", "-m", "seed");
|
||||
return dir;
|
||||
}
|
||||
|
||||
private static void git(Path cwd, String... args) throws Exception {
|
||||
List<String> cmd = new java.util.ArrayList<>(List.of("git"));
|
||||
cmd.addAll(List.of(args));
|
||||
Process p = new ProcessBuilder(cmd).directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git timed out: " + String.join(" ", cmd));
|
||||
assertEquals(0, p.exitValue(), "git " + String.join(" ", args) + " failed:\n" + out);
|
||||
}
|
||||
|
||||
/** Pending changes to {@code file} in {@code cwd}, empty when git considers it unmodified. */
|
||||
private static String status(Path cwd, String file) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "status", "--porcelain", "--", file)
|
||||
.directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git status timed out");
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* The heart of CB-525: a provisioned worktree must not inherit the primary's MCP servers. Without
|
||||
* the isolation step the checked-out {@code .mcp.json} carries them in, and a worker navigating
|
||||
* through the primary's IDE servers edits the primary's tree while building its own.
|
||||
*/
|
||||
@Test
|
||||
void aProvisionedWorktreeInheritsNoMcpServers(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-a", "HEAD");
|
||||
|
||||
Path mcp = Path.of(wt).resolve(".mcp.json");
|
||||
assertTrue(Files.exists(mcp), ".mcp.json must still exist — present and explicitly empty");
|
||||
String body = Files.readString(mcp);
|
||||
assertFalse(body.contains("jetbrains"), "worktree inherited the primary's MCP servers:\n" + body);
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
"expected an explicitly empty server map, got:\n" + body);
|
||||
}
|
||||
|
||||
/** Neutralizing must not look like work in progress, or a worker would commit it into its PR. */
|
||||
@Test
|
||||
void theNeutralizedConfigIsNotAPendingLocalModification(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-b", "HEAD");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"),
|
||||
"the neutralized .mcp.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** Isolation is the worktree's business only; the primary's own checkout must be untouched. */
|
||||
@Test
|
||||
void thePrimaryCheckoutIsLeftAlone(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-525-c", "HEAD");
|
||||
|
||||
assertEquals(WITH_SERVERS, Files.readString(repo.resolve(".mcp.json")),
|
||||
"the primary's .mcp.json was rewritten — isolation reached out of the worktree");
|
||||
}
|
||||
|
||||
/** A repo that commits no {@code .mcp.json} still gets one, so nothing can be inherited later. */
|
||||
@Test
|
||||
void aRepoWithoutAnMcpConfigStillGetsANeutralOne(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-d", "HEAD");
|
||||
|
||||
// Untracked is the normal case here, so the --skip-worktree branch must be skipped rather
|
||||
// than run and fail: `update-index --skip-worktree` on an unknown path exits non-zero.
|
||||
String body = Files.readString(Path.of(wt).resolve(".mcp.json"));
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"), body);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-543's repro: a tracked {@code opencode.json} carries a {@code {file:.secrets/...}} reference
|
||||
* to a gitignored secret that never reaches a worktree, and opencode refuses to start on it. The
|
||||
* worktree's copy must be neutralized and hidden like {@code .mcp.json}.
|
||||
*/
|
||||
@Test
|
||||
void aTrackedOpencodeConfigIsNeutralizedAndHidden(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-a", "HEAD");
|
||||
|
||||
String body = Files.readString(Path.of(wt).resolve("opencode.json"));
|
||||
assertFalse(body.contains(".secrets"),
|
||||
"worktree kept a dangling {file:...} secret reference:\n" + body);
|
||||
assertEquals("{}", body.replaceAll("\\s+", ""),
|
||||
"expected an empty JSON object stub, got:\n" + body);
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"),
|
||||
"the neutralized opencode.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** A config the repo does not carry must be skipped — no stub invented, provisioning still succeeds. */
|
||||
@Test
|
||||
void anAbsentConfigIsSkippedWithoutError(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo")); // only .mcp.json + README are committed
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-b", "HEAD");
|
||||
|
||||
assertFalse(Files.exists(Path.of(wt).resolve("opencode.json")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
assertFalse(Files.exists(Path.of(wt).resolve(".autoenv")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
// .mcp.json's long-standing create-always behaviour must be unchanged.
|
||||
assertTrue(Files.exists(Path.of(wt).resolve(".mcp.json")), ".mcp.json stub was dropped");
|
||||
}
|
||||
|
||||
/** All three protected configs are covered: each one present in a worktree is neutralized and hidden. */
|
||||
@Test
|
||||
void allThreeConfigsAreNeutralizedWhenPresent(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve(".autoenv"), AUTOENV_WITH_DIRECTIVE);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", ".autoenv", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-c", "HEAD");
|
||||
|
||||
assertTrue(Files.readString(Path.of(wt).resolve(".mcp.json"))
|
||||
.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
".mcp.json was not neutralized");
|
||||
assertEquals("{}", Files.readString(Path.of(wt).resolve("opencode.json")).replaceAll("\\s+", ""),
|
||||
"opencode.json was not neutralized");
|
||||
assertEquals("", Files.readString(Path.of(wt).resolve(".autoenv")),
|
||||
".autoenv was not neutralized");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"), ".mcp.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"), "opencode.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), ".autoenv"), ".autoenv still shows as modified");
|
||||
}
|
||||
}
|
||||
@@ -1,496 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-301 / CB-303 acceptance tests for the authoritative session registry, one-shot lifecycle FSM,
|
||||
* and configurable lifecycle limits (idle TTL, context cap, drain).
|
||||
* No live herdr — everything runs against the same {@link FakeHerdr} the rest of the project uses.
|
||||
*/
|
||||
class SessionManagerTest {
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock) {
|
||||
return sessionManager(herdr, clock, 0);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap) {
|
||||
return sessionManager(herdr, clock, contextCap, false);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap,
|
||||
boolean clearAfterTurn) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers, new GitWorktrees(), clock, contextCap, clearAfterTurn);
|
||||
}
|
||||
|
||||
@Test
|
||||
void primaryContactWithNoTerminalIsNotAReadinessSignal() {
|
||||
// The MCP context extractor calls presence.markPresent(p.terminal()) on EVERY request,
|
||||
// and the primary's terminal is null — the presence bridge must treat that as a no-op,
|
||||
// not feed it into the READY transition (which NPEd on the first real primary contact).
|
||||
SessionManager sessions = sessionManager(new FakeHerdr());
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null));
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireRegistersSpawningSessionWithDistinctPaneId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession a = sessions.acquire("ltms-local", "/work/a", "/caller/a", "term_primary");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/work/b", "/caller/b", "term_primary");
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, a.state(), "fresh session starts spawning");
|
||||
assertEquals("ltms-local", a.profile());
|
||||
assertEquals("/work/a", a.cwd(), "explicit requested cwd is recorded");
|
||||
assertEquals("term_primary", a.ownerTerminal());
|
||||
assertTrue(a.spawnedAtNanos() > 0);
|
||||
assertNotNull(a.paneId());
|
||||
assertNotNull(a.terminalId());
|
||||
|
||||
assertNotEquals(a.paneId(), b.paneId(), "no pane reuse");
|
||||
assertNotEquals(a.terminalId(), b.terminalId(), "no terminal reuse");
|
||||
assertEquals(2, sessions.roster().size(), "both sessions are registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullTerminalFromThePrimaryIsANoOpEvenWithSessionsRegistered() {
|
||||
// The primary resolves to a Principal with no terminal, and BridgeMcp's context extractor
|
||||
// forwards that null into markPresent on EVERY MCP call. It only reached the registry scan
|
||||
// once a session existed, so this NPE'd the primary's second spawn while the first passed.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null),
|
||||
"the primary's null terminal must not blow up an unrelated tool call");
|
||||
assertDoesNotThrow(() -> sessions.onDelivered(null));
|
||||
assertDoesNotThrow(() -> sessions.onTurnComplete(null));
|
||||
assertDoesNotThrow(() -> sessions.onTurnFailed(null));
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"and must not transition any registered session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void presenceMovesSpawningToReadyAndDeliveredTurnMovesToDone() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
assertEquals(MemberSession.State.READY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"MCP presence moves SPAWNING → READY");
|
||||
assertTrue(sessions.asPresence().isPresent(terminal), "presence is also recorded");
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
assertEquals(MemberSession.State.BUSY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"delivery moves READY → BUSY");
|
||||
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"turn completion moves BUSY → DONE");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseTearsDownWorkerAndRemovesFromRosterAndIsIdempotent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
String paneId = session.paneId();
|
||||
|
||||
sessions.release(paneId);
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release tears the worker pane down");
|
||||
assertTrue(sessions.get(paneId).isEmpty(), "released session is no longer retrievable");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session is no longer in the roster");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.release(paneId), "a second release is harmless");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedMovesSessionToFailed() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
|
||||
sessions.onTurnFailed(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.FAILED, updated.state(), "turn failure moves to FAILED");
|
||||
assertTrue(sessions.roster().contains(updated), "FAILED is still in acquired-minus-released roster");
|
||||
}
|
||||
|
||||
@Test
|
||||
void recycleProducesNewPaneIdAndOldOneIsGone() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession oldSession = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String oldPane = oldSession.paneId();
|
||||
String oldTerminal = oldSession.terminalId();
|
||||
|
||||
MemberSession fresh = sessions.recycle(oldPane);
|
||||
|
||||
assertNotEquals(oldPane, fresh.paneId(), "recycle yields a new pane id");
|
||||
assertNotEquals(oldTerminal, fresh.terminalId(), "recycle yields a new terminal id");
|
||||
assertEquals(oldSession.profile(), fresh.profile(), "profile is preserved");
|
||||
assertEquals(oldSession.cwd(), fresh.cwd(), "cwd is preserved");
|
||||
assertEquals(oldSession.ownerTerminal(), fresh.ownerTerminal(), "owner is preserved");
|
||||
|
||||
assertTrue(sessions.get(oldPane).isEmpty(), "old pane is deregistered");
|
||||
assertEquals(1, sessions.roster().size(), "only the fresh session remains");
|
||||
assertEquals(fresh.paneId(), sessions.roster().getFirst().paneId());
|
||||
|
||||
// The old session was the first spawn → pane w9:pRoot_1 (CB-519: the registry key is the
|
||||
// uuid id, so teardown is asserted on the real pane coordinate).
|
||||
long paneCloseCount = herdr.calls.stream()
|
||||
.filter(c -> "pane.close".equals(c.method()))
|
||||
.filter(c -> "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
assertEquals(1, paneCloseCount, "the old worker was torn down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void rosterReflectsAcquiredMinusReleased() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession a = sessions.acquire("ltms-local", "/a", "/caller", "ownerA");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/b", "/caller", "ownerB");
|
||||
|
||||
assertEquals(2, sessions.roster().size());
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(a.paneId())));
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(b.paneId())));
|
||||
|
||||
sessions.release(a.paneId());
|
||||
|
||||
assertEquals(1, sessions.roster().size());
|
||||
assertEquals(b.paneId(), sessions.roster().getFirst().paneId());
|
||||
}
|
||||
|
||||
// --- CB-303 lifecycle limits ----------------------------------------------------
|
||||
|
||||
@Test
|
||||
void reapIdleDoesNothingWhenNoSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L);
|
||||
|
||||
assertEquals(0, sessions.reapIdle(10));
|
||||
assertTrue(sessions.roster().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 11;
|
||||
assertEquals(1, sessions.reapIdle(10), "READY session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "reaped session is removed from registry");
|
||||
assertTrue(herdr.called("pane.close"), "reaped session tears the pane down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionWithinIdleTtlSurvives() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 5;
|
||||
assertEquals(0, sessions.reapIdle(10), "READY session within TTL is not reaped");
|
||||
assertEquals(MemberSession.State.READY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"READY session survives");
|
||||
}
|
||||
|
||||
@Test
|
||||
void busySessionPastIdleTtlIsNotReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
|
||||
clock[0] = 100;
|
||||
assertEquals(0, sessions.reapIdle(10), "BUSY session past TTL is never reaped");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"BUSY session remains");
|
||||
}
|
||||
|
||||
@Test
|
||||
void doneSessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
clock[0] = 21;
|
||||
assertEquals(1, sessions.reapIdle(20), "DONE session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "DONE session is removed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIdleReturnsCorrectCountAndSkipsBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "owner1");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "owner2");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId());
|
||||
|
||||
clock[0] = 50;
|
||||
assertEquals(1, sessions.reapIdle(30), "only READY past TTL is reaped");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "READY session is gone");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(busy.paneId()).orElseThrow().state(),
|
||||
"BUSY session is still registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapDisabledSessionSurvivesMultipleTurns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.DONE, updated.state(), "session finishes second turn");
|
||||
assertEquals(2, updated.turnCount(), "turn count tracks both deliveries");
|
||||
long releaseCloseCount = paneCloseCallsFor(herdr, "w9:pRoot_1"); // the real pane coordinate
|
||||
assertEquals(0, releaseCloseCount, "cap disabled — no forced release of the worker pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapTwoReleasesAfterSecondComplete() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 2);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"first turn completes without release");
|
||||
|
||||
sessions.onDelivered(terminal);
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "session released after cap reached");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session leaves roster");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"forced release tears the worker pane down exactly once");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnResetsContextWithoutDoubleCountingTheTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
|
||||
sessions.onDelivered(session.terminalId());
|
||||
assertTrue(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(1, updated.turnCount(), "the reset is housekeeping, not a second delegation");
|
||||
assertEquals(List.of("/clear"), promptTexts(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapReleaseWinsOverClearAfterTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 1, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId());
|
||||
|
||||
assertFalse(sessions.hasPostTurnAction(session.terminalId()),
|
||||
"a session at its cap will be released, not reset for reuse");
|
||||
assertFalse(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty());
|
||||
assertTrue(promptTexts(herdr).isEmpty(), "never send /clear into a worker being torn down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnFalsePreservesCompletionWithoutAControlPrompt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, false);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId());
|
||||
|
||||
sessions.onTurnComplete(session.terminalId());
|
||||
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state());
|
||||
assertTrue(promptTexts(herdr).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllReleasesBusyAndReadySessionsAndWaitsForBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "ownerR");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "ownerB");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId());
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(sessions.roster().isEmpty(), "drain clears the roster");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "ready session is released");
|
||||
assertTrue(sessions.get(busy.paneId()).isEmpty(), "busy session is released after timeout");
|
||||
// ready is the first spawn → pane w9:pRoot_1, busy the second → w9:pRoot_2 (FakeHerdr order).
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"ready worker pane is torn down");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_2"),
|
||||
"busy worker pane is torn down");
|
||||
}
|
||||
|
||||
private static long paneCloseCallsFor(FakeHerdr herdr, String paneId) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "pane.close".equals(c.method()))
|
||||
.filter(c -> paneId.equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
}
|
||||
|
||||
private static List<String> promptTexts(FakeHerdr herdr) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "agent.prompt".equals(c.method()))
|
||||
.map(c -> String.valueOf(((Map<?, ?>) c.params()).get("text")))
|
||||
.toList();
|
||||
}
|
||||
|
||||
// --- CB-306 spawn-readiness gate: no half-registered session on timeout ----------------
|
||||
|
||||
@Test
|
||||
void acquireThrowsPeerUnreachableWhenGateTimesOutAndRegistersNoSession() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never becomes injectable
|
||||
long[] clock = {0};
|
||||
|
||||
// Gate-enabled launcher (1 ms timeout + no-op sleeper that advances clock past deadline)
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
1, () -> clock[0], () -> clock[0] += 10);
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(), () -> 0L, 0);
|
||||
|
||||
assertThrows(PeerUnreachableException.class,
|
||||
() -> sessions.acquire("ltms-local", null, "/caller", "term_primary"),
|
||||
"acquire must throw PeerUnreachableException when spawn times out");
|
||||
|
||||
// No half-registered session — the error happened inside spawn, before
|
||||
// SessionManager could put() anything into the registry.
|
||||
assertTrue(sessions.roster().isEmpty(),
|
||||
"no session is registered when spawn times out (roster empty)");
|
||||
}
|
||||
|
||||
// --- CB-516: release must notify, so a blocked send can be failed --------------------------
|
||||
|
||||
@Test
|
||||
void releaseNotifiesTheListenerWithTheReleasedTerminal() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(released::add);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(java.util.List.of(s.terminalId()), released,
|
||||
"every teardown path funnels through release, so one hook must see the terminal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releasingAnUnknownPaneNotifiesNobody() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(released::add);
|
||||
|
||||
sessions.release("w9:p404"); // idempotent teardown of something already gone
|
||||
|
||||
assertTrue(released.isEmpty(), "no session removed ⇒ no send was waiting on it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aThrowingReleaseListenerDoesNotBlockTheTeardown() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
sessions.onRelease(_ -> {
|
||||
throw new IllegalStateException("listener blew up");
|
||||
});
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||
"a listener failure must never prevent the teardown it is reacting to");
|
||||
assertTrue(sessions.get(s.paneId()).isEmpty(), "and the session is still deregistered");
|
||||
}
|
||||
}
|
||||
@@ -1,264 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-301-ext acceptance tests for worktree provisioning and config-parity overlay.
|
||||
* No live git — every Worktrees call is handled by {@link FakeWorktrees} and every herdr
|
||||
* call by {@link FakeHerdr}, matching the project's fake-based test style.
|
||||
*/
|
||||
class WorktreeSessionManagerTest {
|
||||
|
||||
private static ClaudeCodeLauncher workerService(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
}
|
||||
|
||||
private static String startCwd(FakeHerdr herdr) {
|
||||
@SuppressWarnings("unchecked")
|
||||
// Protocol 19: the worker's cwd rides on pane creation (tab.create), not agent.start.
|
||||
Map<String, Object> start = (Map<String, Object>) herdr.lastCall("tab.create").params();
|
||||
Object cwd = start.get("cwd");
|
||||
return cwd == null ? null : cwd.toString();
|
||||
}
|
||||
|
||||
@Test
|
||||
void sharedTreeAcquireMakesNoWorktreesCallsAndRecordsNullWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary");
|
||||
|
||||
assertTrue(worktrees.addCalls().isEmpty(), "shared-tree acquire never adds a worktree");
|
||||
assertTrue(worktrees.repoRootCalls().isEmpty(), "shared-tree acquire never resolves a repo root");
|
||||
assertTrue(worktrees.overlayCalls().isEmpty(), "shared-tree acquire never overlays parity");
|
||||
assertNull(s.worktree(), "shared-tree session has no worktree");
|
||||
assertNull(s.branch(), "shared-tree session has no branch");
|
||||
assertEquals("/caller/proj", s.cwd(), "shared-tree cwd is the caller's cwd");
|
||||
assertEquals("/caller/proj", startCwd(herdr), "spawn receives the caller's cwd");
|
||||
}
|
||||
|
||||
@Test
|
||||
void worktreeAcquireProvisionsAndRecordsPathAndBranch() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", "term_primary",
|
||||
new WorktreeRequest("cb-999", null));
|
||||
|
||||
assertEquals(1, worktrees.addCalls().size(), "one worktree was added");
|
||||
FakeWorktrees.AddCall add = worktrees.lastAdd();
|
||||
assertNotNull(add);
|
||||
assertEquals("/repo", add.repoRoot());
|
||||
assertTrue(add.branch().startsWith("worker/cb-999-"), "branch is worker/<slug>-<nonce>: " + add.branch());
|
||||
assertNull(add.baseRef(), "null baseRef is passed through (HEAD default)");
|
||||
|
||||
String expectedPath = "/wt/" + add.branch().replace('/', '_');
|
||||
assertEquals(expectedPath, s.worktree(), "session records the returned worktree path");
|
||||
assertEquals(add.branch(), s.branch(), "session records the branch");
|
||||
assertEquals(expectedPath, startCwd(herdr), "spawn receives the worktree path as cwd");
|
||||
assertEquals(expectedPath, s.cwd(), "session cwd is the worktree path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void worktreeAcquireRunsParityOverlayWithProfileDefaults() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt")
|
||||
.track(".envrc")
|
||||
.exists(".claude/settings.local.json");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-888", null));
|
||||
|
||||
assertEquals(1, worktrees.overlayCalls().size());
|
||||
FakeWorktrees.OverlayCall overlay = worktrees.lastOverlay();
|
||||
assertNotNull(overlay);
|
||||
assertEquals("/repo", overlay.repoRoot());
|
||||
assertEquals(List.of(".claude/settings.local.json", ".env", ".envrc"),
|
||||
overlay.requested(), "default parity overlay is used when unset");
|
||||
assertFalse(overlay.requested().contains(".mcp.json"),
|
||||
"CB-525: replicating the primary's MCP config gives a worker the primary's IDE "
|
||||
+ "servers, which navigate its edits out of its own worktree");
|
||||
assertEquals(List.of(".claude/settings.local.json", ".envrc"), overlay.copied(),
|
||||
"existing paths are copied; missing paths are skipped");
|
||||
assertEquals(List.of(".envrc"), overlay.skipWorktree(),
|
||||
"tracked copied paths are --skip-worktree'd");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseRemovesWorktreeButDoesNotDeleteBranch() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-666", null));
|
||||
String paneId = s.paneId();
|
||||
|
||||
sessions.release(paneId);
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release still tears the worker pane down");
|
||||
assertEquals(1, worktrees.removeCalls().size(), "worktree session triggers one remove");
|
||||
FakeWorktrees.RemoveCall remove = worktrees.lastRemove();
|
||||
assertNotNull(remove);
|
||||
assertEquals("/repo", remove.repoRoot());
|
||||
assertEquals(s.worktree(), remove.worktreePath());
|
||||
// The fake records no branch-delete calls because Worktrees.remove only removes the checkout.
|
||||
assertTrue(sessions.get(paneId).isEmpty(), "released session is no longer retrievable");
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllPreservesWorktreeOfIdleSession() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-544", null));
|
||||
sessions.asPresence().markPresent(s.terminalId()); // READY (idle)
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "shutdown drain still stops the worker pane");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"shutdown drain must NOT remove the worktree — it is the only copy of the work");
|
||||
assertTrue(sessions.roster().isEmpty(), "shutdown drain still deregisters the session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllPreservesWorktreeOfSessionStillBusyAtTimeout() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-544", null));
|
||||
String terminal = s.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal); // BUSY, never completes → still BUSY when the timeout hits
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "shutdown drain stops the worker pane");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"a session still BUSY at timeout must preserve its worktree unconditionally");
|
||||
assertTrue(sessions.roster().isEmpty(), "shutdown drain still deregisters the session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sharedTreeReleaseMakesNoWorktreesCalls() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null);
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release tears the worker pane down");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(), "shared-tree release never removes a worktree");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failedWorktreeAddUnwindsWithoutRegisteringSessionOrSpawning() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().failAdd("worktree add failed");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
assertThrows(WorktreeException.class, () ->
|
||||
sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-555", null)));
|
||||
|
||||
assertEquals(0, sessions.size(), "failed acquire leaves no registry entry");
|
||||
assertFalse(herdr.called("agent.start"), "spawn is never reached when add fails");
|
||||
assertTrue(worktrees.removeCalls().isEmpty(), "no worktree was added, so none is removed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoWorktreeAcquiresYieldDistinctBranchesAndPaths() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
MemberSession a = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-444", null));
|
||||
MemberSession b = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-444", null));
|
||||
|
||||
assertNotEquals(a.branch(), b.branch(), "branches are distinct");
|
||||
assertNotEquals(a.worktree(), b.worktree(), "paths are distinct");
|
||||
assertEquals(2, worktrees.addCalls().size());
|
||||
assertEquals(2, sessions.roster().size());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-507 regression. A plain REST spawn supplies neither a requested nor a caller cwd
|
||||
* ({@code BridgedApp} hardcodes {@code callerCwd = null}), and the worktree branch used to
|
||||
* resolve the repo root from just those two — yielding {@code null}, which the real
|
||||
* {@code GitWorktrees} turns into {@code git -C null} and an NPE out of {@code ProcessBuilder}
|
||||
* (HTTP 500).
|
||||
*
|
||||
* <p>Note this asserts on the <em>recorded</em> cwd rather than expecting a throw:
|
||||
* {@link FakeWorktrees#repoRoot} only records its argument and returns a canned root, so a
|
||||
* null flows through the fake harmlessly. That permissiveness is precisely why the whole
|
||||
* suite stayed green while the feature was broken in production — so the assertion has to be
|
||||
* "a usable cwd was passed down", not "an exception was raised".
|
||||
*/
|
||||
@Test
|
||||
void worktreeAcquireWithNoRequestedOrCallerCwdStillResolvesANonNullRepoRoot() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees();
|
||||
SessionManager sessions = new SessionManager(workerService(herdr), worktrees);
|
||||
|
||||
sessions.acquire("ltms-local", null, null, null, new WorktreeRequest("cb-507", null));
|
||||
|
||||
assertFalse(worktrees.repoRootCalls().isEmpty(),
|
||||
"repoRoot should have been called to resolve the repo root");
|
||||
String cwd = worktrees.repoRootCalls().getFirst().cwd();
|
||||
assertNotNull(cwd, "a null cwd here becomes `git -C null` and NPEs in the real GitWorktrees");
|
||||
assertFalse(cwd.isBlank(), "a blank cwd is as unusable as a null one");
|
||||
}
|
||||
|
||||
/**
|
||||
* The same line carried a second, quieter bug: it never consulted the profile's configured
|
||||
* {@code cwd:}, so a worktree spawn silently ignored a pinned per-profile working directory.
|
||||
* Routing through {@code effectiveCwd} honours it.
|
||||
*/
|
||||
@Test
|
||||
void worktreeAcquireHonoursTheProfileConfiguredCwd() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FakeWorktrees worktrees = new FakeWorktrees().withRepoRoot("/repo").withPrefix("/wt");
|
||||
// Argument order matters: configDir is the 4th parameter, cwd the 11th (after mcpUrl).
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, "/pinned/dir", null);
|
||||
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
SessionManager sessions = new SessionManager(launcher, worktrees);
|
||||
|
||||
sessions.acquire("ltms-local", null, null, null, new WorktreeRequest("cb-507b", null));
|
||||
|
||||
assertEquals(1, worktrees.repoRootCalls().size());
|
||||
assertEquals("/pinned/dir", worktrees.repoRootCalls().getFirst().cwd(),
|
||||
"the profile's configured cwd must reach repoRoot, not be ignored");
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,61 +0,0 @@
|
||||
# CB-504 — systemd unit for bridged (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.bridged.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
#
|
||||
# Install (user service — bridged drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/bridged.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now bridged
|
||||
# journalctl --user -u bridged -f
|
||||
|
||||
[Unit]
|
||||
Description=bridged — claude-bridge message server
|
||||
Documentation=https://git.ltms.dev/lms/claude-bridge/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# bridged retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take bridged down with it.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/bridged
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/bridged.jar bridged.yaml
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit bridged → [Service] / Environment=BRIDGED_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/bridged/env
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes bridged fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=bridged
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -1,80 +0,0 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<!--
|
||||
CB-504 — launchd agent for bridged (macOS).
|
||||
|
||||
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
|
||||
no systemd. A systemd unit ships alongside (deploy/bridged.service) for the Linux gateways
|
||||
CB-308 introduces.
|
||||
|
||||
Install:
|
||||
cp deploy/dev.ltms.bridged.plist ~/Library/LaunchAgents/
|
||||
# edit the paths + JAVA_HOME below to match this host, then:
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.bridged.plist
|
||||
launchctl list | grep bridged
|
||||
|
||||
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
|
||||
does systemd in a way that survives a socket appearing late. bridged retries the herdr socket
|
||||
on startup instead, so an agent that comes up before herdr converges rather than dying — that
|
||||
retry is the actual fix; KeepAlive below is the backstop.
|
||||
-->
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>dev.ltms.bridged</string>
|
||||
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged/target/bridged.jar</string>
|
||||
<string>bridged.yaml</string>
|
||||
</array>
|
||||
|
||||
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
|
||||
<key>WorkingDirectory</key>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged</string>
|
||||
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>JAVA_HOME</key>
|
||||
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home</string>
|
||||
<key>HERDR_SOCKET_PATH</key>
|
||||
<string>/Users/CHANGEME/.config/herdr/herdr.sock</string>
|
||||
<!--
|
||||
PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker
|
||||
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
|
||||
source .zprofile/.zshrc, so without this the daemon (and therefore every worker) gets a
|
||||
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
|
||||
-->
|
||||
<key>PATH</key>
|
||||
<string>/Users/CHANGEME/Tool/jdk-25.0.2.jdk/Contents/Home/bin:/Users/CHANGEME/Tool/apache-maven-3.9.16/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
|
||||
<!--
|
||||
Worker/API tokens are NOT set here: this file is committed. Export them from a private
|
||||
launchd override or a wrapper script. bridged reads the API token from the env var named
|
||||
by auth.tokenEnv (default BRIDGED_API_TOKEN) and only in auth.mode: token.
|
||||
-->
|
||||
</dict>
|
||||
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
|
||||
<!-- Restart on crash, but not in a tight loop if the config is bad (bridged fails fast on a
|
||||
non-loopback bind without token auth — that is a config error, not a transient one). -->
|
||||
<key>KeepAlive</key>
|
||||
<dict>
|
||||
<key>SuccessfulExit</key>
|
||||
<false/>
|
||||
</dict>
|
||||
<key>ThrottleInterval</key>
|
||||
<integer>10</integer>
|
||||
|
||||
<key>StandardOutPath</key>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged/logs/bridged.out.log</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/Users/CHANGEME/src/claude-bridge/bridged/logs/bridged.err.log</string>
|
||||
|
||||
<key>ProcessType</key>
|
||||
<string>Background</string>
|
||||
</dict>
|
||||
</plist>
|
||||
@@ -0,0 +1,127 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<!--
|
||||
CB-504 / CB-594 — launchd agent for fleetd (macOS).
|
||||
|
||||
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
|
||||
no systemd. A systemd unit ships alongside (deploy/fleetd.service) for the Linux gateways
|
||||
CB-308 introduces.
|
||||
|
||||
Install:
|
||||
cp deploy/dev.ltms.fleetd.plist ~/Library/LaunchAgents/
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist
|
||||
launchctl list | grep fleetd
|
||||
|
||||
The paths below are already filled in for this host (resolved 2026-08-16 from
|
||||
`/usr/libexec/java_home`... except that reported the system Applet-plugin JVM, not the jenv-
|
||||
managed JDK 25 actually used to build/run fleetd, so JAVA_HOME here is the real one:
|
||||
`JENV_VERSION=25.0.3 java -XshowSettings:properties -version 2>&1 | grep java.home`; `which mvn`;
|
||||
`echo $HOME`). If this file is copied to a different host, re-resolve all three paths and check
|
||||
no placeholder path is left behind; scripts/redeploy-fleetd.sh's check mode does not (and
|
||||
cannot) check this file for you.
|
||||
|
||||
CB-594 — launchd cannot run a login shell (see the PATH comment on EnvironmentVariables below,
|
||||
and scripts/fleetd-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
|
||||
not java directly, so WORKER_GITEA_TOKEN and AI_GATEWAY_TOKEN still get sourced from
|
||||
${SHARED_ENV}/tools/secrets.sh even though launchd itself never sources anything.
|
||||
|
||||
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
|
||||
does systemd in a way that survives a socket appearing late. fleetd retries the herdr socket
|
||||
on startup instead, so an agent that comes up before herdr converges rather than dying — that
|
||||
retry is the actual fix; KeepAlive below is the backstop.
|
||||
|
||||
CB-594 — KeepAlive vs. scripts/redeploy-fleetd.sh: a bare SIGTERM makes this JVM exit 143 even
|
||||
with its shutdown hook running to completion (measured, see the CB-594 report), which
|
||||
SuccessfulExit:false below reads as a crash and races to restart the OLD jar. The redeploy
|
||||
script now detects a loaded agent and uses `launchctl unload`/`load` instead of a raw kill, so
|
||||
only one supervisor ever touches the process at a time — read that script's own output on a
|
||||
redeploy for the confirmation.
|
||||
-->
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>dev.ltms.fleetd</string>
|
||||
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/target/fleetd.jar</string>
|
||||
<string>fleetd.yaml</string>
|
||||
</array>
|
||||
|
||||
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
|
||||
<key>WorkingDirectory</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd</string>
|
||||
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>JAVA_HOME</key>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home</string>
|
||||
<key>HERDR_SOCKET_PATH</key>
|
||||
<string>/Users/dai.ha/.config/herdr/herdr.sock</string>
|
||||
<!--
|
||||
PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker
|
||||
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
|
||||
source .zprofile/.zshrc, so without this the daemon (and therefore every worker) gets a
|
||||
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
|
||||
-->
|
||||
<key>PATH</key>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin:/Users/dai.ha/Softwares/apache-maven/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
|
||||
<!--
|
||||
Worker/API tokens are NOT set here: this file is committed. CB-594 —
|
||||
scripts/fleetd-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
|
||||
them, by execing a login shell that sources ${SHARED_ENV}/tools/secrets.sh before the
|
||||
daemon itself starts. fleetd also reads the API token from the env var named by
|
||||
auth.tokenEnv (default FLEETD_API_TOKEN) and only in auth.mode: token — the wrapper
|
||||
covers that one too, since it is the same login shell.
|
||||
-->
|
||||
</dict>
|
||||
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
|
||||
<!--
|
||||
CB-600 — read this before assuming ThrottleInterval bounds anything. It paces restarts to at
|
||||
most one per 10s; it does NOT cap how many times launchd retries. If fleetd fails fast on
|
||||
every start — a bad fleetd.yaml, for example auth.mode: token with the token env var unset,
|
||||
which throws in main() before the daemon ever binds a port — launchd restarts it forever,
|
||||
once every 10s, until a human intervenes. LaunchAgents have no "give up after N attempts"
|
||||
primitive, so this is not something a config change here can fix.
|
||||
|
||||
That loop stops only two ways: (1) `launchctl unload -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist`,
|
||||
or (2) the underlying cause gets fixed, so the process starts successfully and stays up (no
|
||||
more exits to restart). scripts/redeploy-fleetd.sh does not add a third way — it does not
|
||||
make fleetd self-disable on a config error, on purpose: a fail-fast exit path that
|
||||
sometimes decides "this is unrecoverable, stop trying" is one more thing that can misfire,
|
||||
and a wrongly self-disabled daemon needs the exact same manual `launchctl load -w` recovery
|
||||
this comment already names — so it buys nothing an operator watching for the crash loop
|
||||
doesn't already have, at the cost of a new way to be silently down. Watch for it with
|
||||
`launchctl list dev.ltms.fleetd` (a high restart count) or by tailing fleetd.out for the
|
||||
same startup error repeating every ~10s.
|
||||
-->
|
||||
<key>KeepAlive</key>
|
||||
<dict>
|
||||
<key>SuccessfulExit</key>
|
||||
<false/>
|
||||
</dict>
|
||||
<key>ThrottleInterval</key>
|
||||
<integer>10</integer>
|
||||
|
||||
<!--
|
||||
CB-594 — same file scripts/redeploy-fleetd.sh already tails ($BRIDGED/fleetd.out), and both
|
||||
streams point at it, not two separate log files: the script's fresh-line / ERROR-count checks
|
||||
after a restart read this one path regardless of whether launchd or the script started the
|
||||
process, and a stdout/stderr split would make half of what happens during a launchd-driven
|
||||
restart invisible to it.
|
||||
-->
|
||||
<key>StandardOutPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
|
||||
<key>ProcessType</key>
|
||||
<string>Background</string>
|
||||
</dict>
|
||||
</plist>
|
||||
@@ -0,0 +1,77 @@
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.fleetd.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
#
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/fleetd.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now fleetd
|
||||
# journalctl --user -u fleetd -f
|
||||
|
||||
[Unit]
|
||||
Description=fleetd — claude-bridge message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# fleetd retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down with it.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/fleetd
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/fleetd.jar fleetd.yaml
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put ALL three tokens in a private drop-in
|
||||
# that systemd reads with restrictive permissions. In `systemctl --user edit fleetd`, add:
|
||||
# [Service]
|
||||
# Environment=FLEETD_API_TOKEN=...
|
||||
# Environment=WORKER_GITEA_TOKEN=...
|
||||
# Environment=AI_GATEWAY_TOKEN=...
|
||||
# FLEETD_API_TOKEN protects fleetd's API. WORKER_GITEA_TOKEN lets members open pull requests; if
|
||||
# it is missing, fleetd still starts, but a member fails when it later tries to open a pull request.
|
||||
# AI_GATEWAY_TOKEN authenticates gateway profiles; if it is missing, fleetd still starts, but a
|
||||
# gateway profile later returns HTTP 401. Or, put the same three variables in a 0600 file and add:
|
||||
# EnvironmentFile=%h/.config/fleetd/env
|
||||
# After starting, check which of them actually resolved. The daemon reports every secret a
|
||||
# configured profile references, by name, never by value:
|
||||
# journalctl --user -u fleetd | grep 'startup secret'
|
||||
# A resolved one logs "startup secret NAME: set (profile 'x' tokenEnv)". A missing one logs
|
||||
# "startup secret NAME: MISSING" at WARN — and the daemon starts anyway, which is the whole
|
||||
# problem: without this grep the first sign is a member that cannot open a pull request, hours
|
||||
# later and in a different component.
|
||||
# Note what the report can and cannot tell you. It lists only names some profile actually
|
||||
# references (tokenEnv, gitTokenEnv, and the broker uriEnv). A secret nothing references is never
|
||||
# reported, because nothing needs it.
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes fleetd fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=fleetd
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -1,8 +1,8 @@
|
||||
# LavinMQ — the AMQP broker behind bridged's durable ReplyInbox (CB-307 Stage 2).
|
||||
# LavinMQ — the AMQP broker behind fleetd's durable ReplyInbox (CB-307 Stage 2).
|
||||
#
|
||||
# Why this file exists: the broker was previously run ad hoc and simply vanished from the host,
|
||||
# which takes bridged down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Bridged.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# which takes fleetd down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Fleetd.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# degraded mode. This pins the version, keeps the data, and brings itself back after a reboot.
|
||||
#
|
||||
# Usage:
|
||||
@@ -14,27 +14,27 @@
|
||||
#
|
||||
# Management UI: http://127.0.0.1:15672 (guest / guest)
|
||||
#
|
||||
# This is bridged's OWN broker. Do not point bridged at any other AMQP server on this host —
|
||||
# This is fleetd's OWN broker. Do not point fleetd at any other AMQP server on this host —
|
||||
# notably not the `local-rabbitmq` container, which belongs to a different project and would end
|
||||
# up carrying this project's queues.
|
||||
|
||||
name: bridged-broker
|
||||
name: fleetd-broker
|
||||
|
||||
services:
|
||||
lavinmq:
|
||||
# Pinned deliberately: :latest silently moves the broker under a running daemon.
|
||||
image: cloudamqp/lavinmq:2.9.1
|
||||
container_name: bridged-lavinmq
|
||||
container_name: fleetd-lavinmq
|
||||
|
||||
# The failure this deployment exists to prevent — survive reboots and Docker restarts, but
|
||||
# stay down if it was stopped on purpose.
|
||||
restart: unless-stopped
|
||||
|
||||
# Loopback-bound on purpose. LavinMQ ships a default guest/guest account, which is only
|
||||
# acceptable because nothing off-host can reach it. bridged connects over 127.0.0.1, and
|
||||
# acceptable because nothing off-host can reach it. fleetd connects over 127.0.0.1, and
|
||||
# binding 0.0.0.0 here would expose a broker with default credentials to the network.
|
||||
ports:
|
||||
- "127.0.0.1:5672:5672" # AMQP — bridged.yaml broker.uri points here
|
||||
- "127.0.0.1:5672:5672" # AMQP — fleetd.yaml broker.uri points here
|
||||
- "127.0.0.1:15672:15672" # HTTP management API + UI
|
||||
|
||||
# The whole point of Stage 2. Held-but-unacked replies live here; without a named volume a
|
||||
@@ -57,4 +57,4 @@ services:
|
||||
|
||||
volumes:
|
||||
lavinmq-data:
|
||||
name: bridged-lavinmq-data
|
||||
name: fleetd-lavinmq-data
|
||||
|
||||
@@ -0,0 +1,531 @@
|
||||
# CB-201 and CB-227 refinement
|
||||
|
||||
Date: 2026-09-03
|
||||
|
||||
## Decision
|
||||
|
||||
#201 and #227 are one delivery program, but they are not one implementation unit.
|
||||
|
||||
#201 has a real seam: `CompletionResolver` can publish a typed backend-error event only after its
|
||||
waiter resolution wins. #227 can consume that event without knowing any pane text. The classifier
|
||||
must land before the final #227 wiring. However, the policy engine, roster state, and lead nudge can
|
||||
be built in parallel with the classifier.
|
||||
|
||||
I propose five units. Units 1 to 4 own separate files and can run in parallel. Unit 5 owns all
|
||||
composition files and lands after them. It also depends on the #234 defect 2 fix named in the task.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
U1["Unit 1: typed backend-error classification"] --> U5["Unit 5: wire policy, spawn gate, and fleet views"]
|
||||
U2["Unit 2: credential outage policy"] --> U5
|
||||
U3["Unit 3: lead outage nudge"] --> U5
|
||||
U4["Unit 4: durable member outcome"] --> U5
|
||||
D234["#234 defect 2: fail-loud target resolution"] --> U5
|
||||
```
|
||||
|
||||
*Figure 1. Four file-disjoint foundations feed one composition unit.*
|
||||
|
||||
This split keeps `Fleetd.java` under one owner. It also keeps every other production file under one
|
||||
unit in this plan.
|
||||
|
||||
## Evidence checked in the current branch
|
||||
|
||||
I read both issue pages in full. Each page reports zero comments.
|
||||
|
||||
| Evidence | What the code says now |
|
||||
|---|---|
|
||||
| `inject/CompletionResolver.java:229-237` | A turn below two seconds fails before normal scrape classification. A matching fast backend error is therefore only a generic failure today. |
|
||||
| `inject/CompletionResolver.java:260-276` and `:332-367` | #211 already added raw-screen classification when `lastAssistantBlock` is empty. The “dead-code question” in #201 is stale on this branch. |
|
||||
| `inject/CompletionResolver.java:288-317` | Exhaustion wins before the hard-coded `API Error:` match. A backend error then goes through generic `fail(...)`. |
|
||||
| `inject/CompletionResolver.java:311-316` | The code admits that the pattern is a heuristic. A member report which quotes an API error may match it. |
|
||||
| `inject/CompletionResolver.java:449-467` | Startup coverage exists only for `exhaustedPattern`. |
|
||||
| `inject/ExhaustedPatternLookup.java:13-25` | The current lookup and explicit `none()` value are a good shape for the new classifier seam. |
|
||||
| `Fleetd.java:196-207` | One `BackendQuarantine` is shared by placement and the exhaustion sink. Its cooldown comes from `quarantineCooldownSeconds`. |
|
||||
| `Fleetd.java:322-363` | Pattern compilation, target-to-profile lookup, and the live `ExhaustionSink` are composed in `Fleetd.main`. The sink on this branch still ends in `.ifPresent(...)`. This plan assumes #234 replaces that silent path. |
|
||||
| `placement/BackendQuarantine.java:60-87` | A repeated exhaustion restarts one long quarantine. The store is credential-keyed and uses an injected monotonic clock. |
|
||||
| `member/CompositePeerLauncher.java:260-317` | Explicit and policy-selected spawns have separate gates. Both paths must learn about outage cool-off. |
|
||||
| `member/CompositePeerLauncher.java:347-379` | Exhaustion refusal already checks a credential for explicit spawns and filters policy candidates. Its error text says “exhausted”. |
|
||||
| `placement/PlacementContext.java:10-22` and `PlacementPolicyUtil.java:14-83` | Automatic placement has only one transient exclusion set named `quarantined`. Reusing it would make outage errors say “backend exhausted”. |
|
||||
| `mcp/FleetMcp.java:913-1025` | `fleet_list` sets `free: 0` and adds `credentialId` plus `quarantinedForSeconds` when quarantine is active. |
|
||||
| `session/MemberSession.java:51-59` | The roster has `DONE` and generic `FAILED`, but no backend-error state or stored reason. |
|
||||
| `session/SessionManager.java:695-773` | A normal boundary moves `BUSY` to `DONE`. A failure moves any non-released session to `FAILED`. The async completion resolver can race the `DONE` update. |
|
||||
| `session/SessionManager.java:648-687` | `rosterView` reports the session state, but it reports no terminal reason. |
|
||||
| `msg/MessageService.java:922-940` | CB-588 already nudges for every terminal async ticket, including failures. Current code would report failed tickets, but it would not report one correlated outage. |
|
||||
| `msg/ReplyPushLoop.java:20-48` | Replies, terminal tickets, and questions share one per-lead schedule. This prevents two push sources from injecting competing turns. |
|
||||
| `msg/ReplyPushLoop.java:305-395` | Each push entry point resolves the owning lead through `PrimaryRegistry`. Missing ownership is logged and the durable or pending item remains the backstop. |
|
||||
| `msg/ReplyPushLoop.java:496-547` | One tick builds one combined nudge. Pending items have separate reminder counts. |
|
||||
| `health/FleetHealthMonitor.java:91-143` | Health is a slow periodic observer of members and message-layer facts. It does not receive completion classifications. |
|
||||
| `health/FleetHealthMonitor.java:206-208` | `healthCoverage` means health enabled plus webhook configured. It does not describe lead-pane alerts. |
|
||||
| `Fleetd.java:465-486` | Health stays `detection-only` without the webhook notification setting. |
|
||||
|
||||
I also read the related unit tests for `CompletionResolver`, `ReplyPushLoop`, `BackendQuarantine`,
|
||||
`CompositePeerLauncher`, `PlacementPolicyUtil`, `SessionManager`, `MessageService`, and `FleetMcp`.
|
||||
|
||||
I did not inspect the in-progress #234 branch. I only used the two measured facts in the task. No
|
||||
peer architect was named, so I did not exchange a design with one.
|
||||
|
||||
## Required behaviour
|
||||
|
||||
The policy should use these first values:
|
||||
|
||||
- Threshold: **2** classified backend errors.
|
||||
- Window: **60 seconds**, measured from the first error to the second.
|
||||
- Cool-off: **60 seconds**, starting when the threshold is reached.
|
||||
- Correlation key: `credentialId`, never profile name and never error text.
|
||||
- Incident rule: one active incident per credential. Errors during its cool-off do not extend it and
|
||||
do not create more lead notices.
|
||||
- Rearm rule: after cool-off ends, two fresh errors are needed for another incident.
|
||||
|
||||
Two errors are the smallest threshold which protects the honest one-turn failure. A 60-second window
|
||||
fits the measured two-member outage. A 60-second cool-off blocks immediate repeat spawns without
|
||||
turning a short backend fault into the default 1,800-second exhaustion quarantine.
|
||||
|
||||
A single classified error still fails its send and marks its member `backend_error`. It does not
|
||||
cool a credential and does not send an outage notice. This is what “a single error changes nothing”
|
||||
must mean at the credential level. It cannot mean that the failed member still looks successful.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant R1 as Resolver for member A
|
||||
participant R2 as Resolver for member B
|
||||
participant P as Outage policy
|
||||
participant S as Spawn gate
|
||||
participant N as Lead push loop
|
||||
participant L as Lead pane
|
||||
|
||||
R1->>P: backend error for credential C
|
||||
Note over P: Count 1, no cool-off
|
||||
R2->>P: backend error for credential C within 60s
|
||||
P->>P: Start one 60s incident
|
||||
P->>S: Credential C is cooling off
|
||||
P->>N: Queue one incident notice
|
||||
N->>L: Inject when lead is idle, blocked, or done
|
||||
L->>S: Request another spawn on credential C
|
||||
S-->>L: Refuse and report remaining cool-off
|
||||
```
|
||||
|
||||
*Figure 2. The second independent classification creates the fleet-level event.*
|
||||
|
||||
Against the 2026-09-01 case, the second failed member would start cool-off. `fleet_list` would show
|
||||
zero free capacity and both members as `backend_error`. The push loop would inject one outage notice
|
||||
even if the lead had not polled either ticket yet. The design reports the outage. It does not recover
|
||||
uncommitted work from the members.
|
||||
|
||||
## Unit 1 — Typed backend-error classification
|
||||
|
||||
### Scope
|
||||
|
||||
Replace the direct hard-coded check inside `CompletionResolver` with a lookup and a sink. Keep the
|
||||
public send result as a failed send. The typed internal event is the seam #227 consumes.
|
||||
|
||||
The lookup returns the pattern for a target. The sink receives the target, matched line, and full
|
||||
failure reason. It fires only after `Rendezvous.resolveFailure(...)` wins for that exact captured
|
||||
waiter. This copies the race rule already used by `ExhaustionSink`.
|
||||
|
||||
The classifier must run in all three current paths:
|
||||
|
||||
1. a normal non-empty assistant block;
|
||||
2. the #211 raw scrape fallback;
|
||||
3. a turn inside `MIN_TURN_NANOS`, before it becomes a generic too-fast failure.
|
||||
|
||||
In every path, the order stays: stale-baseline guard, exhaustion, backend error, then generic
|
||||
failure or completion. A fast turn still fails when no configured pattern matches.
|
||||
|
||||
Keep `(?i)\bAPI Error\s*:` as a compatibility pattern for profiles without `errorPattern` until the
|
||||
operator config is updated. Do not call this full coverage. Startup reporting in Unit 5 must name
|
||||
profiles using this weaker legacy default.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorPatternLookup.java`.
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorSink.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/inject/CompletionResolverTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. A target-specific error pattern matches a normal assistant block and resolves the send as failed.
|
||||
2. The same match calls `BackendErrorSink` exactly once after the waiter resolution wins.
|
||||
3. A late classification which loses to `fleet_reply` does not call the sink.
|
||||
4. An exhausted line that also matches the generic error pattern stays `BACKEND_EXHAUSTED`. It calls
|
||||
only `ExhaustionSink`.
|
||||
5. A raw pane with leading Terminal User Interface (TUI) chrome and no assistant marker still uses
|
||||
the #211 fallback and calls the backend-error sink.
|
||||
6. A matching error inside the two-second floor is typed and sent to the sink. A non-matching fast
|
||||
turn stays a generic failure.
|
||||
7. An unchanged delivery baseline which contains old backend-error text is suppressed. It never
|
||||
increments outage evidence.
|
||||
8. A non-match keeps the existing completion result and text.
|
||||
9. Constructors used by current callers keep compiling. They use the legacy default lookup and an
|
||||
explicit inert sink until Unit 5 supplies the production objects.
|
||||
10. Unit tests pass. The developer runs the focused test first, then `mvn clean install` from
|
||||
`fleetd/`.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 1 can run with Units 2, 3, and 4.
|
||||
|
||||
Unit 5 depends on its new lookup, sink, and constructor.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact classifier order in all three paths.
|
||||
- The focused test command and result.
|
||||
- The test which proves a losing waiter race does not publish an event.
|
||||
- The test which proves a fast matching failure is typed.
|
||||
- The final `mvn clean install` result.
|
||||
- Any constructor kept only for transition and where Unit 5 replaces it.
|
||||
|
||||
## Unit 2 — Credential outage policy
|
||||
|
||||
### Scope
|
||||
|
||||
Build a small credential-keyed state machine. It accepts already-classified backend-error events.
|
||||
It does not read pane text, profiles, sessions, or lead state.
|
||||
|
||||
Use an injected monotonic clock. A call records `credentialId`, target, and reason. It returns a new
|
||||
incident only on the threshold crossing. The incident contains a stable event id, credential id,
|
||||
the distinct affected targets, evidence count, window, and remaining cool-off.
|
||||
|
||||
This class owns both correlation and short cool-off. Keeping them together makes threshold crossing
|
||||
and the cool-off deadline one atomic state change.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/placement/BackendOutagePolicy.java`.
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/placement/BackendOutagePolicyTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One error creates no incident and no cool-off.
|
||||
2. Two errors for one credential within 60 seconds create exactly one incident and a 60-second
|
||||
cool-off.
|
||||
3. Two errors more than 60 seconds apart do not create an incident.
|
||||
4. The exact 60-second boundary has a pinned result. Use inclusive `<= 60s` so scheduler delay does
|
||||
not discard evidence at the boundary.
|
||||
5. Different credentials never share evidence.
|
||||
6. Different profiles which supply the same credential id do share evidence. The policy itself only
|
||||
sees the credential id.
|
||||
7. More errors during active cool-off do not extend its deadline and do not return another incident.
|
||||
8. After expiry, old evidence is cleared. Two fresh errors are needed to create the next incident.
|
||||
9. Remaining seconds round up, matching `BackendQuarantine` reporting.
|
||||
10. Concurrent second and third errors cannot return two incidents.
|
||||
11. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 2 can run with Units 1, 3, and 4.
|
||||
|
||||
Unit 5 depends on the policy API.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The state transition table and locking method.
|
||||
- The exact threshold, window, cool-off, and boundary rule.
|
||||
- The test which proves one incident under concurrent calls.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 3 — Lead outage nudge
|
||||
|
||||
### Scope
|
||||
|
||||
Add backend incidents as a fourth pending source in `ReplyPushLoop`. Do not create another scheduler
|
||||
or call `AgentControl.send` from `Fleetd`. The existing combined per-lead schedule is the control
|
||||
which prevents competing injected turns.
|
||||
|
||||
The entry point takes an incident id, affected worker targets, credential id, affected profile
|
||||
names, and remaining cool-off. It resolves distinct owning leads through `PrimaryRegistry`.
|
||||
|
||||
Each `(incidentId, lead)` item is one-shot. It waits while the lead is not injectable. After one
|
||||
successful `agents.send`, remove it. A send exception keeps it pending for a bounded retry. It never
|
||||
uses the repeated reminder behaviour of an uncollected ticket.
|
||||
|
||||
Also add a fail-loud entry point for a classified target that Unit 5 cannot map to a credential. It
|
||||
uses `PrimaryRegistry.nudgeTargetFor(target)` and says that correlation could not run. If no lead is
|
||||
known, log at `WARN`, not `DEBUG`.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/msg/ReplyPushLoop.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/msg/ReplyPushLoopTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One incident affecting two workers owned by one lead causes one successful pane injection.
|
||||
2. Two affected workers owned by two leads cause one successful injection per affected lead. This
|
||||
is one notice per event per lead, not one notice per member.
|
||||
3. Repeating the same incident id is idempotent.
|
||||
4. A busy or unknown lead is not injected. The item stays pending until the lead becomes injectable
|
||||
or its attempt cap is reached.
|
||||
5. After one successful injection, later ticks do not mention that incident again.
|
||||
6. A failed `agents.send` is retried within the existing bound. A successful retry still gives only
|
||||
one successful send.
|
||||
7. A pending ticket and an outage incident for one lead appear in one combined nudge, not two
|
||||
competing turns.
|
||||
8. The text names the credential, profiles, affected workers, and remaining cool-off. It tells the
|
||||
lead to run `fleet_list`.
|
||||
9. An unmapped target produces a direct warning notice when a lead is known. If no lead is known,
|
||||
the code logs a `WARN` naming the target and reason.
|
||||
10. `stop()` clears incident state as it clears other push state.
|
||||
11. Existing reply, ticket, and question tests stay green. The focused tests and
|
||||
`mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. The API uses plain values, not the Unit 2 incident class. This lets Unit 3 run in parallel.
|
||||
|
||||
Unit 5 adapts the Unit 2 incident into this entry point.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact one-shot and retry rules.
|
||||
- The test showing one combined nudge with a failed ticket.
|
||||
- The test showing one successful send for two affected workers.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 4 — Durable member backend-error outcome
|
||||
|
||||
### Scope
|
||||
|
||||
Make a classified backend failure remain visible after its ticket is collected or expires.
|
||||
|
||||
Add `BACKEND_ERROR` to `MemberSession.State`. Add a nullable failure detail to `MemberSession` and
|
||||
render it as `failureReason` in `SessionManager.rosterView`. Add
|
||||
`SessionManager.onBackendError(target, reason)`.
|
||||
|
||||
The transition must handle both completion orderings:
|
||||
|
||||
- `BUSY -> BACKEND_ERROR` when classification wins before the normal completion state update;
|
||||
- `DONE -> BACKEND_ERROR` when the async resolver runs after `SessionManager.onTurnComplete`.
|
||||
|
||||
It must use a compare-and-set retry or another atomic update. `RELEASED` must never return to the
|
||||
roster. A backend-error member is terminal and cannot accept another delivery.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/MemberSession.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/SessionManager.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/session/SessionManagerTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `onBackendError` moves a `BUSY` member to `BACKEND_ERROR` and stores the reason.
|
||||
2. It also moves `DONE` to `BACKEND_ERROR`, covering the resolver race.
|
||||
3. A later normal `onTurnComplete` cannot change `BACKEND_ERROR` back to `DONE`.
|
||||
4. A released or unknown member is not recreated. The unknown case logs at `WARN` and returns an
|
||||
explicit false result to its caller.
|
||||
5. `onDelivered` refuses a `BACKEND_ERROR` member, just as it refuses generic `FAILED`.
|
||||
6. `rosterView` reports `state: backend_error` and `failureReason` after the send ticket is gone.
|
||||
7. Ordinary members do not gain a blank or invented `failureReason` field.
|
||||
8. Existing constructors keep source compatibility for tests and adapters.
|
||||
9. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 4 can run with Units 1, 2, and 3.
|
||||
|
||||
Unit 5 calls the new session method from the production sink.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The two race orderings and the tests for both.
|
||||
- The exact roster JSON shape.
|
||||
- The unknown-target result and log level.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 5 — Production wiring, spawn gate, and fleet views
|
||||
|
||||
### Scope
|
||||
|
||||
Compose Units 1 to 4 in production. This is the only unit which edits `Fleetd.java`.
|
||||
|
||||
Add per-profile `errorPattern` config beside `exhaustedPattern`. Compile both once at startup. A
|
||||
configured pattern wins over the legacy default. Report configured profiles and legacy-default
|
||||
profiles separately at startup. A bad regex must stop startup with the profile and key in the
|
||||
message.
|
||||
|
||||
Wire one production `BackendErrorSink` with this order:
|
||||
|
||||
1. mark the member `backend_error` with its reason;
|
||||
2. resolve the profile and its current `effectiveCredentialId()` through the fail-loud #234 seam;
|
||||
3. record the error in `BackendOutagePolicy`;
|
||||
4. on a new incident, submit one event to `ReplyPushLoop`.
|
||||
|
||||
If target metadata cannot be resolved, do not end in `Optional.ifPresent`. Log an error and call the
|
||||
Unit 3 unmapped-target notice. The failed send still reaches its ticket through CB-588.
|
||||
|
||||
Teach both spawn paths about a separate cool-off source. Exhaustion quarantine has priority when
|
||||
both states are active. Automatic placement needs a distinct `coolingOff` set so its refusal does
|
||||
not say “exhausted”.
|
||||
|
||||
Extend the MCP (Model Context Protocol) views:
|
||||
|
||||
- A cooling profile has `free: 0`, `credentialId`, and `coolingOffForSeconds` in `fleet_list`.
|
||||
- It does not have `quarantinedForSeconds` unless exhaustion quarantine is also active.
|
||||
- `fleet_profiles` has a separate `coolingOff` map, not an entry in `quarantined`.
|
||||
- A direct spawn refusal says the credential is cooling off after repeated backend errors and gives
|
||||
the remaining seconds.
|
||||
|
||||
Do not change `FleetHealthMonitor.coverage`. It still describes the periodic health webhook path.
|
||||
Lead-pane outage delivery is a separate capability.
|
||||
|
||||
### Files owned
|
||||
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/Fleetd.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/ConfigRef.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/member/CompositePeerLauncher.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementContext.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementPolicyUtil.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/FleetConfigTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/ConfigRefTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/member/CompositePeerLauncherTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/placement/PlacementPolicyTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/mcp/FleetMcpTest.java`
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/BackendOutageFlowTest.java`.
|
||||
- `fleetd/fleetd.example.yaml`
|
||||
- `CLAUDE.md`
|
||||
|
||||
No earlier unit edits these files.
|
||||
|
||||
The lead, not a worker, must update `wiki/7-Use-Cases.md`, `wiki/9-Implementation.md`, and
|
||||
`wiki/11-Features.md`. Project rules forbid workers from committing `wiki/`. The portable block in
|
||||
`CLAUDE.md` and `wiki/7-Use-Cases.md` must remain byte-identical.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `errorPattern` binds per profile. Blank uses the legacy default and is reported as degraded
|
||||
coverage. A config reload which changes it is reported as deferred because patterns are compiled
|
||||
at startup.
|
||||
2. A malformed `errorPattern` stops startup and names `profiles.<name>.errorPattern`.
|
||||
3. The real production sink never silently drops an unknown target. A test captures its error log
|
||||
and the fallback notice call.
|
||||
4. One real backend-error classification through `CompletionResolver` marks only its member. It does
|
||||
not cool the credential and does not send an outage notice.
|
||||
5. Two real classifications for one credential within 60 seconds start one incident.
|
||||
6. The integration test then calls the real explicit-profile spawn gate. It is refused before any
|
||||
adapter spawn call, with “cooling off” and remaining seconds in the message.
|
||||
7. The same test calls an automatic placement path. A cooling candidate is skipped. If every
|
||||
candidate is cooling, the error names cool-off rather than exhaustion.
|
||||
8. Profiles sharing the credential are all blocked. A profile on another credential stays usable.
|
||||
9. `fleet_list` from the same fixture shows both members as `backend_error`, preserves each failure
|
||||
reason, and reports `free: 0`, the credential, and `coolingOffForSeconds`.
|
||||
10. `fleet_profiles` reports cool-off separately from quarantine.
|
||||
11. The real `ReplyPushLoop` receives one incident and makes one successful lead-pane send. Existing
|
||||
failed-ticket notice content may share that same combined send.
|
||||
12. Exhaustion still wins when a line matches both patterns. A simultaneous exhaustion quarantine
|
||||
also wins in spawn errors and fleet views.
|
||||
13. After the 60-second cool-off, spawn is allowed again. A new incident needs two fresh errors.
|
||||
14. `healthCoverage` has the same value before and after this change for the same health config.
|
||||
15. `fleetd.example.yaml` explains `errorPattern`, the legacy fallback, 2/60/60 policy, and the
|
||||
difference between cool-off and exhaustion quarantine.
|
||||
16. `CLAUDE.md` tells leads how `fleet_profiles` and `fleet_list` report cool-off. The lead later
|
||||
applies the matching wiki updates and runs the documented byte-sync check.
|
||||
17. The developer records the new end-to-end test failing before implementation, then passing. The
|
||||
focused suites and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
Unit 5 starts only after Units 1 to 4 are merged or rebased into its branch. It also starts after the
|
||||
#234 defect 2 fix lands, because both areas touch the same target-resolution control path.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact commits used for Units 1 to 4 and #234.
|
||||
- The startup coverage line with one configured and one legacy-default profile.
|
||||
- The red test output before the implementation and its green result after.
|
||||
- The explicit and automatic spawn refusal text.
|
||||
- Sample `fleet_list` and `fleet_profiles` JSON for cool-off and exhaustion.
|
||||
- The number and text of lead-pane sends in the real-path test.
|
||||
- The focused test commands and final `mvn clean install` result.
|
||||
- The exact `CLAUDE.md` change and the wiki edits the lead must apply.
|
||||
|
||||
## File ownership summary
|
||||
|
||||
| Area | Unit | Shared edit risk |
|
||||
|---|---:|---|
|
||||
| Completion classification | 1 | Only Unit 1 edits `CompletionResolver` and its test. |
|
||||
| Correlation and cool-off state | 2 | New files only. |
|
||||
| Lead push scheduling | 3 | Only Unit 3 edits `ReplyPushLoop` and its test. |
|
||||
| Member terminal state | 4 | Only Unit 4 edits `MemberSession`, `SessionManager`, and their test. |
|
||||
| Main composition, config, placement, MCP views, shipped prompt | 5 | Only Unit 5 edits `Fleetd`, `FleetConfig`, `CompositePeerLauncher`, placement context, `FleetMcp`, and `CLAUDE.md`. |
|
||||
| Wiki propagation | Lead after Unit 5 | Workers do not commit the wiki submodule. |
|
||||
|
||||
## What I would not build
|
||||
|
||||
1. **Do not reuse `BackendQuarantine` for outages.** Its repeat call restarts a long credential
|
||||
quarantine. Its fields and errors say “exhausted”. That is wrong for a short outage.
|
||||
2. **Do not merge the exhaustion and generic error patterns.** Exhaustion must win because it has a
|
||||
different policy and duration.
|
||||
3. **Do not group by error string.** One outage can produce different text. The shared operational
|
||||
limit is the credential.
|
||||
4. **Do not mark a profile unusable until config changes.** The current classifier cannot safely
|
||||
tell a permanent malformed request from a transient service fault. A permanent state would need
|
||||
a stronger error taxonomy first.
|
||||
5. **Do not quarantine on the first generic backend error.** That would turn one bad request or one
|
||||
false pattern match into a fleet-wide capacity loss.
|
||||
6. **Do not add this to `FleetHealthMonitor`.** The monitor samples slow member health. The exact
|
||||
backend event already exists at completion resolution, and moving it to polling would lose type
|
||||
and time.
|
||||
7. **Do not add another direct lead injector.** `ReplyPushLoop` already owns status gating,
|
||||
per-lead coalescing, retry bounds, and heartbeat stand-down.
|
||||
8. **Do not change `healthCoverage` to `full`.** That field still means a webhook notification sink
|
||||
exists for periodic health. A backend outage nudge does not make every health event visible.
|
||||
9. **Do not persist incident history across daemon restart in this work.** Existing exhaustion
|
||||
quarantine is also in memory. A 60-second state does not justify a new durable store.
|
||||
10. **Do not build work recovery.** The PR-body survival story proves why checkpoint-first work is
|
||||
useful, but these tickets are about detection, capacity, and signalling.
|
||||
11. **Do not remove the legacy `API Error:` fallback in the first release.** Doing so would turn an
|
||||
unedited config back into a false successful completion. Report it as degraded coverage instead.
|
||||
12. **Do not reorder or add the old `visibleTurn` fallback from #201.** #211 already implemented the
|
||||
narrow raw-scrape fallback at `CompletionResolver.classifyRawScrapeFallback`.
|
||||
|
||||
## Riskiest assumption and cheapest experiment
|
||||
|
||||
The riskiest assumption is that a configured error regex means “the backend failed this turn”. The
|
||||
current code and test already show the counterexample: a worker may quote `API Error:` while writing
|
||||
a valid report. Two such false matches on one credential would now remove capacity for 60 seconds.
|
||||
|
||||
The cheapest experiment is a replay corpus before Unit 5 ships:
|
||||
|
||||
1. Save the full pane text from the measured 2026-09-01 outage.
|
||||
2. Produce one safe failure per backend with a disposable invalid endpoint or request.
|
||||
3. Save one valid member report which quotes each error line.
|
||||
4. Replay all samples through the real `CompletionResolver` test fixture.
|
||||
5. Require outage samples to match and quoted-report samples not to match after assistant-block
|
||||
extraction and baseline checks.
|
||||
|
||||
This costs no outage deployment and no real sleep. If quoted reports still match, narrow the profile
|
||||
patterns before enabling correlation. Do not raise the threshold to hide a bad classifier.
|
||||
|
||||
## Sequencing with three developers
|
||||
|
||||
First wave:
|
||||
|
||||
1. Developer A: Unit 1, typed classification.
|
||||
2. Developer B: Unit 2, credential outage policy.
|
||||
3. Developer C: Unit 3, lead outage nudge.
|
||||
|
||||
As soon as one slot is free, start Unit 4. It is file-disjoint from every first-wave unit. Merge and
|
||||
review Units 1 to 4 independently.
|
||||
|
||||
Start Unit 5 only after all four foundations and #234 are available. Unit 5 is the only high-conflict
|
||||
integration branch, so no other active unit should touch its file list.
|
||||
|
||||
## Checks performed for this refinement
|
||||
|
||||
- Read issue #201 and issue #227 through their Gitea pages. Both showed zero comments.
|
||||
- Read the source and tests named in the evidence section.
|
||||
- Ran `git status --short --branch`; the branch was clean before this document was added.
|
||||
- Ran `git log --oneline -12` to identify the branch base.
|
||||
- I did not run Maven because this change adds only a design document.
|
||||
- Rendered both Mermaid blocks with `npx @mermaid-js/mermaid-cli`; both commands succeeded.
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
**Status:** design spec for review → delegate implementation.
|
||||
**Grounded in:** `WorkerService`, `Injector`/`StatusPoller`/`TurnListener`, `MessageService`,
|
||||
`BridgeMcp`, `BridgedApp` (see [wiki 9. Implementation](../wiki/9-Implementation.md)).
|
||||
`FleetMcp`, `FleetApp` (see [wiki 9. Implementation](../wiki/9-Implementation.md)).
|
||||
|
||||
## Problem
|
||||
|
||||
@@ -15,7 +15,7 @@ Consequences today:
|
||||
since when?" without shelling to herdr for a raw agent list (no state, no ownership, no age).
|
||||
- Cleanup of a worker that outlived its owning process depends entirely on the boot-time
|
||||
name-nonce **reaper** (CB-117) — there is no live, authoritative roster during a run.
|
||||
- `bridge_list` (CB-304) can only surface herdr's view, not a bridge-owned roster.
|
||||
- `fleet_list` (CB-304) can only surface herdr's view, not a bridge-owned roster.
|
||||
- There is no seam for per-session policy (checkpoint on teardown → CB-302; idle_ttl /
|
||||
context_cap / drain → CB-303).
|
||||
|
||||
@@ -35,7 +35,7 @@ build on.
|
||||
- **No checkpoint content.** Writing `STATE.md` + commit on teardown is CB-302; CB-301 only exposes
|
||||
the release hook it will attach to.
|
||||
|
||||
"Recycle" under no-reuse is simply **release + fresh acquire** — a helper, not a pool operation.
|
||||
Under no-reuse, a released session is terminal. A new `acquire` always creates a fresh session.
|
||||
|
||||
## Design
|
||||
|
||||
@@ -43,14 +43,9 @@ build on.
|
||||
subscription-guarded spawn/teardown mechanics; `SessionManager` adds the registry, lifecycle, and
|
||||
ownership on top.
|
||||
|
||||
**Package:** new `dev.ltms.bridged.session` — keeps the registry/lifecycle concern separate from
|
||||
**Package:** new `dev.ltms.fleet.session` — keeps the registry/lifecycle concern separate from
|
||||
the `worker` spawn mechanics. Holds `SessionManager` + `WorkerSession`.
|
||||
|
||||
**`recycle` is IN SCOPE for CB-301** (decided): implement `recycle(paneId, …)` = `release` the old
|
||||
session then `acquire` a fresh one, asserting a new distinct paneId (the no-reuse invariant). It is
|
||||
a thin convenience over the two primitives, shipped now so the no-reuse teardown+respawn path is
|
||||
covered by a test from day one.
|
||||
|
||||
### `WorkerSession` (record or small mutable holder)
|
||||
|
||||
| Field | Source | Notes |
|
||||
@@ -88,7 +83,6 @@ SPAWNING|READY|BUSY|DONE --vanished/drop--> FAILED
|
||||
final class SessionManager {
|
||||
WorkerSession acquire(String profile, String requestedCwd, String callerCwd, String ownerTerminal);
|
||||
void release(String paneId); // deterministic teardown + deregister
|
||||
WorkerSession recycle(String paneId, ...); // release + acquire (no-reuse convenience)
|
||||
Optional<WorkerSession> get(String paneId);
|
||||
List<WorkerSession> roster(); // bridge-owned view (CB-304 consumes this)
|
||||
// lifecycle hooks (package-private): onReady/onDelivered/onComplete/onFailed(target)
|
||||
@@ -102,13 +96,13 @@ final class SessionManager {
|
||||
|
||||
### Integration points
|
||||
|
||||
- **`Bridged.main`** — construct `SessionManager(workerService, ...)`; wire it as/decorating the
|
||||
- **`Fleetd.main`** — construct `SessionManager(workerService, ...)`; wire it as/decorating the
|
||||
`TurnListener` alongside `CompletionResolver` so it sees turn boundaries, and give it the
|
||||
`WorkerPresence` signal for `READY`.
|
||||
- **`BridgeMcp.spawn` / `BridgedApp.spawnWorker`** — route spawn through `SessionManager.acquire`
|
||||
(carry `callerTerminal` as `ownerTerminal`). **`bridge_stop` / `DELETE /workers/{paneId}`** →
|
||||
- **`FleetMcp.spawn` / `FleetApp.spawnWorker`** — route spawn through `SessionManager.acquire`
|
||||
(carry `callerTerminal` as `ownerTerminal`). **`fleet_stop` / `DELETE /workers/{paneId}`** →
|
||||
`SessionManager.release`.
|
||||
- **`bridge_list` / `GET /sessions` (CB-304 later)** — read `SessionManager.roster()`.
|
||||
- **`fleet_list` / `GET /sessions` (CB-304 later)** — read `SessionManager.roster()`.
|
||||
- **`MessageService`** — no change required for one-shot; a later CB-303 auto-release hook can call
|
||||
`release` from `onTurnComplete` under policy.
|
||||
|
||||
@@ -120,12 +114,11 @@ final class SessionManager {
|
||||
3. `release` tears the worker down via `WorkerService.stop` and removes it from `roster()`;
|
||||
a second `release` on the same paneId is a harmless no-op.
|
||||
4. `onTurnFailed` / drop moves the session to `FAILED` and it is absent from the live roster.
|
||||
5. `recycle` produces a new paneId and the old one is gone (no-reuse invariant).
|
||||
6. `roster()` reflects exactly the sessions acquired-minus-released, joined with live status.
|
||||
5. `roster()` reflects exactly the sessions acquired-minus-released, joined with live status.
|
||||
|
||||
## Seams left open (deliberately)
|
||||
|
||||
- **CB-302** — attach a checkpoint step (`STATE.md` + commit) to the `release` path.
|
||||
- **CB-303** — a policy loop over `roster()` using `spawnedAtNanos`/state to auto-`release` on
|
||||
`idle_ttl`, or drain on `context_cap`.
|
||||
- **CB-304** — `bridge_list` reads `roster()` for a bridge-owned roster + live join.
|
||||
- **CB-304** — `fleet_list` reads `roster()` for a bridge-owned roster + live join.
|
||||
|
||||
@@ -2,12 +2,12 @@
|
||||
|
||||
**Status:** ✅ shipped — implemented at commit `97ecc71` (per-worker git worktree + config-parity
|
||||
overlay). As-built: `session/GitWorktrees.java` behind the `Worktrees` port, wired in
|
||||
`Bridged.main` and configurable via `worktreeRoot` / per-profile `parityOverlay`
|
||||
(see `bridged.example.yaml`). Branch/worktree surface in `bridge_list` landed with CB-304
|
||||
`Fleetd.main` and configurable via `worktreeRoot` / per-profile `parityOverlay`
|
||||
(see `fleetd.example.yaml`). Branch/worktree surface in `fleet_list` landed with CB-304
|
||||
(`9fe04bf`); the worker-opened-PR checkpoint landed as CB-302 (`64e70ef`).
|
||||
**Extends:** [CB-301 Session Manager](CB-301-Session-Manager.md) (shipped, commit `54d907c`).
|
||||
**Realizes:** the config-parity requirement in [Worker Git Workflow](Worker-Git-Workflow.md).
|
||||
**Grounded in:** `SessionManager`, `WorkerService.spawn/effectiveCwd`, `BridgedConfig.Worker`,
|
||||
**Grounded in:** `SessionManager`, `WorkerService.spawn/effectiveCwd`, `FleetConfig.Worker`,
|
||||
`inject/…LsofPeerPidLookup` (the `ProcessBuilder` exec pattern).
|
||||
|
||||
## Problem
|
||||
@@ -43,7 +43,7 @@ provider.
|
||||
### `WorktreeRequest` (new, nullable = "no worktree")
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.session;
|
||||
package dev.ltms.fleet.session;
|
||||
/** Ask acquire() to provision an isolated worktree. null ⇒ run in the shared primary tree. */
|
||||
public record WorktreeRequest(String ticketSlug, String baseRef) {
|
||||
// ticketSlug seeds the branch name; baseRef null/blank ⇒ current HEAD of the repo.
|
||||
@@ -63,7 +63,7 @@ untouched.
|
||||
### `Worktrees` seam (new)
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.session;
|
||||
package dev.ltms.fleet.session;
|
||||
public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree add <path> -b <branch> <baseRef|HEAD>. Returns the worktree path. */
|
||||
String add(String repoRoot, String branch, String baseRef);
|
||||
@@ -112,7 +112,7 @@ public void release(String paneId) {
|
||||
}
|
||||
```
|
||||
|
||||
### Config — `BridgedConfig.Worker.parityOverlay` + a `worktreeRoot`
|
||||
### Config — `FleetConfig.Worker.parityOverlay` + a `worktreeRoot`
|
||||
|
||||
- Add `List<String> parityOverlay` to the `Worker` record (12th field). Compact-constructor default
|
||||
when null/empty: `[".mcp.json", ".claude/settings.local.json", ".env", ".envrc"]` (missing paths are
|
||||
@@ -122,7 +122,7 @@ public void release(String paneId) {
|
||||
|
||||
### Surface: MCP + REST
|
||||
|
||||
- `bridge_spawn` gains an optional `worktree` arg: `true`, or a ticket slug string. Truthy ⇒ build a
|
||||
- `fleet_spawn` gains an optional `worktree` arg: `true`, or a ticket slug string. Truthy ⇒ build a
|
||||
`WorktreeRequest(slug, null)` and call the 5-arg `acquire`.
|
||||
- `POST /workers` gains `worktree` (+ optional `ticket`) in the body/query, same mapping.
|
||||
- `workerView`/`view(WorkerSession)` include `worktree` and `branch` **when non-null** (omit for
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
# CB-306 — Spawn-Readiness Gate (launcher-owned terminal readiness)
|
||||
|
||||
**Status:** design note / delegation spec (branch `worker/cb-306-readiness`)
|
||||
**Issue:** gitea `lms/claude-bridge` #4
|
||||
**Issue:** gitea `fleet/fleetd` #4
|
||||
**Owner of the behaviour:** `ClaudeCodeLauncher` (the `PeerLauncher` adapter) — NOT core.
|
||||
|
||||
## 1. Problem
|
||||
|
||||
`bridge_spawn` today returns a session the instant the herdr pane is started. The pane is not
|
||||
`fleet_spawn` today returns a session the instant the herdr pane is started. The pane is not
|
||||
yet a usable Claude REPL — it may still be sitting at the folder-trust prompt, or the CLI may
|
||||
never come up at all. Nothing blocks or times out on that. Consequences:
|
||||
|
||||
- A `bridge_send` to a not-yet-ready worker surfaces as a **~60 s MCP-client timeout** (the send
|
||||
- A `fleet_send` to a not-yet-ready worker surfaces as a **~60 s MCP-client timeout** (the send
|
||||
blocks waiting for a turn that can't start) instead of a fast, explicit spawn failure.
|
||||
- A worker stuck at the folder-trust prompt lingers in `SPAWNING` forever; nothing fails it.
|
||||
|
||||
@@ -55,7 +55,7 @@ a crash, and a slow start all present as "never becomes injectable" and all corr
|
||||
or `spawnReadyTimeoutMs` elapses.
|
||||
3. **Ready** → return the `WorkerHandle(paneId, terminalId)` as today.
|
||||
4. **Timeout** → the launcher **closes the pane it started** (and its tab, via the same path
|
||||
`release`/`stop` uses) and throws **`PeerUnreachableException`** (new, in `dev.ltms.bridged.peer`).
|
||||
`release`/`stop` uses) and throws **`PeerUnreachableException`** (new, in `dev.ltms.fleet.peer`).
|
||||
No orphan pane is left behind — the launcher cleans up its own failed birth.
|
||||
|
||||
`spawnReadyTimeoutMs == 0` (or unset) **disables** the gate = legacy non-blocking behaviour, so the
|
||||
@@ -79,8 +79,8 @@ Unit tests (add to the existing `ClaudeCodeLauncher` test):
|
||||
|
||||
## 5. Config
|
||||
|
||||
Add to the launcher-level config (a bridged-level knob, not per-profile) in `bridged.yaml` +
|
||||
`BridgedConfig`:
|
||||
Add to the launcher-level config (a fleetd-level knob, not per-profile) in `fleetd.yaml` +
|
||||
`FleetConfig`:
|
||||
|
||||
```yaml
|
||||
spawn_ready_timeout_ms: 20000 # 0 disables the gate (legacy non-blocking spawn)
|
||||
@@ -88,7 +88,7 @@ spawn_ready_poll_ms: 300
|
||||
```
|
||||
|
||||
Jackson ignores unknown keys, so omitting them in existing YAML is safe; pick sane defaults in code
|
||||
(`20000` / `300`). Keep the names consistent with existing config field style in `BridgedConfig`.
|
||||
(`20000` / `300`). Keep the names consistent with existing config field style in `FleetConfig`.
|
||||
|
||||
## 6. Core / MCP propagation
|
||||
|
||||
@@ -99,11 +99,11 @@ Jackson ignores unknown keys, so omitting them in existing YAML is safe; pick sa
|
||||
`spawn` throws — verify the new exception flows through it (worktree removed, nothing registered).
|
||||
- The **non-worktree** path registers the session only *after* `spawn` returns, so a throw means no
|
||||
half-live `SPAWNING` session is ever registered — confirm this and add a test.
|
||||
- `bridge_spawn` (MCP verb) must return an **error result** carrying the exception message, not a
|
||||
success with a dead session. Trace `BridgeMcp`/`BridgedApp` spawn handlers and make sure the
|
||||
- `fleet_spawn` (MCP verb) must return an **error result** carrying the exception message, not a
|
||||
success with a dead session. Trace `FleetMcp`/`FleetApp` spawn handlers and make sure the
|
||||
exception becomes a clean tool error, not an uncaught 500 with a stack trace.
|
||||
|
||||
**Out of scope (do NOT do here):** gating `bridge_send` on session `READY` (existing status-gate +
|
||||
**Out of scope (do NOT do here):** gating `fleet_send` on session `READY` (existing status-gate +
|
||||
this spawn gate already close the window), MCP-handshake-as-readiness signal, the CB-307 broker,
|
||||
any config `kind:` discriminator, any second adapter.
|
||||
|
||||
@@ -111,7 +111,7 @@ any config `kind:` discriminator, any second adapter.
|
||||
|
||||
- `ClaudeCodeLauncher.spawn` blocks until injectable or throws `PeerUnreachableException` +
|
||||
self-reaps the pane; gate disabled when timeout is 0.
|
||||
- New `PeerUnreachableException` in `dev.ltms.bridged.peer`.
|
||||
- New `PeerUnreachableException` in `dev.ltms.fleet.peer`.
|
||||
- Config knobs wired (`spawn_ready_timeout_ms`, `spawn_ready_poll_ms`) with safe defaults.
|
||||
- Existing `SPAWNING→READY` MCP-contact transition untouched.
|
||||
- New unit tests (ready / timeout+reap / disabled) green; **all existing tests still pass unchanged**.
|
||||
|
||||
@@ -13,28 +13,28 @@ Stage 2 (the AMQP/LavinMQ adapter behind the same port) is explicitly **out of s
|
||||
## 1. The bug this fixes (grounded in current code)
|
||||
|
||||
The reverse (worker→primary) path is `Rendezvous` — a `ConcurrentHashMap<session, CompletableFuture<Resolution>>`
|
||||
of **live blocking waiters only**. No queue, no store. When a worker calls `bridge_reply` and **no send
|
||||
of **live blocking waiters only**. No queue, no store. When a worker calls `fleet_reply` and **no send
|
||||
is currently open** for that worker:
|
||||
|
||||
- `Rendezvous.resolve(session, content)` → `complete(...)` → `waiters.get(session) == null` →
|
||||
returns `false` (`msg/Rendezvous.java:212-215`).
|
||||
- The `content` string is **never retained** — it is dropped. The worker is told it failed:
|
||||
`BridgeMcp.reply` returns `error("no send is awaiting a reply for this worker")` (`mcp/BridgeMcp.java:270-272`);
|
||||
REST returns `409 no_pending_send` (`rest/BridgedApp.java:339-345`).
|
||||
`FleetMcp.reply` returns `error("no send is awaiting a reply for this worker")` (`mcp/FleetMcp.java:270-272`);
|
||||
REST returns `409 no_pending_send` (`rest/FleetApp.java:339-345`).
|
||||
|
||||
This is the observed "communication break": a worker that finishes just after its `bridge_send` timed
|
||||
This is the observed "communication break": a worker that finishes just after its `fleet_send` timed
|
||||
out (the ~60s sync window) replies into the void. There is **no message-id, dedup, or ack** anywhere in
|
||||
the message path today.
|
||||
|
||||
## 2. What to build
|
||||
|
||||
### 2.1 The port — `dev.ltms.bridged.msg.ReplyInbox`
|
||||
### 2.1 The port — `dev.ltms.fleet.msg.ReplyInbox`
|
||||
|
||||
A thin interface owned by the `msg` layer. The in-memory adapter is Stage 1; the AMQP adapter (Stage 2)
|
||||
implements the **same** interface, so keep it broker-agnostic.
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.msg;
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@@ -70,7 +70,7 @@ public interface ReplyInbox {
|
||||
seen-set — your call; preserve insertion order).
|
||||
- `peek` returns an immutable copy; `ack` removes by `msgId`. Thread-safe (concurrent publish vs. drain).
|
||||
- **This is soft-state, NOT persistence.** Lost on a `java -jar` bounce — that is correct and consistent
|
||||
with "bridged stays soft-state." Do **not** add any file/DB backing.
|
||||
with "fleetd stays soft-state." Do **not** add any file/DB backing.
|
||||
|
||||
### 2.3 Publish seam — route reply through the service layer
|
||||
|
||||
@@ -79,7 +79,7 @@ in `MessageService`, which already owns the `Rendezvous` and will own the `Reply
|
||||
|
||||
- Add `MessageService.reply(String session, String content)`:
|
||||
```java
|
||||
/** Route a worker's explicit bridge_reply: resolve an open send, or queue it in the inbox if none. */
|
||||
/** Route a worker's explicit fleet_reply: resolve an open send, or queue it in the inbox if none. */
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
return true; // a live send took it — unchanged fast path
|
||||
@@ -89,12 +89,12 @@ in `MessageService`, which already owns the `Rendezvous` and will own the `Reply
|
||||
}
|
||||
```
|
||||
- Repoint the two callers off the bare `rendezvous.resolve(...)` onto `messages.reply(...)`:
|
||||
- `BridgeMcp.reply` (`mcp/BridgeMcp.java:262-273`) — on success return a normal ack; **remove** the
|
||||
- `FleetMcp.reply` (`mcp/FleetMcp.java:262-273`) — on success return a normal ack; **remove** the
|
||||
`error("no send is awaiting a reply…")` branch (that case is now a successful queue).
|
||||
- `BridgedApp.replyMessage` (`rest/BridgedApp.java:330-346`) — return `200` (queued) instead of
|
||||
- `FleetApp.replyMessage` (`rest/FleetApp.java:330-346`) — return `200` (queued) instead of
|
||||
`409 no_pending_send`.
|
||||
|
||||
**DO NOT touch the QUESTION path.** `bridge_ask` / `rendezvous.resolveQuestion` must keep today's
|
||||
**DO NOT touch the QUESTION path.** `fleet_ask` / `rendezvous.resolveQuestion` must keep today's
|
||||
`NO_WAITER` behaviour — a mid-turn question is **interactive** (the worker blocks synchronously and cannot
|
||||
consume a late answer), so it must **never** be queued. Only terminal `REPLY`s go to the inbox.
|
||||
|
||||
@@ -110,27 +110,27 @@ The primary re-checks a worker it delegated to. Expose a drain keyed by **worker
|
||||
- Add `MessageService.drainReplies(String target)`: `peek` the inbox, `ack` each returned `msgId`, hand
|
||||
back the `List<InboxMessage>` (or just the contents). At-least-once: peek→deliver→ack (ack only after
|
||||
the caller has them, so an in-flight failure re-surfaces them).
|
||||
- **DECISION (required default): expose via the existing poll verb, keyed by target.** Extend `bridge_poll`
|
||||
- **DECISION (required default): expose via the existing poll verb, keyed by target.** Extend `fleet_poll`
|
||||
to accept an optional `target` (worker session) and, when present, return that worker's drained replies —
|
||||
alongside a matching REST route `GET /sessions/{id}/replies`. Do **not** change `send`/`answer` semantics
|
||||
(do not drain inside `send` — that conflates "deliver to worker" with "collect its mail"). Keep the
|
||||
existing ticket-based `bridge_poll(ticket)` path working unchanged. If you see a cleaner surface, still
|
||||
existing ticket-based `fleet_poll(ticket)` path working unchanged. If you see a cleaner surface, still
|
||||
ship this default and note the alternative for review.
|
||||
|
||||
## 3. Config
|
||||
|
||||
**None for Stage 1.** The in-memory adapter is the unconditional default — wire `new InMemoryReplyInbox()`
|
||||
into `MessageService` in `Bridged.main`. Do **not** add a `broker:` config block (that arrives with the
|
||||
into `MessageService` in `Fleetd.main`. Do **not** add a `broker:` config block (that arrives with the
|
||||
Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 4. Acceptance criteria (what the primary will verify)
|
||||
|
||||
1. New `ReplyInbox` + `InboxMessage` + `InMemoryReplyInbox` in `dev.ltms.bridged.msg`.
|
||||
2. `bridge_reply` with **no open send** now **succeeds and queues** (no more `error` / `409`); the reply is
|
||||
1. New `ReplyInbox` + `InboxMessage` + `InMemoryReplyInbox` in `dev.ltms.fleet.msg`.
|
||||
2. `fleet_reply` with **no open send** now **succeeds and queues** (no more `error` / `409`); the reply is
|
||||
later retrievable and identical.
|
||||
3. The queued reply is drainable by the primary keyed by target; draining **acks** it (a second drain
|
||||
returns nothing); dedup by `msgId` (re-publishing the same id does not double-queue).
|
||||
4. **QUESTION path unchanged** — `bridge_ask` with no open send still returns `NO_WAITER` (add/keep a test
|
||||
4. **QUESTION path unchanged** — `fleet_ask` with no open send still returns `NO_WAITER` (add/keep a test
|
||||
proving a question is never queued).
|
||||
5. Completion/failure fallbacks unchanged.
|
||||
6. Unit tests covering: `InMemoryReplyInbox` publish/peek/ack/dedup/FIFO/concurrency; `MessageService.reply`
|
||||
@@ -140,7 +140,7 @@ Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 5. Build & verification (worker side)
|
||||
|
||||
- Build with Maven from the worktree's `bridged/` dir. **Capture the exit code without a masking pipe**
|
||||
- Build with Maven from the worktree's `fleetd/` dir. **Capture the exit code without a masking pipe**
|
||||
(`mvn clean install; echo "MVN_EXIT=$?"` — never `mvn … | tail`, which hides failures).
|
||||
- Read the real test totals from `target/surefire-reports/TEST-*.xml`, not from stdout scroll.
|
||||
- You do **not** have IDE MCP access — do not claim `ide_diagnostics` results. The **primary** runs the
|
||||
@@ -152,7 +152,7 @@ Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
- **`.mcp.json` is `--skip-worktree` in your worktree — never edit, `git add`, or commit it.**
|
||||
- **`wiki/` is a submodule — never run git in it; never touch it.**
|
||||
- Commit only your feature changes (the new port/adapter, the `msg`/`mcp`/`rest` wiring, tests, and if
|
||||
you add config wiring in `Bridged.java`). Nothing else.
|
||||
you add config wiring in `Fleetd.java`). Nothing else.
|
||||
- Work only inside your assigned worktree on your feature branch. The primary fast-forwards `main` after
|
||||
re-gating — do not touch `main`.
|
||||
- Java 25 idioms are welcome (unnamed `_` params, records). Keep the diff minimal and match surrounding style.
|
||||
@@ -171,7 +171,7 @@ removal, msgId dedup, cross-restart redelivery). It is `@Tag("contract")`, so th
|
||||
untouched. Run it explicitly when Docker (or a broker) is available:
|
||||
|
||||
```bash
|
||||
cd bridged
|
||||
cd fleetd
|
||||
mvn -Pcontract test -Dtest=AmqpReplyInboxContractTest # local: spins a RabbitMQ Testcontainers fixture
|
||||
```
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ The design rests on three pieces (the shape this ticket proposes):
|
||||
1. **Dedicated per-agent channels** — every agent has its own addressable inbox on the broker.
|
||||
2. **A federated agent directory** — a global "who/where/status" lookup, assembled from per-host
|
||||
presence, not a central database.
|
||||
3. **A per-host gateway** — each host runs a `bridged` that owns its local herdr, registers/manages
|
||||
3. **A per-host gateway** — each host runs a `fleetd` that owns its local herdr, registers/manages
|
||||
its own sessions, and proxies messages to/from other hosts over the broker.
|
||||
|
||||
## 2. What is single-host today (the assumptions to break)
|
||||
@@ -27,7 +27,7 @@ The design rests on three pieces (the shape this ticket proposes):
|
||||
flowchart TB
|
||||
subgraph host["Single host (today)"]
|
||||
primary["primary<br/>(MCP client)"]
|
||||
daemon["bridged daemon<br/>127.0.0.1:8765"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765"]
|
||||
reg["in-process registry<br/>keyed by PeerHandle.id() == paneId"]
|
||||
herdr["herdr<br/>(local unix-socket PTY mux)"]
|
||||
w1["worker pane wQ:p1"]
|
||||
@@ -48,21 +48,21 @@ Three concrete bake-ins assume one host:
|
||||
|---|---|---|
|
||||
| **herdr is local** | `herdr/` unix socket `~/.config/herdr/herdr.sock` | You cannot drive another host's PTYs → each host **must** own its herdr. This is why a per-host gateway is mandatory. |
|
||||
| **registry is in-process, keyed by `paneId`** | `session/SessionManager` | `paneId` (e.g. `wQ:p2B`) is a herdr-local coordinate — meaningless off-host. Routing needs a host-unique id. |
|
||||
| **loopback, no authn** | `rest/BridgedApp` binds `127.0.0.1:8765` | Fine on one host; the moment a second host can talk to a gateway, that link is a trust boundary. |
|
||||
| **loopback, no authn** | `rest/FleetApp` binds `127.0.0.1:8765` | Fine on one host; the moment a second host can talk to a gateway, that link is a trust boundary. |
|
||||
|
||||
## 3. Target architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A"]
|
||||
gA["gateway = bridged A"]
|
||||
gA["gateway = fleetd A"]
|
||||
regA["local registry + herdr"]
|
||||
primary["primary (MCP client)"]
|
||||
gA --- regA
|
||||
primary --- gA
|
||||
end
|
||||
subgraph hostB["HOST B"]
|
||||
gB["gateway = bridged B"]
|
||||
gB["gateway = fleetd B"]
|
||||
regB["local registry + herdr"]
|
||||
wb["worker panes"]
|
||||
gB --- regB
|
||||
@@ -96,13 +96,13 @@ host's terminals.*
|
||||
delayed-message exchange (the remind/backoff loop for free) — the same reasons CB-307 picked it.
|
||||
|
||||
- **Federated agent directory** = a **soft-state, bridge-owned** roster, *not* a broker-stored
|
||||
database. Per the persistence-boundary decision (bridged is soft-state; the broker owns *message*
|
||||
database. Per the persistence-boundary decision (fleetd is soft-state; the broker owns *message*
|
||||
durability, not *who/where/status*), each gateway announces its local agents `(globalId, host,
|
||||
status, capabilities)` on a `roster.*` presence topic with periodic heartbeats. Every gateway
|
||||
builds an eventually-consistent **union view** — literally CB-304's `rosterView`, federated. A
|
||||
stale entry expires by missed heartbeat (reuses CB-303's idle/TTL thinking).
|
||||
|
||||
- **Per-host gateway** = today's `bridged` daemon, evolved. It already registers/manages sessions
|
||||
- **Per-host gateway** = today's `fleetd` daemon, evolved. It already registers/manages sessions
|
||||
and controls its local herdr; multi-host adds exactly two responsibilities: (a) a broker client
|
||||
that consumes its agents' inboxes and injects into local herdr, and (b) presence announce +
|
||||
union-roster assembly. Evolution, not rewrite.
|
||||
@@ -111,7 +111,7 @@ host's terminals.*
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
send["bridge_send(globalId, msg)"] --> lookup{"directory:<br/>is globalId local?"}
|
||||
send["fleet_send(globalId, msg)"] --> lookup{"directory:<br/>is globalId local?"}
|
||||
lookup -->|"yes"| local["inject via local herdr<br/>(today's Injector path)"]
|
||||
lookup -->|"no"| pub["publish agent.<id>.inbox<br/>(broker routes to owning gateway)"]
|
||||
pub --> consume["owning gateway consumes<br/>→ injects into its local herdr"]
|
||||
@@ -123,7 +123,7 @@ broker. A sender is oblivious to which branch it took.*
|
||||
## 4. What CB-307 already provides vs. what is net-new
|
||||
|
||||
**CB-307 delivers the transport half** and is independently valuable on a single host: the AMQP
|
||||
broker fabric, the `bridged → broker` client/adapter, at-least-once + idempotent (dedup-by-id)
|
||||
broker fabric, the `fleetd → broker` client/adapter, at-least-once + idempotent (dedup-by-id)
|
||||
delivery, DLQ, and delayed-retry (remind). That *is* the "proxy cross-host message" backbone;
|
||||
extending the same broker from "worker→primary reliability" to "gateway↔gateway" is incremental.
|
||||
|
||||
@@ -158,17 +158,17 @@ sequenceDiagram
|
||||
participant BR as broker
|
||||
participant GA as gateway A
|
||||
participant P as primary (host A, MCP client)
|
||||
W->>GB: bridge_reply
|
||||
W->>GB: fleet_reply
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route to A's primary inbox
|
||||
Note over GA: held durably until the primary pulls
|
||||
P->>GA: blocking bridge_send resolves / bridge_poll
|
||||
P->>GA: blocking fleet_send resolves / fleet_poll
|
||||
GA-->>P: reply (then ACK to broker)
|
||||
```
|
||||
|
||||
*Figure 4 — the broker makes the middle hop lossless, ordered, and idempotent; the **final** hop
|
||||
into the primary is still a **pull** (gateway A holds the message until the primary's blocking
|
||||
`bridge_send` or `bridge_poll`). Cross-host neither improves nor worsens this — it just spans hosts.
|
||||
`fleet_send` or `fleet_poll`). Cross-host neither improves nor worsens this — it just spans hosts.
|
||||
This is precisely the gap CB-307 closes on one host and CB-308 stretches across hosts.*
|
||||
|
||||
## 6. Staging & dependencies
|
||||
@@ -207,8 +207,8 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
gateway's host. This extends the single-host invariant — *identity comes from the connection,
|
||||
never an argument* — across the broker: cross-host, identity comes from the key. Complements
|
||||
(not replaces) per-gateway broker logins over TLS.
|
||||
2. **Profiles are owned by the worker's host.** `bridge_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `bridged.yaml`. Gateways advertise their profile names in presence
|
||||
2. **Profiles are owned by the worker's host.** `fleet_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `fleetd.yaml`. Gateways advertise their profile names in presence
|
||||
heartbeats, so a leader sees what each host offers before spawning; an unknown name is a clear
|
||||
error from the target. Secrets (base URLs, tokens) never leave the host that uses them.
|
||||
3. **Repo provisioning — clone from the forge, pinned.** A cross-host spawn names the repo URL and
|
||||
@@ -259,7 +259,7 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
serialize acks fleet-wide. Ordering caveat: a *return* (unroutable) arrives **before** the
|
||||
confirm, so "confirmed" ≠ "routed"; the sender checks the returned-set at confirm time.
|
||||
`mandatory` is false only for `BROADCAST`, where an empty group is legal silence.
|
||||
10. **Queue lifecycle is session lifecycle.** `bridge_stop`/reap deletes the worker's inbox queue
|
||||
10. **Queue lifecycle is session lifecycle.** `fleet_stop`/reap deletes the worker's inbox queue
|
||||
(its `broadcast.*` bindings die with it — no broadcasts to the dead); `x-expires` collects
|
||||
queues orphaned by a crashed gateway (long for main/orchestrator inboxes, short for workers).
|
||||
Queue names carry a version suffix (`.v2`): AMQP refuses to redeclare an existing durable
|
||||
@@ -275,11 +275,11 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
- **Gateway discovery:** how gateways find the broker and each other (static config vs. discovery).
|
||||
- **Control authorization — THE GATE ON U4.** Signing (§7.1) settles *who sent it*; authorization
|
||||
is *who may do what*. **Cross-host spawn must not land before the minimal version exists**: a
|
||||
per-host allowlist in `bridged.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
per-host allowlist in `fleetd.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
publish control to this host, checked against the verified signature. A few lines of config and
|
||||
check; without them, any principal holding broker credentials can start processes on every host
|
||||
in the fleet.
|
||||
- **Key distribution & rotation:** static config (host → public key in each `bridged.yaml`) is
|
||||
- **Key distribution & rotation:** static config (host → public key in each `fleetd.yaml`) is
|
||||
fine at the current 2–3 host scale; rotation is manual. A refinement, not a blocker.
|
||||
- **Gateway death mid-turn:** the roster reaps it by missed heartbeat, and in-flight primary-bound
|
||||
messages survive by broker durability; still open is reconciling *worker* state when the dead
|
||||
|
||||
@@ -7,7 +7,7 @@ the core learning any one peer's environment.
|
||||
## 1. Why
|
||||
|
||||
`claude-bridge` is a **communication bus between heterogeneous AI agents** — its stable surface is
|
||||
the protocol (`bridge_spawn / send / poll / reply / ask / list / stop / status`), and that surface
|
||||
the protocol (`fleet_spawn / send / poll / reply / ask / list / stop / status`), and that surface
|
||||
should stay provider-neutral. Today the daemon can only materialize one kind of peer: an
|
||||
off-subscription Claude Code CLI over herdr. Everything specific to *how that peer is set up*
|
||||
(`ANTHROPIC_BASE_URL`, the subscription guard, `--mcp-config`/system-prompt flags, `claude-*`
|
||||
@@ -60,7 +60,7 @@ Where Claude/herdr specifics actually live today:
|
||||
| `claude-<profile>-<nonce>-<seq>` naming, `WORKER_NAME` regex, orphan reap (CB-117) | `WorkerService` | **→ adapter** (naming is a herdr-label detail) |
|
||||
| `GITEA_TOKEN` / `GITEA_HOST` injection (CB-302 checkpoint) | `WorkerService.spawn` | **→ adapter** + a **capability** (§6) |
|
||||
| tab/pane placement, worker space, tab labels | `WorkerService.spawnInTab/spawnAsPane` via herdr `WorkspaceControl` | **→ adapter** (herdr transport detail) |
|
||||
| `BridgedConfig.Worker` profile shape (`baseUrl`, `model`, `configDir`, …) | `config` | **mostly adapter-shaped** — see §5 |
|
||||
| `FleetConfig.Worker` profile shape (`baseUrl`, `model`, `configDir`, …) | `config` | **mostly adapter-shaped** — see §5 |
|
||||
| FSM, registry, roster, `reapIdle`/`drainAll`/`contextCap`, `rosterView` | `SessionManager` | **stays core** |
|
||||
| turn/completion detection (`TurnListener`, `CompletionResolver`, `StatusPoller`, `WorkerPresence`) | `inject/` | **stays core**, but reads herdr terminal output → transport-coupled (§4b) |
|
||||
| message store & routing | `msg/` | **stays core** |
|
||||
@@ -118,7 +118,7 @@ hard-wire "turns come from herdr".
|
||||
|
||||
## 5. Config shape
|
||||
|
||||
`BridgedConfig.Worker` is Claude-shaped (`baseUrl`, `model`, `configDir`, `tokenEnv`). Rather than
|
||||
`FleetConfig.Worker` is Claude-shaped (`baseUrl`, `model`, `configDir`, `tokenEnv`). Rather than
|
||||
break existing YAML, CB-401 keeps `workers:` exactly as-is and treats those fields as the
|
||||
**ClaudeCodeLauncher's** profile schema. A future peer kind adds a `kind:` discriminator
|
||||
(default `"claude-code"`) selecting the launcher; unknown-kind → clear config error. No migration of
|
||||
@@ -126,7 +126,7 @@ existing configs. (Jackson already ignores unknown keys, so adding `kind` is bac
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MCP as bridge_spawn (MCP/REST)
|
||||
participant MCP as fleet_spawn (MCP/REST)
|
||||
participant SM as SessionManager
|
||||
participant L as PeerLauncher (by profile.kind)
|
||||
participant T as transport (herdr)
|
||||
@@ -150,7 +150,7 @@ degrades gracefully when a launcher lacks one:
|
||||
|
||||
| Capability | Meaning | Claude Code | Codex (likely) | Human |
|
||||
|---|---|---|---|---|
|
||||
| `MID_TURN_ASK` | supports `bridge_ask` rendezvous | ✓ | ? | ✗ |
|
||||
| `MID_TURN_ASK` | supports `fleet_ask` rendezvous | ✓ | ? | ✗ |
|
||||
| `SELF_PR` | can open its own PR at checkpoint (CB-302) | ✓ (opt-in token) | ? | ✗ |
|
||||
| `WORKTREE` | can run in a provisioned git worktree | ✓ | ✓ | ✗ |
|
||||
| `ORPHAN_REAP` | spawner can reconcile orphaned peers on boot | ✓ | ? | ✗ |
|
||||
@@ -185,13 +185,13 @@ messages, not verified facts.
|
||||
Deliverable for CB-401 Stage A — mechanical, behaviour-preserving:
|
||||
|
||||
1. `PeerLauncher` interface + `PeerHandle` (opaque id) + `SpawnRequest` (profile, requestedCwd,
|
||||
callerCwd) + `Capability` enum, new package `dev.ltms.bridged.peer`.
|
||||
callerCwd) + `Capability` enum, new package `dev.ltms.fleet.peer`.
|
||||
2. `ClaudeCodeLauncher implements PeerLauncher` = today's `WorkerService`, adapted: `spawn(...)`
|
||||
returns a `PeerHandle` (id = paneId), `capabilities()` declares
|
||||
`MID_TURN_ASK, SELF_PR(when token), WORKTREE, ORPHAN_REAP`.
|
||||
3. `SessionManager` depends on `PeerLauncher`, not `WorkerService` concretely; routing keys on
|
||||
`PeerHandle.id()` (== paneId today, so zero value change).
|
||||
4. `Bridged.main` wires the concrete `ClaudeCodeLauncher` behind the interface.
|
||||
4. `Fleetd.main` wires the concrete `ClaudeCodeLauncher` behind the interface.
|
||||
5. **No behaviour change, no config change.** Full green gate: `ide_sync` → `ide_diagnostics`
|
||||
(0 errors/0 warnings) → `mvn clean install` with `MVN_EXIT` captured (no masking pipe). All
|
||||
existing tests pass unchanged; add tests only for the new `PeerHandle` indirection.
|
||||
|
||||
@@ -11,7 +11,7 @@ All five increments of §4 are done, including increment 5 (the §5 live checkli
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Prove the [`PeerLauncher`](../bridged/src/main/java/dev/ltms/bridged/peer/PeerLauncher.java) SPI
|
||||
Prove the [`PeerLauncher`](../fleetd/src/main/java/dev/ltms/fleet/peer/PeerLauncher.java) SPI
|
||||
actually holds for a **non-Claude** coding agent by shipping a second, first-class in-tree
|
||||
adapter: **opencode** (`opencode` 1.1.31, a provider-agnostic terminal coding agent).
|
||||
|
||||
@@ -56,8 +56,8 @@ flowchart TB
|
||||
|
||||
*Figure 1 — the two concerns tangled inside today's single launcher; CB-402 splits them.*
|
||||
|
||||
There is also a **Stage-A deferral** to finish: `Bridged.main` still casts
|
||||
`(ClaudeCodeLauncher) workers` at the `BridgeMcp` and `BridgedApp` constructors. Those two
|
||||
There is also a **Stage-A deferral** to finish: `Fleetd.main` still casts
|
||||
`(ClaudeCodeLauncher) workers` at the `FleetMcp` and `FleetApp` constructors. Those two
|
||||
callers only invoke `profiles()`, `defaultProfile()`, and `list()` — **all already on the
|
||||
`PeerLauncher` interface**. The cast survives for one reason only: `PeerLauncher.list()`
|
||||
returns `List<?>` (element type erased) while the callers use `Agent` element methods in their
|
||||
@@ -68,7 +68,7 @@ roster join. Finishing the migration is therefore small and contained (§4.D).
|
||||
## 3. Target design
|
||||
|
||||
Template-Method base + two thin adapters + a routing composite that keeps the Stage-A seam
|
||||
(one `PeerLauncher` reference held by `SessionManager` / `BridgeMcp` / `BridgedApp`) intact.
|
||||
(one `PeerLauncher` reference held by `SessionManager` / `FleetMcp` / `FleetApp`) intact.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
@@ -114,7 +114,7 @@ Two adapter **hooks** (abstract):
|
||||
protected abstract String namePrefix(); // "claude" | "opencode"
|
||||
|
||||
/** Build the peer-specific launch: env map + argv. Runs any pre-spawn guard here. */
|
||||
protected abstract Launch buildLaunch(BridgedConfig.Worker cfg, SpawnRequest req);
|
||||
protected abstract Launch buildLaunch(FleetConfig.Worker cfg, SpawnRequest req);
|
||||
record Launch(Map<String,String> env, List<String> argv) {}
|
||||
```
|
||||
|
||||
@@ -126,7 +126,7 @@ so the opencode adapter never reaps a `claude-*` pane and vice-versa. The compos
|
||||
|
||||
### B. `kind:` config discriminator
|
||||
|
||||
Add one field to `BridgedConfig.Worker`:
|
||||
Add one field to `FleetConfig.Worker`:
|
||||
|
||||
```java
|
||||
String kind // "claude-code" (default) | "opencode"
|
||||
@@ -138,7 +138,7 @@ String kind // "claude-code" (default) | "opencode"
|
||||
so the record stays declarative.)
|
||||
- Keep the existing back-compat constructors; `kind` is additive and optional.
|
||||
|
||||
`bridged.example.yaml` documents a two-kind `workers:` block.
|
||||
`fleetd.example.yaml` documents a two-kind `workers:` block.
|
||||
|
||||
### C. `OpenCodeLauncher` — the adapter hooks for opencode
|
||||
|
||||
@@ -164,13 +164,13 @@ String kind // "claude-code" (default) | "opencode"
|
||||
|
||||
### D. `CompositePeerLauncher` + finish the Stage-A migration
|
||||
|
||||
- `Bridged.main` groups configured profiles by `kind`, instantiates one launcher per kind
|
||||
- `Fleetd.main` groups configured profiles by `kind`, instantiates one launcher per kind
|
||||
present, and wraps them in `CompositePeerLauncher implements PeerLauncher`.
|
||||
- Routing methods (`spawn(req)`, `effectiveCwd(req)`, `parityOverlay(name)`) dispatch by the
|
||||
profile's kind. Fan-out methods (`list()`, `reapOrphanWorkers()`, `capabilities()`,
|
||||
`profiles()`, `defaultProfile()`) merge across sub-launchers. `stop(id)` tries each (teardown
|
||||
only knows the pane id) — already best-effort/idempotent.
|
||||
- **Migrate `BridgeMcp` + `BridgedApp` to the `PeerLauncher` interface**, dropping both
|
||||
- **Migrate `FleetMcp` + `FleetApp` to the `PeerLauncher` interface**, dropping both
|
||||
`(ClaudeCodeLauncher)` casts. Only friction is `list()`'s `List<?>`; resolve by giving the SPI
|
||||
a typed roster element (small neutral `PeerAgent` view exposing `id()`/`name()`/status) that
|
||||
the CB-304 roster join consumes — or, minimally, narrow at the callsite. Prefer the typed view.
|
||||
@@ -179,12 +179,12 @@ String kind // "claude-code" (default) | "opencode"
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant P as Primary
|
||||
participant M as BridgeMcp / REST
|
||||
participant M as FleetMcp / REST
|
||||
participant C as CompositePeerLauncher
|
||||
participant O as OpenCodeLauncher
|
||||
participant B as HerdrPeerLauncher (base)
|
||||
participant H as herdr
|
||||
P->>M: bridge_spawn(profile="oc-impl")
|
||||
P->>M: fleet_spawn(profile="oc-impl")
|
||||
M->>C: spawn(SpawnRequest)
|
||||
C->>C: kind(profile)=="opencode"
|
||||
C->>O: spawn(req)
|
||||
@@ -208,7 +208,7 @@ sequenceDiagram
|
||||
`ClaudeCodeLauncher` extend it with `namePrefix()="claude"` and `buildLaunch()` wrapping
|
||||
today's guard+env+argv logic. Green build, identical tests — pure refactor. *(IDE
|
||||
`refactor` where possible; the primary re-runs the gate workers can't.)*
|
||||
2. **`kind:` discriminator.** Add the field + normalization + `bridged.example.yaml`. Default
|
||||
2. **`kind:` discriminator.** Add the field + normalization + `fleetd.example.yaml`. Default
|
||||
path unchanged (`kind=claude-code`).
|
||||
3. **`OpenCodeLauncher`.** Implement the three hooks; unit-test `buildLaunch` (env has no
|
||||
`ANTHROPIC_BASE_URL`; `OPENCODE_CONFIG` points at a file carrying the bridge MCP block +
|
||||
@@ -226,11 +226,11 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
- **opencode TUI ⇄ herdr injection.** herdr drives a pane by typing into a TUI. Must confirm
|
||||
opencode's TUI accepts injected keystrokes/submit the way `claude` does, and reaches an
|
||||
`injectable` status the CB-306 gate recognizes. *Validation:* spawn one opencode worker,
|
||||
watch the readiness gate pass, `bridge_send` a trivial task.
|
||||
watch the readiness gate pass, `fleet_send` a trivial task.
|
||||
- **Bridge MCP visibility in opencode.** Confirm `OPENCODE_CONFIG` (or `opencode mcp add`)
|
||||
actually surfaces the `bridge_*` tools inside the opencode session, and that `bridge_reply`
|
||||
actually surfaces the `fleet_*` tools inside the opencode session, and that `fleet_reply`
|
||||
is callable — the reply-charter is worthless if the tool isn't mounted. *Validation:* the
|
||||
worker completes a task by calling `bridge_reply`; the reply lands via the CB-307 path.
|
||||
worker completes a task by calling `fleet_reply`; the reply lands via the CB-307 path.
|
||||
- **opencode MCP/config schema drift.** opencode is fast-moving (1.1.31 today). Pin the config
|
||||
schema we generate against the installed version; treat the exact keys (`type: "remote"` vs
|
||||
`"http"`, `instructions` shape) as a dogfood-verified fact, not an assumption.
|
||||
@@ -244,7 +244,7 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
|
||||
- **Unit (hermetic):** base-extraction regression (existing `ClaudeCodeLauncher` tests pass
|
||||
unchanged); `OpenCodeLauncher.buildLaunch` env/argv/config assertions; `kind` normalization
|
||||
in `BridgedConfigTest`; `CompositePeerLauncher` routing + fan-out (merge of `profiles()`,
|
||||
in `FleetConfigTest`; `CompositePeerLauncher` routing + fan-out (merge of `profiles()`,
|
||||
summed `reapOrphanWorkers()`, per-kind reap isolation) with fake sub-launchers.
|
||||
- **Live (dogfood, manual):** the §5 checklist on the running daemon.
|
||||
- **Gate (primary):** IDE diagnostics 0/0 on every changed file, `mvn clean install` green with
|
||||
@@ -269,7 +269,7 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
|
||||
## 8. As-built — live dogfood (2026-07-29)
|
||||
|
||||
Run against `bridged` on `127.0.0.1:8766` at main `19cdf8d`, with opencode **1.18.5** installed
|
||||
Run against `fleetd` on `127.0.0.1:8766` at main `19cdf8d`, with opencode **1.18.5** installed
|
||||
via Homebrew. Every §5 risk is now a verified fact rather than an assumption.
|
||||
|
||||
**The version-drift risk was the real one, and it did not bite.** This adapter was designed against
|
||||
@@ -281,7 +281,7 @@ as a dogfood-verified fact for 1.18.5.
|
||||
| §5 risk | Result |
|
||||
|---|---|
|
||||
| opencode TUI ⇄ herdr injection; CB-306 gate | ✅ `peer pane=wD:p3 reached injectable state` ~0.6s after `agent.start` |
|
||||
| Bridge MCP visible + `bridge_reply` callable | ✅ MCP `initialize` from `Implementation[name=opencode, version=1.18.5]`; worker replied through the tool |
|
||||
| Bridge MCP visible + `fleet_reply` callable | ✅ MCP `initialize` from `Implementation[name=opencode, version=1.18.5]`; worker replied through the tool |
|
||||
| Config schema drift (1.1.31 → 1.18.5) | ✅ unchanged, see above |
|
||||
| Provider credentials | ✅ free tier, zero credentials |
|
||||
|
||||
@@ -291,7 +291,7 @@ Full lifecycle exercised through the REST surface:
|
||||
to `OpenCodeLauncher` (`spawning opencode profile=opencode-free`), pane `wD:p3`.
|
||||
2. Readiness: `{"ready":true,"status":"idle"}`, roster state `ready`.
|
||||
3. `POST /sessions/{id}/message` → **`{"replySource":"reply","reply":"391"}`** — a *structured*
|
||||
`bridge_reply`, not the CB-115 completion-fallback transcript scrape. The clean path.
|
||||
`fleet_reply`, not the CB-115 completion-fallback transcript scrape. The clean path.
|
||||
4. `DELETE /workers/wD:p3` → `204`, roster empty, tolerant teardown (`tab_not_found` ignored —
|
||||
opencode had already closed its own tab).
|
||||
|
||||
@@ -301,5 +301,5 @@ identity (loopback peer PID → herdr pane) classified an **opencode** process a
|
||||
opencode-specific handling — confirming the identity model is peer-kind-agnostic, which is exactly
|
||||
what CB-308 needs when it stretches the roster across hosts.
|
||||
|
||||
CB-502 counters for the same run: `bridged_sends_total{outcome="replied"} 1`,
|
||||
`bridged_replies_total{path="rendezvous"} 1`, `bridged_inbox_depth{...} 0`.
|
||||
CB-502 counters for the same run: `fleet_sends_total{outcome="replied"} 1`,
|
||||
`fleet_replies_total{path="rendezvous"} 1`, `fleet_inbox_depth{...} 0`.
|
||||
|
||||
@@ -38,14 +38,14 @@ never toolchain ownership (§7).
|
||||
flowchart TB
|
||||
human["human (types)"]
|
||||
primary["PRIMARY (Opus)<br/>MCP client — pull-only"]
|
||||
daemon["bridged daemon<br/>127.0.0.1:8765 (single host)"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765 (single host)"]
|
||||
comp["CompositePeerLauncher<br/>routes by kind"]
|
||||
cc["ClaudeCodeLauncher"]
|
||||
oc["OpenCodeLauncher"]
|
||||
w1["worker pane (gx00 vLLM)"]
|
||||
w2["worker pane (ollama)"]
|
||||
human --> primary
|
||||
primary -->|"bridge_send / spawn / ask"| daemon
|
||||
primary -->|"fleet_send / spawn / ask"| daemon
|
||||
daemon --> comp
|
||||
comp --> cc
|
||||
comp --> oc
|
||||
@@ -80,7 +80,7 @@ flowchart TB
|
||||
m1["main A: Opus<br/>MCP client"]
|
||||
m2["main B: cloud module<br/>MCP client"]
|
||||
end
|
||||
subgraph bus["bridged fabric (CB-307/308 substrate)"]
|
||||
subgraph bus["fleetd fabric (CB-307/308 substrate)"]
|
||||
chan["per-agent inbox channels<br/>agent.<globalId>.inbox"]
|
||||
roster["federated roster (union view)"]
|
||||
end
|
||||
@@ -146,11 +146,11 @@ env-manager" rule (§7).*
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant M as main (delegator)
|
||||
participant D as bridged
|
||||
participant D as fleetd
|
||||
participant SL as SandboxLauncher
|
||||
participant SB as sandbox (peer-owned)
|
||||
participant A as agent in sandbox
|
||||
M->>D: bridge_spawn(profile=backend, role=backend)
|
||||
M->>D: fleet_spawn(profile=backend, role=backend)
|
||||
D->>SL: spawn(SpawnRequest)
|
||||
SL->>SB: start entrypoint (image = backend role)
|
||||
Note over SL,SB: bridge injects guarded ANTHROPIC_BASE_URL,<br/>mounts bridge MCP url + reply charter
|
||||
@@ -189,7 +189,7 @@ flowchart TB
|
||||
m2["main B: cloud module<br/>MCP client (pull-only)"]
|
||||
end
|
||||
reg["PrimaryRegistry → multi-slot<br/>(terminal per main)"]
|
||||
subgraph fabric["bridged"]
|
||||
subgraph fabric["fleetd"]
|
||||
ca["agent.A.inbox"]
|
||||
cb["agent.B.inbox"]
|
||||
push["ReplyPushLoop → N terminals"]
|
||||
@@ -211,14 +211,14 @@ transport is already peer-neutral — what was missing is N pull endpoints).*
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MA as main A (Opus)
|
||||
participant BR as bridged / broker
|
||||
participant BR as fleetd / broker
|
||||
participant MB as main B (cloud)
|
||||
MA->>BR: bridge_send(to = main B, msg)
|
||||
MA->>BR: fleet_send(to = main B, msg)
|
||||
BR->>BR: publish agent.B.inbox (durable, msg id)
|
||||
Note over BR: held until B pulls (B is a client too)
|
||||
MB->>BR: blocking bridge_send / poll resolves
|
||||
MB->>BR: blocking fleet_send / poll resolves
|
||||
BR-->>MB: msg (then ACK)
|
||||
MB->>BR: bridge_reply(to = main A)
|
||||
MB->>BR: fleet_reply(to = main A)
|
||||
BR->>BR: publish agent.A.inbox
|
||||
MA->>BR: poll resolves
|
||||
BR-->>MA: reply
|
||||
@@ -233,7 +233,7 @@ push-loop fan-out; relax "orchestration tools only the primary calls" to "any re
|
||||
## 6. Development C — Orchestrator tier
|
||||
|
||||
> **SUPERSEDED — do not implement this model.** The operator rejected a supervisor above the lead.
|
||||
> The human continues to drive the pre-existing lead directly; bridged neither spawns nor resumes that
|
||||
> The human continues to drive the pre-existing lead directly; fleetd neither spawns nor resumes that
|
||||
> lead. What replaced this proposal is **one lead, two short-lived advisory architects, and N workers**:
|
||||
> the lead engages architects sideways for a strong-model assessment, then discards them. Architect
|
||||
> slots are declared in `architects:` (see Gitea issue #16), rather than making leads managed sessions.
|
||||
@@ -243,7 +243,7 @@ push-loop fan-out; relax "orchestration tools only the primary calls" to "any re
|
||||
|
||||
> **Historical alternative retained.** The text and figures below record the considered model and why it
|
||||
> was rejected: it re-rooted the human-facing session above the lead, violating the still-true premise
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by bridged.
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by fleetd.
|
||||
|
||||
The orchestrator is **`SessionManager` recursed one tier up**: today it spawns/names/reaps *worker*
|
||||
sessions; the orchestrator does the same for *main* sessions, and adds **context scoping**.
|
||||
@@ -287,7 +287,7 @@ sequenceDiagram
|
||||
H->>O: high-level goal (large context)
|
||||
O->>O: name/resume main A session
|
||||
O->>MA: task + SCOPED context slice (not the whole history)
|
||||
MA->>W: bridge_send(delegation, carrying only the relevant slice)
|
||||
MA->>W: fleet_send(delegation, carrying only the relevant slice)
|
||||
W-->>MA: result
|
||||
MA-->>O: rollup
|
||||
O->>O: fold into orchestrator context, pick next main/turn
|
||||
@@ -385,7 +385,7 @@ ticket is an extension of an existing pattern (CB-402 for A, CB-308 for B/C).
|
||||
|
||||
The follow-up question — *"clarify the architecture when we have distributed agents in sandboxes"* —
|
||||
resolves the fork left open in §4 and §9. **Decision: Development A (sandbox launcher) and CB-308
|
||||
(per-host federation) *compose*, not compete — each host runs a `bridged` gateway whose launcher
|
||||
(per-host federation) *compose*, not compete — each host runs a `fleetd` gateway whose launcher
|
||||
spawns agents into that host's *local* sandboxes.** A sandbox is never reached across the network; it
|
||||
is reached by the gateway sitting next to it.
|
||||
|
||||
@@ -393,7 +393,7 @@ is reached by the gateway sitting next to it.
|
||||
|
||||
The bus delivers a turn by **herdr keystroke-injection** — `Injector → AgentControl.send` writes into
|
||||
a PTY that its **local** herdr owns. The broker moves *messages and presence*, **never keystrokes**.
|
||||
So an agent's PTY must live in a herdr that *some* `bridged` instance drives locally: a remote
|
||||
So an agent's PTY must live in a herdr that *some* `fleetd` instance drives locally: a remote
|
||||
container with no local herdr **cannot be injected into**. That rules out a central daemon reaching
|
||||
remote PTYs, and collapses the design to a single identity:
|
||||
|
||||
@@ -403,7 +403,7 @@ remote PTYs, and collapses the design to a single identity:
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A — gateway"]
|
||||
mA["main / orchestrator<br/>MCP client → LOCAL gateway"]
|
||||
gA["bridged A<br/>herdr + CompositePeerLauncher<br/>(incl. SandboxLauncher)"]
|
||||
gA["fleetd A<br/>herdr + CompositePeerLauncher<br/>(incl. SandboxLauncher)"]
|
||||
cBEa["sandbox: backend<br/>(local container)"]
|
||||
cFEa["sandbox: frontend<br/>(local container)"]
|
||||
mA --- gA
|
||||
@@ -415,7 +415,7 @@ flowchart TB
|
||||
roster["roster.* (federated presence)"]
|
||||
end
|
||||
subgraph hostB["HOST B — gateway"]
|
||||
gB["bridged B<br/>herdr + SandboxLauncher"]
|
||||
gB["fleetd B<br/>herdr + SandboxLauncher"]
|
||||
cBEb["sandbox: backend<br/>(local container)"]
|
||||
gB -->|"spawn → PTY in B's herdr"| cBEb
|
||||
end
|
||||
@@ -440,13 +440,13 @@ sequenceDiagram
|
||||
participant BR as broker
|
||||
participant GB as gateway B
|
||||
participant SB as sandbox agent (host B, container)
|
||||
MA->>GA: bridge_send(globalId on B, msg)
|
||||
MA->>GA: fleet_send(globalId on B, msg)
|
||||
GA->>GA: directory lookup - is globalId local? NO
|
||||
GA->>BR: publish agent.ID.inbox (durable)
|
||||
BR->>GB: route to the owning gateway
|
||||
GB->>SB: inject via B's LOCAL herdr (keystrokes)
|
||||
Note over GB,SB: SandboxLauncher already spawned the container -<br/>its PTY is in B's herdr, CB-306 readiness passed
|
||||
SB-->>GB: bridge_reply (to B's LOCAL MCP endpoint)
|
||||
SB-->>GB: fleet_reply (to B's LOCAL MCP endpoint)
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route back to A
|
||||
Note over GA: held until the main pulls (the main is a client)
|
||||
@@ -470,7 +470,7 @@ unchanged — the sandbox is transparent to it.*
|
||||
|
||||
| Concern | Provided by |
|
||||
|---|---|
|
||||
| Per-host gateway (owns local herdr + sessions) | **CB-308** (today's `bridged`, evolved) |
|
||||
| Per-host gateway (owns local herdr + sessions) | **CB-308** (today's `fleetd`, evolved) |
|
||||
| Spawn into a local sandbox / role→image | **Development A** `SandboxLauncher` (§4), routed by `CompositePeerLauncher` |
|
||||
| Addressing a remote sandboxed agent | **CB-308** global id + federated roster (host + role as metadata) |
|
||||
| Orphan reap after a gateway restart | **CB-117** per-gateway, summed by the composite — each reaps only its **local** herdr |
|
||||
|
||||
@@ -0,0 +1,467 @@
|
||||
# CB-591 — move the fleet onto the LLM and MCP gateway
|
||||
|
||||
**Status: DONE — the fleet is on the gateway as of 2026-08-15.** `local` runs on `/anthropic` and
|
||||
`gx` on `/v1`, both at `weight: 100`; `local-direct` stays at `weight: 0` as the escape hatch. Getting
|
||||
here took a revert and two upstream fixes — see §7.1, which is the useful part of this document. One
|
||||
risk is **accepted rather than solved**: a stream cut by any mid-response timer arrives as HTTP 200
|
||||
with no terminator, and our third-party members cannot detect it (§7.2).
|
||||
· **Upstream:** [systems/vms wiki → LLM and MCP Gateway](https://git.ltms.dev/systems/vms/wiki/LLM-and-MCP-Gateway)
|
||||
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
||||
|
||||
The gateway went live on 2026-08-15 and replaced Bifrost. This plan says what that means for a
|
||||
**member definition** in `fleetd.yaml`, because that is the part of this repo the change actually
|
||||
touches.
|
||||
|
||||
---
|
||||
|
||||
## 1. What changed upstream
|
||||
|
||||
One front door for every LLM and MCP client: `https://llm.ltms.dev`, one token per consumer.
|
||||
|
||||
| Surface | URL |
|
||||
|---|---|
|
||||
| OpenAI chat | `https://llm.ltms.dev/v1/chat/completions` |
|
||||
| OpenAI models | `https://llm.ltms.dev/v1/models` |
|
||||
| **Anthropic messages** | `https://llm.ltms.dev/anthropic/v1/messages` |
|
||||
| MCP, all servers multiplexed | `https://llm.ltms.dev/mcp` |
|
||||
|
||||
Anything outside that list returns **404 before any token is checked**, on purpose — the gateway must
|
||||
never become a blanket proxy.
|
||||
|
||||
The model backend is unchanged: GX10 vLLM at `10.10.10.26:8000` (`gx00.gw`), model name exactly
|
||||
`deepseek-v4-flash`. The direct LAN path stays open on purpose as an escape hatch.
|
||||
|
||||
---
|
||||
|
||||
## 2. Where claude-bridge sits today
|
||||
|
||||
We do **not** use the gateway. The `local` profile talks straight to the vLLM:
|
||||
|
||||
```yaml
|
||||
local:
|
||||
kind: claude-code
|
||||
baseUrl: http://gx00.gw:8000 # direct vLLM — no auth, LAN only
|
||||
model: deepseek-v4-flash
|
||||
configDir: /Users/dai.ha/.ccs/instances/gx10
|
||||
```
|
||||
|
||||
Three facts about our side that decide the shape of this work:
|
||||
|
||||
1. **`baseUrl` becomes `ANTHROPIC_BASE_URL`** in the member's environment, and `tokenEnv` becomes
|
||||
`ANTHROPIC_AUTH_TOKEN` (the value is read from a host env var and never stored in config).
|
||||
`local` sets no `tokenEnv` today, because a direct vLLM needs no token.
|
||||
2. **`SubscriptionGuard` refuses any host not on an allowlist**, and that allowlist is
|
||||
`guard.offSubscriptionHosts: [gx00.gw]`. It is built once in `Fleetd.java:93` and handed to the
|
||||
launcher, so **it is a restart-required key**, not a hot one. Changing `baseUrl` without changing
|
||||
this makes every `local` spawn throw.
|
||||
3. **The wiki names us as a blocker.** Under *Not done yet*: retiring the shared `legacy` token is
|
||||
blocked because "kb, brain, **claude-bridge** and the workstation still share it. Each needs its
|
||||
own consumer first."
|
||||
|
||||
Context7 is mounted twice today, both times straight at `https://ct7.ltms.dev/mcp` — once in
|
||||
`.mcp.json` (the primary) and once in `opencode.json` (the `sol` and `terra` members).
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph now["Today"]
|
||||
M1["local member<br/>claude-code"] -->|"ANTHROPIC_BASE_URL"| V1["vLLM gx00.gw:8000<br/>no auth, LAN only"]
|
||||
M2["sol / terra<br/>opencode"] --> CT1["ct7.ltms.dev/mcp"]
|
||||
P1["primary"] --> CT1
|
||||
end
|
||||
subgraph after["Proposed"]
|
||||
M3["local member"] -->|"ANTHROPIC_BASE_URL<br/>+ ANTHROPIC_AUTH_TOKEN"| G["llm.ltms.dev/anthropic<br/>consumer: claude-bridge"]
|
||||
G --> V2["vLLM gx00.gw:8000"]
|
||||
M4["local-direct<br/>weight 0, escape hatch"] --> V2
|
||||
end
|
||||
```
|
||||
|
||||
*The member definition is the only thing that moves. The model behind it does not.*
|
||||
|
||||
---
|
||||
|
||||
## 3. The member definition change
|
||||
|
||||
The gateway serves an Anthropic surface *and* an OpenAI surface, so **both member kinds can point at
|
||||
it**. That is the main opportunity here, and it is bigger than the `local` profile alone.
|
||||
|
||||
### 3a. `local` — claude-code, on `/anthropic`
|
||||
|
||||
| Key | Today | After | Note |
|
||||
|---|---|---|---|
|
||||
| `baseUrl` | `http://gx00.gw:8000` | `https://llm.ltms.dev/anthropic` | see the schema warning below |
|
||||
| `tokenEnv` | *(unset)* | `AI_GATEWAY_TOKEN` | new consumer token, `llmk-claude-bridge-<32 hex>` |
|
||||
| `model` | `deepseek-v4-flash` | unchanged | must stay **exact**; a regex match returns an empty `/v1/models` while completions keep working |
|
||||
| `guard.offSubscriptionHosts` | `[gx00.gw]` | `[gx00.gw, llm.ltms.dev]` | **restart required** |
|
||||
|
||||
### 3b. A new opencode profile on `/v1` — no code needed
|
||||
|
||||
`OpenCodeLauncher` already supports a pinned OpenAI-compatible endpoint (CB-508). Given `baseUrl` it
|
||||
writes a custom provider block into the worker's opencode config:
|
||||
|
||||
- `baseUrl` → `options.baseURL`. `openAiBaseUrl` appends `/v1` to a bare host, and takes a URL that
|
||||
already has a path **as-is** — so `https://llm.ltms.dev/v1` works unchanged.
|
||||
- `tokenEnv` → `options.apiKey` (falls back to a placeholder when unset, since a local vLLM ignores it).
|
||||
- `model:` **must** be `<provider>/<model>` when `baseUrl` is set — a bare name is rejected loudly
|
||||
rather than silently falling back to opencode's default gateway.
|
||||
|
||||
So the profile is pure config:
|
||||
|
||||
```yaml
|
||||
gx:
|
||||
kind: opencode
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
model: gx/deepseek-v4-flash # provider id is ours to choose; the half after / is the model
|
||||
argv: ["opencode"]
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
weight: 100 # same tier as `local` — free
|
||||
maxLoad: 2
|
||||
# deliberately NO credentialId — this is our own box, not the shared OpenAI account
|
||||
```
|
||||
|
||||
**Why this matters more than it looks.** Today every opencode member is `sol` or `terra`, and those
|
||||
are two models on **one** OpenAI account sharing `credentialId: openai-shared` — so an exhaustion on
|
||||
either locks out both, and half the fleet's opencode capacity dies at once. A gateway-backed opencode
|
||||
profile is free, is not on that credential, and therefore is not in that quarantine pair. It removes
|
||||
a single point of failure rather than just adding capacity.
|
||||
|
||||
**Note the asymmetry, it is deliberate:** `SubscriptionGuard` does not apply to opencode at all — the
|
||||
guard exists to stop a *Claude* worker borrowing the operator's subscription, and opencode reads its
|
||||
own provider credentials. So 3b needs **no allowlist change**; only 3a does.
|
||||
|
||||
**Both still need a restart, for a different reason.** `tokenEnv` is resolved by
|
||||
`HerdrPeerLauncher.resolveEnv` → `env.apply(name)`, which reads the **daemon's own process
|
||||
environment**. The running `fleetd` inherited its environment when it started, so a variable added to
|
||||
`secrets.sh` afterwards is simply not there — the launcher would inject an empty token and the
|
||||
gateway would answer 401. This is the same failure as trap 1 in `scripts/redeploy-fleetd.sh`
|
||||
(`WORKER_GITEA_TOKEN`), and it has the same fix: **restart from a login shell**, and use
|
||||
`scripts/redeploy-fleetd.sh --check` to confirm the name resolves before restarting anything.
|
||||
|
||||
### 3c. What this does to `ccs`
|
||||
|
||||
Once a profile carries `baseUrl`, `tokenEnv` and `model` itself, the ccs instance stops being what
|
||||
routes a member. Be precise about what is left, though: `configDir` still supplies **folder trust**
|
||||
and `settings.json`, and dropping it is what produced the trust dialog and the wrong-model error
|
||||
recorded in `fleetd.yaml`. So ccs goes from *deciding where the tokens go* to *holding client-side
|
||||
state*. Less load-bearing, not removable.
|
||||
|
||||
### Why `/anthropic` and never `/v1/chat/completions`
|
||||
|
||||
The gateway declares its Anthropic backend as `schema.name: Anthropic`, which means **no
|
||||
translation** — streaming, tool use and thinking blocks pass through exactly as they do against vLLM
|
||||
directly.
|
||||
|
||||
Declared as `OpenAI`, Envoy's translator looks for a `thinking_blocks` field that our vLLM does not
|
||||
send (it sends `reasoning_content`), and **every thinking delta disappears silently**. Claude Code
|
||||
speaks the Anthropic protocol, so `/anthropic` is both correct and the only safe choice.
|
||||
|
||||
This is the exact failure shape this repo keeps hitting: it compiles, it answers, it looks healthy,
|
||||
and a capability is quietly off. Treat it as a `silent-default` risk, not a config preference.
|
||||
|
||||
**Open question for 3b — ANSWERED, 2026-08-15.** The worry was that the OpenAI surface might drop
|
||||
reasoning the way the wiki documents for a mis-declared Anthropic backend. It does not. Checked at
|
||||
the API before any profile was switched:
|
||||
|
||||
| surface | request | result |
|
||||
|---|---|---|
|
||||
| `/anthropic/v1/messages` | `deepseek-v4-flash`, 64 tokens | 200, response carries a real `"type":"thinking"` block |
|
||||
| `/v1/chat/completions` | same | 200, message carries a populated `reasoning_content` (and a `reasoning` field) |
|
||||
| `/v1/models` | — | 200, exactly `["deepseek-v4-flash"]` — the exact-name trap is clear |
|
||||
| `/v1/models`, **no token** | — | **401** — Caddy is gating, as designed |
|
||||
|
||||
So reasoning survives on **both** surfaces, and the `/anthropic` choice for `local` is about protocol
|
||||
correctness rather than a repair for a known loss. The last row matters on its own: the wiki warns
|
||||
the gateway's own `SecurityPolicy` fails open, so it is worth knowing the proxy in front really does
|
||||
refuse an unauthenticated request here.
|
||||
|
||||
---
|
||||
|
||||
## 4. Decisions
|
||||
|
||||
### D1 — switch, but keep the direct path as an explicit profile · **recommended**
|
||||
|
||||
Switching buys four things we do not have:
|
||||
|
||||
- **Free opencode capacity, off the shared credential.** The largest single win. See §3b — it retires
|
||||
a real single point of failure, not just a cost line.
|
||||
- **Per-consumer usage figures.** The cockpit counts requests per consumer. That is the first real
|
||||
measurement of what the fleet consumes, and it feeds [CB-589](https://git.ltms.dev/fleet/fleetd/issues/74) Gap 2 directly.
|
||||
- **Our own revocable token.** One consumer to revoke if a worker ever leaks it, instead of a shared
|
||||
`legacy` token used by four systems.
|
||||
- **It works off-LAN.** `gx00.gw` resolves on the LAN only.
|
||||
|
||||
The cost is honest and worth stating: we add a TLS edge, an auth proxy and a gateway to the path of
|
||||
every member spawn. The wiki keeps the direct route open precisely because "if the gateway breaks,
|
||||
nothing that matters is blocked."
|
||||
|
||||
So keep it. Add a second profile `local-direct` pointing at `http://gx00.gw:8000` with **`weight: 0`**
|
||||
— never auto-selected, still spawnable with an explicit `fleet_spawn{profile: "local-direct"}`.
|
||||
That is exactly what CB-554 made `weight: 0` mean, and it turns the escape hatch into something the
|
||||
lead can actually reach during an incident.
|
||||
|
||||
### D2 — do members also mount the gateway's `/mcp`? · **OPEN, operator's call**
|
||||
|
||||
Not a detail. `CLAUDE.md` states in two places that a member mounts **only** the bridge MCP, and a
|
||||
worker's honesty rule leans on it ("never claim the result of a check you had no way to run").
|
||||
|
||||
- **Keep bridge-only.** The invariant stays true and simple. Workers stay cheap and narrow.
|
||||
- **Add the gateway MCP.** Implementers get context7 documentation lookups, which is genuinely useful
|
||||
for library work. But `mcpUrl` in `FleetConfig.Profile` is a **single `String`**, so a
|
||||
claude-code member can mount exactly one MCP — this needs a code change, not a config edit.
|
||||
|
||||
Note the invariant is **already inaccurate**: `opencode.json` gives `sol` and `terra` both context7
|
||||
and gitea. So the choice is really "make the rule true" or "make the rule match reality". Either is
|
||||
defensible; picking one is not mine to do.
|
||||
|
||||
### D3 — token scope
|
||||
|
||||
One consumer, `claude-bridge`, its token in `${SHARED_ENV}/tools/secrets.sh` as `AI_GATEWAY_TOKEN`,
|
||||
referenced by name only. Never the literal value in `fleetd.yaml` — `tokenEnv` exists for this.
|
||||
|
||||
---
|
||||
|
||||
## 5. Units of work
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
U1["U1 · consumer token<br/>issue via cockpit, add to secrets.sh"]
|
||||
U2["U2 · profile + guard<br/>fleetd.yaml, restart"]
|
||||
U3["U3 · verify live<br/>spawn, prove thinking survives"]
|
||||
U4["U4 · context7 via gateway<br/>.mcp.json + opencode.json"]
|
||||
U5["U5 · docs<br/>CLAUDE.md, wiki 11-Features"]
|
||||
U1 --> U2 --> U3
|
||||
U4 --> U5
|
||||
U3 --> U5
|
||||
```
|
||||
|
||||
| # | Scope | Who | Why |
|
||||
|---|---|---|---|
|
||||
| U1 | Issue the `claude-bridge` consumer at `auth.ltms.dev`; store as `AI_GATEWAY_TOKEN` | **operator** | touches secrets and a host we do not own |
|
||||
| U2a | New `gx` opencode profile on `/v1` — pure config, no guard change | **lead** | `fleetd.yaml` is gitignored, so a worker cannot see or edit it |
|
||||
| U2b | `local` → `/anthropic`; add `local-direct` weight 0; add `llm.ltms.dev` to the guard allowlist | **lead** | same |
|
||||
| U2c | One restart from a **login shell**, after U2a and U2b | **lead** | picks up `AI_GATEWAY_TOKEN` into the daemon env *and* the guard allowlist, in one stop |
|
||||
| U3 | Live spawn on both new profiles; confirm reasoning survives on each surface | **lead** | needs real spawns and the running daemon |
|
||||
| U4 | Point `.mcp.json` and `opencode.json` context7 at the gateway `/mcp`; rename pinned tools | delegatable | tracked files, self-contained |
|
||||
| U5 | Fix the "members mount only the bridge" claim; add a `wiki/11-Features.md` entry | delegatable | writing, clear criteria |
|
||||
|
||||
U1 blocks U2a, U2b and U3. U4 and U5 do not depend on it.
|
||||
|
||||
**Write U2a and U2b, then restart once (U2c), then verify `gx` before `local`.** Since both profiles
|
||||
need the same restart there is no reason to do two, but there is still a reason to *verify* in order:
|
||||
`gx` exercises the token and the gateway with no guard involved, so if it fails the cause is upstream.
|
||||
`local` adds the guard allowlist on top, so a failure there points at our config instead. Testing them
|
||||
in that order separates the two causes instead of confusing them.
|
||||
|
||||
> **U1 status, 2026-08-15:** the operator issued the consumer and exported it as `AI_GATEWAY_TOKEN`
|
||||
> (one key for every agent and MCP client behind `llm.ltms.dev`). Confirmed: it resolves in a login
|
||||
> shell, is 48 characters and carries the documented `llmk-` prefix. The value was never printed.
|
||||
|
||||
---
|
||||
|
||||
## 6. Traps carried over from the wiki
|
||||
|
||||
Each of these cost someone real debugging time upstream. They apply to us.
|
||||
|
||||
1. **Rotating a token restarts the auth proxy, which drops in-flight streaming responses.** For us
|
||||
that means rotating `AI_GATEWAY_TOKEN` kills every live member mid-turn, and an async ticket's
|
||||
report goes with it. This is the same rule as a daemon redeploy: **drain the fleet first**
|
||||
(`fleet_list` → `fleet_poll` anything wanted → `fleet_stop`), then rotate.
|
||||
2. **The gateway's own `SecurityPolicy` fails open.** Standalone `aigw run` accepts it and silently
|
||||
ignores it — an unauthenticated request returned **200**. Auth is the Caddy proxy in front, and
|
||||
nothing else. Never reason as if the gateway authenticates.
|
||||
3. **Exact model name.** A regex match routes fine but returns an **empty** `/v1/models` list while
|
||||
completions keep working. A wrong name returns a bare 404 that reads exactly like a dead gateway.
|
||||
4. **MCP tool names changed prefix separator.** Bifrost used one dash (`ct7-resolve-library-id`); the
|
||||
gateway uses **two underscores** (`ct7__resolve-library-id`). Relevant only if U4 is done.
|
||||
5. **`/v1/models` 404 vs empty list are different faults.** 404 means no route loaded at all; empty
|
||||
means the model match is a regex. Do not conflate them when diagnosing.
|
||||
|
||||
---
|
||||
|
||||
## 7. Verification — what would prove this works
|
||||
|
||||
Merging config is not proving it. The checks, in order:
|
||||
|
||||
1. `fleet_spawn{profile: "gx"}` succeeds and the member completes a real turn ending in
|
||||
`fleet_reply`. This is the first proof of the token, the URL and the model name, and it risks
|
||||
nothing the fleet depends on.
|
||||
2. `fleet_spawn{profile: "local"}` succeeds. If the guard allowlist was missed, this **throws** — a
|
||||
loud, self-correcting failure, which is the good kind. If the restart was missed, it also throws,
|
||||
for the same reason.
|
||||
3. A `local` member completes a turn. That exercises streaming through two TLS edges, the auth proxy
|
||||
and the gateway.
|
||||
4. **Reasoning survives, checked separately on each surface.** For `local` on `/anthropic` this is
|
||||
the check that catches the `/v1` versus `/anthropic` mistake, and it is the only one that does —
|
||||
nothing else distinguishes a working passthrough from a translator quietly dropping thinking
|
||||
deltas. For `gx` on `/v1`, this answers the open question in §3 rather than assuming it.
|
||||
5. The cockpit at `auth.ltms.dev` shows requests counted against the `claude-bridge` consumer, not
|
||||
`legacy`. That is the whole point of taking our own token.
|
||||
6. `fleet_spawn{profile: "local-direct"}` still works, so the escape hatch is real rather than
|
||||
theoretical.
|
||||
7. `fleet_list` shows `gx` carrying no `credentialId`, so a `sol`/`terra` exhaustion cannot
|
||||
quarantine it. This is the single-point-of-failure claim in §3b, checked rather than asserted.
|
||||
|
||||
---
|
||||
|
||||
## 7.1 What the live run actually found — 2026-08-15
|
||||
|
||||
U1–U2c were done, the daemon restarted onto them, and both new profiles were spawned for real. The
|
||||
migration was then **reverted**. This section is the result, so none of it has to be re-derived.
|
||||
|
||||
### The blocker
|
||||
|
||||
`llm.ltms.dev` answers **HTTP 413 Request Entity Too Large** above **32 KiB (32768 bytes)**, on both
|
||||
surfaces:
|
||||
|
||||
```
|
||||
/v1 32695 bytes -> 200 /anthropic 32095 bytes -> 200
|
||||
/v1 32795 bytes -> 413 /anthropic 32855 bytes -> 413
|
||||
```
|
||||
|
||||
32 KiB is far below one real agent turn.
|
||||
|
||||
**Root cause — confirmed by the systems/vms side, 2026-08-15.** My guess that it was a Caddy
|
||||
`request_body max_size` was **wrong**. It is Envoy, inside `aigw` on `llm.vm`. Envoy Gateway defaults
|
||||
a listener's `per_connection_buffer_limit_bytes` to **32768**, and the AI Gateway buffers the *whole*
|
||||
request body before it can route on the model name — so that default is not a network tuning knob
|
||||
here, it is a hard ceiling on prompt size. Read out of the live Envoy `config_dump`:
|
||||
|
||||
```
|
||||
listener default/llm/http per_connection_buffer_limit_bytes: 32768
|
||||
```
|
||||
|
||||
Nobody chose 32 KiB; it was inherited from the default. Both TLS edges are innocent: the same
|
||||
boundary reproduces on the LAN path and the internet path, and both 413s carry an `x-llm-consumer`
|
||||
header their auth proxy sets only *after* authenticating — so the body cleared both edges and the
|
||||
auth. Directly on `llm.vm`, `aigw` 413s at 39 KB while the vLLM backend accepts the same 39 KB and
|
||||
answers 200.
|
||||
|
||||
**Do not plan around 32 KiB.** The intended ceiling is far higher. Their fix — a `ClientTrafficPolicy`
|
||||
setting `bufferLimit: 8Mi` — is written but **not deployed** as of this note, pending their operator's
|
||||
approval. I have not re-tested and will not until they confirm, so as not to measure a half-changed
|
||||
system. Fixed in **systems/vms**, not here.
|
||||
|
||||
### The part worth remembering
|
||||
|
||||
Two members were spawned at the same moment with the same message:
|
||||
|
||||
| | `local` (claude-code, `/anthropic`) | `gx` (opencode, `/v1`) |
|
||||
|---|---|---|
|
||||
| READY → BUSY | 19:07:26 | 19:07:45 |
|
||||
| BUSY → DONE | **19:08:51 (66s)** | **never — 10+ min, ticket FAILED** |
|
||||
|
||||
**`local` passed.** It passed only because the probe was three trivial questions in a fresh session,
|
||||
so the request fit under 32 KiB. The profile looked healthy and was a landmine set to fire on the
|
||||
first turn that reads a file.
|
||||
|
||||
So §7's checklist was not wrong, it was **too easy**. Any future run of it must use a task that reads
|
||||
a real file. A liveness probe proves the token and the URL; it does not prove the path.
|
||||
|
||||
`gx` did not fail loudly either. Reproduced outside the bridge by running `opencode` by hand with the
|
||||
launcher's own generated config:
|
||||
|
||||
```
|
||||
Error: Request Entity Too Large
|
||||
...compacts context, retries...
|
||||
Error: Request Entity Too Large
|
||||
```
|
||||
|
||||
opencode **catches the 413, compacts, and retries — indefinitely**. A member that fails loudly costs
|
||||
one turn; this one costs the whole task and is indistinguishable from a slow worker.
|
||||
|
||||
> **Diagnosing a stuck opencode member.** Do not read its pane. The launcher writes its config to a
|
||||
> temp dir and passes it as `OPENCODE_CONFIG` — find it with
|
||||
> `ls -dt /var/folders/*/*/T/fleetd-opencode-* | head -1`, check the provider block and the key's
|
||||
> length and prefix (never its value), then reproduce with `opencode run --auto -m <provider>/<model>`
|
||||
> using the same `OPENCODE_CONFIG`. That is what turned "it hangs" into a one-line error.
|
||||
|
||||
### What checked out, and needs no re-testing
|
||||
|
||||
- Token accepted on both surfaces. **Unauthenticated → 401**, so the Caddy proxy really does gate —
|
||||
the wiki's "SecurityPolicy fails open" warning is about the gateway itself, not the edge.
|
||||
- `/v1/models` returns exactly `["deepseek-v4-flash"]`, so trap 3 is clear.
|
||||
- **Reasoning survives both surfaces** — see §3b above.
|
||||
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
||||
key rather than the `fleetd-local-noauth` placeholder.
|
||||
- `SubscriptionGuard` accepted `llm.ltms.dev` after the allowlist edit and the restart: `local`
|
||||
spawned without throwing, which is the check that catches a missed restart.
|
||||
|
||||
### Resolution — both ceilings fixed, migration completed
|
||||
|
||||
systems/vms fixed both, and each was re-checked from this side rather than taken on trust:
|
||||
|
||||
| ceiling | was | now | our own check |
|
||||
|---|---|---|---|
|
||||
| listener buffer | 32 KiB | 32 Mi | 1.2 MB body → **200** (was 413) |
|
||||
| LLM route timeout | 60s | 86400s | the request that truncated: **101s, `message_stop` present, 4000/4000** |
|
||||
|
||||
The timeout moved in two steps on 2026-08-15: 60s → 1800s, then 1800s → **86400s (24 hours)** after
|
||||
the truncation risk below was discussed. They tried `request: 0s` first, which removes the
|
||||
total-duration timer completely. It works, but on an `AIGatewayRoute` the **idle timeout is derived
|
||||
from the request timeout**, so `0s` also removed any bound on a stalled connection. 86400s keeps a
|
||||
reaper for dead connections while putting the truncation timer out of practical reach.
|
||||
|
||||
Neither was deliberate. The 32 KiB was Envoy Gateway's default `per_connection_buffer_limit_bytes`;
|
||||
the 60s was Envoy AI Gateway's own documented default. The 60s bounded **generation** as well as
|
||||
prompt size — a tiny prompt with a long answer returned 504 at 60.05s.
|
||||
|
||||
Two configuration facts worth keeping, from their bisection:
|
||||
|
||||
- **`ClientTrafficPolicy` is honoured in standalone `aigw run`; `BackendTrafficPolicy` is NOT.** A
|
||||
`BackendTrafficPolicy` setting `requestTimeout` is accepted, logs nothing, and leaves the routes
|
||||
unchanged (upstream `envoyproxy/gateway#9513`). What works is `timeouts: {request: …}` on each
|
||||
`AIGatewayRoute` rule. Nothing from the outside distinguishes the two — the same silent-default
|
||||
shape as their `SecurityPolicy` caveat.
|
||||
- In that stack, "the config was accepted" proves nothing. Read the live `config_dump`.
|
||||
|
||||
## 7.2 The risk we accepted, and why we could not remove it
|
||||
|
||||
Raising the timeout made the failure **rare, not impossible**, and the residual failure is silent.
|
||||
|
||||
On a mid-response timeout over chunked HTTP/1.1, Envoy ends the chunked encoding *cleanly* instead of
|
||||
resetting the connection, so the client receives what looks like a complete transfer
|
||||
(`envoyproxy/envoy#17186` — acknowledged as a bug in 2021, closed by a stale bot, never fixed). The
|
||||
December 2025 fix `envoyproxy/envoy#42269` changes locally-originated resets from `NO_ERROR` to
|
||||
`INTERNAL_ERROR`, but it is **HTTP/2 only** and SSE clients here speak HTTP/1.1.
|
||||
|
||||
Measured on our side while the timeout was still 60s:
|
||||
|
||||
```
|
||||
HTTP 200 61.07s 141992 bytes
|
||||
message_stop 0 message_delta 0 error events 0
|
||||
emitted 2473 of 4000, ending on a WELL-FORMED SSE frame
|
||||
```
|
||||
|
||||
A syntactically valid stream that simply stops. Any timer firing mid-stream — route timeout, idle
|
||||
timeout, `max_stream_duration` — fails this same way.
|
||||
|
||||
**The recommended defence does not transfer to us.** The right fix is to treat a stream with no
|
||||
`message_stop` / `[DONE]` / `finish_reason` as failed. We cannot: our members are Claude Code and
|
||||
opencode, third-party clients whose SSE parsing we do not own, and there is no seam to insert the
|
||||
check. Whether either detects a missing terminator is unverified — and opencode's handling of the 413
|
||||
(swallow, compact, retry forever, never surface an error) does not suggest it is strict.
|
||||
|
||||
So the honest statement of our position:
|
||||
|
||||
> Gateway traffic is acceptable at 86400s because a single request would have to run for 24 hours to
|
||||
> trip the bug — **not** because we could detect it if it did.
|
||||
|
||||
At 86400s our **own** limit binds first, which is the ordering we want. `MessageService.ASYNC_TIMEOUT_MS`
|
||||
caps a turn at 30 minutes, so a runaway request ends as a clean `FAILED` ticket that we raised, rather
|
||||
than as a silently truncated `200` that we cannot see. While the gateway sat at 1800s the two numbers
|
||||
were equal and did not nest, so a gateway-side stall could have been misread as a bug in our own ticket
|
||||
handling. That ambiguity is now gone.
|
||||
|
||||
**If a member ever returns a confident but truncated answer, suspect this before anything in our own
|
||||
code.** That is the whole reason this section exists.
|
||||
|
||||
---
|
||||
|
||||
## 8. Related
|
||||
|
||||
- [CB-589 / #74](https://git.ltms.dev/fleet/fleetd/issues/74) — cost-first placement and a
|
||||
gateway that reports live capacity. The per-consumer figures this migration unlocks are the first
|
||||
input that ticket actually needs.
|
||||
- `docs/CB-500-Multi-Tier-Coordination.md` §11 — the distributed-sandbox topology this gateway is
|
||||
part of.
|
||||
+18
-18
@@ -12,12 +12,12 @@ not after it.
|
||||
|
||||
## 1. Why this stage is not optional bookkeeping
|
||||
|
||||
`bridged` today has **exactly one security control: the loopback bind**. Every other guarantee
|
||||
`fleetd` today has **exactly one security control: the loopback bind**. Every other guarantee
|
||||
rests on it.
|
||||
|
||||
The identity model (`mcp/ConnectionIdentity.java`) resolves a caller from the connection alone —
|
||||
the OS reports the connecting PID, herdr owns the PID→pane map, so a worker cannot forge another
|
||||
worker. Its own javadoc is explicit: *"Single-host only (the herd shares the `bridged` host); the
|
||||
worker. Its own javadoc is explicit: *"Single-host only (the herd shares the `fleetd` host); the
|
||||
token path is the split-host fallback."* The token path does not exist yet.
|
||||
|
||||
That leaves a seam that is **latent today and load-bearing the moment the bind moves**:
|
||||
@@ -67,7 +67,7 @@ it will be disabled and the stage is wasted. So:
|
||||
auth:
|
||||
mode: loopback-trust # default — behaves exactly like today: loopback ⇒ PRIMARY, no token needed
|
||||
# mode: token # every non-worker caller must present a valid bearer token
|
||||
# tokenEnv: BRIDGED_API_TOKEN # host env var holding the token; never the literal value
|
||||
# tokenEnv: FLEETD_API_TOKEN # host env var holding the token; never the literal value
|
||||
```
|
||||
|
||||
`mode: loopback-trust` is the current behaviour, named honestly and now *chosen* rather than
|
||||
@@ -112,9 +112,9 @@ The roadmap says "systemd unit". **This host is macOS — there is no systemd on
|
||||
not found), and the daemon that has been dogfooded for weeks runs as a bare foreground
|
||||
`java -jar`. Ship **both**:
|
||||
|
||||
- `deploy/dev.ltms.bridged.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
- `deploy/dev.ltms.fleet.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
ordered start after herdr.
|
||||
- `deploy/bridged.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
- `deploy/fleetd.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
|
||||
Ordering after herdr is advisory in both: the herdr socket may not exist at boot, so the daemon
|
||||
must **retry the socket rather than exit** — supervision ordering is a nicety, socket-retry is the
|
||||
@@ -158,10 +158,10 @@ configured. It already leaks nothing but herdr's version and up/down.
|
||||
### 3.1 There are TWO entry paths, and only one of them has identity today
|
||||
|
||||
The wiki describes MCP as "a thin adapter over the REST core". **At the code level that is not
|
||||
literally true, and the difference is security-relevant.** `BridgeMcp` calls `MessageService` /
|
||||
literally true, and the difference is security-relevant.** `FleetMcp` calls `MessageService` /
|
||||
`SessionManager` *directly*; it never issues an HTTP request against a Javalin route. And `/mcp` is
|
||||
mounted as a raw servlet on Jetty's `ServletContextHandler`
|
||||
(`BridgedApp.build → cfg.jetty.modifyServletContextHandler`), so it does **not** pass through
|
||||
(`FleetApp.build → cfg.jetty.modifyServletContextHandler`), so it does **not** pass through
|
||||
Javalin's `before` filters at all.
|
||||
|
||||
The current split is the mirror image of what you'd expect:
|
||||
@@ -172,7 +172,7 @@ The current split is the mirror image of what you'd expect:
|
||||
| REST routes | ❌ **none at all** — the session id is taken from the URL path and trusted | ❌ none |
|
||||
|
||||
So REST is the *more* exposed surface: `POST /sessions/{id}/reply` accepts any `{id}` from the
|
||||
path, whereas the MCP `bridge_reply` derives the worker from the connection and refuses to read it
|
||||
path, whereas the MCP `fleet_reply` derives the worker from the connection and refuses to read it
|
||||
from an argument. Loopback-only bind is what makes this safe today.
|
||||
|
||||
**Therefore CB-505 must enforce on both paths against one shared resolver** — not at a single
|
||||
@@ -188,15 +188,15 @@ Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
|
||||
| Metric | Type | Why it exists |
|
||||
|---|---|---|
|
||||
| `bridged_sends_total{outcome}` | counter | outcome ∈ replied\|completion_fallback\|timeout\|failed — the completion-fallback rate is the health signal for turn detection (CB-115/116/118) |
|
||||
| `bridged_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `bridged_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `bridged_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `bridged_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `bridged_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `bridged_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `bridged_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
| `bridged_auth_failures_total{reason}` | counter | only meaningful once CB-501 lands; catches misconfigured workers |
|
||||
| `fleet_sends_total{outcome}` | counter | outcome ∈ replied\|completion_fallback\|timeout\|failed — the completion-fallback rate is the health signal for turn detection (CB-115/116/118) |
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
| `fleet_auth_failures_total{reason}` | counter | only meaningful once CB-501 lands; catches misconfigured workers |
|
||||
|
||||
---
|
||||
|
||||
@@ -226,7 +226,7 @@ was answered from the forge itself. Kept here as the decision record.*
|
||||
|
||||
1. ✅ **TLS scope (D3) — confirmed.** Bearer auth + the fail-fast bind guard ship in the daemon;
|
||||
TLS terminates at a reverse proxy, documented with a worked example. AMQP gets TLS via an
|
||||
`amqps://` URI. No keystore handling in `bridged`.
|
||||
`amqps://` URI. No keystore handling in `fleetd`.
|
||||
2. ✅ **Micrometer (D4) — confirmed dropped.** Zero-dependency Prometheus text renderer, for the
|
||||
reasons in D4 (pom reconciliation burden + the mandated CVE gate being un-runnable this
|
||||
session). Revisit if a push-gateway or JVM-metrics requirement appears; the endpoint is the
|
||||
|
||||
@@ -0,0 +1,972 @@
|
||||
# M4 - Fleet health, recovery, routing, and capacity
|
||||
|
||||
**Status:** Design accepted on 2026-08-15. CB-573 part 1 has shipped the classification model and
|
||||
the `fleet_list` capacity view; the remaining M4 units are not yet shipped. See
|
||||
[Unit 2 — what has landed so far](#unit-2---what-has-landed-so-far) before planning Unit 2 work:
|
||||
some of its criteria were met by separate CB tickets, and one of them contradicts the unit text.
|
||||
**Scope:** Fleet evidence, safe mechanical repair, lead routing, capacity reporting, and optional
|
||||
human notification.
|
||||
**Grounded in:** `health/FleetHealth`, `health/PaneBudget`, `inject/StatusPoller`,
|
||||
`inject/StatusRefiner`, `inject/CompletionResolver`, `inject/Injector`, `session/SessionManager`,
|
||||
`msg/MessageService`, `msg/ReplyInbox`, `msg/ReplyPushLoop`, `msg/LeadHeartbeatLoop`,
|
||||
`mcp/PrimaryRegistry`, and `herdr/AgentControl`.
|
||||
|
||||
## 1. Problem and decision boundary
|
||||
|
||||
The operator asked the bridge to detect idle agents, exceptions, stopped work, and broken
|
||||
communication. The bridge may read an agent pane from time to time. It must notify a person when
|
||||
the fleet cannot move forward.
|
||||
|
||||
The four operator terms are not four equal health states. `IDLE` is a normal mode. An exception is
|
||||
sometimes visible only as pane text. Stopped work may look the same as slow work. Broken
|
||||
communication can occur on several links.
|
||||
|
||||
M4 uses this boundary:
|
||||
|
||||
- The bridge detects facts and joins evidence.
|
||||
- The bridge repairs only mechanical failures with no judgement.
|
||||
- The lead decides whether to stop, retry, replace, or reassign a member.
|
||||
- A human is notified only when no healthy lead can act.
|
||||
- n8n may route an outbound incident. It never classifies state or chooses recovery.
|
||||
|
||||
An inbound n8n decider would need bridge authority. No narrow machine-decider role exists. Giving a
|
||||
workflow engine lead authority is unsafe, while adding a new role is a separate authorization
|
||||
design. An outbound sink needs no bridge role.
|
||||
|
||||
The bridge must never replay a delivered task. That task may already have changed files, pushed a
|
||||
branch, opened a pull request, or changed external state. A replay can run those side effects twice.
|
||||
This rule must remain true even if later code stores delivered prompt text.
|
||||
|
||||
## 2. Evidence model
|
||||
|
||||
A health state is mainly a comparison between two views:
|
||||
|
||||
- **herdr view:** current agents and raw live status from one `AgentControl.list()` call.
|
||||
- **bridge view:** session FSM, MCP presence, accepted turns, tasks, inbox state, and lead ownership.
|
||||
|
||||
A strong fault often appears as a disagreement between those views. For example, `BUSY` in the
|
||||
session FSM and `DONE` in herdr means the bridge missed a turn boundary. Pane reads support this
|
||||
model, but they are not the main monitor.
|
||||
|
||||
`SessionManager.rosterView` already joins session state and live status. `AgentControl.list()`
|
||||
already gets the whole live fleet in one call. M4 makes that join persistent and adds timers,
|
||||
accepted-turn state, and incident state.
|
||||
|
||||
### 2.1 Real traces behind the design
|
||||
|
||||
The first trace was an architect that stopped making progress:
|
||||
|
||||
```text
|
||||
profile=opus role=architect state=busy liveStatus=done
|
||||
```
|
||||
|
||||
The session moved from `DONE` to `BUSY` for turn 2. Eighteen minutes later, the session still said
|
||||
`BUSY`, herdr still said `DONE`, the async task still said `PENDING`, and no completion fallback had
|
||||
run. This is `TURN_BOUNDARY_LOST`, not a general slow-turn guess.
|
||||
|
||||
The second trace had two async sends to the same pane, one second apart. The pane was then stopped.
|
||||
One ticket became failed. The other stayed `pending - worker unknown`. Current
|
||||
`MessageService.abandon` resolves only `Rendezvous.currentWaiter(target)`, while async tasks live in
|
||||
a separate ticket map. CB-568 is intended to fix that bug. M4 still keeps an independent
|
||||
post-teardown invariant so a later regression becomes `DELEGATION_ORPHANED`.
|
||||
|
||||
### 2.2 Corrections made during design
|
||||
|
||||
The first state table missed `BUSY` in fleetd plus `IDLE` or `DONE` in herdr. It would have found
|
||||
the real trace only through a late, weak stall timer. The final model adds
|
||||
`TURN_BOUNDARY_LOST` as a strong disagreement state.
|
||||
|
||||
The first notification design also required a webhook before `health.enabled` could turn on. That
|
||||
removed useful local detection to avoid a narrower human-notification gap. The final design splits
|
||||
detection from notification. Missing human escalation is shown as partial coverage instead of
|
||||
disabling health.
|
||||
|
||||
## 3. Classification precedence
|
||||
|
||||
Evidence is applied in this order. A lower rule cannot hide a higher one.
|
||||
|
||||
1. **Control link:** failed fleet list plus failed ping becomes `CONTROL_LINK_DOWN`.
|
||||
2. **Definitive target loss:** `_not_found` becomes `GONE` or `LEAD_UNREACHABLE` when the control
|
||||
link is healthy.
|
||||
3. **Startup and teardown invariants:** readiness expiry becomes `NEVER_READY`; surviving tasks
|
||||
after teardown become `DELEGATION_ORPHANED`.
|
||||
4. **Bridge/live disagreement:** `BUSY` plus stable raw `IDLE` or `DONE` becomes
|
||||
`TURN_BOUNDARY_LOST`.
|
||||
5. **Known screen evidence:** a tested fatal signature becomes `ERROR_ON_SCREEN`.
|
||||
6. **Timed suspicion:** unchanged sparse pane probes may become `STALL_SUSPECTED`.
|
||||
7. **Communication quality:** completion fallback becomes `MUTE`; an old inbox entry becomes
|
||||
`REPLY_STRANDED`.
|
||||
8. **Normal mode:** `STARTING`, `IDLE`, `WORKING`, `WORK_PENDING`, or `BLOCKED_AMBIGUOUS`.
|
||||
|
||||
The member flow in Figure 1 shows lifecycle states and the main fault exits. Fault states are
|
||||
reported beside the session FSM; most are not new FSM values.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Registered["Member registered"] --> Starting["STARTING"]
|
||||
Starting -->|"MCP presence"| Idle["IDLE"]
|
||||
Starting -->|"Readiness grace expires"| NeverReady["NEVER_READY"]
|
||||
Idle -->|"Accepted delivery"| Working["WORKING"]
|
||||
Working -->|"Trusted turn boundary"| Idle
|
||||
Working -->|"Bridge BUSY and herdr IDLE or DONE"| Lost["TURN_BOUNDARY_LOST"]
|
||||
Working -->|"Known fatal screen"| Error["ERROR_ON_SCREEN"]
|
||||
Working -->|"Long age and unchanged sparse probes"| Stall["STALL_SUSPECTED"]
|
||||
Working -->|"Target not found"| Gone["GONE"]
|
||||
Idle -->|"Inbox or queued delivery exists"| Pending["WORK_PENDING"]
|
||||
Pending -->|"Delivery or collection finishes"| Idle
|
||||
Idle -->|"Raw BLOCKED with an open turn"| Blocked["BLOCKED_AMBIGUOUS"]
|
||||
Lost -->|"Strict guarded repair"| Repaired["DONE with reconciled completion"]
|
||||
Lost -->|"Repair refused"| LeadDecision["Lead decision required"]
|
||||
```
|
||||
|
||||
*Figure 1. The member lifecycle and the main health exits. Pane-based states never authorise an
|
||||
automatic retry of the task.*
|
||||
|
||||
## 4. State model
|
||||
|
||||
### 4.1 Normal and transitional member states
|
||||
|
||||
| State | Exact evidence | Meaning and certainty |
|
||||
|---|---|---|
|
||||
| `STARTING` | Session is `SPAWNING`; MCP presence is absent | Normal inside the startup grace. MCP contact is the readiness signal. |
|
||||
| `IDLE` | Session is `READY` or `DONE`; live status is `IDLE` or `DONE`; no open turn or inbox item exists | Normal. Idle is not a fault. |
|
||||
| `WORKING` | Session is `BUSY`; raw live status is `WORKING`; the accepted turn is open | Certain that herdr sees work. It does not prove useful progress. |
|
||||
| `WORK_PENDING` | Queued delivery or inbox content exists while the target is injectable | Transitional. Existing injector or push logic should move it. |
|
||||
| `BLOCKED_AMBIGUOUS` | An open turn exists and raw live status is `BLOCKED` | The bridge cannot tell whether this is permission, input, or a settled screen. |
|
||||
|
||||
Idle may drive configured resource cleanup. It never opens an incident and never pages a person.
|
||||
|
||||
### 4.2 Member fault and quality states
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `NEVER_READY` | `SPAWNING`, no MCP presence, and an accepted delivery waits through the existing readiness grace | Delivery never became possible. The exact cause is unknown. Fail the send, stop the process, and preserve a provisioned worktree. |
|
||||
| `GONE` | Per-target herdr call returns `_not_found` while fleet list or ping works | Certain target loss. Fail all target work. Do not replay it. |
|
||||
| `TURN_BOUNDARY_LOST` | Same session turn stays `BUSY`; same accepted task stays open; two raw snapshots show `IDLE` or `DONE` | Strong disagreement. Strict reconciliation may repair it. |
|
||||
| `ERROR_ON_SCREEN` | Suspicious non-working state survives grace; `detection` matches a tested adapter-specific fatal signature | Certain only for the matched signature. A bare word such as `Exception` is not enough. |
|
||||
| `STALL_SUSPECTED` | Open turn is older than the configured threshold; two normalised `recent_unwrapped` digests are unchanged; no boundary or reply occurs | Not certain. A long valid API call can look the same. Lead decides. |
|
||||
| `MUTE` | Turn resolves through completion fallback instead of `fleet_reply` | Certain that no structured reply won. It does not prove an MCP failure. A single event is a metric, not an incident. |
|
||||
| `REPLY_STRANDED` | Typed reply or health message remains after owning-lead push reaches its cap | Collection failed. This does not explain whether the lead is busy, dead, or ignoring the nudge. |
|
||||
| `DELEGATION_ORPHANED` | Target is gone, failed, or released, but one or more tasks remain `PENDING` after reconciliation grace | Certain bridge invariant failure. This is not an inbox-drain fault. |
|
||||
| `WORK_PRODUCT_AT_RISK` | Provisioned branch has commits after its recorded base; member is `DONE`, `FAILED`, or preserved after release; no turn or inbox item remains; long-idle threshold passed | A warning, not proof of loss. Work may already have an open pull request or a squash merge. |
|
||||
|
||||
`MUTE` opens an incident only after a small fixed rate threshold for one target or profile, or when
|
||||
it appears with another fault.
|
||||
|
||||
`WORK_PRODUCT_AT_RISK` must not become `WORK_PRODUCT_UNCOLLECTED`. The bridge does not know pull
|
||||
request or merge state. If committed work appears with `REPLY_STRANDED` or
|
||||
`DELEGATION_ORPHANED`, the existing incident gains `committedWorkAtRisk: true`.
|
||||
|
||||
### 4.3 Control-link state
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `CONTROL_LINK_DOWN` | Two full-fleet `agent.list` calls fail across the grace, and herdr `ping` also fails | Certain for the fleetd-to-herdr link. Retry calls, record the incident, and use human escalation if no lead can be reached. |
|
||||
|
||||
A failed fleet list alone is not a dead-member claim. A single `_not_found` with a healthy global
|
||||
link is a target fault, not a control-link fault.
|
||||
|
||||
### 4.4 Lead states
|
||||
|
||||
| State | Exact evidence | Meaning and action |
|
||||
|---|---|---|
|
||||
| `LEAD_IDLE` | Expected lead is present with raw injectable status; no actionable state waits | Normal. Existing heartbeat may run under its own policy. |
|
||||
| `LEAD_WORKING` | Expected lead is present with raw `WORKING`; stall threshold is not met | Reachable and busy. Never inject into the live turn. |
|
||||
| `LEAD_STATUS_UNKNOWN` | Expected lead is present with raw `UNKNOWN` | Neither dead nor a healthy routing target. Retain evidence and retry. |
|
||||
| `LEAD_UNREACHABLE` | Expected lead is absent from two successful live-agent snapshots while ping works, or targeted lookup returns `_not_found` with a healthy control link | Route to a healthy peer. If none exists, use human escalation. |
|
||||
| `LEAD_UNRESPONSIVE` | Actionable state waits; lead stays injectable; bounded nudges exhaust; inbox remains uncollected | Route to a healthy peer or a person. |
|
||||
| `LEAD_STALL_SUSPECTED` | Lead stays `WORKING` past threshold; two sparse pane probes show no progress | Not certain. Never kill or restart automatically. Route to peer or person. |
|
||||
|
||||
The monitor retains the lead name and terminal, last successful sighting, raw status and age,
|
||||
consecutive list absences, targeted errors, pane-probe facts, pending incident age, and nudge
|
||||
outcomes. Current heartbeat and push loops discard much of this history.
|
||||
|
||||
Expected lead identity comes from the same supplier used by `CallerResolver`. It is not liveness
|
||||
evidence. `LeadTabScanner` keeps cached identity after a failed scan, so the health monitor compares
|
||||
that identity with a fresh successful agent list. A dynamic identity also survives a two-successful-
|
||||
snapshot retirement grace. This stops a dead lead from escaping health by disappearing from one map.
|
||||
|
||||
### 4.5 Evidence limits
|
||||
|
||||
M4 cannot tell these cases apart with current evidence:
|
||||
|
||||
- A valid long call and a hung call may have the same status and pane digest.
|
||||
- `BLOCKED` does not explain which input is needed.
|
||||
- An idle prompt after failure may look like an idle prompt after success.
|
||||
- A missing structured reply does not prove a broken MCP connection.
|
||||
- An undrained inbox does not explain why the lead did not collect it.
|
||||
- Arbitrary pane text cannot safely classify arbitrary exceptions.
|
||||
- A branch ahead of its base does not prove that work was not collected.
|
||||
|
||||
Logs are outputs, not classifier inputs. The monitor never parses its own logs.
|
||||
|
||||
## 5. Automatic action and lead action
|
||||
|
||||
### 5.1 Actions the bridge may take
|
||||
|
||||
The bridge may:
|
||||
|
||||
- retry transient herdr status, list, ping, and pane-read failures with bounded backoff;
|
||||
- re-submit Enter after the existing paste/submit race;
|
||||
- fail queued delivery after `NEVER_READY`;
|
||||
- stop a never-ready process while preserving its provisioned worktree;
|
||||
- fail all queued, accepted, and async tasks for a gone or released target;
|
||||
- reconcile one lost boundary when every strict gate in Section 8 passes;
|
||||
- hold typed messages, nudge the owning lead, and stop at the configured cap;
|
||||
- use the existing bounded idle-lead heartbeat;
|
||||
- deduplicate, route, update, and resolve incidents.
|
||||
|
||||
These actions do not choose new work and do not replay old work.
|
||||
|
||||
### 5.2 Decisions reserved for the lead
|
||||
|
||||
Only the lead may:
|
||||
|
||||
- stop or continue `BLOCKED_AMBIGUOUS`;
|
||||
- stop, inspect, or wait on `ERROR_ON_SCREEN`;
|
||||
- kill or continue `STALL_SUSPECTED`;
|
||||
- spawn a replacement or reassign work;
|
||||
- retry a delivered task;
|
||||
- choose how to use partial work in a worktree;
|
||||
- restart herdr or change network, model, credentials, backend, or configuration.
|
||||
|
||||
Reports include literal safe tool calls such as `fleet_status(sessionId="...")`,
|
||||
`fleet_poll(ticket="...")`, `fleet_list()`, and optional `fleet_stop(paneId="...")`. A judgement
|
||||
state never presents stop as the only action.
|
||||
|
||||
### 5.3 Release causes and worktree safety
|
||||
|
||||
| Release cause | Process action | Provisioned worktree |
|
||||
|---|---|---|
|
||||
| `SPAWN_ROLLBACK` before registration or delivery | Stop and clean up | Remove |
|
||||
| `COMPLETED` for `READY` or `DONE` without pending work, idle TTL, or successful context-cap completion | Stop | Remove only if clean; preserve a dirty worktree (CB-576) |
|
||||
| `NEVER_READY` | Stop | Preserve |
|
||||
| `GONE` | Best-effort stop | Preserve |
|
||||
| `TURN_FAILED` or lead abort while `BUSY` or `FAILED` | Stop | Preserve |
|
||||
| `RELEASE_WITH_PENDING_TASKS` | Stop | Preserve |
|
||||
| `SHUTDOWN` | Stop | Preserve |
|
||||
|
||||
Explicit stop is state-aware. `SPAWNING`, `BUSY`, `FAILED`, or any target with pending tasks uses a
|
||||
preserving cause.
|
||||
|
||||
Before abnormal release removes the live session, M4 writes an atomic manifest under the worktree
|
||||
root. It records session identity, owner, role, profile, repository, path, branch, base commit,
|
||||
release cause, release time, state, and pending task ids. `fleet_list.preservedWorktrees` loads these
|
||||
manifests after restart. Stop output and WARN logs also name the path and cause. M4 never
|
||||
auto-deletes a preserved worktree.
|
||||
|
||||
## 6. Fleet health monitor
|
||||
|
||||
Add `FleetHealthMonitor`. Do not widen `StatusPoller` into a policy loop.
|
||||
|
||||
`StatusPoller` has a 250 ms delivery cadence and samples only injector targets with outstanding
|
||||
work. Health needs all sessions, all leads, task state, inbox age, and global control evidence. One
|
||||
loop cannot serve both cadences safely.
|
||||
|
||||
Build the monitor like `LeadHeartbeatLoop`:
|
||||
|
||||
- pure `decide(snapshot, priorState, now)` logic;
|
||||
- a thin scheduler;
|
||||
- an injected clock;
|
||||
- edge-triggered state changes;
|
||||
- no network work in the pure function;
|
||||
- no sleeping in tests.
|
||||
|
||||
Each enabled fleet tick reads:
|
||||
|
||||
- one `AgentControl.list()` result for the whole fleet;
|
||||
- one in-memory `SessionManager.roster()` snapshot;
|
||||
- accepted turns and async task state;
|
||||
- typed inbox depth, kind, and age;
|
||||
- push and heartbeat outcomes;
|
||||
- configured and discovered leads.
|
||||
|
||||
Existing failure paths publish structured evidence to the monitor. The monitor does not infer events
|
||||
from log text.
|
||||
|
||||
### 6.1 Pane budget
|
||||
|
||||
Healthy idle members, recent working members, and quiet leads cause no pane reads.
|
||||
|
||||
A pane is eligible only for a stable lost boundary, sustained `BLOCKED` or `UNKNOWN`, work older
|
||||
than the suspect threshold, or one final evidence read for a confirmed fault when the pane exists.
|
||||
|
||||
Compiled brakes apply even if config asks for more:
|
||||
|
||||
- per-target pane cooldown is at least 60 seconds;
|
||||
- working age before the first progress probe is at least 300 seconds;
|
||||
- at most two pane reads occur in one fleet tick;
|
||||
- targets rotate fairly;
|
||||
- only a normalised digest and optional clipped local excerpt are stored;
|
||||
- no pane excerpt leaves fleetd in a human webhook.
|
||||
|
||||
Use `detection` for tested screen signatures. Use normalised `recent_unwrapped` only for progress
|
||||
comparison.
|
||||
|
||||
## 7. Typed inbox and routing
|
||||
|
||||
### 7.1 Semantic record
|
||||
|
||||
The typed inbox record carries:
|
||||
|
||||
```text
|
||||
schemaVersion
|
||||
kind: reply | health
|
||||
msgId, target, subjectTerminal, recipientLead
|
||||
severity, state, evidence
|
||||
createdAtEpochMillis, firstSeenEpochMillis, lastSeenEpochMillis
|
||||
recoveryTried, suggestedToolCalls, content
|
||||
```
|
||||
|
||||
A health message never calls `Rendezvous.resolve`. It cannot look like the member's task result.
|
||||
|
||||
Both inbox adapters share field preservation, first-id-wins dedup, FIFO among decoded messages,
|
||||
explicit ownership, ack, and release rules. The in-memory adapter stores typed records directly. It
|
||||
does not copy AMQP migration logic.
|
||||
|
||||
### 7.2 AMQP migration
|
||||
|
||||
The reader uses AMQP `content_type`, never body sniffing:
|
||||
|
||||
```text
|
||||
Legacy v0: text/plain
|
||||
Typed family: application/vnd.ltms.fleet.inbox-message+json
|
||||
```
|
||||
|
||||
A legacy reply may begin with `{`. It remains plain text because its media type is `text/plain`.
|
||||
Legacy text becomes `kind=reply` with exact UTF-8 content and absent typed metadata.
|
||||
|
||||
Typed JSON has required integer `schemaVersion: 1`. Version 1 ignores unknown optional fields.
|
||||
Missing required fields, invalid enums, malformed UTF-8 or JSON, and property/body identity mismatch
|
||||
are invalid data.
|
||||
|
||||
An unknown schema version is not partly decoded. It remains unacknowledged on the original queue and
|
||||
creates one operator-visible `unsupported_version` failure. A newer daemon may read it later.
|
||||
|
||||
Invalid known-format data is copied byte-for-byte to durable queue
|
||||
`agent.<target>.inbox.quarantine`. A dedicated confirm-mode publisher confirms the persistent copy
|
||||
before the original is acknowledged. A failed quarantine handoff leaves the original unacknowledged.
|
||||
The raw body never enters logs.
|
||||
|
||||
Decode failure creates a redacted WARN, metric, `fleet_list` summary, and routed health incident.
|
||||
One bad entry never escapes the consumer callback and never stops later valid messages.
|
||||
|
||||
Safe downgrade is not supported. The previous build ignores `content_type` and would show typed JSON
|
||||
as ordinary reply text. If drained, it would acknowledge the message and lose typed meaning. Typed
|
||||
queues must be drained or preserved before an old jar runs.
|
||||
|
||||
The existing contract suite uses RabbitMQ. Production uses LavinMQ. The migration and lead-key
|
||||
ownership cases must run once against production LavinMQ before release, or the release must state
|
||||
that LavinMQ was not checked.
|
||||
|
||||
### 7.3 Member routing
|
||||
|
||||
A member incident first goes to the exact lead that owns its accepted delegation.
|
||||
`PrimaryRegistry` needs a no-fallback `delegatingLeadFor(memberTarget)` query. Health routing must not
|
||||
use the old singular-primary fallback when several leads exist.
|
||||
|
||||
Publish the incident under the affected member target. Trigger the existing bounded push route. The
|
||||
push waits until the owning lead is injectable, so it does not interrupt a live lead turn.
|
||||
|
||||
### 7.4 Peer lead routing
|
||||
|
||||
Figure 2 shows the route from incident to lead, peer, or person.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Incident["Open incident"] --> Member{"Member incident?"}
|
||||
Member -->|"yes"| Known{"Exact delegation owner known?"}
|
||||
Known -->|"no"| Sink{"Human webhook enabled and healthy?"}
|
||||
Known -->|"yes"| Owner{"Owner lead healthy?"}
|
||||
Owner -->|"yes"| OwnerInbox["Publish to owner lead path"]
|
||||
Owner -->|"no"| PeerSet["Build healthy peer candidate set"]
|
||||
Member -->|"no, lead incident"| PeerSet
|
||||
PeerSet --> Peer{"Healthy peer exists?"}
|
||||
Peer -->|"yes"| Select["Choose fewest assigned incidents<br/>then stable name and terminal id"]
|
||||
Select --> PeerInbox["Publish to peer lead inbox<br/>and status-gated push"]
|
||||
Peer -->|"no"| Sink
|
||||
Sink -->|"yes"| Webhook["Send classified outbound incident"]
|
||||
Sink -->|"no"| Passive["Keep incident open<br/>show partial coverage on local surfaces"]
|
||||
```
|
||||
|
||||
*Figure 2. Routing keeps delegation ownership separate from temporary peer fallback.*
|
||||
|
||||
Peer candidates exclude the incident subject, failed owner, absent leads, raw-unknown leads, and
|
||||
leads with an open unhealthy state. A reachable `WORKING` peer may be selected; its push waits for an
|
||||
injectable window.
|
||||
|
||||
Choose the candidate with the fewest assigned foreign incidents. Break ties by stable lead name,
|
||||
then terminal id. Pin the recipient. Reassign only if that peer becomes unhealthy or retires. A
|
||||
routing generation marks a reassignment, and old pending assignments become superseded.
|
||||
|
||||
`fleet_list` lead rows show health, health age, assigned foreign incident count, and a bounded list
|
||||
of incident id, subject, state, severity, age, and routing generation. The top-level view also shows
|
||||
owner, recipient, and routing reason.
|
||||
|
||||
A peer incident is published under the recipient lead's inbox key, not the failed subject's key. Its
|
||||
status-gated nudge names the failed lead and gives the exact
|
||||
`fleet_poll(target="<recipient-terminal>")` call.
|
||||
|
||||
### 7.5 Lead inbox ownership
|
||||
|
||||
Add `LeadInboxRegistry`, driven by the same expected-lead supplier as `CallerResolver`.
|
||||
|
||||
It calls `replyInbox.own(leadTerminal)` at startup for configured leads, after successful discovery,
|
||||
after config adds a lead, and before publication. Ownership is not an authorization side effect.
|
||||
|
||||
A missing lead keeps its key owned. Release happens only after confirmed retirement, all incidents
|
||||
are reassigned or resolved, typed health messages move or ack, and the queue is empty. Own a
|
||||
replacement terminal before moving messages from the old key. Never release a non-empty in-memory
|
||||
lead key, because in-memory release clears local data.
|
||||
|
||||
### 7.6 Single-lead deployment
|
||||
|
||||
One lead and no peer is a normal mode, not an edge case.
|
||||
|
||||
An idle, reachable lead may receive the existing bounded nudge. An unreachable or stalled sole lead
|
||||
has no safe in-loop recovery. The bridge must not restart or replace it. A new lead would not have the
|
||||
failed lead's plan or context, and an uncertain relaunch could create two orchestrators.
|
||||
|
||||
With no webhook, only `fleet_list`, `/healthz`, metrics, WARN logs, and the incident journal remain.
|
||||
These are passive surfaces. They are not a human notification.
|
||||
|
||||
## 8. Lost-boundary reconciliation
|
||||
|
||||
This is the only M4 path that reconstructs a result. It must prefer a visible stall over a fabricated
|
||||
reply.
|
||||
|
||||
### 8.1 Why normal completion rules are not enough
|
||||
|
||||
Current `CompletionResolver.resolve` has two fail-open rules. It resolves when the delivery baseline
|
||||
is missing. It also resolves an empty completion when the pane read fails. Those choices are valid
|
||||
after a trusted `WORKING -> IDLE` boundary because the bridge knows the turn ran. They are unsafe
|
||||
when health only guesses that a boundary was lost.
|
||||
|
||||
M4 gives each accepted send an internal `TurnToken`. It ties target, exact waiter, session turn,
|
||||
delivery baseline, and task outcome together.
|
||||
|
||||
### 8.2 Delivery baseline
|
||||
|
||||
Capture the baseline immediately after prompt send and before the delivery future completes. Store:
|
||||
|
||||
```text
|
||||
TurnToken
|
||||
exact waiter identity
|
||||
capture time and pane source
|
||||
normalised assistant block clipped to MAX_SCRAPE_CHARS
|
||||
whether a supported assistant marker was recognised
|
||||
capture result: PRESENT | READ_FAILED | UNRECOGNISED
|
||||
```
|
||||
|
||||
A failed or missing baseline never authorises repair. A late baseline is not valid evidence. After a
|
||||
daemon restart, the old waiter, task, token, and baseline are gone, so the old turn cannot be
|
||||
repaired.
|
||||
|
||||
Automatic repair is enabled only for agent kinds with tested assistant-block fixtures. Current
|
||||
extraction is Claude Code-specific and falls back to arbitrary raw text without `⏺`. That raw fallback
|
||||
cannot authorise repair. OpenCode repair stays disabled until live pane fixtures exist.
|
||||
|
||||
### 8.3 Strict gates and resolver result
|
||||
|
||||
Figure 3 shows the repair gates. Any failed gate keeps the waiter unchanged.
|
||||
|
||||
The two raw snapshots must describe the same `TurnToken` and session turn. No `WORKING`,
|
||||
`BLOCKED`, `UNKNOWN`, missing-agent, reply, failure, or new-delivery observation may occur between
|
||||
them.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Candidate["TURN_BOUNDARY_LOST candidate"] --> Stable{"Same TurnToken and BUSY turn<br/>across two raw IDLE or DONE snapshots?"}
|
||||
Stable -->|"no"| Resnapshot["Take a fresh snapshot"]
|
||||
Stable -->|"yes"| Waiter{"Exact captured waiter<br/>still open by identity?"}
|
||||
Waiter -->|"no"| Stale["STALE_TURN or ALREADY_RESOLVED"]
|
||||
Waiter -->|"yes"| Baseline{"Successful recognised<br/>delivery baseline exists?"}
|
||||
Baseline -->|"no"| Refuse["Refuse repair<br/>leave ticket pending"]
|
||||
Baseline -->|"yes"| Read{"Fresh pane read succeeds?"}
|
||||
Read -->|"no"| Refuse
|
||||
Read -->|"yes"| Output{"Recognised non-blank assistant block<br/>differs from clipped baseline?"}
|
||||
Output -->|"no"| Refuse
|
||||
Output -->|"yes"| Resolve["Shared CompletionResolver guard core<br/>resolves exact waiter"]
|
||||
Resolve -->|"won race"| Repaired["RECONCILED_COMPLETION<br/>same turn becomes DONE"]
|
||||
Resolve -->|"lost race"| Resnapshot
|
||||
```
|
||||
|
||||
*Figure 3. Repair needs stronger evidence than a normal observed turn boundary.*
|
||||
|
||||
Refactor the current resolver into one guard core with two policies:
|
||||
|
||||
```text
|
||||
resolveCaptured(target, inFlight, OBSERVED_BOUNDARY)
|
||||
resolveCaptured(target, inFlight, LOST_BOUNDARY_REPAIR)
|
||||
```
|
||||
|
||||
The health monitor calls only:
|
||||
|
||||
```text
|
||||
CompletionResolver.reconcileLostBoundary(target, expectedTurnToken)
|
||||
```
|
||||
|
||||
It returns `REPAIRED`, `ALREADY_RESOLVED`, `REFUSED_NO_CAPTURE`, `REFUSED_NO_BASELINE`,
|
||||
`REFUSED_UNREADABLE`, `REFUSED_UNCHANGED`, `REFUSED_AMBIGUOUS_OUTPUT`, `STALE_TURN`, or
|
||||
`RACE_LOST`.
|
||||
|
||||
Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move that turn from `BUSY` to `DONE`. A
|
||||
per-target reconciliation gate stops a queued second send from being accepted between waiter
|
||||
resolution and the FSM transition.
|
||||
|
||||
### 8.4 Lead-visible marker and refusal
|
||||
|
||||
A repaired result uses distinct `RECONCILED_COMPLETION` values in `Rendezvous`, `MessageService`,
|
||||
task poll source, and metrics. The lead sees:
|
||||
|
||||
```text
|
||||
[repaired completion - fleetd detected a lost turn boundary. The member did not call
|
||||
fleet_reply; pane-derived text follows and may be partial]
|
||||
```
|
||||
|
||||
Clipped text also keeps the existing clipped-tail marker.
|
||||
|
||||
A refused repair leaves `TURN_BOUNDARY_LOST` open and the ticket pending. The report states that no
|
||||
reply was reconstructed and no task was replayed. `UNCHANGED`, `UNREADABLE`, and
|
||||
`AMBIGUOUS_OUTPUT` get at most one delayed retry for the same token. Missing capture or baseline gets
|
||||
no retry. After two refused scrapes, automatic repair stops for that token.
|
||||
|
||||
### 8.5 Target-wide teardown invariant
|
||||
|
||||
CB-568 owns the multi-ticket cancellation mechanism. M4 routes every terminal cause through that one
|
||||
idempotent operation and checks this independent invariant after teardown:
|
||||
|
||||
- no injector entry exists for the target;
|
||||
- no accepted turn or completion record exists;
|
||||
- no rendezvous waiter or ask exists;
|
||||
- every async task is terminal or was already terminal;
|
||||
- no thread waiting for the target send lock can later accept it;
|
||||
- new sends fail immediately;
|
||||
- each old task has one terminal outcome and one metric count.
|
||||
|
||||
A violation becomes `DELEGATION_ORPHANED`. The monitor may call the same idempotent target-wide
|
||||
failure operation once. It never recreates the task.
|
||||
|
||||
## 9. Capacity and utilisation
|
||||
|
||||
Capacity is a view, not a health state.
|
||||
|
||||
`fleet_list` adds one block per profile:
|
||||
|
||||
```text
|
||||
profile, maxLoad, live, free, reclaimable
|
||||
```
|
||||
|
||||
For an unlimited profile, `maxLoad` and `free` are null. `free` is
|
||||
`max(0, maxLoad - live)` for a capped profile.
|
||||
|
||||
The view must use the exact live-count function used by placement. A second calculation could show a
|
||||
free slot that placement then refuses. Member rows add `idleForSeconds` only when state is `READY` or
|
||||
`DONE`, no accepted turn exists, and the inbox is empty. `reclaimable` means only that the member
|
||||
holds capacity without open bridge work.
|
||||
|
||||
The existing idle-lead nudge gains a bounded capacity summary. It lists per-profile live, cap, free,
|
||||
and reclaimable counts, plus at most three long-idle members. Capacity does not make
|
||||
`FleetState.hasPending()` true. A changed capacity fingerprint may re-arm one capped heartbeat
|
||||
sequence. The fingerprint excludes changing idle durations, so a static idle fleet cannot reset the
|
||||
cap forever. Reply-push stand-down remains first.
|
||||
|
||||
The bridge must never:
|
||||
|
||||
- spawn a member because a slot is free;
|
||||
- generate a task or acceptance criteria;
|
||||
- move queued work to another member or profile;
|
||||
- treat a free slot or idle member as an incident;
|
||||
- stop an idle member only to improve utilisation.
|
||||
|
||||
The bridge knows capacity facts but has no work list. Only the lead has the plan, task context,
|
||||
side-effect history, and acceptance criteria.
|
||||
|
||||
Capacity calculation is in memory and adds no pane reads. Work-product checks run on a terminal
|
||||
session edge, not every fleet tick.
|
||||
|
||||
This capacity design adds no automatic stop. The accepted `NEVER_READY` cleanup can still stop a
|
||||
very slow startup after the existing grace, which is a known risk. Free capacity and long idle time
|
||||
never trigger that path.
|
||||
|
||||
## 10. Human escalation and notification
|
||||
|
||||
### 10.1 Escalation rule
|
||||
|
||||
Notify a person only when no healthy lead can act:
|
||||
|
||||
- `CONTROL_LINK_DOWN` survives grace;
|
||||
- a lead is unhealthy and no healthy peer can receive the incident;
|
||||
- a member incident has no known owning lead;
|
||||
- the only owning lead becomes unreachable, unresponsive, or stalled;
|
||||
- incident publication or routing itself fails.
|
||||
|
||||
Do not page a person for a member fault while a healthy owning lead exists. An uncollected member
|
||||
incident feeds lead-health evidence. If the lead then becomes unhealthy, peer or human routing starts.
|
||||
|
||||
### 10.2 Detection and notification switches
|
||||
|
||||
`health.enabled` controls detection and bridge-local reporting. It does not require a webhook.
|
||||
|
||||
`health.notifications.mode` is `disabled` or `webhook`. Disabled is valid and is the default.
|
||||
Webhook mode requires a resolved environment variable. Turning notification off stops outbound
|
||||
attempts but keeps incidents. Turning it back on resumes still-open human incidents.
|
||||
|
||||
Without a sink, `fleet_list.healthCoverage` states that human escalation is unavailable. `/healthz`
|
||||
keeps its existing HTTP liveness result and adds a nested `fleetHealth.status=partial` component.
|
||||
Metrics and one startup or reload WARN expose the same limit.
|
||||
|
||||
### 10.3 Incident and delivery deduplication
|
||||
|
||||
One open incident uses this key:
|
||||
|
||||
```text
|
||||
(scope, subjectStableId, state, causeFingerprint)
|
||||
```
|
||||
|
||||
The cause fingerprint includes stable error codes, dependency names, signature ids, or invariant
|
||||
names. It excludes times, ages, retry counts, pane text, and changing digests. A later recurrence
|
||||
after resolution gets a new generation and incident id.
|
||||
|
||||
Each outbound event uses:
|
||||
|
||||
```text
|
||||
Idempotency-Key = hash(incidentId, eventType, eventRevision)
|
||||
```
|
||||
|
||||
Event types are `open`, `severity_changed`, `reminder`, and `resolved`. Transport retries keep the
|
||||
same key.
|
||||
|
||||
An atomic owner-only journal beside the active config stores open incidents, routing, delivered
|
||||
revisions, retry state, and resolution state. It stores no pane or task content. Journal failure does
|
||||
not stop detection, but notification coverage becomes degraded.
|
||||
|
||||
### 10.4 Retry, reminder, and resolve
|
||||
|
||||
Send the first event immediately. Retry network errors, timeouts, HTTP 408, HTTP 429, and HTTP 5xx
|
||||
with full-jitter exponential backoff:
|
||||
|
||||
```text
|
||||
base: 5 seconds
|
||||
factor: 3
|
||||
maximum delay: 15 minutes
|
||||
one outstanding attempt per event
|
||||
```
|
||||
|
||||
Respect `Retry-After` up to 15 minutes. Other HTTP 4xx responses are permanent for that event until
|
||||
config changes or a person requests replay.
|
||||
|
||||
Transport retry is not an incident reminder. `humanRepeatSeconds` creates a new reminder revision
|
||||
for an unresolved critical incident after the last successful human event. Disabled mode does not
|
||||
build an unbounded reminder queue.
|
||||
|
||||
Send `resolved` only if at least one human event for that incident was delivered. If an incident
|
||||
resolves before its first successful delivery, cancel the pending open event and record local
|
||||
resolution.
|
||||
|
||||
### 10.5 Outbound payload boundary
|
||||
|
||||
An outbound payload may contain incident id and event type, severity, state, scope, stable bridge
|
||||
ids, role or profile, times, duration, structured evidence type and counts, recovery attempted,
|
||||
routing reason, coverage, and safe tool calls.
|
||||
|
||||
It must never contain:
|
||||
|
||||
- raw pane text, pane excerpts, or pane digests;
|
||||
- task briefs, prompts, or member reply content;
|
||||
- source files, diffs, or worktree file content;
|
||||
- worktree paths;
|
||||
- environment values, tokens, credentials, headers, or webhook URL;
|
||||
- raw exception messages or stack traces;
|
||||
- arbitrary model output.
|
||||
|
||||
The sink response body is ignored. A webhook cannot direct recovery. n8n remains outbound-only.
|
||||
|
||||
### 10.6 Metrics
|
||||
|
||||
M4 adds bounded-label series:
|
||||
|
||||
```text
|
||||
fleet_health_incidents{scope,state,severity}
|
||||
fleet_health_incidents_total{event}
|
||||
fleet_health_notifications_total{event,outcome}
|
||||
fleet_health_notification_queue_depth
|
||||
fleet_health_notification_last_success_seconds
|
||||
fleet_health_notification_capability{mode,status}
|
||||
fleet_lead_health{lead,state}
|
||||
fleet_lead_assigned_incidents{lead}
|
||||
```
|
||||
|
||||
Metric labels never include terminal ids, incident ids, URLs, or error text.
|
||||
|
||||
## 11. Configuration
|
||||
|
||||
The optional `health:` block is absent or disabled by default. The dormant monitor scheduler does no
|
||||
herdr or pane work while disabled. Every listed key is hot because the monitor reads `ConfigRef` on
|
||||
each tick or notification.
|
||||
|
||||
| Key | Class | Default and hard bound | Purpose |
|
||||
|---|---|---|---|
|
||||
| `health.enabled` | Hot | `false` | Enable detection and bridge-local reporting. |
|
||||
| `health.snapshotIntervalSeconds` | Hot | default 30, minimum 15 | Whole-fleet comparison cadence. |
|
||||
| `health.workingSuspectAfterSeconds` | Hot | default 600, minimum 300 | Age before working-pane probes. |
|
||||
| `health.paneProbeIntervalSeconds` | Hot | default 60, minimum 60 | Per-target pane cooldown. |
|
||||
| `health.leadUnresponsiveAfterSeconds` | Hot | default 300, minimum 120 | Delay after exhausted actionable nudges before lead fault. |
|
||||
| `health.humanRepeatSeconds` | Hot | default 3600, minimum 900 | Minimum repeat period for one open human incident. |
|
||||
| `health.capacityLongIdleAfterSeconds` | Hot | default 900, minimum 300 | Long-idle threshold for capacity summaries. |
|
||||
| `health.includePaneExcerpt` | Hot | `false` | Allow a clipped excerpt in local lead reports only. Human payloads still exclude it. |
|
||||
| `health.notifications.mode` | Hot | `disabled` | Select `disabled` or `webhook`. |
|
||||
| `health.notifications.webhookUrlEnv` | Hot | required in webhook mode | Name of the environment variable that holds the sink URL. |
|
||||
| `health.notifications.requestTimeoutMs` | Hot | default 10000, range 1000-30000 | Whole webhook request limit. |
|
||||
|
||||
Two consecutive snapshots are compiled floors for lost boundary, lead disappearance, and control
|
||||
link failure. The two-pane-reads-per-tick limit is also compiled and cannot be weakened by config.
|
||||
|
||||
## 12. Delivery units and acceptance
|
||||
|
||||
### Unit 1 - Evidence model and fleet snapshot
|
||||
|
||||
Scope: health state model, fleet join, clocks, evidence retention, and pane budget.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. One `agent.list` call covers one enabled fleet tick.
|
||||
2. Pure decision tests cover every state and every evidence limit in Section 4.
|
||||
3. `BUSY` plus stable raw `DONE` opens `TURN_BOUNDARY_LOST` after two snapshots.
|
||||
4. Healthy fleet snapshots perform zero pane reads.
|
||||
5. Pane cooldown, two-read fleet budget, and fair rotation cannot be disabled by config.
|
||||
6. Logs are outputs only; no log parsing exists.
|
||||
7. Fleet snapshots expose the same profile live-count calculation that placement uses.
|
||||
8. Capacity rows report cap, live, free, and reclaimable values without opening incidents.
|
||||
|
||||
### Unit 2 - Lost boundary and task reconciliation
|
||||
|
||||
Scope: accepted-turn identity, guarded repair, target-wide teardown, release causes, and preserved
|
||||
worktree discovery.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Every accepted send receives a stable `TurnToken` tied to target, exact waiter, and delivery
|
||||
baseline.
|
||||
|
||||
**Corrected during implementation (2026-08-15).** This criterion first also required the session
|
||||
turn number and the task outcome. That is not implementable at this layer, and the implementer
|
||||
refused it three times rather than fabricate a value — correctly. The reason is an ordering fact
|
||||
that is invisible from any single class: `MessageService` owns acceptance and holds the waiter and
|
||||
the async `Task`, but it learns nothing about delivery, because the delivery event goes to
|
||||
`CompletionResolver` through `TurnListener.onDelivered`. And `CompletionResolver.onDelivered` runs
|
||||
*before* `SessionManager.onDelivered`, so the session turn number does not exist yet at the only
|
||||
point where the token could capture it.
|
||||
|
||||
Two ways out were rejected. A shared registry keyed by target reintroduces exactly the "whichever
|
||||
send happens to be waiting" ambiguity the token exists to remove — the same weak claim
|
||||
`Rendezvous.currentWaiter` warns about. Injecting a turn counter into `MessageService` adds a
|
||||
required cross-layer dependency to populate a field that nothing in this slice reads, which is
|
||||
speculative coupling across a boundary already shown to be fragile.
|
||||
|
||||
So the token identifies the **accepted send**, and `SessionManager` keeps verifying its own
|
||||
delivery separately. Repair (criterion 2) does need the session turn; binding it means resolving
|
||||
that acceptance-versus-delivery ordering first, and that work belongs to the repair unit, not
|
||||
here. The token record carries a comment saying the field is deliberately absent.
|
||||
2. Repair requires the same `BUSY` token, two raw `IDLE` or `DONE` snapshots, no conflicting
|
||||
observation, exact open waiter, successful baseline, and new recognised assistant output.
|
||||
3. Missing, failed, late, or post-restart baseline never authorises repair.
|
||||
4. Repair is enabled only for agent kinds with tested assistant-block extraction. Raw-text fallback
|
||||
without a recognised marker refuses repair.
|
||||
5. Normal completion and repair use one resolver guard core. Waiter, scrape, clipping, unchanged, and
|
||||
exact-turn guards are not duplicated.
|
||||
6. `reconcileLostBoundary` returns every typed result named in Section 8.3.
|
||||
7. Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move the same turn to `DONE`.
|
||||
8. A per-target reconciliation gate blocks a queued second send during repair and FSM update.
|
||||
9. Repaired completion has distinct rendezvous kind, message outcome, poll source, lead marker, and
|
||||
metric. Clipping keeps its extra marker.
|
||||
10. Unchanged, unreadable, or ambiguous evidence gets at most one delayed retry. Missing capture or
|
||||
baseline gets none.
|
||||
11. Refusal leaves the ticket pending and tells the lead that no result was rebuilt or replayed.
|
||||
12. Release, gone, never-ready, and abnormal stop use CB-568's one idempotent target-wide failure
|
||||
operation.
|
||||
13. The post-teardown invariant in Section 8.5 is tested independently of CB-568 internals.
|
||||
14. A violated teardown invariant creates `DELEGATION_ORPHANED` and retries only the idempotent
|
||||
failure operation.
|
||||
15. `SPAWN_ROLLBACK` and normal `COMPLETED` remove worktrees. Abnormal and shutdown causes preserve
|
||||
them.
|
||||
16. Explicit stop is state-aware. Any pending task or non-terminal state preserves the worktree.
|
||||
17. Atomic preserved-worktree manifests reload after restart and appear in lead-only
|
||||
`fleet_list.preservedWorktrees`.
|
||||
18. Manifest failure preserves the worktree and opens an operator-visible health failure.
|
||||
19. Provision records the base commit. Terminal, long-idle worktrees report
|
||||
`WORK_PRODUCT_AT_RISK` only under the evidence in Section 4.2 and never auto-delete work.
|
||||
20. No path replays a delivered task, rebuilds its brief, or retargets it, even when prompt text is
|
||||
available.
|
||||
21. Tests cover both real traces, all repair refusals, clipping, explicit-reply and next-turn races,
|
||||
restart without capture, concurrent send and release, and preserved discovery after restart.
|
||||
|
||||
#### Unit 2 - what has landed so far
|
||||
|
||||
Checked against `main` at `e09cac6` on 2026-08-15. Unit 2 was written as one block, but parts of it
|
||||
have since been built by separate CB tickets. Read this before planning the rest, or that work gets
|
||||
done twice.
|
||||
|
||||
The check was a symbol survey of `fleetd/src/main/java` plus the merge history. It tells you whether
|
||||
the machinery exists at all. It is **not** a line-by-line audit of whether each criterion is fully
|
||||
met, and I did not run one.
|
||||
|
||||
| Criterion | Marker searched for | Found in main source | Reading |
|
||||
|---|---|---|---|
|
||||
| 1 | `TurnToken` | 8 files | **Done** — unit 2a, merged as `fec284e`. Criterion 1 was corrected first; see the note under it. |
|
||||
| 2-5, 9 | `REPAIRED` | 0 files | Not started. The whole guarded-repair path is absent. |
|
||||
| 6, 7, 10 | `reconcileLostBoundary` | 0 files | Not started. |
|
||||
| 12 | CB-568 failure operation | via CB-580 | **Partial.** CB-580 (`0af902e`) routes `GONE` and `NEVER_READY` into the one idempotent target-wide failure. I did not check that release and abnormal stop go through the same call. |
|
||||
| 14 | `DELEGATION_ORPHANED` | 3 files | **Partial.** The health state exists. The teardown-invariant check that creates it, and the retry rule, do not. |
|
||||
| 15 | `SPAWN_ROLLBACK` | 0 files | **Contradicted — see below.** |
|
||||
| 16 | — | — | Partial at best. CB-576 made release preserve a dirty worktree; whether explicit stop is state-aware is not checked. |
|
||||
| 17, 18 | `preservedWorktrees` | 0 files | Not started. No manifest, and no lead-only `fleet_list` field. |
|
||||
| 19 | `WORK_PRODUCT_AT_RISK` | 0 files | Not started. |
|
||||
|
||||
**Criterion 15 no longer matches the code, and the code is right.** It says "normal `COMPLETED`
|
||||
remove worktrees". Since CB-576 (`500bfa2`) that is false on purpose: a `COMPLETED` release now
|
||||
preserves the worktree when it still holds uncommitted work, because deleting it destroys work
|
||||
nobody can get back. CB-576 was filed after exactly that loss. CB-581 goes further — if the
|
||||
dirty-check itself fails, the worktree is preserved rather than removed, since "we could not tell"
|
||||
must not be treated as "it is clean".
|
||||
|
||||
So criterion 15 should be rewritten as: `SPAWN_ROLLBACK` and a `COMPLETED` release with a **clean**
|
||||
worktree remove it; abnormal causes, shutdown, a dirty worktree, and a failed dirty-check all
|
||||
preserve it. `SPAWN_ROLLBACK` itself does not exist yet.
|
||||
|
||||
### Unit 3 - Typed inbox and member routing
|
||||
|
||||
Scope: semantic record, AMQP migration, both adapters, member routing, polling, and member health in
|
||||
`fleet_list`.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. AMQP selects legacy or typed decoding only from `content_type`; it never sniffs the body.
|
||||
2. Persistent `text/plain` from the old build becomes `kind=reply` with exact UTF-8 content,
|
||||
including content beginning with `{`.
|
||||
3. New entries use the vendor media type, `schemaVersion: 1`, UTF-8, persistent delivery, and AMQP
|
||||
message ids.
|
||||
4. Version 1 ignores unknown optional fields but rejects missing fields and identity mismatch.
|
||||
5. Unknown versions are not decoded or acked. They remain on the original queue and create one
|
||||
deduplicated failure.
|
||||
6. Invalid known data never escapes the callback, appears as a reply, or blocks later valid messages.
|
||||
7. Invalid data reaches durable per-target quarantine before original ack. Failed handoff leaves the
|
||||
original unacked.
|
||||
8. Decode failures create redacted WARN, metric, `fleet_list` summary, and routed incident without
|
||||
raw content.
|
||||
9. Both adapters pass one semantic contract for fields, FIFO, dedup, ownership, ack, and release.
|
||||
10. Lead keys require explicit ownership. Publication never claims a queue.
|
||||
11. Unit codec tests cover legacy `{`, Unicode, malformed UTF-8, typed round trip, additive fields,
|
||||
malformed JSON, missing fields, identity mismatch, media type, version, and dedup.
|
||||
12. A live broker contract writes old wire data and reads it with the new adapter after reconnect.
|
||||
13. Live contract tests cover mixed entries, quarantine confirm-before-ack, unsupported redelivery,
|
||||
later progress past poison, property persistence, lead ownership, and ack removal.
|
||||
14. Safe downgrade is documented as unsupported.
|
||||
15. RabbitMQ contract tests pass with `mvn test -Pcontract`. The same cases run once on production
|
||||
LavinMQ, or the release states that LavinMQ was not checked.
|
||||
16. Member incidents route to the exact delegating lead and never resolve a task rendezvous.
|
||||
17. `fleet_list` shows compact member health and capacity without pane content. Member
|
||||
`idleForSeconds` is present only when no accepted turn or inbox item exists.
|
||||
|
||||
### Unit 4 - Lead health and peer routing
|
||||
|
||||
Scope: lead evidence, exact ownership, peer selection, explicit-recipient push, and lead inbox
|
||||
lifecycle.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Lead identity uses the `CallerResolver` supplier. Liveness uses successful current agent data.
|
||||
2. Two successful-list absences with healthy ping become `LEAD_UNREACHABLE`; global link failure does
|
||||
not.
|
||||
3. Raw `WORKING`, raw `UNKNOWN`, first-seen time, failures, last success, and error class persist
|
||||
across ticks.
|
||||
4. Heartbeat and push publish status and nudge outcomes before safe no-injection decisions.
|
||||
5. Dynamic lead identity survives a two-successful-snapshot retirement grace.
|
||||
6. Member incidents first use exact delegation ownership with no singular-primary fallback.
|
||||
7. Peer selection follows the exclusions, load rule, and stable tie break in Section 7.4.
|
||||
8. A selected working peer is not interrupted. Its push waits for an injectable window.
|
||||
9. Recipient assignment stays pinned. Reassignment increments generation and supersedes old pending
|
||||
assignment.
|
||||
10. `fleet_list` shows bounded foreign assignments, recipient, reason, and generation without pane
|
||||
content.
|
||||
11. `LeadInboxRegistry` owns configured and discovered lead keys before publication.
|
||||
12. Missing leads keep ownership. Retirement needs an empty queue and handled incidents.
|
||||
13. Replacement owns the new key before messages move. Non-empty in-memory keys are not released.
|
||||
14. Tests cover dead versus busy, unknown, global failure, stale scan cache, disappearance, one peer,
|
||||
several peers, reassignment, and no peer.
|
||||
15. Adapter tests cover lead ownership, restart re-ownership, retirement, and terminal replacement.
|
||||
LavinMQ is checked or named as unchecked.
|
||||
16. A sole unreachable or stalled lead is never restarted or replaced. Without a sink, only passive
|
||||
evidence remains and every coverage surface says so.
|
||||
|
||||
### Unit 5 - Human sink, hot config, metrics, and operator coverage
|
||||
|
||||
Scope: generic webhook, config split, incident journal, retry, resolve, metrics, example config, and
|
||||
operator documentation.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. `health.enabled` works without a human sink.
|
||||
2. Notification mode is hot, defaults to disabled, and supports disabled or webhook.
|
||||
3. Webhook mode requires a resolved environment value. Bad notification config does not disable an
|
||||
already valid detector.
|
||||
4. Mode changes keep open incidents. Re-enable resumes eligible incidents.
|
||||
5. `fleet_list`, `/healthz`, metrics, and one WARN show partial coverage without a sink. HTTP
|
||||
liveness behavior stays unchanged.
|
||||
6. One-lead, no-sink coverage states that lead failure has no active notification or recovery.
|
||||
7. Incident and outbound dedupe use the stable keys in Section 10.3.
|
||||
8. The owner-only local journal survives restart and contains no pane or task content.
|
||||
9. Journal failure keeps detection running but marks notification coverage degraded.
|
||||
10. Retry tests cover network failure, timeout, 408, 429, `Retry-After`, 5xx, permanent 4xx, jitter,
|
||||
delay cap, config re-arm, and one outstanding attempt.
|
||||
11. Reminders and transport retries remain separate. Disabled mode does not build an unbounded queue.
|
||||
12. Resolve sends only after an earlier human event succeeded. Resolve-before-delivery cancels stale
|
||||
open delivery.
|
||||
13. Metrics use bounded labels and exclude ids, URLs, and error text.
|
||||
14. Payload tests reject every content type forbidden in Section 10.5.
|
||||
15. Webhook response bodies are ignored and cannot direct recovery.
|
||||
16. Tests cover disabled mode, one lead without sink, open/update/reminder/resolve, restart, dedup,
|
||||
reassignment, disable/re-enable, and sink failure while local health continues.
|
||||
17. `fleetd.example.yaml` documents all hot keys and compiled floors.
|
||||
18. The operator Features wiki is updated separately. The portable `CLAUDE.md` block is checked and
|
||||
changed only if shipped tool or inbox semantics make it untrue.
|
||||
19. `mvn clean install` passes.
|
||||
|
||||
## 13. Not checked and release gates
|
||||
|
||||
These limits are part of the design, not optional follow-up notes.
|
||||
|
||||
- **OpenCode pane status and assistant markers were not checked.** OpenCode lost-boundary repair is
|
||||
disabled until live fixtures exist.
|
||||
- **Permission-prompt status was not checked** for Claude Code or OpenCode. `BLOCKED` remains
|
||||
ambiguous and has no automatic action.
|
||||
- **`recent_unwrapped` stability was not checked** across all supported agent kinds. If normalisation
|
||||
is not stable, `STALL_SUSPECTED` must say its evidence is weaker.
|
||||
- **The real `BUSY + DONE` trace was not replayed against live herdr.** The design uses the observed
|
||||
production trace and current poller behavior.
|
||||
- **CB-568 was not present when Unit 2 was designed.** Unit 2 must inspect the landed API and keep its
|
||||
independent teardown invariant.
|
||||
- **Production LavinMQ was not checked.** Existing durable-inbox contracts use RabbitMQ. Migration,
|
||||
quarantine, redelivery, lead ownership, and reassignment must run on LavinMQ before release or be
|
||||
recorded as unchecked.
|
||||
- **Live multi-lead routing was not checked.** Peer choice and reassignment are design rules backed by
|
||||
fake-clock and adapter tests until a live exercise runs.
|
||||
- **A live sole-lead failure with a webhook was not checked.** The no-peer path is a design result,
|
||||
not a tested recovery.
|
||||
- **No n8n, Slack, PagerDuty, or other receiver was checked.** The webhook remains generic and
|
||||
outbound-only.
|
||||
- **Deployment supervisor behavior for nested `/healthz` fields was not checked.** HTTP liveness
|
||||
status stays unchanged to reduce this risk.
|
||||
- **Incident-journal crash behavior was not checked** because the journal does not exist yet. Unit 5
|
||||
must test atomic replacement and restart recovery.
|
||||
- **Worktree merge state cannot be checked reliably** without forge or explicit collection evidence.
|
||||
`WORK_PRODUCT_AT_RISK` stays a warning.
|
||||
|
||||
## 14. Locked exclusions
|
||||
|
||||
M4 does not expose `agent.read` as a bridge tool. It does not add a workflow engine, inbound n8n
|
||||
authority, automatic task assignment, task replay, automatic lead replacement, or automatic member
|
||||
spawn for free capacity.
|
||||
|
||||
The bridge remains a message bus with evidence and bounded mechanical repair. The lead remains the
|
||||
place where judgement and work planning happen.
|
||||
+129
-315
@@ -1,296 +1,162 @@
|
||||
# MCP Contract — `bridged`'s unified gateway
|
||||
# MCP flows and error model — `fleetd`
|
||||
|
||||
> **Status:** 🟡 Design (2026-07-14). Greenfield — no MCP code exists yet; the pom carries
|
||||
> only Javalin/Jackson. This page defines the tool surface that CB-104 and its followers
|
||||
> implement. It supersedes nothing; it fills the "MCP server face" left open by the
|
||||
> [Architecture](1-Architecture) page.
|
||||
|
||||
`bridged` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
> **What this page is.** The **flows**: how a delegation, a clarification, a detached task and a
|
||||
> silent member each travel through `fleetd`. These shapes are what shipped, and they are hard to
|
||||
> read off the code because they span the MCP face, the rendezvous registry, the `Injector` and
|
||||
> herdr.
|
||||
>
|
||||
> **What this page is NOT: a tool reference.** It deliberately holds no tool catalogue, no
|
||||
> parameter tables and no REST paths. **The live MCP schema is the authority** — each tool's own
|
||||
> description and parameters, as mounted — with the intent→tool table in `CLAUDE.md` as the short
|
||||
> form.
|
||||
>
|
||||
> That absence is the fix for fleetd #114 (CB-609), and it is worth stating why. This page used to
|
||||
> carry a full tool catalogue written in July 2026, before any MCP code existed. The code shipped;
|
||||
> the page did not follow. By August it named two tools that do not exist, omitted five that do,
|
||||
> had the wrong name for nearly every parameter, pointed at REST paths the daemon does not serve,
|
||||
> and — worst — still described an identity model (*"any connection that does not map to a known
|
||||
> worker is treated as a primary"*) that was a real privilege bug, fixed since by the ancestry
|
||||
> walk in fleetd #161. Every one of those errors is the same error: **a second, hand-maintained
|
||||
> copy of something the code already states**. So the second copy is gone rather than corrected.
|
||||
> Only the flows remain, because a flow is a shape rather than a name, and shapes are what this
|
||||
> page was ever good for.
|
||||
>
|
||||
> The names that do appear below are checked by `McpContractDocTest`, which fails if this page
|
||||
> names a `fleet_*` tool the server does not register. That test is the whole reason it is safe to
|
||||
> write a tool name here at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
## 1. Rendezvous flows
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
### 1.1 Delegation — happy path
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `bridged` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `bridged` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`bridged` owns policy; herdr owns PTYs.** MCP tools express *intent*; `bridged`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["bridged — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>bridge_send · bridge_reply<br/>bridge_ask · bridge_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"bridge_send (blocks)"| MCP
|
||||
W -.->|"bridge_reply / bridge_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `bridged` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `bridged` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `bridge_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `bridge_reply` / `bridge_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`bridged` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http bridged http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`bridge_send`](#bridge_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`bridge_reply`](#bridge_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`bridge_ask`](#bridge_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`bridge_status`](#bridge_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`bridge_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`bridge_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`bridge_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`bridge_read`](#bridge_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`bridge_cancel`](#bridge_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `bridge_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `bridge_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `bridge_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `bridge_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `bridge_status` on a split-host primary.
|
||||
|
||||
#### `bridge_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `bridged` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `bridge_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `bridge_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `bridge_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`bridge_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`bridge_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`bridge_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `bridge_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `bridge_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `bridge_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
One blocking call, zero polls. The lead's call is held open by `fleetd` until the member answers.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as bridged (MCP + Injector)
|
||||
participant P as "Lead (primary)"
|
||||
participant B as "fleetd (MCP + Injector)"
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: bridge_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B->>B: "register waiter(sessionId)"
|
||||
B->>H: "agent.send — only in an injectable window"
|
||||
H-->>W: "prompt injected"
|
||||
W->>W: "works the turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B->>B: "resolve waiter"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`bridge_ask`)
|
||||
**The cap that matters:** a blocking `fleet_send` is bounded by the *caller's own* MCP client
|
||||
timeout, about 60 seconds — not by the task. Anything slower than that must use the detached flow
|
||||
in §1.3, or the lead's call returns while the member is still working.
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
### 1.2 Clarification — reverse rendezvous
|
||||
|
||||
The member pauses mid-turn to ask, the lead answers, and the member resumes **the same turn** with
|
||||
its context intact.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant W as Worker
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: bridge_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: bridge_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve bridge_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: bridge_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B-->>W: "content injected"
|
||||
W->>B: "fleet_ask{question} — member blocks"
|
||||
B-->>P: "{ outcome: question, turnId }"
|
||||
P->>B: "fleet_send{turnId, content} — answers THIS turn"
|
||||
B-->>W: "fleet_ask returns the answer"
|
||||
W->>W: "resumes the same turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
**Answer with `turnId`, never `sessionId`.** A `sessionId` send starts a new turn; it does not
|
||||
resolve the waiting `fleet_ask`.
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
**The window is about 55 seconds and no nudge extends it.** So never brief a member to "ask me":
|
||||
decide before delegating, or give the member an explicit default to fall back on.
|
||||
|
||||
### 1.3 Detached delegation — the lead does not block
|
||||
|
||||
The lead gets a ticket immediately and collects the answer later. This is the flow for any real
|
||||
task, because of the ~60s cap in §1.1.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant W as Worker
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: bridge_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
P->>B: "fleet_send{sessionId, content, wait:false}"
|
||||
B-->>P: "accepted — ticket"
|
||||
P->>P: "continues its own work"
|
||||
W->>B: "fleet_reply{content}"
|
||||
Note over B: "no waiter is blocked — the reply is held"
|
||||
B->>B: "nudge the lead's own pane (status-gated)"
|
||||
P->>B: "fleet_poll{ticket}"
|
||||
B-->>P: "the member's report"
|
||||
P->>B: "fleet_ack{target, msgId}"
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
A terminal ticket nudges the lead's pane by itself, so a detached task does not need watching. The
|
||||
nudge needs an injectable lead pane and is capped, so it is a convenience rather than a guarantee.
|
||||
|
||||
A worker that never calls `bridge_reply` still returns a result: `bridged` reads its terminal
|
||||
tail when the turn completes.
|
||||
### 1.4 The member never replies — turn-done fallback
|
||||
|
||||
A member that ends its turn without `fleet_reply` still produces something: `fleetd` reads its
|
||||
pane tail. This is a **fallback, not a channel** — it is lossy in three separate ways, and every
|
||||
one of them has produced a wrong answer in practice.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant W as Worker
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls bridge_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
P->>B: "fleet_send — blocks or detaches"
|
||||
B-->>W: "content injected"
|
||||
W->>W: "works, never calls fleet_reply"
|
||||
B->>B: "StatusPoller sees the turn end"
|
||||
B->>B: "read the pane tail"
|
||||
B->>B: "classify: exhausted? echoed brief? real report?"
|
||||
B-->>P: "{ outcome: turn_done } or a named failure"
|
||||
```
|
||||
|
||||
The three ways it goes wrong, and what each looks like now:
|
||||
|
||||
| What happened | What the lead used to get | What it gets today |
|
||||
|---|---|---|
|
||||
| The report is longer than the scrape window | The **end** silently cut off | Still clipped, but marked partial |
|
||||
| The member never started — spent credential | The lead's **own brief** echoed back as a report | A named failure: backend exhausted |
|
||||
| The member is simply slow | A tail of work in progress | Unchanged — read it as a hint, not a result |
|
||||
|
||||
The echoed-brief case is the one to remember: it reads as a long, on-topic report with nothing in
|
||||
it from the member. It is suppressed now, but the general rule stands — **check the member's
|
||||
worktree with `git log` before believing a report you did not watch arrive.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
## 2. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `bridge_send` is simply its producer.
|
||||
Delivery only happens in a safe window. `fleet_send` is a producer for the `Injector`, which
|
||||
already enforces this through `AgentStatus.injectable()`.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
IDLE --> WORKING: "message delivered, picked up"
|
||||
WORKING --> IDLE: "turn done"
|
||||
WORKING --> BLOCKED: "awaits input"
|
||||
BLOCKED --> WORKING: "input delivered"
|
||||
IDLE --> UNKNOWN: "detection glitch"
|
||||
BLOCKED --> UNKNOWN: "detection glitch"
|
||||
UNKNOWN --> IDLE: "re-detected"
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
@@ -306,69 +172,17 @@ stateDiagram-v2
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
**At most one message per turn.** After a send, the `Injector` waits for a `WORKING` pickup before
|
||||
delivering the next, with a grace-poll fallback for turns that finish faster than the poll
|
||||
interval.
|
||||
|
||||
---
|
||||
Two consequences a lead feels directly:
|
||||
|
||||
## 8. Error model
|
||||
- **A second send to a busy member never lands.** It reports as queued and times out. The member
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
|
||||
| Condition | `bridge_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `bridge_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`bridge_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `bridge_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `bridge_reply` / `bridge_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `bridge_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `bridge_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `bridge_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `BridgedApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`bridge_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `bridge_send` (keeps the catalog
|
||||
small) vs. a separate `bridge_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `bridge_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `bridge_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `bridge_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `bridge_reply` / `bridge_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`bridge_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `bridge_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
unreadable pane as a ready one.
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
# v1.0.0 — One leader, one host, complete
|
||||
|
||||
This is the first release of **`bridged`**.
|
||||
This is the first release of **`fleetd`**.
|
||||
|
||||
`bridged` lets one main Claude Code session (the **leader**, on your Pro/Max subscription)
|
||||
`fleetd` lets one main Claude Code session (the **leader**, on your Pro/Max subscription)
|
||||
run a team of **workers** — extra Claude Code sessions on a cheaper or local model, and
|
||||
non-Claude agents too. The leader's own session is never touched: it stays on subscription,
|
||||
with a clean environment.
|
||||
@@ -19,12 +19,12 @@ of this one.
|
||||
## One gateway for all messages
|
||||
|
||||
- **Everyone talks through the same door.** The leader and every worker connect to the same
|
||||
MCP server and use only its tools: `bridge_whoami` · `bridge_profiles` · `bridge_spawn` ·
|
||||
`bridge_list` · `bridge_status` · `bridge_send` · `bridge_reply` · `bridge_ask` ·
|
||||
`bridge_poll` · `bridge_ack` · `bridge_stop`.
|
||||
MCP server and use only its tools: `fleet_whoami` · `fleet_profiles` · `fleet_spawn` ·
|
||||
`fleet_list` · `fleet_status` · `fleet_send` · `fleet_reply` · `fleet_ask` ·
|
||||
`fleet_poll` · `fleet_ack` · `fleet_stop`.
|
||||
- **You are who your connection says you are.** The bridge finds out who is calling from the
|
||||
connection itself, never from a name the caller sends. So a worker cannot pretend to be
|
||||
someone else, and `bridge_whoami` tells each agent its own role — no guessing.
|
||||
someone else, and `fleet_whoami` tells each agent its own role — no guessing.
|
||||
- **The subscription line cannot be crossed.** Only a spawned worker gets
|
||||
`ANTHROPIC_BASE_URL`; the leader never does. Each worker profile has a list of allowed
|
||||
model hosts, checked before anything starts.
|
||||
@@ -59,7 +59,7 @@ had to ask for its replies. That gap is now closed on a single machine:
|
||||
- **The leader gets a tap on the shoulder.** When a reply lands, the bridge nudges the
|
||||
leader's own pane — only when the leader is free, and only a few times. If the leader is on
|
||||
another machine, this quietly falls back to pick-up mode; the reply still waits.
|
||||
- **Workers can ask questions.** With `bridge_ask`, a worker can pause mid-task, ask the
|
||||
- **Workers can ask questions.** With `fleet_ask`, a worker can pause mid-task, ask the
|
||||
leader something, and continue the *same* task with the answer.
|
||||
|
||||
## More than one kind of worker
|
||||
|
||||
+21
-21
@@ -1,9 +1,9 @@
|
||||
# Team — lead orchestrating a mixed Claude + local-LLM fleet
|
||||
|
||||
The message server (`bridged`) delivers **one turn into one worker**. A **team** is the
|
||||
The message server (`fleetd`) delivers **one turn into one worker**. A **team** is the
|
||||
layer above it: a **Claude team-lead** that fans a job out across a **mixed fleet** of
|
||||
workers — some on Claude, some on the remote local LLM — and reduces their replies. Same
|
||||
`bridged` delivery, same subscription boundary; this doc is only about **orchestration** —
|
||||
`fleetd` delivery, same subscription boundary; this doc is only about **orchestration** —
|
||||
who the workers are, how the lead picks one, and how it runs many at once.
|
||||
|
||||
> Delivery mechanics (blocking `POST /message`, status-gated reply envelope) live in the
|
||||
@@ -12,12 +12,12 @@ who the workers are, how the lead picks one, and how it runs many at once.
|
||||
## The team
|
||||
|
||||
- **Team-lead** — the primary **Opus** (Claude Code, env **CLEAN**, on Pro/Max). Not a
|
||||
worker; a **thin client of `bridged`**. It plans, routes, dispatches, and integrates, and
|
||||
worker; a **thin client of `fleetd`**. It plans, routes, dispatches, and integrates, and
|
||||
never sets `ANTHROPIC_BASE_URL`.
|
||||
- **Workers** — a herd of `claude` panes in herdr, each an addressable `bridged` session
|
||||
- **Workers** — a herd of `claude` panes in herdr, each an addressable `fleetd` session
|
||||
with its **own model/env**:
|
||||
- **Claude workers** (clean env, e.g. Sonnet) — reasoning-heavy or high-accuracy subtasks.
|
||||
- **Local workers** (`ANTHROPIC_BASE_URL=https://ollama.ltms.dev`) — bulk, cheap, or
|
||||
- **Local workers** (`ANTHROPIC_BASE_URL=https://llm.ltms.dev/anthropic`) — bulk, cheap, or
|
||||
embarrassingly parallel subtasks.
|
||||
|
||||
Every worker is still a *real Claude Code process* (inherits `CLAUDE.md`, hooks, skills,
|
||||
@@ -28,14 +28,14 @@ MCP) — only its model differs. Scale each kind horizontally by adding panes.
|
||||
```mermaid
|
||||
flowchart TB
|
||||
LEAD["lead — Opus<br/>(Claude Code, env CLEAN)"]
|
||||
BD["bridged<br/>message server + router"]
|
||||
BD["fleetd<br/>message server + router"]
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
WC1["w-claude-1<br/>Sonnet · CLEAN"]
|
||||
WC2["w-claude-2<br/>Sonnet · CLEAN"]
|
||||
WL1["w-local-1<br/>ANTHROPIC_BASE_URL set"]
|
||||
WL2["w-local-2<br/>ANTHROPIC_BASE_URL set"]
|
||||
ANT["api.anthropic.com<br/>(Pro/Max)"]
|
||||
OLL["ollama.ltms.dev<br/>(local model)"]
|
||||
OLL["llm.ltms.dev<br/>(gateway to the local model)"]
|
||||
|
||||
LEAD -->|"blocking POST /message (target role)"| BD
|
||||
BD -->|"Unix socket · send_text · events.subscribe"| HERDR
|
||||
@@ -62,14 +62,14 @@ flowchart TB
|
||||
| `w-local-*` | `ANTHROPIC_BASE_URL` set | local LLM | task is bulk / cheap / embarrassingly parallel |
|
||||
|
||||
The lead applies this rubric itself, guided by its `CLAUDE.md` team charter (below). Worker
|
||||
selection is **policy in the lead**, not a `bridged` concern — `bridged` just delivers to
|
||||
selection is **policy in the lead**, not a `fleetd` concern — `fleetd` just delivers to
|
||||
the session the lead names.
|
||||
|
||||
## Subscription boundary in a team
|
||||
|
||||
Unchanged from the base architecture, and it scales with the fleet: **only local-worker
|
||||
panes** launch with `ANTHROPIC_BASE_URL`. The lead and every Claude worker stay env-clean on
|
||||
the subscription. `bridged` enforces which panes may carry the off-subscription env, so
|
||||
the subscription. `fleetd` enforces which panes may carry the off-subscription env, so
|
||||
adding workers never widens the boundary.
|
||||
|
||||
## Parallel fan-out (map / reduce)
|
||||
@@ -80,7 +80,7 @@ different workers at once, then results are gathered.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant L as lead (Opus)
|
||||
participant B as bridged
|
||||
participant B as fleetd
|
||||
participant WC as w-claude-1
|
||||
participant WL as w-local-1
|
||||
|
||||
@@ -100,26 +100,26 @@ sequenceDiagram
|
||||
```
|
||||
|
||||
- **Map:** the lead issues N concurrent blocking `POST /message` calls (one per subtask → its
|
||||
chosen worker). Each call blocks only *that* request; `bridged` holds it open until the
|
||||
chosen worker). Each call blocks only *that* request; `fleetd` holds it open until the
|
||||
worker's turn completes (status-gated) and returns the reply envelope.
|
||||
- **Reduce:** the lead collects the N envelopes and integrates. A slow local worker never
|
||||
blocks a fast Claude worker — wall-clock ≈ the slowest single subtask, not the sum.
|
||||
- **Detached / long jobs** use the async broker path instead of a held request (Channel 2 in
|
||||
the base architecture), so the lead never busy-polls across turns.
|
||||
|
||||
Fan-out is bounded by the herd size (pane count) and `bridged`'s concurrency policy, not by
|
||||
Fan-out is bounded by the herd size (pane count) and `fleetd`'s concurrency policy, not by
|
||||
the lead.
|
||||
|
||||
## Knowing the roster
|
||||
|
||||
The lead discovers its team from `bridged` (session list / roles) rather than hard-coding
|
||||
The lead discovers its team from `fleetd` (session list / roles) rather than hard-coding
|
||||
pane ids, so workers can be added or restarted without editing the lead. A minimal charter
|
||||
in the lead's `CLAUDE.md` turns Opus into the orchestrator:
|
||||
|
||||
```markdown
|
||||
## Your team (via bridged)
|
||||
You are the team-lead. Delegate through the bridged client — never launch workers yourself.
|
||||
Roster: ask bridged for current sessions/roles.
|
||||
## Your team (via fleetd)
|
||||
You are the team-lead. Delegate through the fleetd client — never launch workers yourself.
|
||||
Roster: ask fleetd for current sessions/roles.
|
||||
- w-claude-* — Claude Sonnet. Reasoning-heavy / high-accuracy subtasks.
|
||||
- w-local-* — remote local LLM. Bulk, cheap, or parallelizable subtasks.
|
||||
|
||||
@@ -133,22 +133,22 @@ tool instead of hand-rolling the HTTP request.
|
||||
|
||||
## What this layer does NOT change
|
||||
|
||||
- **Delivery** is still `bridged` → herdr `pane.send_text` + status events (Message-Server).
|
||||
- **Delivery** is still `fleetd` → herdr `pane.send_text` + status events (Message-Server).
|
||||
- **Completion timing** is still the worker status event; **reply content** still rides the
|
||||
worker `Stop`-hook envelope.
|
||||
- **Single-host** still applies: herdr's socket is local, so the whole herd lives on the
|
||||
`bridged` host. The lead may be remote — it only needs HTTP to `bridged`.
|
||||
`fleetd` host. The lead may be remote — it only needs HTTP to `fleetd`.
|
||||
|
||||
## Open questions
|
||||
|
||||
- **Routing intelligence:** rubric-in-`CLAUDE.md` (lead decides) vs. a `bridged` role-router
|
||||
- **Routing intelligence:** rubric-in-`CLAUDE.md` (lead decides) vs. a `fleetd` role-router
|
||||
(label-based). Start with the former; promote to the latter if routing logic grows.
|
||||
- **Backpressure:** per-role concurrency caps in `bridged` so a fan-out can't exhaust the
|
||||
- **Backpressure:** per-role concurrency caps in `fleetd` so a fan-out can't exhaust the
|
||||
local gateway.
|
||||
- **Result schema:** whether reply envelopes should carry structured metadata (worker, model,
|
||||
tokens) to help the lead's reduce step.
|
||||
|
||||
## Status
|
||||
|
||||
🟡 Design (2026-07-11). Orchestration layer over the selected `bridged` server; inherits
|
||||
🟡 Design (2026-07-11). Orchestration layer over the selected `fleetd` server; inherits
|
||||
herdr (chosen) + AgentAPI (fallback). Delivery unchanged — see the Message-Server design.
|
||||
|
||||
+10
-10
@@ -20,7 +20,7 @@ requirement, not a nice-to-have.**
|
||||
- **Worktree provisioned by the daemon** — `SessionManager` creates a dedicated git worktree +
|
||||
branch per session, **hydrates it to full config parity** (below), and tears it down on release.
|
||||
- **Worker opens its own PR** — the worker commits, pushes its branch, and opens the PR/MR itself,
|
||||
returning the PR URL in its `bridge_reply`.
|
||||
returning the PR URL in its `fleet_reply`.
|
||||
|
||||
## Why worktrees (the hazard being fixed)
|
||||
|
||||
@@ -76,7 +76,7 @@ worktree checks out anyway. Amber is the real gap — untracked local config the
|
||||
3. **Never overlay the git plumbing** — the worktree's own `.git` file/branch is what gives
|
||||
isolation; that's the *one* thing that must differ from the main tree.
|
||||
|
||||
The overlay set lives in config (`BridgedConfig.Worker.parityOverlay` — a list of repo-relative
|
||||
The overlay set lives in config (`FleetConfig.Worker.parityOverlay` — a list of repo-relative
|
||||
paths, with sane defaults) so it's auditable and per-repo tunable.
|
||||
|
||||
> **Trust note (deliberate).** Hydrating local config means the primary's local secrets/tokens
|
||||
@@ -103,7 +103,7 @@ sequenceDiagram
|
||||
Note over W: implement in the isolated worktree
|
||||
W->>G: git commit + git push (SSH, same user)
|
||||
W->>G: open PR (branch to main)
|
||||
W-->>P: bridge_reply (prUrl, branch, summary, tests)
|
||||
W-->>P: fleet_reply (prUrl, branch, summary, tests)
|
||||
P->>SM: release(paneId)
|
||||
SM->>G: git worktree remove wt
|
||||
Note over G: branch + PR persist for review/merge
|
||||
@@ -115,9 +115,9 @@ earlier `STATE.md` idea — a PR is reviewable, mergeable, and self-describing.*
|
||||
|
||||
## Infra facts (verified this session)
|
||||
|
||||
- **Remote:** `ssh://git@git.ltms.dev:2224/lms/claude-bridge.git` (gitea). Push is over **SSH** —
|
||||
- **Remote:** `ssh://git@git.ltms.dev:2224/fleet/fleetd.git` (gitea). Push is over **SSH** —
|
||||
a worker running as the same user with the same keys can `git push` **with no extra credential**.
|
||||
- **gitea is NOT in the project `.mcp.json`** (only `jetbrains`, `intellij-index`, `bridged`). The
|
||||
- **gitea is NOT in the project `.mcp.json`** (only `jetbrains`, `intellij-index`, `fleetd`). The
|
||||
primary's gitea MCP comes from a global/user config, so **workers do not inherit it**. A worker
|
||||
gets only the `bridge` MCP mounted (via `--mcp-config` launch flag).
|
||||
- **No gitea CLI** (`tea`) installed; `glab` is present but is the GitLab CLI (wrong backend).
|
||||
@@ -128,7 +128,7 @@ earlier `STATE.md` idea — a PR is reviewable, mergeable, and self-describing.*
|
||||
|
||||
| Option | Mechanism | Trade-off |
|
||||
|---|---|---|
|
||||
| **A. gitea REST + token** | Worker `curl`s `POST /api/v1/repos/lms/claude-bridge/pulls` with a scoped token injected by the daemon into the worker env | Minimal, no new server; token lives in the off-subscription worker's env (scope it tightly) |
|
||||
| **A. gitea REST + token** | Worker `curl`s `POST /api/v1/repos/fleet/fleetd/pulls` with a scoped token injected by the daemon into the worker env | Minimal, no new server; token lives in the off-subscription worker's env (scope it tightly) |
|
||||
| **B. mount gitea MCP into workers** | Add the gitea MCP to the worker's `--mcp-config` alongside `bridge` | Clean tool call, but the gitea MCP's own auth/token must be provisioned per worker; more moving parts |
|
||||
| **C. install `tea` CLI** | Worker runs `tea pr create` with a token | Another dependency to install + configure; same token question as A |
|
||||
|
||||
@@ -140,7 +140,7 @@ and the token is a single scoped secret the daemon injects like it already injec
|
||||
|
||||
- Off-subscription workers already *could* push (SSH, same user). The **incremental grant is
|
||||
PR-create**, i.e. a gitea API token.
|
||||
- Scope the token **minimally**: the `lms/claude-bridge` repo, `write:repository` (create branch +
|
||||
- Scope the token **minimally**: the `fleet/fleetd` repo, `write:repository` (create branch +
|
||||
PR), **not** merge/admin/org. A leaked token can open PRs, not merge them — the primary/human is
|
||||
still the merge gate.
|
||||
- Inject via the daemon (env var, e.g. `GITEA_TOKEN`), never written to the worker's config dir —
|
||||
@@ -153,11 +153,11 @@ and the token is a single scoped secret the daemon injects like it already injec
|
||||
|---|---|---|
|
||||
| Worktree provision/teardown | **CB-301 ext** — `SessionManager.acquire`/`release`; `WorkerSession` gains `worktree`, `branch` | daemon shells out to `git worktree add/remove` |
|
||||
| **Config-parity overlay** | **CB-301 ext** — `SessionManager.acquire`, after `git worktree add` | symlink/copy the `parityOverlay` set into the worktree so the worker is a full peer; **this is what makes worktrees viable, not a dead-end** |
|
||||
| Overlay config | `BridgedConfig.Worker.parityOverlay` — repo-relative paths, sane defaults | auditable, per-repo tunable; keep explicit + minimal (trust) |
|
||||
| Overlay config | `FleetConfig.Worker.parityOverlay` — repo-relative paths, sane defaults | auditable, per-repo tunable; keep explicit + minimal (trust) |
|
||||
| Branch naming | `worker/<ticket-slug>-<nonce>` off `main` (or a configured base) | one branch per session |
|
||||
| Commit + push + PR handoff | **CB-302** — worker-driven, guided by the skill | push = SSH; PR = option A |
|
||||
| Implementer skill | `.claude/skills/implementer/SKILL.md` | worktree-aware playbook (see below); mounts automatically since workers inherit repo cwd |
|
||||
| gitea token injection | `WorkerService` env + `BridgedConfig` | repo-scoped, minimal perms |
|
||||
| gitea token injection | `WorkerService` env + `FleetConfig` | repo-scoped, minimal perms |
|
||||
| PR review + merge | Primary (has gitea MCP + judgment) | merge on green; the human/primary gate stays |
|
||||
|
||||
## Implementer skill (outline)
|
||||
@@ -170,7 +170,7 @@ A worker-facing playbook (sibling to the existing `reviewer` skill):
|
||||
3. **Push** your branch (`git push -u origin HEAD`).
|
||||
4. **Open a PR** to `main` (option A `curl`, or the decided mechanism) with a title/body describing
|
||||
the change and referencing the ticket.
|
||||
5. **Reply** via `bridge_reply` with the **PR URL**, branch name, files changed, and test names —
|
||||
5. **Reply** via `fleet_reply` with the **PR URL**, branch name, files changed, and test names —
|
||||
that reply is the whole handoff.
|
||||
6. Do **not** merge; do **not** touch `.mcp.json` or `wiki/`.
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Worker startup: working directory & the folder-trust prompt
|
||||
|
||||
When `bridged` spawns a worker, the worker CLI may show an **interactive startup prompt** before it
|
||||
When `fleetd` spawns a worker, the worker CLI may show an **interactive startup prompt** before it
|
||||
is ready to accept a task — most importantly a *"Do you trust the files in this folder?"* dialog. An
|
||||
unattended worker parked on that prompt never becomes injectable: the status-gated injector waits for
|
||||
`idle`/`blocked`, the task is never delivered, and (worst case) a stray Enter answers the dialog
|
||||
@@ -14,11 +14,11 @@ rule that **a worker inherits the primary's directory** (never `$HOME`), and how
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A["bridge_spawn / POST /workers"] --> B{"explicit cwd?<br/>(profile cwd or spawn arg)"}
|
||||
A["fleet_spawn / POST /workers"] --> B{"explicit cwd?<br/>(profile cwd or spawn arg)"}
|
||||
B -->|"yes — told otherwise"| C["use that cwd"]
|
||||
B -->|"no"| D{"caller PID resolvable?<br/>(MCP peer PID)"}
|
||||
D -->|"yes"| E["cwd = the primary's cwd<br/>lsof -a -p PID -d cwd"]
|
||||
D -->|"no (REST / off-host)"| F["cwd = bridged daemon cwd<br/>(never $HOME by assumption)"]
|
||||
D -->|"no (REST / off-host)"| F["cwd = fleetd daemon cwd<br/>(never $HOME by assumption)"]
|
||||
C --> G["ensureWorkspace → tab.create → agent.start {cwd}"]
|
||||
E --> G
|
||||
F --> G
|
||||
@@ -51,10 +51,10 @@ only affect the seed shell, which the bridge closes).
|
||||
| # | Source | When |
|
||||
|---|--------|------|
|
||||
| 1 | Explicit `cwd` — a per-profile `cwd:` in config, or a spawn argument | "told otherwise" — pin a fixed workdir |
|
||||
| 2 | The **primary's cwd**, auto-detected from the `bridge_spawn` caller | normal MCP spawn from the primary |
|
||||
| 3 | The `bridged` daemon's own cwd | REST spawn / off-host caller — **never `$HOME`** |
|
||||
| 2 | The **primary's cwd**, auto-detected from the `fleet_spawn` caller | normal MCP spawn from the primary |
|
||||
| 3 | The `fleetd` daemon's own cwd | REST spawn / off-host caller — **never `$HOME`** |
|
||||
|
||||
The primary's cwd (source 2) is discoverable with no new plumbing: `bridged` already resolves the MCP
|
||||
The primary's cwd (source 2) is discoverable with no new plumbing: `fleetd` already resolves the MCP
|
||||
caller's loopback **peer PID** for connection identity (`ConnectionIdentity` → `LsofPeerPidLookup`);
|
||||
the same PID yields its cwd via `lsof -a -p <pid> -d cwd -Fn` (the `n…` line). The primary maps to no
|
||||
worker pane (it is not a worker), but its PID and cwd are still readable.
|
||||
@@ -62,10 +62,10 @@ worker pane (it is not a worker), but its PID and cwd are still readable.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as "Primary (main)"
|
||||
participant B as "bridged"
|
||||
participant B as "fleetd"
|
||||
participant O as "OS (lsof)"
|
||||
participant H as "herdr"
|
||||
P->>B: "bridge_spawn {profile} (no cwd)"
|
||||
P->>B: "fleet_spawn {profile} (no cwd)"
|
||||
B->>O: "peer PID for this connection's port"
|
||||
O-->>B: "pid"
|
||||
B->>O: "cwd of pid (lsof -d cwd)"
|
||||
@@ -77,9 +77,9 @@ sequenceDiagram
|
||||
|
||||
*Figure 2 — a no-cwd spawn inherits the primary's directory from the caller's PID.*
|
||||
|
||||
> **Status:** implemented (CB-112). `bridged` threads the resolved `cwd` onto **`agent.start {cwd}`**
|
||||
> **Status:** implemented (CB-112). `fleetd` threads the resolved `cwd` onto **`agent.start {cwd}`**
|
||||
> (verified: the worker process is rooted there), keeping the single shared worker space. On an MCP
|
||||
> `bridge_spawn` the primary's cwd is auto-detected from the caller's PID; over REST (no MCP caller)
|
||||
> `fleet_spawn` the primary's cwd is auto-detected from the caller's PID; over REST (no MCP caller)
|
||||
> it is the explicit `cwd` param else the daemon's cwd. Both placements (`tab` and legacy `pane`)
|
||||
> carry it, since it rides `agent.start`.
|
||||
|
||||
@@ -133,5 +133,5 @@ unattended.
|
||||
|
||||
## See also
|
||||
|
||||
- `docs/MCP-Contract.md` — the tool surface (`bridge_spawn`, `bridge_profiles`, …).
|
||||
- `docs/MCP-Contract.md` — the tool surface (`fleet_spawn`, `fleet_profiles`, …).
|
||||
- `wiki/2-Message-Server.md` — the herdr `agent.*` / `workspace.*` schema (`workspace.create {cwd}`).
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
# Wiki audit for #168
|
||||
|
||||
**Source checked:** `.wiki-snapshot/` at `68e32c6` (2026-08-31). I did not use
|
||||
`wiki/`. Code references below are from the current `fleetd` source tree. A quoted
|
||||
line is a concrete claim that needs correction, unless the table says `KEEP`.
|
||||
|
||||
| Page | Verdict | One-line reason |
|
||||
|---|---|---|
|
||||
| `Home.md` | REVISE | Good overview, but it still names the retired product. |
|
||||
| `_Sidebar.md` | REVISE | The heading still says `claude-bridge`. |
|
||||
| `1-Architecture.md` | REBUILD | Its component contract mixes current names with removed tools, routes, and planned backends. |
|
||||
| `2-Message-Server.md` | REBUILD | The claimed MCP schema, mount command, REST/SSE surface, and fallback paths are pre-build design. |
|
||||
| `3-Approaches.md` | REVISE | Useful research history, but it presents unbuilt AgentAPI as a selectable fallback. |
|
||||
| `4-Setup.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `5-Operations.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `6-Team.md` | REBUILD | It teaches role-addressed sends and a Claude-only team model that the shipped API does not have. |
|
||||
| `7-Use-Cases.md` | REBUILD | Its flagship flow depends on removed `ccs` profiles and removed send parameters. |
|
||||
| `8-Roadmap.md` | REBUILD | It is a historical plan, but it presents old implementation choices and planned work as the current stack. |
|
||||
| `9-Implementation.md` | REBUILD | Its package, class, endpoint, and outcome map has drifted from the source. |
|
||||
| `10-Cross-Host-Messaging.md` | REVISE | It labels most federation work proposed, but misses the shipped `coordinator:` lead channel. |
|
||||
| `11-Features.md` | REVISE | It is the right catalogue, but code-path names are old and it misses the second-herdr-daemon capability. |
|
||||
| `12-Claude-to-OpenCode.md` | REVISE | The porting guide is mostly current, but calls the product and spawned-member path a bridge. |
|
||||
| `13-User-Guide.md` | REVISE | It is the best operator page, but needs the product rename and the second-herdr-daemon setup. |
|
||||
|
||||
## Pages needing work
|
||||
|
||||
### `Home.md` — REVISE
|
||||
|
||||
- Quote: `# claude-bridge` (line 1) and `` `claude-bridge` keeps`` (line 11).
|
||||
The product is `fleet` / `fleetd`. The MCP server identifies itself as `fleet` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:313-315`.
|
||||
- Quote: `AgentAPI ... swappable fallback injector` (lines 73-76).
|
||||
There is no AgentAPI implementation under `fleetd/src/main/java`; the actual
|
||||
launchers are selected by `Profile.kind` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java:265-270`.
|
||||
|
||||
### `_Sidebar.md` — REVISE
|
||||
|
||||
- Quote: `### 📖 claude-bridge` (line 1).
|
||||
Rename it to `fleet`. `FleetMcp` registers the current product-facing tool set at
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:301-326`.
|
||||
|
||||
### `1-Architecture.md` — REBUILD
|
||||
|
||||
- Quote: `` `claude-bridge` lets`` (line 3). The product was renamed; the MCP
|
||||
server name is `fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: ``fleet_read`` in the tool list (line 102). No such tool is registered.
|
||||
The complete registered list is `fleet_send` through `fleet_whoami` at
|
||||
`FleetMcp.java:301-326`; `fleet_read` is absent.
|
||||
- Quote: `SSE (GET /events)` (line 143). `FleetApp.build()` registers no `/events`
|
||||
route; its routes are listed at `FleetApp.java:143-159`.
|
||||
- Quote: `Redis Streams / NATS JetStream, or an embedded queue` (line 106).
|
||||
The shipped durable inbox is AMQP, configured by `broker`, at
|
||||
`FleetConfig.java:49-50` and `FleetConfig.java:655-714`.
|
||||
- Quote: `AgentAPI (fallback)` (line 107). No AgentAPI adapter exists; shipped
|
||||
launcher kinds are `claude-code` and `opencode` (`FleetConfig.java:265-270`).
|
||||
|
||||
### `2-Message-Server.md` — REBUILD
|
||||
|
||||
- Quote: `claude mcp add --transport http bridge http://127.0.0.1:8080/mcp`
|
||||
(line 67). The daemon defaults to port `8765` in `FleetConfig.java:183-187`,
|
||||
and identifies its server as `fleet` at `FleetMcp.java:313-315`.
|
||||
- Quote: ``fleet_send(message, target?, {block, timeout_seconds, auto_spawn,
|
||||
turn_id})`` (line 80). The real parameters are `sessionId`, `content`,
|
||||
`timeoutMs`, `wait`, `turnId`, and `coordId` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: ``fleet_read(target, source)`` (line 85). It is not registered; see the
|
||||
complete registration at `FleetMcp.java:301-326`.
|
||||
- Quote: `docs/MCP-Contract.md ... normative` (lines 87-88). That is not a valid
|
||||
reference: only §6 is current, as the current operator guide itself says at
|
||||
`.wiki-snapshot/13-User-Guide.md:466`.
|
||||
- Quote: `SSE (GET /events)` (line 45). No route exists in the built REST surface,
|
||||
`FleetApp.java:143-159`.
|
||||
|
||||
### `3-Approaches.md` — REVISE
|
||||
|
||||
- Quote: `AgentAPI ... remains a swappable fallback injector` (lines 78-84).
|
||||
It was never built. The shipped adapter selection is only `claude-code` or
|
||||
`opencode` (`FleetConfig.java:265-270`). Keep it as discarded research, not an
|
||||
operational fallback.
|
||||
- Quote: `claude-bridge` (line 109). Rename the product to `fleet`; the runtime
|
||||
package is `dev.ltms.fleet`, for example `FleetMcp.java:1`.
|
||||
|
||||
### `4-Setup.md` — RETIRE
|
||||
|
||||
It is a 25-line redirect and says its procedure was never written (lines 3-9).
|
||||
Chapter 13 is the maintained install procedure. Keeping a second navigation page
|
||||
adds no working documentation.
|
||||
|
||||
### `5-Operations.md` — RETIRE
|
||||
|
||||
It is a 35-line redirect and says its runbook was never written (lines 3-14).
|
||||
Chapter 13 now owns run and recovery instructions.
|
||||
|
||||
### `6-Team.md` — REBUILD
|
||||
|
||||
- Quote: `fleet_send {role: w-claude, prompt: A}` (line 98). `fleet_send` accepts
|
||||
`sessionId` and `content`, not `role` or `prompt` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: `some on Claude, some on the remote local LLM` (lines 3-5) and `Every
|
||||
worker is ... Claude Code` (line 25). `opencode` is a first-class launcher kind,
|
||||
not a Claude worker (`FleetConfig.java:265-270`).
|
||||
- Quote: `fleetd's concurrency policy` (line 121). The configured capacity control
|
||||
is per-profile `maxLoad` (`FleetConfig.java:251-264`), not the role routing model
|
||||
described here.
|
||||
|
||||
### `7-Use-Cases.md` — REBUILD
|
||||
|
||||
- Quote: `ccs profile` (line 10), `ccs + herdr` (line 22), and `ccs-spawn`
|
||||
(line 45). The configuration has `profiles` and `fleet`, not `ccs`:
|
||||
`FleetConfig.java:34-58` and `FleetConfig.java:81-101`.
|
||||
- Quote: `fleet_send({"to", "kind", "body", "block"})` (lines 55-62).
|
||||
None of those are the shipped send parameters. The schema is
|
||||
`FleetMcp.java:1096-1108`.
|
||||
- Quote: `fleet_list() → { "profiles": ... }` (lines 74-80). `fleet_list` is a
|
||||
roster view; `fleet_profiles` is the configured-backend view, as registered at
|
||||
`FleetMcp.java:307-311` and described at `FleetMcp.java:1176-1182`.
|
||||
|
||||
### `8-Roadmap.md` — REBUILD
|
||||
|
||||
- Quote: `Java 21+` (line 43). The current project guidance and source use Java 25;
|
||||
the `FleetConfig` source itself uses Java 25 unnamed lambda parameters, for
|
||||
example `FleetConfig.java:102`.
|
||||
- Quote: `herdr 0.7.0 / protocol 14` (line 46). The current REST health endpoint
|
||||
reports the live protocol returned by herdr (`FleetApp.java:240-244`), while the
|
||||
current operator guide records protocol 19 at
|
||||
`.wiki-snapshot/13-User-Guide.md:76-85`.
|
||||
- Quote: `ccs <profile> claude` and `ccs env <profile>` (lines 47-48). Shipped
|
||||
configuration uses `Profile` records and launcher `kind`,
|
||||
`FleetConfig.java:313-330` and `FleetConfig.java:265-270`.
|
||||
- Quote: `Redis Streams via Lettuce` (line 50). The actual durable inbox is AMQP
|
||||
`broker`, `FleetConfig.java:655-714`.
|
||||
|
||||
### `9-Implementation.md` — REBUILD
|
||||
|
||||
- Quote: `rest.FleetdApp` and `mcp.BridgeMcp` (lines 29-30). The classes are
|
||||
`rest.FleetApp` and `mcp.FleetMcp` (`FleetApp.java:46`; `FleetMcp.java:67`).
|
||||
- Quote: `dev.ltms.fleetd` (line 67). The source package is `dev.ltms.fleet`
|
||||
(`FleetMcp.java:1`).
|
||||
- Quote: `WorkerPresence` (line 110). The current class is `MemberPresence`, as
|
||||
imported and used by `FleetMcp` at `FleetMcp.java:12` and `465-469`.
|
||||
- Quote: the outcome list ending in `STALE_TURN` (lines 128-131). The code also
|
||||
has `BACKEND_EXHAUSTED` (`FleetMcp.java:550-554`) and async `ASKING` handling
|
||||
(`FleetMcp.java:664-668`).
|
||||
- Quote: `FleetdApp` (line 207) and `FleetdConfig` (line 211). These names do not
|
||||
resolve; current classes are `FleetApp` and `FleetConfig`.
|
||||
|
||||
### `10-Cross-Host-Messaging.md` — REVISE
|
||||
|
||||
- Quote: the chapter says the cross-host fabric is proposed except for the
|
||||
single-host inbox (lines 3-8). Cross-host **lead-to-lead** delivery shipped:
|
||||
`fleet_send` accepts `coordId` (`FleetMcp.java:1094-1107`) and publishes it at
|
||||
`FleetMcp.java:616-641`; configuration has `coordinator` at
|
||||
`FleetConfig.java:74-78` and `99-101`.
|
||||
- Quote: `bridge.dlx` (line 90). This product name is stale. The shipped lead path
|
||||
uses `LeadChannel`, not the proposed exchange flow (`FleetMcp.java:95-96` and
|
||||
`616-641`). Keep the proposed federation design, but add a clear shipped/proposed
|
||||
boundary for CB-637.
|
||||
|
||||
### `11-Features.md` — REVISE
|
||||
|
||||
- Quote: `mcp/BridgeMcp` (line 22), `config/FleetdConfig` (lines 25-27), and other
|
||||
index references. These paths no longer resolve; the source classes are
|
||||
`mcp/FleetMcp` (`FleetMcp.java:67`) and `config/FleetConfig`
|
||||
(`FleetConfig.java:81`).
|
||||
- Quote: `fleet_whoami` returns only `primary` or `worker` (lines 99-100).
|
||||
It also returns `architect` (`FleetMcp.java:1235-1244`).
|
||||
- The page needs the missing separate member-herdr-daemon feature listed below.
|
||||
|
||||
### `12-Claude-to-OpenCode.md` — REVISE
|
||||
|
||||
- Quote: `same bridge mount` (line 5) and `a bridge-spawned worker` (line 94).
|
||||
Rename the product path to `fleet`. The daemon exposes the MCP server as `fleet`
|
||||
(`FleetMcp.java:313-315`), and profiles select OpenCode with `kind: opencode`
|
||||
(`FleetConfig.java:332-335`).
|
||||
- Quote: the sample mount name is `fleetd` (line 67). The server name is `fleet`;
|
||||
update the sample to avoid teaching a second product name.
|
||||
|
||||
### `13-User-Guide.md` — REVISE
|
||||
|
||||
- Quote: `The bridge is the only channel` (line 63). The invariant is correct, but
|
||||
the product term needs the `fleet` rename. The daemon's MCP server name is
|
||||
`fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: it describes one herdr socket (lines 72-85). It needs the optional
|
||||
`memberHerdrSocket` setup and two-daemon health meaning. The config key is in
|
||||
`FleetConfig.java:34-37`, and `/healthz` checks both daemons when configured at
|
||||
`FleetApp.java:210-245`.
|
||||
|
||||
## MISSING
|
||||
|
||||
`11-Features.md` has a body section for **routing members through a separate herdr daemon**
|
||||
(`## memberHerdrSocket`, line 2174), but **no row in the index table** at the top of the page
|
||||
(lines 20-95). That table is how the page is meant to be read, so a capability absent from it is
|
||||
effectively undiscoverable. Lead note: this is my own omission — I added the section on 2026-08-31
|
||||
and did not add the matching row. Fixed in the wiki at `68e32c6`'s successor.
|
||||
|
||||
The original audit stated the feature had no entry at all. That was wrong: the section exists. The
|
||||
gap is the index row. Recorded here rather than silently corrected, because the difference matters —
|
||||
"undocumented" and "documented but unindexed" are different jobs.
|
||||
|
||||
Evidence for the feature itself: `FleetConfig.java:34-37` and `FleetApp.java:103-115`, `210-245`,
|
||||
and `247-263`.
|
||||
|
||||
## Audit method and coverage
|
||||
|
||||
I checked all 15 pages. I checked concrete tool, route, config, class, file, and
|
||||
product-name claims claim-by-claim on 11 pages: Home, Sidebar, 1, 2, 4, 5, 6, 7, 9,
|
||||
11, and 13. I skimmed the remaining four long historical or research pages (3, 8, 10,
|
||||
12), then checked their concrete claims that affect the verdict. This is an audit of
|
||||
the supplied snapshot, not a wiki rewrite.
|
||||
+6
-6
@@ -2,11 +2,11 @@
|
||||
|
||||
A standard, repeatable **live** end-to-end test of the two-way channel: it drives a real
|
||||
multi-turn conversation between a primary and an off-subscription worker **through the
|
||||
running `bridged` daemon**, captures the full transcript, and grades the channel.
|
||||
running `fleetd` daemon**, captures the full transcript, and grades the channel.
|
||||
|
||||
This is the committed form of the ad-hoc channel test that discovered the CB-115 gaps
|
||||
(herdr `unknown` misclassification wedging delivery, dirty completion scrapes, and workers
|
||||
never calling `bridge_reply` in conversation). Run it after any change to the injector,
|
||||
never calling `fleet_reply` in conversation). Run it after any change to the injector,
|
||||
status handling, completion/failure paths, or the worker reply charter.
|
||||
|
||||
## What it exercises
|
||||
@@ -20,7 +20,7 @@ construction** — it only calls the bridge's loopback REST face.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant T as conversation_test.py
|
||||
participant B as bridged (REST)
|
||||
participant B as fleetd (REST)
|
||||
participant W as worker (off-sub)
|
||||
T->>B: POST /workers (spawn)
|
||||
T->>B: GET /sessions/{id}/status (await ready)
|
||||
@@ -28,7 +28,7 @@ sequenceDiagram
|
||||
T->>B: POST /sessions/{id}/message {wait:false}
|
||||
B-->>T: ticket
|
||||
B->>W: inject prompt (status-gated)
|
||||
W-->>B: bridge_reply
|
||||
W-->>B: fleet_reply
|
||||
T->>B: GET /tasks/{ticket} (poll)
|
||||
B-->>T: done + reply
|
||||
end
|
||||
@@ -37,7 +37,7 @@ sequenceDiagram
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- `bridged` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
- `fleetd` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
profile configured and its backend reachable.
|
||||
- herdr is up (the daemon needs it).
|
||||
- Python 3 (standard library only — no pip installs).
|
||||
@@ -68,7 +68,7 @@ Per-turn grade:
|
||||
|
||||
| Grade | Meaning |
|
||||
|------------|---------------------------------------------------------------------|
|
||||
| `OK` | delivered and resolved by an explicit `bridge_reply` (`source=reply`) |
|
||||
| `OK` | delivered and resolved by an explicit `fleet_reply` (`source=reply`) |
|
||||
| `DEGRADED` | delivered and answered, but resolved via completion-scrape fallback |
|
||||
| `EMPTY` | turn completed but the reply was empty |
|
||||
| `FAILED` | the worker's turn ended in failure (`phase=failed`) |
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Sustained back-and-forth bridge test — ONE primary, ONE worker, many dependent turns
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `bridged` daemon.
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `fleetd` daemon.
|
||||
|
||||
Where conversation_test.py proves a handful of turns work and issue_hunt_test.py proves
|
||||
fan-out isolation, this proves the channel stays healthy under a *sustained, stateful*
|
||||
@@ -24,7 +24,7 @@ Usage:
|
||||
--turn-timeout per-turn max wait, seconds (default 150)
|
||||
--keep-worker do not stop the worker at the end
|
||||
|
||||
Exit code: 0 if every turn in the window resolved via a clean bridge_reply with no channel
|
||||
Exit code: 0 if every turn in the window resolved via a clean fleet_reply with no channel
|
||||
break; 1 otherwise. A live per-turn log streams to stdout so the run can be watched.
|
||||
"""
|
||||
import argparse
|
||||
@@ -47,12 +47,12 @@ STEPS = [7, 3, 11, 5, 9, 4, 13, 6, 8, 2]
|
||||
RULES = (
|
||||
"Let's play a running-total game across several messages. The total starts at 0. "
|
||||
"In each message I'll tell you to add a number; keep the running total yourself and "
|
||||
"reply via bridge_reply with ONLY the current total as a plain integer — no words, no "
|
||||
"reply via fleet_reply with ONLY the current total as a plain integer — no words, no "
|
||||
"punctuation, just the number. Do not restate the arithmetic. First move: add {step}."
|
||||
)
|
||||
NEXT = ("Add {step}. Reply via bridge_reply with only the new running total.")
|
||||
NEXT = ("Add {step}. Reply via fleet_reply with only the new running total.")
|
||||
REANCHOR = ("Let's re-sync — the running total is {total}. Now add {step}. Reply via "
|
||||
"bridge_reply with only the new running total.")
|
||||
"fleet_reply with only the new running total.")
|
||||
|
||||
|
||||
def parse_int(reply):
|
||||
@@ -142,15 +142,15 @@ def main():
|
||||
print("=" * 72)
|
||||
print(f"SUSTAINED CONVERSATION SUMMARY — 1 primary <-> 1 worker over {dur}s (~{dur/60:.1f} min)")
|
||||
print(f" turns: {turns}")
|
||||
print(f" clean bridge_reply exchanges: {oks + drifts}/{turns} (channel breaks: {breaks})")
|
||||
print(f" clean fleet_reply exchanges: {oks + drifts}/{turns} (channel breaks: {breaks})")
|
||||
print(f" arithmetic correct (continuity held): {oks}/{turns} (drifts: {drifts})")
|
||||
print(f" latency: avg {avg}s over {turns} turns")
|
||||
ok = breaks == 0 and turns >= 2
|
||||
if ok and drifts == 0:
|
||||
print(" RESULT: PASS — every turn resolved via bridge_reply and the worker held the "
|
||||
print(" RESULT: PASS — every turn resolved via fleet_reply and the worker held the "
|
||||
"running total across the whole window.")
|
||||
elif ok:
|
||||
print(f" RESULT: PASS (channel) — every turn resolved via bridge_reply for the full "
|
||||
print(f" RESULT: PASS (channel) — every turn resolved via fleet_reply for the full "
|
||||
f"window; {drifts} arithmetic drift(s) (worker recovered after re-anchor).")
|
||||
else:
|
||||
print(" RESULT: FAIL — the channel broke on at least one turn (see CHANNEL BREAK above).")
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge conversation test — a multi-turn primary↔worker exchange through
|
||||
the running `bridged` daemon, fully captured, with automatic gap analysis.
|
||||
the running `fleetd` daemon, fully captured, with automatic gap analysis.
|
||||
|
||||
This is the repeatable form of the ad-hoc channel test that surfaced the CB-115 gaps
|
||||
(herdr `unknown` misclassification, dirty completion scrape, workers not calling
|
||||
bridge_reply). It drives a real off-subscription worker over the live gateway exactly
|
||||
fleet_reply). It drives a real off-subscription worker over the live gateway exactly
|
||||
as a primary Opus session would (async fire-and-poll), records every turn, and grades
|
||||
the channel.
|
||||
|
||||
@@ -130,9 +130,9 @@ def grade(rec):
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit bridge_reply"
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call bridge_reply)"
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
@@ -206,7 +206,7 @@ def main():
|
||||
ok = all(g in ("OK", "DEGRADED") for g in grades)
|
||||
reply_clean = all(g == "OK" for g in grades)
|
||||
if reply_clean:
|
||||
print(" RESULT: PASS — every turn delivered and got a clean bridge_reply.")
|
||||
print(" RESULT: PASS — every turn delivered and got a clean fleet_reply.")
|
||||
elif ok:
|
||||
print(" RESULT: PASS (with notes) — every turn delivered & replied, but some via fallback.")
|
||||
else:
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Live bridge_ask test — the REVERSE rendezvous (CB-205), watched end to end.
|
||||
"""Live fleet_ask test — the REVERSE rendezvous (CB-205), watched end to end.
|
||||
|
||||
Every other harness drives the forward path: primary `bridge_send` → worker `bridge_reply`.
|
||||
Every other harness drives the forward path: primary `fleet_send` → worker `fleet_reply`.
|
||||
This drives the one that runs the other way. A worker is told to pause its delegated turn,
|
||||
ask the primary a question via `bridge_ask`, and only finish once it has the answer — so the
|
||||
ask the primary a question via `fleet_ask`, and only finish once it has the answer — so the
|
||||
turn round-trips primary→worker→primary→worker inside a SINGLE delegation.
|
||||
|
||||
The mechanics that only this path exercises:
|
||||
|
||||
• a worker's mid-turn question surfacing on the primary's *own* blocked send (Outcome.QUESTION),
|
||||
• the `turnId` correlation that lets the primary answer the exact paused turn,
|
||||
• the answer resuming that same turn and the worker's final `bridge_reply` landing on the
|
||||
• the answer resuming that same turn and the worker's final `fleet_reply` landing on the
|
||||
re-opened forward waiter (never a stale or cross-wired one).
|
||||
|
||||
It is two blocking REST calls, no polling:
|
||||
@@ -24,7 +24,7 @@ Like the rest of the suite it talks ONLY to the bridge's REST face on loopback
|
||||
ANTHROPIC_BASE_URL and never touches herdr, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 bridge_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
python3 fleet_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
[--send-timeout SECS] [--answer-timeout SECS] [--keep-worker]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
@@ -57,14 +57,14 @@ ANSWER_COLOR = "blue"
|
||||
|
||||
# A task that CANNOT be completed without asking: the worker is not told which color to choose,
|
||||
# only that the primary will name one when asked. So a correct final reply is only reachable by
|
||||
# actually calling bridge_ask and using the answer.
|
||||
# actually calling fleet_ask and using the answer.
|
||||
TASK_PROMPT = (
|
||||
"You are a bridge worker in a quick coordination game. You do NOT know which color to pick — "
|
||||
"only the primary does. Do exactly this, in order:\n"
|
||||
"1. Call the `bridge_ask` tool with EXACTLY this question: \"PICK A COLOR: red or blue?\"\n"
|
||||
"1. Call the `fleet_ask` tool with EXACTLY this question: \"PICK A COLOR: red or blue?\"\n"
|
||||
"2. The primary will answer with one color word. Take that color and uppercase it.\n"
|
||||
"3. Call `bridge_reply` with EXACTLY one line: CHOSEN=<COLOR> (e.g. CHOSEN=GREEN if told green).\n"
|
||||
"Do not guess a color. Do not call bridge_reply before bridge_ask has returned an answer. "
|
||||
"3. Call `fleet_reply` with EXACTLY one line: CHOSEN=<COLOR> (e.g. CHOSEN=GREEN if told green).\n"
|
||||
"Do not guess a color. Do not call fleet_reply before fleet_ask has returned an answer. "
|
||||
"Do nothing else — no file reads, no other tools."
|
||||
)
|
||||
|
||||
@@ -134,7 +134,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
rec.update(tid=tid, pane=pane, spawned=True)
|
||||
await_ready(base, tid)
|
||||
|
||||
# 1) Delegate the ask-forcing task. This blocks until the worker calls bridge_ask, at which
|
||||
# 1) Delegate the ask-forcing task. This blocks until the worker calls fleet_ask, at which
|
||||
# point our own send unblocks carrying the question and the turnId to answer on.
|
||||
print(f"[{now()}] delegating task (blocks until the worker asks; up to {send_timeout}s)…")
|
||||
t0 = time.time()
|
||||
@@ -159,7 +159,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
rec.update(phase="no_turnid", detail="question surfaced without a turnId to answer on")
|
||||
return rec
|
||||
|
||||
# 2) Answer on that exact turn. This blocks again until the resumed worker calls bridge_reply.
|
||||
# 2) Answer on that exact turn. This blocks again until the resumed worker calls fleet_reply.
|
||||
print(f"[{now()}] answering '{ANSWER_COLOR}' on turn {rec['turnId']} (blocks until reply; up to {answer_timeout}s)…")
|
||||
t1 = time.time()
|
||||
rec["phase"] = "awaiting_reply"
|
||||
@@ -188,7 +188,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
def grade(rec):
|
||||
"""PASS only if the worker asked, the turn resumed, and the reply reflects the answer."""
|
||||
if rec["phase"] == "no_question":
|
||||
return "NO_ASK", "the worker finished/stalled without ever calling bridge_ask"
|
||||
return "NO_ASK", "the worker finished/stalled without ever calling fleet_ask"
|
||||
if rec["phase"] in ("send_error", "answer_error", "spawn"):
|
||||
return "ERROR", rec.get("detail") or "transport error before the round-trip completed"
|
||||
if rec["phase"] == "no_turnid":
|
||||
@@ -203,26 +203,26 @@ def grade(rec):
|
||||
return "OK", "asked, resumed the same turn, and the reply reflected the primary's answer"
|
||||
if reflected:
|
||||
return "DEGRADED", f"reply reflected the answer but resolved via {rec['replySource']} " \
|
||||
"(worker did not call bridge_reply cleanly)"
|
||||
"(worker did not call fleet_reply cleanly)"
|
||||
return "WRONG_ANSWER", f"the worker replied but did not reflect '{ANSWER_COLOR}' — " \
|
||||
f"the answer may not have reached the resumed turn: {rec['reply']!r}"
|
||||
return "WEDGE", f"unexpected terminal phase {rec['phase']}: {rec.get('detail')}"
|
||||
|
||||
|
||||
def write_transcript(out_dir, rec, meta):
|
||||
path = out_dir / "bridge_ask_transcript.md"
|
||||
path = out_dir / "fleet_ask_transcript.md"
|
||||
g, note = grade(rec)
|
||||
with path.open("w") as f:
|
||||
f.write(f"# Live bridge_ask — reverse rendezvous — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"# Live fleet_ask — reverse rendezvous — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"One worker paused its delegated turn to ask the primary, then resumed with the "
|
||||
f"answer (profile `{meta['profile']}`). Result: **`{g}`**.\n\n")
|
||||
f.write("## Round-trip\n\n")
|
||||
f.write(f"1. **primary → worker** (delegation): the ask-forcing task.\n")
|
||||
f.write(f"2. **worker → primary** (`bridge_ask`, {rec.get('ask_latency')}s): "
|
||||
f.write(f"2. **worker → primary** (`fleet_ask`, {rec.get('ask_latency')}s): "
|
||||
f"{rec.get('question')!r} — surfaced on the primary's blocked send as a "
|
||||
f"`question` with `turnId={rec.get('turnId')}`.\n")
|
||||
f.write(f"3. **primary → worker** (answer on that turn): `{ANSWER_COLOR}`.\n")
|
||||
f.write(f"4. **worker → primary** (`bridge_reply`, {rec.get('answer_latency')}s, "
|
||||
f.write(f"4. **worker → primary** (`fleet_reply`, {rec.get('answer_latency')}s, "
|
||||
f"source={rec.get('replySource')}): {rec.get('reply')!r}\n\n")
|
||||
f.write(f"> **{g}:** {note}\n")
|
||||
if rec.get("detail"):
|
||||
@@ -231,7 +231,7 @@ def write_transcript(out_dir, rec, meta):
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Live bridge_ask reverse-rendezvous test (CB-205)")
|
||||
ap = argparse.ArgumentParser(description="Live fleet_ask reverse-rendezvous test (CB-205)")
|
||||
ap.add_argument("--base", default="http://127.0.0.1:8765")
|
||||
ap.add_argument("--profile", default=None)
|
||||
ap.add_argument("--repo", default=str(REPO_ROOT))
|
||||
@@ -241,7 +241,7 @@ def main():
|
||||
ap.add_argument("--keep-worker", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
print(f"[{now()}] live bridge_ask: 1 primary, 1 worker "
|
||||
print(f"[{now()}] live fleet_ask: 1 primary, 1 worker "
|
||||
f"(profile={args.profile or 'default'}, repo={args.repo})\n")
|
||||
|
||||
rec = {"spawned": False, "pane": None}
|
||||
@@ -262,7 +262,7 @@ def main():
|
||||
|
||||
print()
|
||||
print("=" * 72)
|
||||
print("LIVE bridge_ask SUMMARY — reverse rendezvous (CB-205)")
|
||||
print("LIVE fleet_ask SUMMARY — reverse rendezvous (CB-205)")
|
||||
print(f" asked: {rec.get('question')!r} (turnId={rec.get('turnId')}, {rec.get('ask_latency')}s)")
|
||||
print(f" answered: {ANSWER_COLOR!r}")
|
||||
print(f" replied: {rec.get('reply')!r} (source={rec.get('replySource')}, {rec.get('answer_latency')}s)")
|
||||
@@ -1,12 +1,12 @@
|
||||
# Live bridge_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
# Live fleet_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
|
||||
One worker paused its delegated turn to ask the primary, then resumed with the answer (profile `default`). Result: **`OK`**.
|
||||
|
||||
## Round-trip
|
||||
|
||||
1. **primary → worker** (delegation): the ask-forcing task.
|
||||
2. **worker → primary** (`bridge_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
2. **worker → primary** (`fleet_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
3. **primary → worker** (answer on that turn): `blue`.
|
||||
4. **worker → primary** (`bridge_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
4. **worker → primary** (`fleet_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
|
||||
> **OK:** asked, resumed the same turn, and the reply reflected the primary's answer
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge fan-out test — ONE primary vs MANY workers, concurrently, for
|
||||
issue hunting through the running `bridged` daemon, fully captured, with gap analysis.
|
||||
issue hunting through the running `fleetd` daemon, fully captured, with gap analysis.
|
||||
|
||||
Where conversation_test.py exercises a single worker over multiple turns, this drives
|
||||
the path that only appears under fan-out: the primary spawns N workers, sends each a
|
||||
@@ -12,7 +12,7 @@ and collects every reply concurrently. That stresses what a single worker never
|
||||
• reply routing under concurrency (worker A's answer must never resolve worker B's send),
|
||||
|
||||
and, as the payload, whether a fleet of off-subscription workers can actually surface
|
||||
real issues in the repo and report them back structurally via bridge_reply.
|
||||
real issues in the repo and report them back structurally via fleet_reply.
|
||||
|
||||
It talks ONLY to the bridge's REST face on loopback — it never sets ANTHROPIC_BASE_URL
|
||||
and never touches herdr directly, so it is subscription-safe by construction.
|
||||
@@ -53,18 +53,18 @@ REPO_ROOT = HERE.parent
|
||||
# hot files this project has been iterating on, so a real issue is plausible to find.
|
||||
DEFAULT_ASSIGNMENTS = [
|
||||
{"id": "completion", "probe": "CompletionResolver",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/inject/CompletionResolver.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java"},
|
||||
{"id": "worker", "probe": "WorkerService",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/worker/WorkerService.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/worker/WorkerService.java"},
|
||||
{"id": "rendezvous", "probe": "Rendezvous",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/msg/Rendezvous.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/msg/Rendezvous.java"},
|
||||
]
|
||||
|
||||
PROMPT_TMPL = (
|
||||
"You are one of several issue-hunting workers in the claude-bridge repo (it is your "
|
||||
"current working directory). Your assignment: inspect the file `{target}` and find the "
|
||||
"SINGLE most important real bug, correctness gap, or risk in it. Read the file before "
|
||||
"answering. Reply via bridge_reply with EXACTLY these four lines:\n"
|
||||
"answering. Reply via fleet_reply with EXACTLY these four lines:\n"
|
||||
"1. {target}:<line>\n"
|
||||
"2. issue: <one sentence>\n"
|
||||
"3. fix: <one line>\n"
|
||||
@@ -147,9 +147,9 @@ def grade(rec):
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit bridge_reply"
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call bridge_reply)"
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
||||
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
||||
fleetd.yaml
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -5,10 +5,10 @@
|
||||
|
||||
## Why this exists
|
||||
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `bridge_send` is
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `fleet_send` is
|
||||
open: it lands in `agent.<target>.inbox` on the broker and survives a daemon bounce. But
|
||||
delivery is still **pull** — the primary only sees the reply if it happens to call
|
||||
`bridge_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
`fleet_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
the primary works on something else.
|
||||
|
||||
This layer makes delivery **active**: the bridge *pushes* a nudge to the primary the moment
|
||||
@@ -26,11 +26,11 @@ pointed at the primary's pane instead.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
W["worker"] -->|"bridge_reply (no open send)"| MS["MessageService.reply"]
|
||||
W["worker"] -->|"fleet_reply (no open send)"| MS["MessageService.reply"]
|
||||
MS -->|"inbox.publish"| INBOX[("agent.<target>.inbox<br/>(durable, LavinMQ)")]
|
||||
MS -->|"notify"| LOOP["ReplyPushLoop"]
|
||||
LOOP -->|"status-gated inject"| PANE["primary's herdr pane"]
|
||||
PANE -->|"primary drains"| DRAIN["bridge_poll(target)<br/>= peek + ack"]
|
||||
PANE -->|"primary drains"| DRAIN["fleet_poll(target)<br/>= peek + ack"]
|
||||
DRAIN -->|"inbox now empty"| LOOP
|
||||
LOOP -.->|"still non-empty →<br/>re-inject on backoff"| PANE
|
||||
classDef store fill:#2c5282,stroke:#1a365d,color:#ffffff;
|
||||
@@ -51,13 +51,13 @@ the caller runs in a herdr pane on this host. Today it's discarded for the prima
|
||||
(`presence.markPresent` is a no-op on it).
|
||||
|
||||
**Plan:** a single-slot `PrimaryRegistry` (thread-safe) holding the primary's `terminal_id`.
|
||||
Populate it from the **orchestration-side** MCP tools — `bridge_send`, `bridge_spawn`,
|
||||
`bridge_poll`, `bridge_list`, `bridge_status`, `bridge_profiles` — capturing
|
||||
Populate it from the **orchestration-side** MCP tools — `fleet_send`, `fleet_spawn`,
|
||||
`fleet_poll`, `fleet_list`, `fleet_status`, `fleet_profiles` — capturing
|
||||
`callerTerminal(exchange)` when it is (a) non-null and (b) **not** a registered worker
|
||||
session in `SessionManager`. That caller is, by construction, the primary. Worker-side tools
|
||||
(`bridge_reply`, `bridge_ask`) never set it.
|
||||
(`fleet_reply`, `fleet_ask`) never set it.
|
||||
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `BridgedConfig`
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `FleetConfig`
|
||||
(nested record, same shape as `Broker`). Lets an operator pin it, or supply it when
|
||||
derivation can't (see degrade case).
|
||||
- **Degrade:** if the primary is off-host or in a non-herdr terminal, `terminalForPid`
|
||||
@@ -71,13 +71,13 @@ A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
(`MessageService.reply` → the `inbox.publish` branch, `MessageService.java:192`).
|
||||
|
||||
- **Inject a nudge, not the payload.** The injected turn tells the primary *to drain*
|
||||
(e.g. "Worker `<target>` returned a reply — run `bridge_poll(target=<target>)` to collect
|
||||
(e.g. "Worker `<target>` returned a reply — run `fleet_poll(target=<target>)` to collect
|
||||
it"), it does **not** carry the reply text. Rationale: replies can be large/multiline and
|
||||
terminal injection would mangle them; the drain response is the clean transport. Keeps the
|
||||
push idempotent — re-nudging is harmless.
|
||||
- **Ack = drain.** The primary draining (`drainReplies` = peek + ack) is the acknowledgement.
|
||||
The loop's **stop condition is `inbox.peek(target).isEmpty()`** — the reply is gone from the
|
||||
inbox because it was acked. No new `bridge_ack` tool needed for v1 (see Increment 3).
|
||||
inbox because it was acked. No new `fleet_ack` tool needed for v1 (see Increment 3).
|
||||
- **Status-gated injection (mechanism (b), chosen).** A dedicated lightweight scheduled loop,
|
||||
**not** the worker `Injector`. It injects via `AgentControl.send(primaryTerminal, nudge)`
|
||||
(the same herdr `agent.send` = `pane send-text` + submit that delivers to workers) only when
|
||||
@@ -93,10 +93,10 @@ A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
the reply remains in the durable inbox and the next natural poll (or a later worker reply's
|
||||
nudge) still surfaces it. Bounded so the bridge never spams the primary.
|
||||
|
||||
### Increment 3 — optional per-`msgId` `bridge_ack` tool (deferred)
|
||||
### Increment 3 — optional per-`msgId` `fleet_ack` tool (deferred)
|
||||
|
||||
Drain-as-ack is coarse: it clears *all* pending replies for a target at once. If finer
|
||||
control is ever needed (ack one reply, leave others held), add a `bridge_ack(msgId)` tool
|
||||
control is ever needed (ack one reply, leave others held), add a `fleet_ack(msgId)` tool
|
||||
mapping to `inbox.ack(target, msgId)` — the port already supports per-`msgId` ack. Not built
|
||||
in v1; the stop-on-empty loop is sufficient.
|
||||
|
||||
@@ -106,7 +106,7 @@ in v1; the stop-on-empty loop is sufficient.
|
||||
resolved terminal is non-null **and not a registered worker session**, seen on an
|
||||
orchestration-side tool. This never mislabels a worker (workers are in `SessionManager`)
|
||||
and needs no new env var or argument (identity stays connection-derived, per the existing
|
||||
`BridgeMcp` invariant).
|
||||
`FleetMcp` invariant).
|
||||
|
||||
2. **Readiness-gate mismatch → dedicated loop.** The existing `Injector` gates delivery on
|
||||
`ready.test(target)` = `WorkerPresence` (the *worker's* MCP connected). The primary is not
|
||||
@@ -132,7 +132,7 @@ boundary**. The bridge is signalling the primary that it has mail — not drivin
|
||||
non-null terminal AND not a registered session" predicate; the loop's stop-on-empty and
|
||||
bounded-reminder logic with an injected clock + a fake injector (no real herdr).
|
||||
- **Live dogfood (primary-side):** with the daemon on the broker jar + a real worker,
|
||||
delegate a task, let the worker reply after the `bridge_send` window closes, and observe the
|
||||
delegate a task, let the worker reply after the `fleet_send` window closes, and observe the
|
||||
bridge inject a drain nudge into *this* primary pane; confirm draining stops the reminders;
|
||||
confirm an unreachable primary (registry empty) degrades to pull with no loss.
|
||||
|
||||
@@ -0,0 +1,851 @@
|
||||
# fleetd configuration (example). Copy to fleetd.yaml and adjust.
|
||||
#
|
||||
# fleetd is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — fleetd REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
|
||||
# API authentication (CB-501). Governs how a caller that is NOT an on-host worker pane proves it
|
||||
# is the primary. Worker identity never depends on this: a loopback peer PID that maps to a herdr
|
||||
# pane is unforgeable and is always honoured, so turning auth on cannot lock the fleet out.
|
||||
#
|
||||
# mode: loopback-trust → DEFAULT, and the historical behaviour: any loopback caller that is not
|
||||
# a worker is the primary, no credential needed. Sound ONLY because the
|
||||
# OS refuses remote connections to a loopback socket.
|
||||
# mode: token → such a caller must send `Authorization: Bearer <token>`; without it it
|
||||
# is anonymous and authorized for nothing. REQUIRED for a non-loopback
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# FLEETD_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
# location / { proxy_pass http://127.0.0.1:8765; proxy_set_header Authorization $http_authorization; }
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: FLEETD_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from fleet_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
# pushReminders: 5 # max nudges before giving up (default 5)
|
||||
# pushBackoffMs: 15000 # delay between nudges (default 15000)
|
||||
|
||||
# CB-530: MORE THAN ONE LEAD. `primary:` above is singular by construction — every other pane
|
||||
# resolves as a worker — which is right for one lead driving a fleet and wrong the moment two leads
|
||||
# (say a Claude lead and an opencode lead) work as peers: the second is silently demoted and refused
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# tab → the ONLY field identity depends on (CB-579); the exact label of the tab hosting the lead.
|
||||
# Label the tab yourself, or let fleetd label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by fleet_whoami
|
||||
#
|
||||
# A lead's tab must already carry its label (or be launched by fleetd, which labels it) — there is
|
||||
# no terminal id to paste in and nothing to re-pin when the session restarts: the tab survives, so
|
||||
# the same label resolves the same lead again on the next scan.
|
||||
# `fleet_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
# destination for its nudges, and is a separate mechanism from lead identity — see `fleet.leaders:`.
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing fleetd places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open fleet_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
# Opt-in on purpose — it SPENDS the operator's subscription on its own initiative (each nudge starts
|
||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||
# block = feature off, exactly as before.
|
||||
#
|
||||
# Three knobs, each with a default that errs on the side of not burning context:
|
||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||
# leadHeartbeat:
|
||||
# idleAfterSeconds: 300
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
# per tick.
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
# Math.max(15, intervalSeconds), so a lower value is silently raised, not
|
||||
# rejected.
|
||||
# workingSuspectAfterSeconds → age before a BUSY member is suspected of a stall (default 600).
|
||||
# ENFORCED floor of 300: a lower value is silently raised.
|
||||
# paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by anything. Setting it changes
|
||||
# nothing right now. It exists so a later build can start honouring it without
|
||||
# another config-shape change.
|
||||
# notifications.mode → "webhook" flips what fleet_list REPORTS (healthCoverage: "full" instead
|
||||
# of "detection-only") — it does NOT make fleetd send any webhook call; no
|
||||
# delivery mechanism is implemented yet. Any other value, or omitting the
|
||||
# block, reports "detection-only".
|
||||
# health:
|
||||
# enabled: true
|
||||
# intervalSeconds: 30 # floor 15
|
||||
# workingSuspectAfterSeconds: 600 # floor 300 — how long BUSY with no activity means STALL_SUSPECTED
|
||||
# paneProbeIntervalSeconds: 60 # parsed, but nothing reads it yet — changing it changes nothing
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# Optional socket for member panes. Omit this to use herdrSocket for both leads and members.
|
||||
# memberHerdrSocket: /Users/member/.config/herdr/herdr.sock
|
||||
|
||||
# fleetd #213: the login shell the member OS user (memberHerdrSocket above) actually runs. ONLY
|
||||
# read when memberHerdrSocket is set — fleetd's own $SHELL says nothing about a pane running
|
||||
# under a different OS user, and there is no channel to ask herdr for that user's shell, so this
|
||||
# must be told rather than guessed. Absent, blank, or anything not ending in "zsh" is treated the
|
||||
# same as "not zsh": the memberCredentials.policy: allow-list ZDOTDIR scrub (see worktreeGroup
|
||||
# below) is skipped in favour of the weaker CB-596 sentinel overlay — a degraded control, never a
|
||||
# refusal to spawn. When memberHerdrSocket is absent this key is never consulted at all.
|
||||
# memberLoginShell: /bin/zsh
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
# it is for; that is the member's role, and roles live under `fleet:` below. Which profile an
|
||||
# unqualified spawn lands on comes from that role's pool, not from a global default.
|
||||
#
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → fleetd mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# ideMcpUrl → opt-in (CB-634), default off. When set, fleetd mounts the IDE Index MCP as a
|
||||
# second inline server named `intellij`, and adds an IDE charter that pins every
|
||||
# ide_* call to the member's own worktree. A URL, not a boolean — host and port
|
||||
# are host-specific. Set it only on a host where the IDE actually runs.
|
||||
# ideProjectDir → repo-relative module dir the IDE opens and the overlay pins (CB-634). Only read
|
||||
# when ideMcpUrl is set. This repo's Maven pom lives in `fleetd/`, not at the
|
||||
# worktree root, so opening the root imports no module and ide_* resolves nothing;
|
||||
# set this to `fleetd`. Omit for a repo whose project is the worktree root.
|
||||
# ideOpenCommand → host command that opens ideProjectDir in the IDE at spawn (CB-634 auto-open).
|
||||
# Only read when ideMcpUrl is set. `{dir}` is replaced with the absolute module
|
||||
# dir and the command runs through `/bin/sh -c`, so set env inline if needed —
|
||||
# e.g. `env DISPLAY=:10.0 idea {dir}`. Best-effort: a failure is logged, never
|
||||
# fails the spawn. Omit to open the member's module by hand. There is no close
|
||||
# half yet — an opened module stays open until the operator closes it.
|
||||
# autoCompactWindow → opt-in, default off. A bounded token window that forces a spawned member to
|
||||
# compact its context instead of running on the backend's own default and dying
|
||||
# mid-turn (losing its fleet_reply — the whole point of the turn — with it).
|
||||
# Validated at config load to [100000, 1000000] — the band Claude Code's own
|
||||
# --autocompact flag accepts.
|
||||
# CROSS-BACKEND SEMANTICS DIFFER: on claude-code this is a launch-time
|
||||
# `--autocompact <tokens>` flag — the member compacts AT this window. opencode
|
||||
# has no equivalent flag (it only forces `compaction.auto: true`, unconditionally,
|
||||
# already), so this is instead applied as the model's `limit.context` in the
|
||||
# generated opencode.json — the member compacts WITHIN this window, not exactly
|
||||
# at it — and only when this profile's `model:` is in `provider/model` form; if it
|
||||
# isn't, fleetd logs a WARN naming the profile rather than silently doing nothing.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
# cwd on an MCP spawn, else the daemon's cwd — never $HOME. See
|
||||
# docs/Worker-Startup-and-Trust.md.
|
||||
# configDir → CLAUDE_CONFIG_DIR for the worker, so it inherits that profile's
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.env] only (CB-148). .envrc is left out of the default on purpose: it is
|
||||
# executable shell that direnv runs on every cd, so copying it carries
|
||||
# behaviour into the worker, not just values, unlike .env. An operator who
|
||||
# wants it copied can still write parityOverlay: [.env, .envrc] explicitly.
|
||||
# (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
# handed a worker the primary's IDE servers, which are bound to the primary's
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# fleetd neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
# Opt-in by design — omit and the worker gets no PR-create grant (push over
|
||||
# SSH is unaffected). The token value itself is never stored in this file.
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# exhaustedPattern → regex matched against a completion-fallback scrape (CB-578 stage A) to
|
||||
# classify a turn that ended with no fleet_reply as the backend having
|
||||
# refused on a subscription usage limit, rather than a real answer. Opt-in —
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into fleetd itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
# BACKEND_EXHAUSTED classification fires. Two profiles that set the SAME
|
||||
# credentialId share one quarantine — the case this exists for is two models
|
||||
# on one account (e.g. sol and terra both billing one OpenAI credential): an
|
||||
# exhaustion on either one must lock out both, or the fleet just walks onto
|
||||
# the same dead account under the sibling's name. Opt-in — omit and this
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# errorPattern → fleetd #201 / #227: regex matched against a completion-fallback scrape to
|
||||
# classify a turn that ended with no fleet_reply as a BACKEND ERROR — a
|
||||
# credential outage or a provider 5xx — rather than a real answer or a
|
||||
# usage-limit exhaustion (exhaustedPattern above always wins when a line
|
||||
# matches both). Opt-in. Omit it and this profile falls back to fleetd's
|
||||
# built-in legacy pattern `(?i)\bAPI Error\s*:` — classification still
|
||||
# happens, just without a profile-specific match; every backend words its
|
||||
# failure differently, so a hardcoded sentence would only ever match one
|
||||
# of them.
|
||||
# DEFERRED: compiled once into a startup pattern map, same as exhaustedPattern
|
||||
# — editing it needs a daemon restart.
|
||||
# # errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage
|
||||
#
|
||||
# What happens once a match fires (BackendOutagePolicy, credentialId-keyed,
|
||||
# SEPARATE from the CB-578 stage B quarantine above and never merged with it):
|
||||
# - threshold 2 — TWO DISTINCT TARGETS (never raw events) on the same
|
||||
# effective credential inside a 60-second window start an "incident" and a
|
||||
# 60-second cool-off for that credential. One member repeating the same
|
||||
# classified line twice never cools anything off — a real outage hits
|
||||
# every target on that credential, so requiring a second, independent
|
||||
# target loses nothing against the case this guards against, while
|
||||
# protecting against a heuristic misfire on one flaky member.
|
||||
# - a fresh error while a credential is already cooling off is ignored
|
||||
# outright: it neither extends the 60s deadline nor starts a new incident.
|
||||
# - `fleet_list`/`fleet_profiles` report a cooling credential with
|
||||
# `coolingOffForSeconds` (never `quarantinedForSeconds`, unless CB-578
|
||||
# exhaustion quarantine is ALSO independently active for the same
|
||||
# credential — the two checks can both fire at once). A spawn onto a
|
||||
# cooling profile is refused with a message naming the credential and
|
||||
# remaining seconds — "cooling off", never "exhausted", so an operator can
|
||||
# tell a short transient fault from a spent subscription at a glance.
|
||||
# - the lead gets ONE nudge per incident (not one per affected target), via
|
||||
# the same push loop that already delivers ticket/question reminders.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. fleetd hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# fleetd now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.fleet.plist and deploy/fleetd.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
# entry cannot repoint a worker past the SubscriptionGuard — which is checked
|
||||
# against `baseUrl` alone.
|
||||
# Put `defaultMode: "auto"` in each ccs profile so the worker runs autonomously.
|
||||
profiles:
|
||||
gx10: # ccs profile name (NOT a hostname)
|
||||
kind: claude-code # which adapter spawns this profile (default; may omit)
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: FLEETD_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
# weight: relative selection weight for automatic placement (weighted, round-robin, and
|
||||
# fixed's fallback walk). Absent defaults to 1.0. An explicit 0 or negative value means
|
||||
# "never auto-select this profile" (CB-554) — it stays reachable via an explicit
|
||||
# `fleet_spawn{profile:"gx10"}`, which bypasses placement entirely; only automatic
|
||||
# selection skips it.
|
||||
weight: 0.5
|
||||
# maxLoad: max live workers on this profile. Omit for unlimited. An explicit 0 (CB-585) caps
|
||||
# the profile at zero live members — it is excluded from automatic placement and an explicit
|
||||
# `fleet_spawn{profile:"gx10"}` against it is refused too; a cap holds even when the profile
|
||||
# is named directly. Negative is refused at config load — there is no sane meaning for it.
|
||||
maxLoad: 2
|
||||
# subscription: true
|
||||
# THE KNOB THAT DECIDES WHO PAYS (CB-539). Default false. When true, this profile's members
|
||||
# run on the OPERATOR'S OWN Claude subscription instead of a metered endpoint — every spawn
|
||||
# bills your plan and eats your usage limit. Off-subscription is the whole point of this
|
||||
# daemon, so treat `true` as a deliberate exception, not a convenience.
|
||||
#
|
||||
# What changes when it is set (ClaudeCodeLauncher):
|
||||
# - no ANTHROPIC_BASE_URL and no ANTHROPIC_AUTH_TOKEN are injected — the member inherits
|
||||
# the operator's own Claude Code auth, which is exactly why it bills the plan;
|
||||
# - SubscriptionGuard never vets it, because there is no baseUrl to vet;
|
||||
# - no token is required, so `tokenEnv` is irrelevant here.
|
||||
#
|
||||
# MUTUALLY EXCLUSIVE with `baseUrl` — setting both is refused at config load (CB-542). On the
|
||||
# subscription path no guard would vet the URL, so allowing both would be a way around the
|
||||
# guard rather than a configuration.
|
||||
#
|
||||
# GOTCHA 1 — it is invisible to the startup secret check. `Fleetd.reportRequiredSecrets`
|
||||
# skips subscription profiles on purpose (they need no token), so a boot log that reports
|
||||
# every secret as fine says nothing about these profiles.
|
||||
#
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
#
|
||||
# GOTCHA 3 (fleetd #176, corrected by fleetd #257) — `maxLoad` counts members, never the lead
|
||||
# itself. The lead is a live `claude` session on this SAME account (a lead is never moved
|
||||
# off-subscription, whatever its own profile says), so it already holds one seat before any
|
||||
# member spawns. If a lead's `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER
|
||||
# `subscription: true` profile that shares this one's account (see THE SENTINEL, just below,
|
||||
# next to `credentialId:`) — `fleet_list` reports that seat count under `leadSeats`; see
|
||||
# `profile:` under THE FLEET below. `free` itself is NEVER reduced by `leadSeats`: `free` means
|
||||
# "what the real placement gate (`CompositePeerLauncher#enforceMaxLoad`) will actually grant a
|
||||
# fresh `fleet_spawn` right now", and that gate only ever compares live members against
|
||||
# `maxLoad` — it has no notion of the lead's own seat. An earlier cut of this feature
|
||||
# subtracted `leadSeats` from `free` on the theory it made `free` describe the true ceiling on
|
||||
# the account, but no backend seat ceiling shared with the lead has ever actually been
|
||||
# measured, and the subtraction just made `free` disagree with the one thing it is supposed to
|
||||
# describe — the fleetd #257 fix. `maxLoad: 3` means 3 member slots, full stop; a lead sharing
|
||||
# the account is a fact you can see in `leadSeats`, not a reason `free` undercounts spawns that
|
||||
# will, in practice, succeed.
|
||||
#
|
||||
# THE SENTINEL (fleetd #176 stage 2, correcting an inert stage 1 fix): every `subscription:
|
||||
# true` profile that leaves `credentialId` unset shares ONE implicit account-wide credential
|
||||
# id with every other such profile on this host — because a subscription profile doesn't
|
||||
# authenticate with a credential of its own, it authenticates as the operator's own Claude
|
||||
# login, and there is exactly one of those. So on a typical host, `opus` (the lead's profile)
|
||||
# and `sonnet` (the members' profile) are linked automatically, with NOTHING to set here — that
|
||||
# is what makes GOTCHA 3 above work without also writing matching `credentialId:` values on
|
||||
# both. This linkage is not just cosmetic: it is the same key `BackendQuarantine`/cool-off use,
|
||||
# so a usage-limit hit on `opus` now quarantines `sonnet` too (and vice versa) — correct, since
|
||||
# they are one Claude account, but worth knowing before you wonder why an unrelated-looking
|
||||
# profile went quarantined.
|
||||
#
|
||||
# WHEN TO OVERRIDE — set explicit, DIFFERENT `credentialId:` values on two `subscription: true`
|
||||
# profiles only when they are genuinely two separate Claude logins on the same host (a real,
|
||||
# if unusual, setup). An explicit `credentialId` always wins over the sentinel, so this is the
|
||||
# one way to keep two subscription profiles from being treated as one account for lead-seat
|
||||
# counting AND for quarantine/cool-off grouping alike.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage (fleetd #201/#227) — see the key doc above
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".env"] # the default; add ".envrc" explicitly if you want it copied too (CB-148) — never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
# autoCompactWindow: 250000 # opt-in: bound member context; claude-code compacts AT this, opencode within it (model limit.context)
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
# Pin an auto-compact window BELOW the served model's context ceiling. The global
|
||||
# ~/.claude/settings.json value is shared by every ccs instance and the primary, so the
|
||||
# per-profile override belongs here. Equal to the ceiling means auto-compact never fires
|
||||
# before the server rejects the prompt, which kills a worker mid-turn (CB-523).
|
||||
env:
|
||||
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "280000"
|
||||
# CB-402: a second coding-agent kind, proving the PeerLauncher SPI is provider-neutral.
|
||||
# opencode is provider-agnostic and uses NONE of Claude's private seams: no ANTHROPIC_BASE_URL /
|
||||
# SubscriptionGuard (so it needs no `guard` host entry), no --mcp-config / --append-system-prompt.
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → fleet_send →
|
||||
# structured fleet_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
#
|
||||
# CB-508: point an opencode profile at your OWN OpenAI-compatible endpoint (local vLLM, llama.cpp,
|
||||
# LM Studio, TGI…) instead of opencode's gateway. Setting `baseUrl` on a `kind: opencode` profile
|
||||
# makes the bridge emit a custom `provider` block into the generated opencode.json — opencode has
|
||||
# no ANTHROPIC_BASE_URL seam, so this is how the endpoint is pinned.
|
||||
# baseUrl → a bare host:port gets `/v1` appended (where these servers mount the API); a URL that
|
||||
# already has a path is used verbatim, so a custom mount point still works.
|
||||
# model → MUST be "<provider>/<model>". The provider half names the generated block; the model
|
||||
# half must match an id the server reports at /v1/models. One field drives both the
|
||||
# declaration and the `-m` flag, so they cannot drift apart. A bare model name with a
|
||||
# baseUrl set is rejected at spawn rather than silently using the default gateway.
|
||||
# tokenEnv → optional; its value becomes the provider apiKey. Most local servers ignore the key,
|
||||
# so a placeholder is used when unset (the AI SDK still requires a non-empty one).
|
||||
# NOTE: no `guard` entry is needed even with a baseUrl set. The SubscriptionGuard exists to stop a
|
||||
# worker borrowing the primary's Anthropic subscription, and an opencode process has no Anthropic
|
||||
# credential path at all.
|
||||
# opencode-local:
|
||||
# kind: opencode
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
# How an unqualified spawn chooses a profile: fixed (default, reproduces pre-CB-518 behaviour),
|
||||
# round-robin, or weighted. Omitting this key is a strict no-op for existing configs.
|
||||
#
|
||||
# `weighted` IS NOT "cheapest first" — read this before you set weights (CB-589).
|
||||
# It is smooth weighted round-robin: it spreads spawns across EVERY profile that has a free slot,
|
||||
# in weight ratio. It has no idea which profile costs money. So with local:10 / paid:2 you do not
|
||||
# get "use local, overflow to paid" — you get roughly one spawn in six going to the paid profile
|
||||
# while the local box still has a free slot.
|
||||
#
|
||||
# There is a sharper second effect. The policy's running score map lives for the daemon's whole
|
||||
# life. While a profile is at maxLoad it is filtered out and its score FREEZES, so the paid
|
||||
# profiles keep accumulating against it. When the local slot frees up it returns with a stale
|
||||
# score and can LOSE the next pick — a paid spawn while the free box sits idle.
|
||||
#
|
||||
# Until a real cost-first policy exists, the workaround is to make the ratio decisive rather than
|
||||
# proportional: give the free profile a weight so large that it wins every pick it is eligible
|
||||
# for, and paid profiles only ever take genuine overflow. On this host that is local weight 100
|
||||
# against paid weights of ~1.
|
||||
#
|
||||
# The gotcha with that workaround: it expresses a PREFERENCE ORDER through a RATIO knob. Add a
|
||||
# future profile at weight 150 and it silently outranks the free box, with nothing to warn you.
|
||||
# Re-check the weights whenever you add a profile.
|
||||
placement: weighted
|
||||
|
||||
# How long a credential sits out after a BACKEND_EXHAUSTED classification (CB-578 stage B), in
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
#
|
||||
# This does NOT govern the fleetd #201 / #227 backend-error cool-off documented under errorPattern
|
||||
# above — that mechanism is a separate, shorter-lived, NOT-configurable policy (threshold 2 distinct
|
||||
# targets, 60-second window, 60-second cool-off), on purpose: it exists to survive a brief transient
|
||||
# fault, not to replace this 30-minute exhaustion quarantine. Do not conflate the two when reading
|
||||
# fleet_list/fleet_profiles — coolingOffForSeconds and quarantinedForSeconds are independent facts.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded fleetd keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. fleetd checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
# Not every key can move under a running daemon, and the difference is about what already exists
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Fleetd.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
# with nothing in the deferred list — but has NO effect until you restart. Treat it
|
||||
# as deferred in practice, even though today's reload output does not say so.
|
||||
# DEFERRED → accepted into the new config, but the wiring built at startup keeps the old value
|
||||
# until you restart: `lifecycle:`, `leadHeartbeat:`, `guard:`, `worktreeRoot:`,
|
||||
# `spawnReadyTimeoutMs` / `spawnReadyPollMs`, `quarantineCooldownSeconds` (CB-578
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern, errorPattern (fleetd #201 / #227 —
|
||||
# compiled once into a startup pattern map the same way exhaustedPattern is). The
|
||||
# launcher takes a copy of `profiles:` at startup and resolves every spawn out of
|
||||
# that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
# bound, the broker connection is open, and the auth mode decides who may reach the
|
||||
# port that is already listening.
|
||||
#
|
||||
# A changed COLD key refuses the WHOLE reload — not the hot half applied and the cold half warned
|
||||
# about. A half-applied reload would leave the daemon matching no file on disk, which is the worst
|
||||
# thing a reload can do to an operator debugging one. A file that fails to parse or fails a startup
|
||||
# validator is refused the same way, and the running config stays live.
|
||||
# configReload:
|
||||
# enabled: true
|
||||
# intervalSeconds: 10
|
||||
|
||||
# THE FLEET (CB-557) — who the daemon may run, and under which role. This one block replaced four
|
||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||
#
|
||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
||||
# file, the playbook skill and the authz row.
|
||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||
# which is why the two cannot be one field.
|
||||
#
|
||||
# The ROLE IS THE CONTAINING KEY, not a `role:` field. That is not only tidier: a misspelled role
|
||||
# used to parse into a member with no contract at all, while a misspelled pool name here simply
|
||||
# declares nothing.
|
||||
#
|
||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
||||
# with being listed here; the entry key just names the entry.
|
||||
fleet:
|
||||
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
||||
# reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put secrets
|
||||
# here: a later launch step writes this text to a world-readable temp file, and ${ENV} interpolation
|
||||
# is deliberately not supported.
|
||||
charters:
|
||||
architect: |-
|
||||
You are an architect in this fleet. You refine work before anyone builds it:
|
||||
scope, acceptance criteria, risks, and a unit split. You read the repo and
|
||||
write analysis. You never commit production code and never open a PR.
|
||||
A design task is worked by two architects. Design alone first, then exchange
|
||||
and say plainly where you disagree. Do not concede just to agree.
|
||||
dev: |-
|
||||
You implement the one unit you were given, and nothing else. You test it,
|
||||
commit it, and open your own pull request. You never merge.
|
||||
reviewer: |-
|
||||
You review the diff you were given. You report bugs, risks and missing tests.
|
||||
You do not change code.
|
||||
|
||||
# Optional. Template for a member tab's label; {role}, {profile}, {model} and {n} are substituted.
|
||||
# {n} counts per role+profile, so `dev: sonnet #2` really is the second sonnet dev. Because {role}
|
||||
# comes from a closed enum, a generated label can never begin with a lead's tabPrefix.
|
||||
# tabLabel: "{role}: {profile} #{n}"
|
||||
|
||||
# Panes that orchestrate rather than are orchestrated. A lead may now be CREATED as well as
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `profile:` has a SECOND job as of fleetd #176, even for a recognise-only lead you never want
|
||||
# auto-launched: it is also how fleetd learns which account this lead's own session shares. A
|
||||
# `subscription: true` profile bills the operator's Claude account, and the lead itself is always
|
||||
# a live `claude` session on that same account — `maxLoad` never counted that seat. If a lead
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list` reports the
|
||||
# lead's live seat(s) on that worker profile under `leadSeats` — informational only, as of fleetd
|
||||
# #257 it is NEVER subtracted from `free` (see GOTCHA 3, next to `maxLoad:`, in THE WORKERS above,
|
||||
# for why). "Shares the account" is decided by matching `effectiveCredentialId()`, which (fleetd
|
||||
# #176 stage 2 — see THE SENTINEL, next to `credentialId:`, in THE WORKERS above) means: an
|
||||
# explicit, matching `credentialId:` on both, OR — the common case, needing NO extra config — both
|
||||
# being `subscription: true` with `credentialId` left unset, since those all share one implicit
|
||||
# account-wide id. A lead on `opus` and workers on `sonnet` link automatically this way; they do
|
||||
# NOT need the same profile name. Setting `profile:` on an already-running, recognise-only lead is
|
||||
# safe — the daemon only launches the SHORTFALL below `instances`, so naming a profile here does
|
||||
# not, by itself, start anything. Omit it and fleetd has no way to derive the sharing — there is
|
||||
# no other reliable signal on the daemon's side — so that lead's seat never appears in `leadSeats`.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
# inside it restarts — the terminal id changes; the tab, and its label, do not, so no config edit
|
||||
# follows a restart.
|
||||
#
|
||||
# A lead the daemon launches is labelled BY the daemon with this same `tab:` value, so it is found
|
||||
# by the same scan. A lead counts as live only when herdr also reports a running agent in that
|
||||
# tab — a label left behind by a session that died does not block the relaunch, and a tab that is
|
||||
# gone entirely drops out of the next scan rather than being remembered forever.
|
||||
#
|
||||
# An auto-launched lead is NOT a member: it gets no worker reply charter, is never registered with
|
||||
# the session lifecycle (the idle reaper would kill your orchestrator), and stays on the
|
||||
# subscription — ANTHROPIC_BASE_URL/AUTH_TOKEN are stripped from its env whatever the profile says.
|
||||
#
|
||||
# GET THE `tab:` VALUE RIGHT. A pane that does not match any configured `tab:` (a typo, a renamed
|
||||
# tab, a pane no entry names at all) is not recognised as a lead — it resolves as an ordinary
|
||||
# WORKER instead, silently, and every orchestration call it makes (spawn/stop/send/drain) is
|
||||
# refused. There is no error at startup for this: an unmatched pane is simply not a lead. If your
|
||||
# primary suddenly can't spawn or send, check this section first.
|
||||
# leaders:
|
||||
# opus-5.0:
|
||||
# profile: opus # omit to never create this lead, only recognise it
|
||||
# instances: 1 # desired live count; only the shortfall is launched. 0 = off
|
||||
# tab: "lead: opus-5.0" # REQUIRED — the exact tab label this lead lives in
|
||||
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
||||
# # this convention at startup; plays no part in matching a lead
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
||||
# kind: claude # descriptive; reported by fleet_whoami
|
||||
# gpt-sol-5.6:
|
||||
# tab: "lead: gpt-sol-5.6"
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
# architect-2:
|
||||
# profile: sol # a different vendor on purpose — two architects that share a
|
||||
# # model share its blind spots
|
||||
developers:
|
||||
gx10:
|
||||
profile: gx10
|
||||
# reviewers:
|
||||
# gx10:
|
||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Every profile above must have its host listed here.
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- gx01.gw
|
||||
|
||||
# Member credential policy (CB-596, gitea issue #82). A herdr pane runs a LOGIN shell, and that
|
||||
# shell re-sources the operator's own secret store — so a spawned member inherits every credential
|
||||
# the operator's shell holds, not just the ones fleetd means to give it. Measured on this host:
|
||||
# 31 credential names, all set, with only ONE (GITEA_ACCESS_TOKEN) blocked before this — and that
|
||||
# block was a single name hardcoded in HerdrPeerLauncher.java, not driven by this file. This block
|
||||
# replaces that hardcoded shadow with a config-driven list of names.
|
||||
#
|
||||
# ROUND-2 CORRECTION, measured live: the pane-creation env overlay below (applied at tab.create /
|
||||
# pane.split, BEFORE the pane's login shell runs) does NOT survive that login shell for any name
|
||||
# secrets.sh actually exports — the shell re-exports it afterwards and overwrites the sentinel.
|
||||
# Proof: GITEA_ACCESS_TOKEN comes back blocked only because secrets.sh itself carries a guarded
|
||||
# export (`[ -n "${BRIDGED_MEMBER:-}" ] || export GITEA_ACCESS_TOKEN=...`) — that guard, not this
|
||||
# file, is what wins. No other name in `known` below has a matching guard in secrets.sh yet (1
|
||||
# guard measured against 33 export lines there). So today this block's overlay is REAL protection
|
||||
# only for a name secrets.sh does not export, or a peer kind whose pane never runs a login shell —
|
||||
# for everything secrets.sh exports and guards, the guard in secrets.sh (out of scope for this
|
||||
# ticket) is what actually blocks it, not this list. An exec-time fix (winning after the login
|
||||
# shell finishes, before the agent process starts) was attempted and found to have no seam in the
|
||||
# current herdr protocol — AgentControl.start takes a fixed `kind` (herdr resolves the executable)
|
||||
# plus trailing CLI args for that binary, not an arbitrary argv or an env map; only tab.create /
|
||||
# pane.split accept `env`, and that is this same pane-creation overlay. See gitea #82 for the open
|
||||
# design question this leaves.
|
||||
#
|
||||
# DENY-BY-DEFAULT, NOT A DENY-LIST. A deny-list (name the bad ones, let everything else through) is
|
||||
# silently wrong the moment the operator's store gains a new secret — nothing would ever report it.
|
||||
# Deny-by-default inverts that: `known` bounds the blast radius to names actually enumerated below,
|
||||
# and EVERY one of them is blocked UNLESS it is also in `allow`. Omitting this block entirely (the
|
||||
# shipped default) blocks NOTHING — unlike most optional blocks in this file, absence here is a real
|
||||
# gap, not a safe "feature off". A name that is neither `known` nor `allow`-ed is not silently let
|
||||
# through either: the daemon logs a WARN naming any credential-shaped env var it finds on neither
|
||||
# list (never its value), so a secret added to the store later does not go unnoticed forever.
|
||||
#
|
||||
# policy → "deny-by-default" (the default; also accepted spelled "deny-list") overlays each
|
||||
# known-but-not-allowed name BEFORE the pane's login shell runs — real protection only
|
||||
# where that shell does not re-export the name (see ROUND-2 CORRECTION above). An
|
||||
# unrecognized value refuses to start, naming it.
|
||||
# policy → "allow-list" (CB-633) moves the control to a per-spawn ZDOTDIR directory the daemon
|
||||
# generates and passes through tab.create's env map. Each generated startup file sources
|
||||
# its ~/ counterpart FIRST and then runs the scrub, so the scrub happens after the
|
||||
# operator's whole chain and no sourced file can undo it.
|
||||
# The scrub is sourced from BOTH the generated .zshrc and the generated .zlogin, because
|
||||
# herdr does not open the same kind of shell everywhere: macOS panes run a LOGIN zsh (so
|
||||
# .zlogin runs), Linux panes run a plain interactive zsh (so .zlogin never runs at all).
|
||||
# A scrub in .zlogin alone would be a control that silently does nothing on Linux.
|
||||
# The allow-list is DERIVED, never typed:
|
||||
# every profile's tokenEnv/gitTokenEnv/gitHostEnv values and env-map keys, plus an
|
||||
# infrastructure set (PATH HOME SHELL TERM LANG LC_* TMPDIR USER LOGNAME PWD SHLVL EDITOR
|
||||
# PAGER JAVA_HOME XDG_* ZDOTDIR), plus whatever keys this spawn's own env overlay carries.
|
||||
# Adding a profile can therefore only widen the list, never break another spawn's scrub.
|
||||
# Under this policy `known`/`allow` below become REPORTING ONLY — they feed the gap WARN,
|
||||
# they are no longer a control. If the member's login shell is NOT zsh, the daemon logs a
|
||||
# loud WARN saying protection is off and falls back to deny-by-default's overlay.
|
||||
# Each pane writes a scrub-report.txt naming how many variables it kept of how many it
|
||||
# saw; the daemon logs that "allowed N of M" line when the pane stops. If the report is
|
||||
# MISSING the daemon logs a WARN instead — the scrub cannot then be confirmed to have
|
||||
# run, and a silently-dead control is exactly what this policy exists to prevent.
|
||||
# allow → credential names a member legitimately needs. Under deny-by-default, left OUT of the
|
||||
# pane's env overlay entirely, so the value the pane's own (login) shell exports passes
|
||||
# through untouched. Under allow-list: reporting only.
|
||||
# known → every credential name the operator's store is known to export. Under deny-by-default,
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAgentEnv → whether SSH_AUTH_SOCK may pass through under allow-list ("inherit") or is omitted
|
||||
# from the member environment ("omit", the default). Omitting it only omits the
|
||||
# inherited ssh-agent path. It discourages automatic use of the operator's agent.
|
||||
# It does not deny same-user access to that socket. It also does not block SSH keys that
|
||||
# are readable on disk. Git over SSH may still work from inside a member. Keep the block:
|
||||
# it is correct and costs nothing, but it is not a control. A member runs as the same OS
|
||||
# user as the lead. Inside one uid, ordinary Unix permissions provide no meaningful
|
||||
# confidentiality boundary. A real boundary needs a different OS user or OS-level
|
||||
# confinement, such as a container or VM. That is the open question in fleetd #184.
|
||||
#
|
||||
# Still do not set this to "inherit" casually. SSH_AUTH_SOCK is a live handle to YOUR
|
||||
# ssh-agent, so a member holding it can sign with EVERY key the agent holds. It sits in
|
||||
# no secret file and looks like no credential, which is why it slipped past three
|
||||
# earlier tickets (gitea #110). Blocking it does not contain a member, but allowing it
|
||||
# hands one a signing capability for no gain — the block costs nothing, so keep it.
|
||||
#
|
||||
# Both halves of this are measured, not argued. 2026-08-28: a member with
|
||||
# SSH_AUTH_SOCK blanked pushed to the forge over SSH successfully, because `ssh -G`
|
||||
# resolves an IdentityFile outside ~/.ssh that is readable and has no passphrase. An
|
||||
# earlier version of this comment claimed blocking the socket BREAKS git over SSH. It
|
||||
# does not. That claim came from looking only in ~/.ssh, which holds nothing but four
|
||||
# `Include` lines — looking in one place and concluding about the whole host.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAgentEnv: omit # allow-list only; see the sshAgentEnv note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
# - WORKER_GITEA_TOKEN # the repo-scoped forge token a member needs to open its own PR (CB-302)
|
||||
# - CONTEXT7_TOKEN # already decided as allowed by CB-593
|
||||
# - GITEA_HOST # not a credential — a hostname, paired with the forge token above
|
||||
# known:
|
||||
# - AI_GATEWAY_TOKEN
|
||||
# - BESZEL_ADMIN_EMAIL
|
||||
# - BESZEL_ADMIN_PASSWORD
|
||||
# - BESZEL_HUB_URL
|
||||
# - BESZEL_KEY
|
||||
# - BESZEL_UNIVERSAL_TOKEN
|
||||
# - BRAIN_MCP_TOKEN
|
||||
# - CF_ACCOUNT_ID
|
||||
# - CF_API_TOKEN
|
||||
# - CF_USER_TOKEN
|
||||
# - CONFLUENCE_API_TOKEN
|
||||
# - CONFLUENCE_USERNAME
|
||||
# - CONTEXT7_TOKEN
|
||||
# - GITEA_ACCESS_TOKEN
|
||||
# - GITEA_HOST
|
||||
# - GITLAB_OAUTH_CLIENT_SECRET
|
||||
# - GITLAB_PERSONAL_ACCESS_TOKEN
|
||||
# - GRAFANA_ADMIN_PASSWORD
|
||||
# - GRAFANA_ADMIN_USER
|
||||
# - HASS_TOKEN
|
||||
# - HW_PASSWORD
|
||||
# - HW_USER
|
||||
# - LTMS_API_KEY
|
||||
# - MEMORY_MCP_TOKEN
|
||||
# - METRICS_PUSH_TOKEN
|
||||
# - OPENCODE_AUTOMODE_MODEL
|
||||
# - TELEGRAM_BOT_TOKEN
|
||||
# - TELEGRAM_CHAT_ID
|
||||
# - TS_API_KEY
|
||||
# - TS_AUTHKEY
|
||||
# - WORKER_GITEA_TOKEN
|
||||
|
||||
# Spawn-readiness gate (CB-306). The launcher blocks until the worker's herdr status is
|
||||
# injectable (IDLE/BLOCKED/DONE) or the timeout elapses. 0 disables the gate.
|
||||
# NOTE: keys are camelCase — config is bound by plain Jackson with no naming strategy and
|
||||
# unknown keys are ignored, so a snake_case key would be silently dropped (default kept).
|
||||
# spawnReadyTimeoutMs: 20000
|
||||
# spawnReadyPollMs: 300
|
||||
|
||||
# Worktree provisioning root (CB-301-ext). Where per-worker git worktrees are checked out so
|
||||
# each worker owns an isolated branch instead of sharing the primary's tree. Omit to default
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Worktree group sharing (fleetd #185 stage 3). OPTIONAL, off by default. Names an OS group
|
||||
# that a provisioned worktree's repo is made group-writable for (git config
|
||||
# core.sharedRepository group, plus a one-time chgrp/chmod/setgid fix-up), so a member spawned
|
||||
# under a DIFFERENT OS user (see memberHerdrSocket) can write its own worktree, its
|
||||
# per-worktree git metadata, and its own commit objects — without it, every file GitWorktrees
|
||||
# creates is owned by fleetd's own uid and unwritable by another user.
|
||||
# CAUTION: this isolates credentials, not the repository — a member in the group can still
|
||||
# write the operator's git objects and refs in the shared repo. The operator running fleetd
|
||||
# must already be a member of the named group, or every provisioning spawn fails loudly.
|
||||
#
|
||||
# fleetd #213: this is also the ONE group the memberCredentials.policy: allow-list ZDOTDIR scrub
|
||||
# reuses when memberHerdrSocket is set — deliberately not a second config key. Under
|
||||
# memberHerdrSocket, the scrub directory is generated under worktreeRoot (never java.io.tmpdir,
|
||||
# which the member OS user cannot reach) and shared read-only with this group. If worktreeGroup
|
||||
# is unset while memberHerdrSocket is set, the scrub cannot be guaranteed reachable by the member,
|
||||
# so fleetd falls back to the weaker CB-596 sentinel overlay instead (a WARN names the gap).
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
# contextCap → force-release a session after this many delegated turns
|
||||
# drainTimeoutSeconds → seconds to wait for BUSY sessions on shutdown before forced teardown
|
||||
# clearAfterTurn → whether a reusable worker discards its conversation context after every
|
||||
# completed delegated turn (default false). Works for claude-code workers
|
||||
# only — any other peer kind (e.g. opencode) logs "context reset is
|
||||
# unsupported for peer kind …" once and the reset is a no-op.
|
||||
# lifecycle:
|
||||
# idleTtlSeconds: 300
|
||||
# contextCap: 10
|
||||
# drainTimeoutSeconds: 5
|
||||
# clearAfterTurn: false
|
||||
|
||||
# Durable reply delivery (CB-307 Stage 2). OMIT this block entirely to keep the default
|
||||
# in-memory, soft-state reply inbox (late worker replies are held only until a daemon bounce).
|
||||
# Set a broker uri to swap in the AMQP-backed inbox: worker replies with no open send are held
|
||||
# on a durable per-target queue (agent.<target>.inbox) and survive a restart — the broker
|
||||
# redelivers anything the primary had not yet drained. Production default is LavinMQ; a stock
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# uriEnv → CB-151: name of a host env var holding the AMQP URI, preferred over `uri` (wins
|
||||
# whenever set). The URI carries `user:pass@` inline, so naming a variable keeps the
|
||||
# password out of fleetd.yaml — same pattern as auth.tokenEnv/Profile.tokenEnv. A
|
||||
# uriEnv that resolves to an unset or blank variable is treated as NOT configured and
|
||||
# the daemon falls back to the in-memory inbox, warning loudly.
|
||||
# prefetch → CB-527: consumer basicQos, capping how many unacked messages the inbox holds
|
||||
# in-heap per owned target (the rest sits on the broker's durable queue instead of
|
||||
# growing the JVM heap). Default 32 when omitted.
|
||||
# broker:
|
||||
# uriEnv: LAVINMQ_URI
|
||||
# prefetch: 32
|
||||
|
||||
# Shared cross-host LEADER coordination broker. OMIT this block to leave lead-to-lead messaging
|
||||
# off entirely (config-only in this ticket — nothing here wires it into a live LeadMailbox yet).
|
||||
# This is a SEPARATE AMQP vhost from `broker:` above: member/worker inboxes always stay on the
|
||||
# per-fleet `broker:` vhost, and this vhost carries only leader-to-leader traffic, so two fleets
|
||||
# whose members must never see each other can still share one coordination vhost for their leads.
|
||||
# uriEnv → name of a host env var holding the coordination AMQP URI, same convention as
|
||||
# broker.uriEnv (keeps the credential out of fleetd.yaml). Wins over `uri` when set.
|
||||
# selfId → this daemon's own lead coord-id — the name its mailbox is owned under
|
||||
# (lead.<selfId>.inbox), e.g. "mac-opus" or "fleet01-lead". Must be globally unique
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
# terminal → pin the primary's herdr terminal id. Omit to learn it from the connection on
|
||||
# the first orchestration-side MCP call (the normal case). An off-host or
|
||||
# non-herdr primary leaves this unresolved → the loop is a no-op and delivery
|
||||
# degrades to pull; the reply is still never lost.
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just fleetd-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off fleet_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
# primary:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
@@ -5,17 +5,17 @@
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
<groupId>dev.ltms</groupId>
|
||||
<artifactId>bridged</artifactId>
|
||||
<artifactId>fleetd</artifactId>
|
||||
<version>1.0.0</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<name>bridged</name>
|
||||
<description>claude-bridge message server: sole gateway between primary/worker Claude sessions and herdr</description>
|
||||
<name>fleetd</name>
|
||||
<description>fleet message server: sole gateway between a lead session, its members, and herdr</description>
|
||||
|
||||
<properties>
|
||||
<maven.compiler.release>25</maven.compiler.release>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<mainClass>dev.ltms.bridged.Bridged</mainClass>
|
||||
<mainClass>dev.ltms.fleet.Fleetd</mainClass>
|
||||
|
||||
<jackson.version>2.19.0</jackson.version>
|
||||
<javalin.version>6.7.0</javalin.version>
|
||||
@@ -28,6 +28,8 @@
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
<archunit.version>1.5.0</archunit.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -44,6 +46,12 @@
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
|
||||
fleetd #206: org.xerial:sqlite-jdbc 3.53.4.0 (added for OpenCodeSessionDiscovery) — the
|
||||
only known advisory against this artifact is CVE-2023-32697 (RCE via an attacker-controlled
|
||||
JDBC URL), fixed in 3.41.2.2; 3.53.4.0 is well past that fix and OSV.dev reports no open
|
||||
advisory against it. Checked via the OSV.dev API (no Mend.io/JetBrains IDE MCP mount
|
||||
available from this worktree) on 2026-08-31.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
@@ -106,7 +114,7 @@
|
||||
</dependency>
|
||||
|
||||
<!-- MCP server: the SERVER face. Streamable-HTTP servlet mounted on Javalin's Jetty at
|
||||
/mcp, exposing bridge_send/bridge_reply/bridge_status as thin adapters over REST. -->
|
||||
/mcp, exposing fleet_send/fleet_reply/fleet_status as thin adapters over REST. -->
|
||||
<dependency>
|
||||
<groupId>io.modelcontextprotocol.sdk</groupId>
|
||||
<artifactId>mcp</artifactId>
|
||||
@@ -123,6 +131,17 @@
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #206: opencode moved its session store from a JSON tree to SQLite
|
||||
(opencode.db). This is the JDBC driver OpenCodeSessionDiscovery uses to read it
|
||||
read-only. Ships bundled native libraries (linux/mac/windows, several archs), so it
|
||||
is a heavier jar than most deps here — see the pom's dependency-security note below
|
||||
for the size/CVE tradeoff actually measured. -->
|
||||
<dependency>
|
||||
<groupId>org.xerial</groupId>
|
||||
<artifactId>sqlite-jdbc</artifactId>
|
||||
<version>${sqlite-jdbc.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -158,10 +177,21 @@
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #131: package-boundary and cycle enforcement (PackageCyclesTest). -->
|
||||
<dependency>
|
||||
<groupId>com.tngtech.archunit</groupId>
|
||||
<artifactId>archunit-junit5</artifactId>
|
||||
<version>${archunit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
<finalName>bridged</finalName>
|
||||
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
||||
launchd plist together. The installed plist names fleetd/target/fleetd.jar and
|
||||
KeepAlive is armed, so this name, the plist, and the wrapper must move as one. -->
|
||||
<finalName>fleetd</finalName>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
@@ -203,7 +233,7 @@
|
||||
</configuration>
|
||||
</plugin>
|
||||
|
||||
<!-- Runnable fat jar: java -jar target/bridged.jar -->
|
||||
<!-- Runnable fat jar: java -jar target/fleetd.jar -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-shade-plugin</artifactId>
|
||||
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
+2
-2
@@ -1,9 +1,9 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||
*
|
||||
* <p>Most of these rules are already true de facto — {@code BridgeMcp} derives a worker's identity
|
||||
* <p>Most of these rules are already true de facto — {@code FleetMcp} derives a worker's identity
|
||||
* from the connection rather than reading it from an argument, so a worker has never been able to
|
||||
* reply <em>as</em> another worker over MCP. What was missing is that the REST surface trusted the
|
||||
* session id in the URL path, and neither surface checked role at all. This class makes the
|
||||
+51
-39
@@ -1,16 +1,18 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Resolves every caller to a {@link Principal}, for both entry paths into the core (CB-501).
|
||||
*
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code BridgeMcp}
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code FleetMcp}
|
||||
* calls the service layer directly and is mounted as a raw servlet (so it never passes through a
|
||||
* Javalin filter), while the REST routes historically resolved no identity at all. Both now
|
||||
* delegate here, so the authorization rules are stated once instead of drifting apart.
|
||||
@@ -57,17 +59,19 @@ public final class CallerResolver {
|
||||
* <p>Like {@link #leadTerminals}, a supplier rather than a fixed map, so a binding injected
|
||||
* after startup — when the later spawn lifecycle establishes a live architect session, or an
|
||||
* operator pins one — takes effect without a restart. Consulted per resolve; today's wiring
|
||||
* in {@code Bridged} reads a constant from config, which is the degenerate live case.
|
||||
* in {@code Fleetd} reads a constant from config, which is the degenerate live case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> architectTerminals;
|
||||
private final Function<String, MemberRole> memberSlotRoles;
|
||||
private final Function<String, String> memberSlotNames;
|
||||
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. */
|
||||
public CallerResolver(ConnectionIdentity identity) {
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity) {
|
||||
this(identity, false, null, Map.of());
|
||||
}
|
||||
|
||||
/** As {@link #CallerResolver(ConnectionIdentity, boolean, String, Map)} with no leads pinned. */
|
||||
public CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token) {
|
||||
/** As {@link #CallerResolver(ConnectionIdentity, boolean, String, Map)} with no leads pinned. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token) {
|
||||
this(identity, tokenMode, token, Map.of());
|
||||
}
|
||||
|
||||
@@ -82,8 +86,8 @@ public final class CallerResolver {
|
||||
* @param pinnedPrimaryTerminal the primary's own herdr {@code terminal_id}
|
||||
* ({@code null}/blank = unpinned)
|
||||
*/
|
||||
public static CallerResolver pinnedTo(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token, String pinnedPrimaryTerminal) {
|
||||
static CallerResolver pinnedTo(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token, String pinnedPrimaryTerminal) {
|
||||
return new CallerResolver(identity, tokenMode, token,
|
||||
pinnedPrimaryTerminal == null || pinnedPrimaryTerminal.isBlank()
|
||||
? Map.of() : Map.of(pinnedPrimaryTerminal, "primary"));
|
||||
@@ -98,21 +102,11 @@ public final class CallerResolver {
|
||||
* {@link Role#PRIMARY} — rather than a worker. Empty = nothing pinned,
|
||||
* so every pane resolves as a worker.
|
||||
*/
|
||||
public CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals) {
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals) {
|
||||
this(identity, tokenMode, token, fixed(leadTerminals), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Map-form of both registries (CB-548): lead terminals and the initial architect terminal
|
||||
* bindings, each snapshotted at construction (a handed-over map is not offered as live state).
|
||||
*/
|
||||
public CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals,
|
||||
Map<String, String> architectTerminals) {
|
||||
this(identity, tokenMode, token, fixed(leadTerminals), fixed(architectTerminals));
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-registry form: {@code leadTerminals} is consulted on every resolve, so leads discovered
|
||||
* after startup (CB-531's tab scan) take effect without a restart.
|
||||
@@ -121,26 +115,26 @@ public final class CallerResolver {
|
||||
* {@link #pinnedTo}: {@code Map} and {@code Supplier} overloads are ambiguous for a literal
|
||||
* {@code null}.
|
||||
*/
|
||||
public static CallerResolver withLeads(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token,
|
||||
Supplier<Map<String, String>> leadTerminals) {
|
||||
static CallerResolver withLeads(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token,
|
||||
Supplier<Map<String, String>> leadTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-registry form for both {@code leadTerminals} and the CB-548 architect registry: both
|
||||
* are consulted on every resolve, so a slot binding injected after startup takes effect
|
||||
* without a restart.
|
||||
* Live registry form that can confirm a bound slot is an architect slot.
|
||||
*
|
||||
* <p>A static factory rather than a constructor overload, for the same reason as
|
||||
* {@link #pinnedTo}: too many {@code Map}/{@code Supplier} combinations to make {@code null}
|
||||
* unambiguous.
|
||||
* <p>This is the only public construction path. It keeps terminal bindings and slot roles in
|
||||
* the same {@link MemberRegistry}, so a configured architect can resolve as an architect.
|
||||
*/
|
||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals, architectTerminals);
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
MemberRegistry members) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
||||
members == null ? null : members::snapshot,
|
||||
members == null ? null : members::roleForSlot,
|
||||
members == null ? null : members::nameForSlot);
|
||||
}
|
||||
|
||||
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
||||
@@ -151,6 +145,21 @@ public final class CallerResolver {
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, memberSlotRoles, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles,
|
||||
Function<String, String> memberSlotNames) {
|
||||
if (tokenMode && (token == null || token.isBlank())) {
|
||||
throw new IllegalArgumentException(
|
||||
"auth.mode=token requires a non-empty token; check that the env var named by "
|
||||
@@ -161,6 +170,8 @@ public final class CallerResolver {
|
||||
this.expectedToken = tokenMode ? token.getBytes(StandardCharsets.UTF_8) : null;
|
||||
this.leadTerminals = leadTerminals == null ? Map::of : leadTerminals;
|
||||
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
||||
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
||||
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -206,11 +217,12 @@ public final class CallerResolver {
|
||||
return Principal.leader(lead, c.terminal(), c.pid());
|
||||
}
|
||||
String slot = architectTerminals.get().get(c.terminal());
|
||||
if (slot != null) {
|
||||
if (slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT) {
|
||||
// The config/live binding names this pane as an architect slot's own. Same
|
||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||
// Checked before the generic worker fallback, per the CB-548 precedence order.
|
||||
return Principal.architect(slot, c.terminal(), c.pid());
|
||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||
// escalating a dev or reviewer into an architect. Checked before the worker fallback.
|
||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||
}
|
||||
@@ -223,7 +235,7 @@ public final class CallerResolver {
|
||||
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (BridgedConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
/** A slot held before a member process starts. */
|
||||
record SlotReservation(String slot, String profile) { }
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return role; // no registry configured — nothing to bind against, so the request stands
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
// no registry configured — nothing to validate against, so nothing is refused
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||
*
|
||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||
* normal operation this fallback should not happen once a reservation has been bound.
|
||||
* It remains the honest answer if a caller has no reservation, or if binding a
|
||||
* reservation unexpectedly fails.
|
||||
*/
|
||||
MemberRole acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
|
||||
/**
|
||||
* Refuse an acquire before anything spawns when {@code role} requires a live slot binding and
|
||||
* no configured slot carries {@code profile} (CB-619 / fleetd #123). A no-op for a role with
|
||||
* no slot-binding semantics.
|
||||
*
|
||||
* @throws IllegalArgumentException naming the role, the profile, and the pools that do carry it
|
||||
*/
|
||||
void requireSlotFor(MemberRole role, String profile);
|
||||
|
||||
/** Reserve a matching slot before launch, or refuse before a charter can be delivered. */
|
||||
SlotReservation reserve(MemberRole role, String profile);
|
||||
|
||||
/** Convert a reservation into a live terminal binding. */
|
||||
boolean bind(SlotReservation reservation, String terminal);
|
||||
|
||||
/** Return an unbound reservation after a failed launch. */
|
||||
void release(SlotReservation reservation);
|
||||
}
|
||||
+148
-8
@@ -1,12 +1,16 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
@@ -29,7 +33,9 @@ import java.util.Map;
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
*/
|
||||
public final class MemberRegistry {
|
||||
public final class MemberRegistry implements MemberLifecycle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MemberRegistry.class);
|
||||
|
||||
/**
|
||||
* One flattened {@code fleet:} entry.
|
||||
@@ -50,11 +56,13 @@ public final class MemberRegistry {
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(BridgedConfig.Fleet fleet) {
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
@@ -89,7 +97,7 @@ public final class MemberRegistry {
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code bridge_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* {@code fleet_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
@@ -125,6 +133,12 @@ public final class MemberRegistry {
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
@@ -155,7 +169,7 @@ public final class MemberRegistry {
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
if (terminalToSlot.containsValue(slot) || reservedSlots.contains(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
@@ -189,4 +203,130 @@ public final class MemberRegistry {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind only architect sessions to a free slot with the resolved profile.
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*
|
||||
* <p>CB-619 / fleetd #123: the return value is the role this session actually holds, and the
|
||||
* caller is required to record THAT — never the requested {@code role} — on the session. Before
|
||||
* this fix the caller kept the requested role regardless of whether the bind below succeeded, so
|
||||
* a demoted session's {@code GET /members} row still said {@code "architect"} while
|
||||
* {@code fleet_whoami} (which reads the live binding, not the request) correctly said
|
||||
* {@code "worker"} — three sources of truth that disagreed about one live member, silently.
|
||||
*/
|
||||
@Override
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return role;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return MemberRole.ARCHITECT;
|
||||
}
|
||||
}
|
||||
// fleetd #123: at least WARN — a role downgrade that the roster must now also reflect is
|
||||
// not routine bookkeeping. requireSlotFor already refuses the config-gap case (no slot at
|
||||
// all carries this profile) before a process ever spawns; reaching here means the config DID
|
||||
// carry a matching slot but every one of them was already bound to a different terminal — a
|
||||
// race this pre-spawn check cannot close on its own (see requireSlotFor's javadoc).
|
||||
log.warn("member slot: no free architect slot for profile={} terminal={}; holding the session "
|
||||
+ "as {} instead of the architect it asked for — every configured slot for this "
|
||||
+ "profile is already bound to a different terminal", profile, terminal,
|
||||
MemberRole.DEV.wireName());
|
||||
return MemberRole.DEV;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||
* documented operator override, not a defect.
|
||||
*
|
||||
* <p>This closes the config-gap case, not the live-capacity case: a profile that DOES carry a
|
||||
* slot can still lose the race to a concurrent spawn between this check and the actual
|
||||
* {@link #bind}, which is why {@link #acquired} must still answer honestly even after this
|
||||
* check has passed.
|
||||
*/
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return;
|
||||
}
|
||||
boolean hasSlot = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.anyMatch(e -> Objects.equals(profile, e.profile()));
|
||||
if (hasSlot) {
|
||||
return;
|
||||
}
|
||||
List<String> pools = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.map(Entry::profile)
|
||||
.distinct()
|
||||
.toList();
|
||||
throw new IllegalArgumentException(
|
||||
"no " + role.wireName() + " slot for profile '" + profile + "' — an architect's "
|
||||
+ "identity IS the slot it is bound to, so there is nothing to bind this "
|
||||
+ "session's identity to. fleet." + role.configKey() + " carries profiles: "
|
||||
+ (pools.isEmpty() ? "(none configured)" : String.join(", ", pools))
|
||||
+ "; add profile '" + profile + "' there, or spawn " + role.wireName()
|
||||
+ " on one of those profiles instead");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if ((profile == null || profile.isBlank() || Objects.equals(profile, entry.profile()))
|
||||
&& !terminalToSlot.containsValue(entry.key()) && reservedSlots.add(entry.key())) {
|
||||
return new SlotReservation(entry.key(), entry.profile());
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("no free architect slot for profile '" + profile
|
||||
+ "' — every matching slot is already bound or reserved");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
if (reservation == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!reservedSlots.remove(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
if (!isSlot(reservation.slot()) || terminalToSlot.containsKey(terminal)
|
||||
|| terminalToSlot.containsValue(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
terminalToSlot.put(terminal, reservation.slot());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
if (reservation != null) {
|
||||
synchronized (terminalToSlot) {
|
||||
reservedSlots.remove(reservation.slot());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
String slot = slotForTerminal(terminal);
|
||||
if (slot != null) {
|
||||
unbind(slot, terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
+15
-5
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* A resolved caller: its {@link Role}, and — for a worker — the herdr {@code terminal_id} that
|
||||
@@ -39,16 +39,16 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
*
|
||||
* <p>Carries {@link Role#PRIMARY}: a lead <em>is</em> a primary as far as authorization goes,
|
||||
* so every existing {@code isPrimary()} gate keeps working unchanged and the role table needed
|
||||
* no new entry. The name is reporting only — it lets {@code bridge_whoami} say <em>which</em>
|
||||
* no new entry. The name is reporting only — it lets {@code fleet_whoami} say <em>which</em>
|
||||
* lead is asking once more than one is configured.
|
||||
*
|
||||
* <p><strong>CB-532: a lead now carries the terminal it was matched by.</strong> Under CB-530 it
|
||||
* deliberately did not, because {@code terminal} meant "which worker pane" everywhere and a
|
||||
* non-null one would have enrolled the lead in the worker presence map. That reading was what
|
||||
* made a lead unaddressable: {@link #ownsSession} could never be true for it, so
|
||||
* {@code bridge_reply} was refused and one lead could send to another but never be answered.
|
||||
* {@code fleet_reply} was refused and one lead could send to another but never be answered.
|
||||
* The terminal now means "which pane is this caller", the presence map keys on
|
||||
* {@link #isWorker()} instead, and a lead is a peer that can both send and receive.
|
||||
* {@link #isSpawnedMember()} instead, and a lead is a peer that can both send and receive.
|
||||
*/
|
||||
public static Principal leader(String name, String terminal, long pid) {
|
||||
return new Principal(Role.PRIMARY, terminal, pid, name);
|
||||
@@ -63,7 +63,7 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
* An architect (CB-548), identified by the slot it occupies and the pane bound to it.
|
||||
*
|
||||
* <p>Carries {@link Role#ARCHITECT}. {@code slotName} is reporting only — it lets
|
||||
* {@code bridge_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* {@code fleet_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* (future) spawn lifecycle reads a profile back from. Identity is the {@code terminal}: like a
|
||||
* worker's it comes from the connection and the live terminal→slot binding, so
|
||||
* {@code ownsSession} works exactly as it does for a worker — an architect acts as its own
|
||||
@@ -85,6 +85,16 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
return role == Role.WORKER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller is a spawned member with its own pane.
|
||||
*
|
||||
* <p>Both workers and architects are spawned members. A lead is excluded because recording it
|
||||
* as present would count it as an available member in the roster.
|
||||
*/
|
||||
public boolean isSpawnedMember() {
|
||||
return role == Role.WORKER || role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isAnonymous() {
|
||||
return role == Role.ANONYMOUS;
|
||||
}
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* What a caller is allowed to be on the bus (CB-501).
|
||||
+62
-28
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.config;
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -16,7 +16,7 @@ import java.util.function.Supplier;
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link BridgedConfig}, and read through {@link #get()} at the point
|
||||
* <p>Consumers hold this, not a {@link FleetConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
@@ -27,16 +27,27 @@ import java.util.function.Supplier;
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool and {@code tabLabel}), {@code placement:}, and an existing
|
||||
* profile's {@code weight} / {@code maxLoad}. Those three are read through a supplier on
|
||||
* {@code CompositePeerLauncher}, which is what makes them hot — not the fact that they are
|
||||
* config.</li>
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Fleetd.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code guard:},
|
||||
* {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl} and the rest.
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), {@code errorPattern} (fleetd #201 Unit 5 — compiled once into
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup, the same way), and the rest.
|
||||
* {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
@@ -56,30 +67,30 @@ import java.util.function.Supplier;
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<BridgedConfig> current;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
|
||||
public ConfigRef(Path path, BridgedConfig initial) {
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(BridgedConfig cfg) {
|
||||
public static ConfigRef fixed(FleetConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public BridgedConfig get() {
|
||||
public FleetConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
@@ -119,7 +130,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart bridged to apply them.";
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
@@ -140,16 +151,17 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
BridgedConfig old = current.get();
|
||||
BridgedConfig fresh;
|
||||
FleetConfig old = current.get();
|
||||
FleetConfig fresh;
|
||||
try {
|
||||
fresh = BridgedConfig.load(path);
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
@@ -172,7 +184,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
private static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
@@ -180,6 +192,9 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.memberHerdrSocket(), fresh.memberHerdrSocket())) {
|
||||
changed.add("memberHerdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
@@ -192,7 +207,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
private static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
@@ -210,9 +225,15 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
Map<String, BridgedConfig.Profile> before =
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, FleetConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, BridgedConfig.Profile> after =
|
||||
Map<String, FleetConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
@@ -231,7 +252,7 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
BridgedConfig.Profile now = after.get(name);
|
||||
FleetConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
@@ -245,10 +266,12 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight} and {@code maxLoad} are excluded because those are
|
||||
* read live by the placement policy and really do take effect on the next spawn.
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(BridgedConfig.Profile a, BridgedConfig.Profile b) {
|
||||
private static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
@@ -258,12 +281,23 @@ public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
// CB-634: the IDE MCP mount is a launch flag, fixed at spawn like mcpUrl — a
|
||||
// reload changes it only for members spawned after, so a changed value is deferred.
|
||||
&& Objects.equals(a.ideMcpUrl(), b.ideMcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription());
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern())
|
||||
// fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error
|
||||
// pattern map at startup (see BackendErrorPatternLookup wiring), the same way
|
||||
// exhaustedPattern is — a reload never re-reads it either.
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern());
|
||||
}
|
||||
}
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.config;
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -11,7 +11,7 @@ import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Polls {@code bridged.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* Polls {@code fleetd.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* (CB-559). Opt-in through {@code configReload.enabled}.
|
||||
*
|
||||
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
||||
File diff suppressed because it is too large
Load Diff
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user