Compare commits
407 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| a22480c117 | |||
| 24f404f989 | |||
| 17af61e8dd | |||
| 6af87b6ad6 | |||
| fc655e78c2 | |||
| 11c3ff67b6 | |||
| d867c87100 | |||
| b0c4cedfab | |||
| 85417d5215 | |||
| 4accc746bd | |||
| 21c4c8cbef | |||
| 430f5b0dae | |||
| 42731833d0 | |||
| 65acf066ad | |||
| ee8f570fd7 | |||
| fa3f910d44 | |||
| 46ac6e4e38 | |||
| 3bfa82839b | |||
| 65ccf2e4ad | |||
| 7a3b27f76f | |||
| 82e7be564c | |||
| c5e24197bf | |||
| bcb402b688 | |||
| 7c4170ff6d | |||
| a6095743f0 | |||
| 66e776d178 | |||
| 26f64cba45 | |||
| 51047848f1 | |||
| d8c0b657e8 | |||
| 312c0584ce | |||
| da2625acfb | |||
| 97d9cebc59 | |||
| 3ce76a5d69 | |||
| 2e138a199b | |||
| 450a5ed9c5 | |||
| edabccd885 | |||
| 1dbe3a03fc | |||
| 6058b8472b | |||
| f29968c334 | |||
| 7b98cca967 | |||
| 2757bc7185 | |||
| 7655f1b51a | |||
| a5efb7c676 | |||
| 5a8cf4cb4d | |||
| a3fc7e4df8 | |||
| d811b30df3 | |||
| 3f4ac2b24e | |||
| 4644359128 | |||
| 95a8dbcea9 | |||
| 83b50753fe | |||
| 6f968d59a4 | |||
| d432df8e5c | |||
| 4b822731e6 | |||
| e38eac1a33 | |||
| 8101290933 | |||
| 6e7fc12f89 | |||
| 50df14a50f | |||
| ccd882f2fc | |||
| ecc590f344 | |||
| 3b7cdf9365 | |||
| 9b50dd69d8 | |||
| 08f1a79800 | |||
| 51bdec22e7 | |||
| d43f670285 | |||
| c135583402 | |||
| 8d2893b67e | |||
| 555715ced9 | |||
| 446a9d395c | |||
| 8311f2db7f | |||
| f4570ff274 | |||
| 3aea4e1ec6 | |||
| 516366b796 | |||
| d82d0157cb | |||
| 1deb0c90d4 | |||
| 41a3114d03 | |||
| 76e4b577a0 | |||
| 03473a286b | |||
| 5cabd09705 | |||
| cc47672b7c | |||
| 37edd9134b | |||
| 80f167b1f7 | |||
| bd6547fca3 | |||
| 7e97f5bff5 | |||
| 7d41ccccee | |||
| bf616e192a | |||
| f8bd5d0c51 | |||
| ac40de1d30 | |||
| 7930a31b94 | |||
| 850fb12807 | |||
| b14b66ab03 | |||
| 15ff6bcde5 | |||
| 7822772905 | |||
| 0efe1567c0 | |||
| 837fed7690 | |||
| aa4ee64a34 | |||
| bb750cdba3 | |||
| d56c77b368 | |||
| 83e2ff06cf | |||
| f5deaafd06 | |||
| fe46311266 | |||
| 08968bb1b7 | |||
| 27bbd11f06 | |||
| 48d7841fbf | |||
| cc1df11f69 | |||
| fdfd4ac491 | |||
| d5dd5639ae | |||
| 7a120b3256 | |||
| 863d477966 | |||
| cec48832be | |||
| a36b7ccd7c | |||
| 32bf324a1e | |||
| 16de9df000 | |||
| 65c9deb4d1 | |||
| 28ae27b8e1 | |||
| 81a0cf4710 | |||
| 8a837a2830 | |||
| 0a2b3a4a56 | |||
| 613ece92dc | |||
| 8d4206c2b5 | |||
| 03286a589b | |||
| 88b9503c3b | |||
| c553d795d8 | |||
| 3ba6d6784c | |||
| 78ca24dc3f | |||
| 4fa6553db5 | |||
| 2124e043ce | |||
| e4f3620acb | |||
| 032a59a34d | |||
| e689090024 | |||
| 1cc34888fd | |||
| 0331ecd5d3 | |||
| 831a918c30 | |||
| 3db5277ae8 | |||
| 6939e0cbbc | |||
| f0095bf8b2 | |||
| 5206679efd | |||
| ac044e7573 | |||
| 6ebad2a91f | |||
| 3d10ed385c | |||
| 6d0c94dbdb | |||
| e01563a550 | |||
| 30e3225a3c | |||
| ef186a1516 | |||
| 5d5b3bdc76 | |||
| 2f8c98dac9 | |||
| 2db7189067 | |||
| 94476ac109 | |||
| 5fe02b7c98 | |||
| a1052f4fd1 | |||
| 8c9904a7c4 | |||
| 23af5dfdff | |||
| 3a10f6ad17 | |||
| a3842c873d | |||
| c29c3f063d | |||
| 758d62a396 | |||
| 6725642274 | |||
| a1f4dc365a | |||
| 165b62ee20 | |||
| 72d3481de3 | |||
| 29ccb747ad | |||
| 4749e27453 | |||
| e501d39988 | |||
| 6b6cf25862 | |||
| 180de840eb | |||
| 8fd2d7e5e7 | |||
| 129dd4a838 | |||
| e09cac6f1f | |||
| c00a86b32c | |||
| d3ae0350a2 | |||
| 2c2196a1f1 | |||
| 35ade14630 | |||
| 541df87272 | |||
| 337b6ccd6e | |||
| 42f46dfe9a | |||
| 9088d2b2c5 | |||
| 500bfa2c33 | |||
| 9ca9c43dfa | |||
| b525b0f08f | |||
| 1966c69994 | |||
| 976eff8ad1 | |||
| 0af902ec43 | |||
| 9118ce2537 | |||
| 2f48e08f1f | |||
| fec284e7cb | |||
| 8e2e4c5e73 | |||
| 0edc6615fc | |||
| b745e159de | |||
| aac29d604c | |||
| c884802b13 | |||
| 5f5573a24e | |||
| 74b0087ebb | |||
| 927e0151d4 | |||
| 5275922d1d | |||
| e186c7945a | |||
| 33a6e77f0e | |||
| 4e47489d53 | |||
| c76b2f149e | |||
| 826fffe05b | |||
| 75f57cdba7 | |||
| f556af5d4e | |||
| d5f33f0c6e | |||
| 16e17b32ad | |||
| 988494e18b | |||
| c4549a5e20 | |||
| 46fa4f38d5 | |||
| 20e0e68ad7 | |||
| 17468a234a | |||
| 8a53d5bfc6 | |||
| 695da7418e | |||
| abd26c796b | |||
| 24559d81ac | |||
| 6c1c2c3994 | |||
| 01fab15713 | |||
| 0cd00e71c3 | |||
| bf0ff2adbf | |||
| ed4bbc1c56 | |||
| c456402cc5 | |||
| 4f0bf667b1 | |||
| 61af9aa574 | |||
| 3b59b34e76 | |||
| 65b38997f7 | |||
| 7510f7649c | |||
| 619792a81c | |||
| cea1183f75 | |||
| e2af4c5ae4 | |||
| dfd5f82894 | |||
| e81944cef6 | |||
| 7bdd39ab9a | |||
| f4b38f040e | |||
| 799668e129 | |||
| 890190263e | |||
| f8522edacd | |||
| b958747855 | |||
| 078bde2c02 | |||
| 0b10ea987b | |||
| 1553d38182 | |||
| d04b075996 | |||
| 644927636d | |||
| 95e45007aa | |||
| 7a583c4045 | |||
| 65bce058bb | |||
| 48b437083b | |||
| fa0612859b | |||
| 4ffbcd0b7d | |||
| a0cd053fd9 | |||
| f9a5e066b5 | |||
| e9bc192160 | |||
| e0a57988ad | |||
| b414a74c26 | |||
| 2c3796d598 | |||
| 7e0ff9ab06 | |||
| 6d1565d8b6 | |||
| e18e002d2f | |||
| c18572ea9d | |||
| e9d02bc5e8 | |||
| a2108a8a14 | |||
| 57f8fa257a | |||
| 61944fc045 | |||
| 4b48d2d921 | |||
| 3aa145cbef | |||
| 4875127daa | |||
| d14a624421 | |||
| 246f50b778 | |||
| 15b53c6cfa | |||
| 3a7ef0adbd | |||
| 293a305748 | |||
| e5038c6d13 | |||
| 13ea79f6fd | |||
| f55d3c203b | |||
| 37c4d47d3a | |||
| 04e21c9243 | |||
| 83cac07f6e | |||
| f472e0f782 | |||
| 91c9f981c5 | |||
| dd906526c0 | |||
| ec3001796a | |||
| 86cf4c285a | |||
| cd18887b69 | |||
| 4637c68295 | |||
| 0114bd1fa7 | |||
| 6dee84ca71 | |||
| f004a0c654 | |||
| 6123576c68 | |||
| 21cfc09f8e | |||
| 509530e235 | |||
| d146a01422 | |||
| 91332723db | |||
| 244fbd98a5 | |||
| 5afe8e14d9 | |||
| 73f6b12dd8 | |||
| 793f2e7157 | |||
| cf48983046 | |||
| d038516750 | |||
| f8182e4514 | |||
| bc13b8e92c | |||
| fbe79258bf | |||
| ef1e014b41 | |||
| e3c8393d1b | |||
| 379e03f9d0 | |||
| e13921aa8a | |||
| 8a549d8610 | |||
| 84081b2bd8 | |||
| defe3365c4 | |||
| 4e6201ecd1 | |||
| ccf50f950e | |||
| a7f0211e2f | |||
| 049ce4828c | |||
| c1173346ef | |||
| 7ace184fe6 | |||
| 54b314ace5 | |||
| 224b344445 | |||
| 0b28b4cb0f | |||
| 83129e165c | |||
| 871b595954 | |||
| b67b1585c2 | |||
| 979b2b5632 | |||
| cf4ad186ab | |||
| 5100f215cf | |||
| f129e9b7cd | |||
| 4aed45de19 | |||
| 1e6daa5c73 | |||
| 4c015d76b7 | |||
| 37a11cd168 | |||
| 22ad24db6c | |||
| e94c1b8841 | |||
| cc0ec65714 | |||
| d67d30c58a | |||
| d75ee1cca5 | |||
| 6804676a96 | |||
| 3e5d742ac7 | |||
| 6da2a71050 | |||
| e32ac39faf | |||
| daa243d37a | |||
| 2773ab600d | |||
| 19cdf8dc9f | |||
| 9daf1ec5ba | |||
| c9f0ca9359 | |||
| 84c8a2d2f0 | |||
| ded226abfe | |||
| 9b8d55bc18 | |||
| 11f8709286 | |||
| b034f105c0 | |||
| 6e37722383 | |||
| ffce30afa2 | |||
| e724a59f2d | |||
| 4bf855d225 | |||
| d4c9704007 | |||
| 7c252b5f5f | |||
| f756933879 | |||
| a1aecbf4fc | |||
| 131e7b1ccd | |||
| d0ac6c435f | |||
| 2bc5f3a057 | |||
| ba6b4a5da9 | |||
| da5a987df0 | |||
| 7dd6c46156 | |||
| 3a5cdc5108 | |||
| 3aa69a9e32 | |||
| e056c7e1fa | |||
| d63273d082 | |||
| 0efb65ca0a | |||
| 9fe04bfb08 | |||
| 09d3948acf | |||
| 954351a80b | |||
| 8d51066ddd | |||
| 84102baab4 | |||
| 64e70efdf1 | |||
| 97ecc7136e | |||
| f9073e2320 | |||
| 54d907c314 | |||
| 82c7d6553a | |||
| 19b10e3216 | |||
| 0f79e6bed5 | |||
| aa0cf814ff | |||
| 4aa9d03da2 | |||
| 37abfdfd6e | |||
| d5c3ede215 | |||
| 426855e378 | |||
| 358c6970b5 | |||
| 55ebd5b949 | |||
| 2a61fe69f1 | |||
| ff6aacdc78 | |||
| 5f0ec034d9 | |||
| 31e34d177b | |||
| b2d85af78b | |||
| 968a5c68b6 | |||
| 3c05823f19 | |||
| 2fb46f670f | |||
| 37f7ad4185 | |||
| 926724a279 | |||
| 959f04bc96 | |||
| 7f1b6b3a0d | |||
| ab771ea24e | |||
| a629a7ee73 | |||
| 2052929768 | |||
| a97c287aee | |||
| 8ed2370fbe | |||
| 5b26caca0c | |||
| 6988bfe88f | |||
| 07722a6007 | |||
| fa570ab32a | |||
| 46f87b9cd3 | |||
| 8ffbfd1f06 | |||
| 597ac2e562 | |||
| bc04637694 | |||
| c7b58f2195 | |||
| 01cfda7965 |
@@ -0,0 +1,18 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
}
|
||||
]
|
||||
}
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
name: architect
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
name: dev
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,274 @@
|
||||
---
|
||||
name: fleets-status
|
||||
description: Report the status of every fleet that shares one LavinMQ instance. Use for local daemon health, broker-wide fleet presence, and cross-host lead coordination checks.
|
||||
---
|
||||
|
||||
# Status of every fleet on the shared LavinMQ instance
|
||||
|
||||
**The headline: always report what is missing.** This skill starts with the local fleet, then adds
|
||||
broker-wide facts when its read-only credential exists. A missing fleet must appear as `unknown` or
|
||||
`not reachable`, with the reason and the fix. Never leave it out.
|
||||
|
||||
The known topology has one LavinMQ instance on `10.10.20.13` (`fleet01`). AMQP uses port `5672`,
|
||||
and the management API uses port `15672`. The Mac fleet owns vhost `/mac`. The fleet01 fleet owns
|
||||
vhost `/fleet01`.
|
||||
|
||||
## 1. Protect credentials before any probe
|
||||
|
||||
**Hard rule — never print `LAVINMQ_URI`.** It is an AMQP URI with its password inline. It only
|
||||
resolves in a login shell because `${SHARED_ENV}/tools/secrets.sh` supplies it. A non-login shell
|
||||
can make every broker probe look empty.
|
||||
|
||||
- Never run `echo "$LAVINMQ_URI"`.
|
||||
- Never put `${LAVINMQ_URI:-something}` in output. That form expands to the secret value when set.
|
||||
- Parse the user, host, and password into shell or Python variables. Use them without printing them.
|
||||
- Prefer `resolves` or `does not resolve` over any part of the value.
|
||||
- Every command that can read `LAVINMQ_URI` must send all output through this redaction before it
|
||||
reaches the report:
|
||||
|
||||
```bash
|
||||
sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
```
|
||||
|
||||
**The `g` flag is not optional.** Without it `sed` replaces only the first match on each line, so a
|
||||
line carrying two URIs leaks the second one. `scripts/redeploy-fleetd.sh --check` prints lines like
|
||||
that. Checked on 2026-08-27: without `g`, `amqp://u1:p1@h1/mac and http://u2:p2@h2:15672/api`
|
||||
redacts the first pair and prints `u2:p2` in the clear.
|
||||
|
||||
Keep `pipefail` on when applying that filter. Otherwise the filter can hide a failed probe. Apply
|
||||
the same no-print rule to the management password below, even though it is not in an AMQP URI.
|
||||
|
||||
## 2. Tier 1 — this fleet (always run)
|
||||
|
||||
Start here even when the broker tier is blocked. Work from the local fleetd checkout.
|
||||
|
||||
First run the read-only deployment check. It already checks the daemon process, deployed jar versus
|
||||
the checkout `HEAD`, launchd state, and whether each configured token resolves in a login shell.
|
||||
Do not copy those checks into new shell code. The script reads `LAVINMQ_URI`, so redact all output:
|
||||
|
||||
```bash
|
||||
set -o pipefail
|
||||
scripts/redeploy-fleetd.sh --check 2>&1 \
|
||||
| sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
git rev-parse HEAD
|
||||
```
|
||||
|
||||
Treat jar drift as a top-level warning. A merge is not a deployment. State the running jar result
|
||||
as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into a match.
|
||||
|
||||
Report the process identifier (PID) and uptime too:
|
||||
|
||||
```bash
|
||||
PIDS="$(pgrep -f 'target/fleetd.jar' || true)"
|
||||
if [ -z "$PIDS" ]; then
|
||||
printf '%s\n' 'fleetd: not running'
|
||||
else
|
||||
for PID in $PIDS; do
|
||||
ps -p "$PID" -o pid=,etime=,lstart=,command=
|
||||
done
|
||||
fi
|
||||
```
|
||||
|
||||
Read the full health response. Keep the HTTP status because `503` means fleetd is running but herdr
|
||||
is not reachable. Report both `herdr.version` and `herdr.protocol` when present:
|
||||
|
||||
```bash
|
||||
curl -sS --max-time 5 -w '\nHTTP %{http_code}\n' http://127.0.0.1:8765/healthz
|
||||
```
|
||||
|
||||
Call `fleet_whoami`, then call `fleet_list`. Preserve its sections in the report:
|
||||
|
||||
- `leads`, including which row is this lead;
|
||||
- `members`, including state, role, profile, branch, and worktree when present;
|
||||
- every per-profile `capacity` row, including `maxLoad`, `live`, `free`, and quarantine facts;
|
||||
- the exact `healthCoverage` value.
|
||||
|
||||
Do not describe an empty `members` list as an empty fleet. It says only that no members are spawned.
|
||||
Also do not hide a profile with `free: 0`; say whether load or credential quarantine caused it.
|
||||
|
||||
Show only WARN, ERROR, and SEVERE lines after the last `fleetd listening` line. This anchor stops an
|
||||
old incident from looking current:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
path = Path("fleetd/fleetd.out")
|
||||
if not path.exists():
|
||||
print("cannot check current WARN/ERROR: fleetd/fleetd.out does not exist")
|
||||
else:
|
||||
lines = path.read_text(errors="replace").splitlines()
|
||||
starts = [i for i, line in enumerate(lines) if "fleetd listening" in line]
|
||||
if not starts:
|
||||
print("cannot anchor WARN/ERROR: no 'fleetd listening' line exists")
|
||||
else:
|
||||
current = lines[starts[-1]:]
|
||||
alerts = [line for line in current if re.search(r"\b(?:WARN|ERROR|SEVERE)\b", line)]
|
||||
print(f"current WARN/ERROR/SEVERE count: {len(alerts)}")
|
||||
for line in alerts[-50:]:
|
||||
print(line)
|
||||
PY
|
||||
```
|
||||
|
||||
**What this tier cannot see:** it proves facts only about the Mac daemon at `127.0.0.1:8765`.
|
||||
It cannot show the fleet01 daemon, broker queue depth, or broker consumers. The fleet01 REST service
|
||||
at `10.10.20.13:8765` is not reachable from the Mac. Say this in the report rather than omitting
|
||||
fleet01.
|
||||
|
||||
**But fleet01 IS reachable over SSH — checked 2026-08-28.** An older version of this line said SSH
|
||||
was denied. That is true only for the user `dai.ha`. The host alias `fleet01` maps to user `ltms`,
|
||||
and `ssh fleet01` works with key auth:
|
||||
|
||||
```bash
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=6 fleet01 'echo $(id -un)@$(hostname)'
|
||||
```
|
||||
|
||||
So fleet01's daemon PID, uptime, jar and `/healthz` **can** be reported — over SSH, not over REST.
|
||||
Do that rather than writing `not reachable`. `ltms` also has passwordless sudo there.
|
||||
|
||||
## 3. Tier 2 — the shared broker (run when management access exists)
|
||||
|
||||
**This tier is blocked today.** The AMQP user in `LAVINMQ_URI` can connect on port `5672`, but gets
|
||||
HTTP `401` from the management API on port `15672`. An AMQP connection does not grant monitoring
|
||||
access.
|
||||
|
||||
The operator must create a separate, read-only LavinMQ management user with the `monitoring` tag.
|
||||
It needs access to inspect both `/mac` and `/fleet01`. Store its values as
|
||||
`LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD` in
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Do not reuse or print the AMQP URI. Full multi-fleet status stays
|
||||
blocked until this user exists.
|
||||
|
||||
When both variables resolve, run this from a login shell. It calls `GET /api/overview`,
|
||||
`GET /api/vhosts`, `GET /api/queues`, and `GET /api/connections`. It prints selected status fields,
|
||||
but never the user, password, Authorization header, or AMQP URI:
|
||||
|
||||
```bash
|
||||
zsh -lc 'python3 - "$@"' -- <<'PY'
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
base = "http://10.10.20.13:15672"
|
||||
user = os.environ.get("LAVINMQ_MANAGEMENT_USER", "")
|
||||
password = os.environ.get("LAVINMQ_MANAGEMENT_PASSWORD", "")
|
||||
if not user or not password:
|
||||
print("broker tier: BLOCKED — management credential does not resolve in a login shell")
|
||||
sys.exit(0)
|
||||
|
||||
token = base64.b64encode(f"{user}:{password}".encode()).decode()
|
||||
|
||||
def get(path):
|
||||
request = urllib.request.Request(
|
||||
base + path,
|
||||
headers={"Authorization": "Basic " + token, "Accept": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=5) as response:
|
||||
return json.load(response)
|
||||
|
||||
try:
|
||||
overview = get("/api/overview")
|
||||
vhosts = get("/api/vhosts")
|
||||
queues = get("/api/queues")
|
||||
connections = get("/api/connections")
|
||||
except urllib.error.HTTPError as error:
|
||||
print(f"broker tier: BLOCKED — management API returned HTTP {error.code}")
|
||||
sys.exit(0)
|
||||
except Exception as error:
|
||||
print(f"broker tier: BLOCKED — management API is not reachable: {type(error).__name__}")
|
||||
sys.exit(0)
|
||||
|
||||
fleet_names = {"/mac": "Mac fleet", "/fleet01": "fleet01 fleet"}
|
||||
print(json.dumps({
|
||||
"overview": {
|
||||
"lavinmq_version": overview.get("lavinmq_version"),
|
||||
"rabbitmq_version": overview.get("rabbitmq_version"),
|
||||
"queue_totals": overview.get("queue_totals", {}),
|
||||
"object_totals": overview.get("object_totals", {}),
|
||||
},
|
||||
"fleets": [
|
||||
{
|
||||
"fleet": fleet_names.get(vhost.get("name"), "UNKNOWN FLEET"),
|
||||
"vhost": vhost.get("name"),
|
||||
"queues": [
|
||||
{
|
||||
"name": queue.get("name"),
|
||||
"messages": queue.get("messages", 0),
|
||||
"messages_ready": queue.get("messages_ready", 0),
|
||||
"messages_unacknowledged": queue.get("messages_unacknowledged", 0),
|
||||
"consumers": queue.get("consumers", 0),
|
||||
}
|
||||
for queue in queues if queue.get("vhost") == vhost.get("name")
|
||||
],
|
||||
"connections": [
|
||||
{
|
||||
"name": connection.get("name"),
|
||||
"peer_host": connection.get("peer_host"),
|
||||
"state": connection.get("state"),
|
||||
}
|
||||
for connection in connections if connection.get("vhost") == vhost.get("name")
|
||||
],
|
||||
}
|
||||
for vhost in vhosts
|
||||
],
|
||||
}, indent=2, sort_keys=True))
|
||||
PY
|
||||
```
|
||||
|
||||
Map `/mac` to the Mac fleet and `/fleet01` to the fleet01 fleet. Keep any other vhost in the
|
||||
report as `UNKNOWN FLEET`; do not drop it. For each vhost, total the ready, unacknowledged, and all
|
||||
messages. Report every queue's consumer count and each live connection.
|
||||
|
||||
**A vhost with queues but zero consumers means that fleet's daemon is down while its durable state
|
||||
survives. Call this out as a top-level warning.** This is the main reason to use the management API
|
||||
instead of calling each remote daemon.
|
||||
|
||||
**What this tier cannot see:** without the new `monitoring` credential it cannot enumerate any
|
||||
vhost, queue, depth, consumer, or connection. With the credential it still cannot report fleet01's
|
||||
daemon PID, uptime, jar revision, `/healthz`, herdr version, or member capacity. Those need reachable
|
||||
fleet01 REST or SSH access, which the Mac does not have today.
|
||||
|
||||
## 4. Tier 3 — cross-fleet lead coordination
|
||||
|
||||
Use the queue data from Tier 2. Select queues whose names match `lead.<coordId>.inbox`. Report each
|
||||
queue's vhost, depth, consumer count, and the `coordId` between the prefix and suffix.
|
||||
|
||||
- A lead inbox with a consumer shows that a lead mailbox is live on that vhost.
|
||||
- A durable lead inbox with zero consumers shows saved coordination state, but no live receiver.
|
||||
- No lead inbox is not proof that coordination is disabled. The daemon may be down before declaring
|
||||
its queue, or this account may not be allowed to see the vhost.
|
||||
|
||||
This Mac fleet currently sets both `broker.uriEnv` and `coordinator.uriEnv` to the same variable,
|
||||
`LAVINMQ_URI`. Therefore its coordinator connects to `/mac`. Cross-host `fleet_send{coordId}` routes
|
||||
only when both leads share the same coordinator vhost. If the fleet01 lead uses `/fleet01` for its
|
||||
coordinator, the leads cannot see each other and the send will not route.
|
||||
|
||||
**Open question:** the fleet01 coordinator vhost has not been checked. Surface this question in
|
||||
every report until a live `lead.<coordId>.inbox` consumer or fleet01's config proves the answer. Do
|
||||
not claim that fleet01 uses `/fleet01` just because its member queues do.
|
||||
|
||||
Also compare these broker facts with the `leads` rows from local `fleet_list`. A missing remote lead
|
||||
is `not visible from this coordinator`, not `down`, unless the broker consumer facts prove it.
|
||||
|
||||
**What this tier cannot see:** without Tier 2 management access it cannot list lead inboxes or their
|
||||
consumers. Even with that access, a stopped fleet01 daemon leaves only durable queue history. That
|
||||
history cannot prove which coordinator URI its current config would use after restart.
|
||||
|
||||
## 5. Report all fleets
|
||||
|
||||
Use one row per known or discovered fleet. Include blocked rows.
|
||||
|
||||
| Fleet | Daemon | Deployment | Herdr | Members/capacity | Queues/consumers | Lead coordination | Cannot check |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| Mac (`/mac`) | PID + uptime | jar vs `HEAD` | health + version + protocol | `fleet_list` + `healthCoverage` | facts or blocked reason | inbox facts or open question | exact missing facts |
|
||||
| fleet01 (`/fleet01`) | reachable/down/unknown | value or `not reachable` | value or `not reachable` | value or `not reachable` | facts or blocked reason | inbox facts plus coordinator-vhost question | exact missing facts and fix |
|
||||
|
||||
Add rows for unknown vhosts. End with three short sections: `Current warnings`, `Checks that were
|
||||
blocked`, and `Operator action`. Until the management user exists, `Operator action` must say:
|
||||
|
||||
> Create a read-only LavinMQ management user with the `monitoring` tag and access to `/mac` and
|
||||
> `/fleet01`. Put its user and password in `${SHARED_ENV}/tools/secrets.sh` as
|
||||
> `LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD`.
|
||||
@@ -0,0 +1,137 @@
|
||||
---
|
||||
name: implementer
|
||||
description: Implementer-role procedure for a fleetd worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over fleetd.
|
||||
---
|
||||
|
||||
# Implementer worker — procedure
|
||||
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge, never commit `.mcp.json` or `wiki/`) is in **`CLAUDE.md` → Bridge communication →
|
||||
Worker** and already applies. This skill is only the *implement-and-hand-off procedure*.
|
||||
|
||||
You run in an **isolated git worktree on your own branch** — a full peer of the primary (same repo,
|
||||
`CLAUDE.md`, skills), differing in the model behind you and the branch you sit on. Your MCP surface
|
||||
is **only what your launcher mounted** (the bridge): the primary's IDE and forge servers are not
|
||||
yours, and the worktree's `.mcp.json` is deliberately emptied so you cannot inherit them.
|
||||
The worktree model is documented in [`docs/Worker-Git-Workflow.md`](../../../docs/Worker-Git-Workflow.md).
|
||||
|
||||
## 1. Confirm where you are — then never leave
|
||||
|
||||
Before touching anything:
|
||||
|
||||
```bash
|
||||
git rev-parse --show-toplevel # your worktree root — NOT the primary's main tree
|
||||
git branch --show-current # your dedicated branch: worker/<ticket>-<nonce>
|
||||
git status # should be clean at the start
|
||||
```
|
||||
|
||||
Do **all** work here, on this branch. Never `git checkout main`, never rebase onto or push to
|
||||
`main`. The branch is your isolation — respect it.
|
||||
|
||||
**Every path you read, edit, or build is relative to that root.** Work from `$PWD`; if a tool, a
|
||||
brief, or your own memory hands you an absolute path, check it starts with your worktree root
|
||||
before you touch it, and stop if it doesn't. An absolute path pointing anywhere else is the
|
||||
primary's checkout — editing there while building here means **every build you run is of code that
|
||||
does not contain your changes**, and it passes while your work goes nowhere. This has happened:
|
||||
a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
|
||||
```bash
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
in your reply instead of widening it.
|
||||
- Match the surrounding code's style, naming, and idioms.
|
||||
|
||||
**Acceptance criterion — a green build, quoted.** Your work is not done until this passes *inside
|
||||
your worktree*:
|
||||
|
||||
```bash
|
||||
cd "$(git rev-parse --show-toplevel)/fleetd" && mvn clean install
|
||||
echo "exit=$?"
|
||||
```
|
||||
|
||||
Read its **full** output — never pipe it through `tail`/`head`/`grep`, which hide a failure behind
|
||||
a zero exit. Then quote the real `Tests run: … Failures: … Errors: …` line and the
|
||||
`BUILD SUCCESS`/`FAILURE` verbatim in your reply. If it does not go green, say so with the actual
|
||||
error; a failing build honestly reported is a usable result, a claimed-green one is not. You have
|
||||
no IDE MCP tools, so `mvn` is your only verification — never claim a check you had no way to run.
|
||||
|
||||
## 3. Commit
|
||||
|
||||
```bash
|
||||
git add <the files you changed> # explicitly — never `git add -A` / `git add .`
|
||||
git commit -m "<ticket>: <clear one-line summary>"
|
||||
```
|
||||
|
||||
`.mcp.json` is neutralized and `--skip-worktree` in your worktree — never `git add` it, and never
|
||||
"restore" it from the primary's copy. Same for `wiki/` (a submodule with its own remote).
|
||||
|
||||
## 4. Push
|
||||
|
||||
```bash
|
||||
git push -u origin HEAD
|
||||
```
|
||||
|
||||
Push is over SSH as the same user — no extra credential needed. Never force-push over anything
|
||||
you did not create.
|
||||
|
||||
## 5. Open your own PR to `main`
|
||||
|
||||
Via the gitea REST API. The daemon injected a **repo-scoped token** (`GITEA_TOKEN`) and the forge
|
||||
host (`GITEA_HOST`) into your env for exactly this — the token can create a PR but **cannot
|
||||
merge**.
|
||||
|
||||
```bash
|
||||
API="${GITEA_HOST%/}/api/v1/repos/fleet/fleetd/pulls"
|
||||
BRANCH="$(git branch --show-current)"
|
||||
curl -sS -X POST "$API" \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
-H "Content-Type: application/json" \
|
||||
-d "$(cat <<JSON
|
||||
{"head": "${BRANCH}", "base": "main",
|
||||
"title": "<ticket>: <concise change summary>",
|
||||
"body": "<what changed and why; reference the ticket; note tests run and their result>"}
|
||||
JSON
|
||||
)"
|
||||
```
|
||||
|
||||
The response JSON carries `"html_url"` — that is your PR URL. On a non-2xx, read the error body,
|
||||
fix it if the cause is yours (e.g. branch not pushed yet), and report the failure rather than
|
||||
inventing a URL. If `GITEA_TOKEN` is unset your profile was not granted PR-create: push the branch
|
||||
and report its name so the lead opens the PR.
|
||||
|
||||
## 6. Hand off — what goes in `fleet_reply`
|
||||
|
||||
The reply is the entire handoff; the lead cannot see your terminal.
|
||||
|
||||
```
|
||||
PR: <html_url from step 5, or "not created: <reason>" + branch name>
|
||||
branch: <your branch>
|
||||
root: <git rev-parse --show-toplevel — proves you worked in your own worktree>
|
||||
files: <worktree-relative paths you changed>
|
||||
build: <the verbatim "Tests run: …" and BUILD SUCCESS/FAILURE lines — or "not run: <why>">
|
||||
summary: <2-3 lines: what you implemented and any caveat the reviewer needs>
|
||||
```
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant L as Lead
|
||||
participant I as Implementer (you)
|
||||
participant G as git / gitea
|
||||
|
||||
L->>I: delegated task (you are in a worktree on your branch)
|
||||
I->>I: "implement here — every path under $PWD"
|
||||
I->>I: "mvn clean install in this worktree, unpiped, until green"
|
||||
I->>G: git commit (never .mcp.json / wiki)
|
||||
I->>G: git push -u origin HEAD
|
||||
I->>G: POST /pulls (GITEA_TOKEN) — open PR to main
|
||||
G-->>I: html_url
|
||||
I->>L: fleet_reply(PR url, branch, files, tests)
|
||||
Note over L,G: the lead reviews the PR and merges on green — you never merge
|
||||
```
|
||||
|
||||
*The implement turn: work in the worktree, commit → push → open the PR, hand off the URL.*
|
||||
@@ -0,0 +1,180 @@
|
||||
---
|
||||
name: port-to-opencode
|
||||
description: Make an OpenCode session a first-class participant in a Claude Code workspace — instructions, MCP servers, and secrets — without duplicating config. Use when onboarding opencode to a project that already has CLAUDE.md and .mcp.json, or when an opencode peer needs the same tools and rules as the Claude session.
|
||||
---
|
||||
|
||||
# Porting a Claude Code workspace to OpenCode
|
||||
|
||||
**The headline: there is almost nothing to port.** OpenCode reads `CLAUDE.md` natively. The only
|
||||
artifact you create is one `opencode.json` mapping MCP servers. Do not translate instructions, do
|
||||
not generate a second rules file, and do not install a sync tool — every one of those makes the
|
||||
workspace worse.
|
||||
|
||||
Everything below was verified against `opencode 1.18.16` and the OpenCode docs.
|
||||
|
||||
## 1. Know what you get for free
|
||||
|
||||
OpenCode's instruction search order:
|
||||
|
||||
```
|
||||
1. walking up from cwd: AGENTS.md , then CLAUDE.md
|
||||
2. global: ~/.config/opencode/AGENTS.md
|
||||
3. Claude Code global: ~/.claude/CLAUDE.md (unless disabled)
|
||||
```
|
||||
|
||||
*"The first matching file wins in each category."*
|
||||
|
||||
Consequences that decide the whole procedure:
|
||||
|
||||
- **A project `CLAUDE.md` is already read.** No port needed.
|
||||
- **Your user-level `~/.claude/CLAUDE.md` is already read too.** Global preferences carry over.
|
||||
- **An `AGENTS.md` in the repo SHADOWS `CLAUDE.md`.** If one exists from a previous Codex port,
|
||||
**delete it** — otherwise opencode reads the stale translated copy instead of the real rules.
|
||||
This is the single most likely way to get this wrong.
|
||||
|
||||
## 2. Create `opencode.json` for MCP servers only
|
||||
|
||||
Project config lives at `opencode.json` in the repo root; the global one is
|
||||
`~/.config/opencode/opencode.json`. **Configs merge, they do not replace** — so machine-local
|
||||
servers belong in the global file and shared ones in the project file.
|
||||
|
||||
Map each entry from `.mcp.json`:
|
||||
|
||||
| `.mcp.json` | `opencode.json` |
|
||||
|---|---|
|
||||
| `"type": "http"` / `"sse"` | `"type": "remote"`, `"url"` |
|
||||
| `"type": "stdio"` | `"type": "local"`, `"command": ["bin", "arg"]` |
|
||||
| `"command"` + `"args"` | single `"command"` array |
|
||||
| `"env"` | `"environment"` |
|
||||
| `"headers"` | `"headers"` |
|
||||
|
||||
```json
|
||||
{
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"instructions": ["CLAUDE.md"],
|
||||
"mcp": {
|
||||
"fleetd": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"context7": { "type": "remote", "url": "https://example.dev/mcp", "enabled": true,
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" } },
|
||||
"gitea": { "type": "local", "command": ["gitea-mcp", "-t", "stdio"], "enabled": true,
|
||||
"environment": { "GITEA_ACCESS_TOKEN": "{env:GITEA_ACCESS_TOKEN}" } }
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
Set `instructions` explicitly even though `CLAUDE.md` is found anyway — the fallback only applies
|
||||
while no `AGENTS.md` exists, and being explicit survives someone adding one later.
|
||||
|
||||
## 3. Reference secrets, never embed them
|
||||
|
||||
OpenCode substitutes at load time, in both `headers` and `environment`:
|
||||
|
||||
```
|
||||
{env:VARIABLE_NAME} value from the environment
|
||||
{file:~/.secrets/token} value read from a file
|
||||
```
|
||||
|
||||
**No credential ever belongs in `opencode.json`.** With `{env:…}` there is no reason to write one,
|
||||
which is what makes this file safe to commit — and it must be committable, because a peer running
|
||||
in a git worktree receives tracked files only.
|
||||
|
||||
**But `{env:…}` reads OpenCode's *process* environment — and OpenCode has no env store of its
|
||||
own.** A Claude Code `env` block in `~/.claude/settings.json` or `.claude/settings.local.json` does
|
||||
**not** reach it: those files are Claude Code's, and the variables exist only inside processes
|
||||
Claude Code spawned. Verify with a clean login shell, not the shell your agent hands you:
|
||||
|
||||
```bash
|
||||
env -u MY_TOKEN zsh -lc 'echo "${MY_TOKEN:-NOT IN PROFILE}"'
|
||||
```
|
||||
|
||||
So `{env:…}` only works if something puts the variable in the environment first. Pick the supply
|
||||
route by who launches opencode:
|
||||
|
||||
| Launcher | Route |
|
||||
|---|---|
|
||||
| a human, from a terminal | one central store, sourced by the login shell |
|
||||
| a spawner (bridge, CI, IDE) | `{env:…}`, with the spawner injecting the variable |
|
||||
|
||||
**For the human case, keep every credential in one file the login shell sources.** Here that file
|
||||
is `${SHARED_ENV}/tools/secrets.sh`, sourced from `${SHARED_ENV}/.ltms`, kept at mode 600 and never
|
||||
committed. `opencode.json` then names variables and holds no values:
|
||||
|
||||
```json
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" }
|
||||
```
|
||||
|
||||
Both routes end at the same syntax, and that is the point. The file does not change when a human
|
||||
launches opencode instead of the bridge.
|
||||
|
||||
**`{file:…}` also works, and this project moved away from it.** Opencode resolves a relative
|
||||
`{file:}` path against the project root, so a gitignored `.secrets/` beside `opencode.json` needs no
|
||||
shell setup at all. It did not fail; the problem is that it makes a second copy of the token. The
|
||||
same secret then lives in two places, and the copy you forget is the one that leaks or goes stale.
|
||||
One store with many references is easier to rotate and to audit.
|
||||
|
||||
**One catch survives either choice, so state it out loud:** what a human's shell exports does not
|
||||
reach a spawned peer, and neither does a gitignored `.secrets/` — a git worktree receives tracked
|
||||
files only. Spawned peers must be fed through `{env:…}` by whatever launches them. Check the
|
||||
variable *names* match: a spawner often injects under a different name than your shell uses, and
|
||||
the config has no fallback. In this repo the bridge goes further: it neutralizes a worktree's
|
||||
`opencode.json`, so a member cannot inherit the primary's credentials by accident.
|
||||
|
||||
## 4. Do not port machine-local MCP servers
|
||||
|
||||
IDE indexes, language servers, editor bridges — anything bound to *your* checkout — stay out of the
|
||||
project file. Put them in `~/.config/opencode/opencode.json` if you want them personally.
|
||||
|
||||
A committed project config reaches every worktree. A worker that mounts servers whose paths point
|
||||
into the primary's checkout will edit the primary's files while building in its own — every build
|
||||
passes, every change lands in the wrong tree.
|
||||
|
||||
Port the servers the work needs. Leave the rest.
|
||||
|
||||
## 5. Verify against the running agent, not the file
|
||||
|
||||
A file on disk proves nothing about what the agent loaded.
|
||||
|
||||
```bash
|
||||
opencode run "In one line: state a rule from this project's instructions."
|
||||
```
|
||||
|
||||
The answer must reflect the actual `CLAUDE.md`. If it answers generically, the instructions did not
|
||||
reach the model and everything after this is built on sand.
|
||||
|
||||
Then confirm the tools are mounted:
|
||||
|
||||
```bash
|
||||
opencode mcp list
|
||||
```
|
||||
|
||||
**"connected" does not mean "working".** A stdio server with a missing credential still completes
|
||||
the MCP handshake and reports green; only a real tool call reveals it. Verified: `gitea` showed
|
||||
`✓ connected` with no token, then failed the first call with `token is required`. A remote server
|
||||
is more honest (`⚠ needs authentication`), but do not rely on that difference — **exercise one
|
||||
authenticated tool per server**:
|
||||
|
||||
```bash
|
||||
opencode run "Call <server>'s <tool>. Report the result or the exact error. One line."
|
||||
```
|
||||
|
||||
Check too that no server you deliberately withheld is present.
|
||||
|
||||
## 6. Report
|
||||
|
||||
State what you changed, which servers crossed and which you withheld and why, and quote the
|
||||
verification answer verbatim. If any server failed to connect, say so plainly — a partially mounted
|
||||
peer is worse than a missing one, because it looks configured.
|
||||
|
||||
## Gotchas
|
||||
|
||||
- **A leftover `AGENTS.md` silently wins over `CLAUDE.md`.** Check for one before anything else.
|
||||
- **`opencode.json` is merged, not overridden** — a global entry and a project entry with the same
|
||||
server name both matter; keep names distinct unless you intend to layer them.
|
||||
- **`enabled: false`** turns a server off without deleting its config — prefer it over removal when
|
||||
you may want the server back.
|
||||
- **OAuth-based servers** store tokens in `~/.local/share/opencode/mcp-auth.json` after
|
||||
`opencode mcp auth <server>`; that is machine state, never config to commit.
|
||||
- **Skills and subagents do not port.** OpenCode uses its own agent markdown under
|
||||
`.opencode/agents/`; `.claude/skills/**` is not read. If a delegation brief tells a peer to load a
|
||||
skill by name, that instruction has no effect on an opencode peer — spell the procedure out in the
|
||||
brief, or author the equivalent agent file.
|
||||
@@ -0,0 +1,51 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Reviewer-role procedure for a fleetd worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over fleetd.
|
||||
---
|
||||
|
||||
# Reviewer worker — procedure
|
||||
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies. This
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
A review that fires on a snippet misses the caller that makes it safe (or the one that makes it
|
||||
a bug). Reviewing part of the scope and guessing the rest is the most common way a reviewer is
|
||||
wrong.
|
||||
|
||||
## 2. Stay in the scope
|
||||
|
||||
- Review **only** what you were assigned. Something elsewhere looks wrong? One line in your
|
||||
reply — do not go hunt it. Wandering is how two reviewers report the same thing and neither
|
||||
covers what it was given.
|
||||
- Do **not** edit files or run the build. You review; the owner acts.
|
||||
|
||||
## 3. Reach for `fleet_ask` only for a genuine fork
|
||||
|
||||
Ambiguous requirement, a missing acceptance criterion, "intended or a bug?", or two defensible
|
||||
fixes with different consequences — those are the lead's call, and guessing produces a
|
||||
confident-but-wrong finding. Anything you could settle by reading more code is yours to settle.
|
||||
|
||||
## 4. The finding — what goes in `fleet_reply`
|
||||
|
||||
Report the **single most important** real issue in the scope, in these four lines, under
|
||||
~90 words:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence — what is wrong and why it matters>
|
||||
3. fix: <one line — the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
- **Nothing real after reading?** Reply `NO ISSUE` and one line saying why. A clean review is a
|
||||
valid result; a fabricated issue is worse than none.
|
||||
- **Severity:** `high` = wrong result, data loss, security, or a hang/crash on a real path ·
|
||||
`medium` = a real bug on an edge path, or a correctness risk under load/concurrency ·
|
||||
`low` = clarity, a latent foot-gun, or a smell with no current failure.
|
||||
- Be specific and verifiable: a line number and a one-line repro beat an adjective. If you can't
|
||||
point at where it goes wrong, you haven't found it yet.
|
||||
@@ -0,0 +1,105 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
pull_request:
|
||||
branches: [main]
|
||||
|
||||
jobs:
|
||||
build:
|
||||
runs-on: ubuntu-latest
|
||||
steps:
|
||||
# The wiki submodule is docs only and is not needed to build — leave it unfetched so CI
|
||||
# does not depend on the wiki repo being reachable.
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The runner image ships an older default-jdk; fleetd sets maven.compiler.release=25, so
|
||||
# provision the JDK explicitly rather than apt-installing whatever "default" means today.
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '25'
|
||||
cache: maven
|
||||
|
||||
# setup-java provisions the JDK only — it does NOT install Maven, and the runner image has
|
||||
# no mvn on PATH (a bare `mvn` exits 127). Install it separately. apt pulls a default JRE as
|
||||
# a dependency; JAVA_HOME from setup-java still wins, which the version check below proves.
|
||||
- name: Install Maven
|
||||
run: |
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
- name: Build and test
|
||||
working-directory: fleetd
|
||||
# This IS the mock-socket surface CB-503 asks for: the pom's `default-excludes` profile
|
||||
# already sets excludedGroups=contract, so the @Tag("contract") tests — which need a live
|
||||
# herdr socket and a RabbitMQ container — are excluded without any flag here. Everything
|
||||
# that runs does so against the fake UDS herdr and fake ccs/claude stubs.
|
||||
run: mvn -B clean install
|
||||
|
||||
# Deliberately NOT actions/upload-artifact: this Gitea instance presents as GHES, and
|
||||
# @actions/artifact v2+ (i.e. upload-artifact@v4) refuses to run there —
|
||||
# "GHESNotSupportedError ... not currently supported on GHES", which red-Xes an otherwise
|
||||
# green build. Since the artifact could not be retrieved anyway, dump the failing tests into
|
||||
# the log instead, where they are actually readable.
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
grep -qE "Failures: [1-9]|Errors: [1-9]" "$f" && { echo "===== $f ====="; cat "$f"; }
|
||||
done
|
||||
exit 0
|
||||
|
||||
# CB-521 — actually run the AMQP contract test in CI, against a REAL broker. The broker is a
|
||||
# RabbitMQ SERVICE CONTAINER, not Testcontainers-with-Docker: the runner image has no Docker, so
|
||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||
# straight to it — no Docker, no skipped tests. This separation (build job hermetic and
|
||||
# Docker-free; contract job broker-provided) is deliberate — see the default-excludes/contract
|
||||
# profiles in fleetd/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# exactly as in the build job above.
|
||||
contract:
|
||||
runs-on: ubuntu-latest
|
||||
services:
|
||||
rabbitmq:
|
||||
image: rabbitmq:3.13 # same AMQP 0-9-1 engine the local Testcontainers fixture uses
|
||||
env:
|
||||
RABBITMQ_DEFAULT_USER: guest
|
||||
RABBITMQ_DEFAULT_PASS: guest
|
||||
env:
|
||||
# Service containers are reachable from the job by their network alias on their internal port.
|
||||
AMQP_URI: amqp://guest:guest@rabbitmq:5672
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
with:
|
||||
distribution: temurin
|
||||
java-version: '25'
|
||||
cache: maven
|
||||
|
||||
- name: Install Maven
|
||||
run: |
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
# The `contract` profile clears the default-excludes group, so the @Tag("contract") AMQP test
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
- name: Contract tests
|
||||
working-directory: fleetd
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
grep -qE "Failures: [1-9]|Errors: [1-9]" "$f" && { echo "===== $f ====="; cat "$f"; }
|
||||
done
|
||||
exit 0
|
||||
+21
@@ -0,0 +1,21 @@
|
||||
# No secret belongs in this repo any more: every credential lives in one shell-level store
|
||||
# (${SHARED_ENV}/tools/secrets.sh), and opencode.json reads it as {env:...}. This line stays as a
|
||||
# backstop, so a workspace-scoped copy that someone re-creates by habit still cannot be committed.
|
||||
.secrets/
|
||||
|
||||
# Settings backups inherit the env block — and secrets with it.
|
||||
.claude/settings.local.json.bak*
|
||||
|
||||
# The default profile parityOverlay copies these primary→worktree, so they appear in EVERY worker
|
||||
# worktree. Two reasons they must be ignored. They hold environment values, which is reason enough.
|
||||
# And since CB-576 a release preserves any worktree that `git status --porcelain` calls dirty —
|
||||
# untracked files included, deliberately. An untracked overlay file would therefore make every
|
||||
# COMPLETED release preserve its worktree, and worktrees would pile up with no error to notice.
|
||||
.env
|
||||
.envrc
|
||||
|
||||
# Daemon runtime artefacts. fleetd appends its log wherever it is launched from, so both the
|
||||
# repo root and fleetd/ collect one; neither belongs in git.
|
||||
fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
[submodule "wiki"]
|
||||
path = wiki
|
||||
url = ssh://git@git.ltms.dev:2224/lms/claude-bridge.wiki.git
|
||||
url = ssh://git@git.ltms.dev:2224/fleet/fleetd.wiki.git
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -1,13 +1,321 @@
|
||||
# claude-bridge — project instructions
|
||||
|
||||
## Bridge communication (enforced — read this first)
|
||||
|
||||
> **Canonical block.** Everything down to §Layering is the portable bridge charter, copied verbatim
|
||||
> into every project that mounts the bridge MCP. Keep it byte-identical with the template in the
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
`fleetd` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **member**) mount the *same* MCP server and talk only
|
||||
through its `fleet_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
|
||||
### Which role am I? — settle this before acting
|
||||
|
||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||
|
||||
**Call `fleet_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
||||
was bound to. Don't infer what you can ask.
|
||||
|
||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||
fires: the reply charter in your system prompt (*"You are a spawned member in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; fleet tools prefixed `mcp__fleet__*` ⇒ **spawned
|
||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||
`.mcp.json`, so it varies — and a member spawned before CB-632 still says `mcp__bridge__*`); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
||||
only `fleet_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
### Invariants — both roles, no exceptions
|
||||
|
||||
1. **Never set, export, or forward `ANTHROPIC_BASE_URL`** (or `ANTHROPIC_AUTH_TOKEN`). The primary
|
||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||
never move a session across that boundary.
|
||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||
side cannot see your screen. An answer that isn't in a `fleet_*` call is silently discarded.
|
||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
||||
outside your role is refused, not queued.
|
||||
4. **Delivery is status-gated: one message per turn.** Don't busy-poll a peer's terminal and don't
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
|
||||
### Primary (lead) — run this on every task, in order
|
||||
|
||||
**Delegate by default — that is the job.** With the bridge mounted you are an orchestrator on a
|
||||
metered subscription, and workers are cheap, parallel, and disposable. The default answer to "who
|
||||
does this?" is **a worker**, not you. Reach for `fleet_send` before you reach for `Edit`. The steps
|
||||
below are the procedure — run them in order, every task, not only the big ones.
|
||||
|
||||
0. **Know your role** — `fleet_whoami`, once per session, before anything else.
|
||||
1. **Split.** Write the unit list. Every unit carries: scope · the files or PR in question ·
|
||||
acceptance criteria · exactly what to report back. A unit with no acceptance criteria is not
|
||||
ready to delegate — refine it or keep it.
|
||||
2. **Gate each unit** on one question: **"can I write a brief good enough for a worker to
|
||||
succeed?"** — *not* "could I do this faster myself?" (usually you could; doing it yourself costs
|
||||
your context and your subscription, while a wasted worker turn costs a worker turn). Yes ⇒
|
||||
delegate. The keep-list is closed: the conversation with the user, decomposition and planning,
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
brief instead. The brief is self-contained — the worker sees your message and the repo, nothing
|
||||
of your context, your plan, or your screen.
|
||||
5. **Collect** — `fleet_poll{ticket}` → `fleet_ack{target, msgId}`. Answer a worker's `fleet_ask`
|
||||
with `fleet_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||
forge tools it appears to have hold a blocked credential and fail, and a piped command
|
||||
(`… | tail`) hides failures behind a zero exit — never promote a worker's "clean" to a fact.
|
||||
7. **Review — fan out.** Spawn reviewers against the diff, one per dimension or per file, with
|
||||
`wait:false`. Never the implementer of the scope it reviews, and brief them from the diff — not
|
||||
from the implementer's rationale, which carries its own blind spot. Dispatch each PR's reviewers
|
||||
as it lands; don't wait for the last implementer. Under ~50 changed lines, skip the fan-out and
|
||||
read it yourself.
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
prefer `wait:false` + `fleet_poll` for anything non-trivial: a blocking `fleet_send` is capped by
|
||||
*your own* MCP client call timeout (~60s), well below the task's real runtime.
|
||||
|
||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||
the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
|
||||
| Intent | Tool |
|
||||
|---|---|
|
||||
| Confirm your own role | `fleet_whoami` |
|
||||
| See backends available | `fleet_profiles` |
|
||||
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `fleet_status{sessionId}` |
|
||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
`fleet_list` returns `leads` alongside `members`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own members, and its own judgment. An empty
|
||||
`members` array means no members are spawned; it says nothing about peers.
|
||||
|
||||
**A lead never assigns a task to another lead.** Work goes to members — only ever downward, never
|
||||
sideways. Sending a peer a brief with acceptance criteria is a category error: a brief is a member's
|
||||
artefact, and a peer is not yours to task. If a unit needs doing and it falls in your area, spawn a
|
||||
member and delegate it yourself; if it falls in the peer's area, say so and let the peer assign it.
|
||||
The traffic between leads is coordination and nothing else:
|
||||
|
||||
1. **Divide the map, not the work.** Agree who owns which area, then each of you assigns inside your
|
||||
own. Split by **context ownership** — whoever already holds the context owns that area — and say
|
||||
who takes what, in one message, before either of you starts. Two leads silently working the same
|
||||
unit is the failure mode here, and neither notices until the merge.
|
||||
2. **Share findings, hazards, and corrections.** What you have already discovered, what broke, what
|
||||
the next person will trip on. This is the traffic that actually pays for the channel: it costs one
|
||||
message and saves a peer a rediscovery.
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
1. **Load the playbook skill the lead named** before doing anything else.
|
||||
2. **Do the assigned scope only.** Note anything you spot outside it in one line; don't go hunt it.
|
||||
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||
Don't ask what you could decide yourself.
|
||||
4. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||
lead with its end cut off.
|
||||
5. **Report honestly.** State only what you actually ran and its real output, including failures,
|
||||
and never claim the result of a check you had no way to run. **Measure your own tools; do not
|
||||
assume them.** What you mount depends on your backend: an opencode member gets the bridge and
|
||||
nothing else, while a Claude Code member also inherits the operator's user-scope MCP servers,
|
||||
which the bridge never chose for you. Two rules follow. The primary's IDE tooling is still not
|
||||
yours, whatever you see. And **a mounted tool is not a working tool** — the forge server you may
|
||||
find there holds a deliberately blocked credential and fails every call, by design.
|
||||
6. **Never merge.** Stage files explicitly — never `git add -A` — and leave alone anything the
|
||||
project marks as not-yours-to-commit.
|
||||
|
||||
### Where each rule lives (don't duplicate — extend the right layer)
|
||||
|
||||
| Layer | Scope | Reaches |
|
||||
|---|---|---|
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `fleet_reply`* | every spawned member, at launch, every peer kind — never a lead |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every member that reads the repo — tracked in git, so worktrees inherit it |
|
||||
| role agent definition files | role contract and per-job procedure | a member whose launcher binds its role to the matching file in its worktree |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a member told to load one |
|
||||
| the bridge's own docs | design detail, flows, error model | on demand |
|
||||
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. A member without a
|
||||
repo checkout still gets the launcher's reply charter, which is why that one rule stays there.
|
||||
Peers that don't read `CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they*
|
||||
must obey belongs in the charter, not here.
|
||||
|
||||
## Project addendum — claude-bridge (not part of the canonical block)
|
||||
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace) and
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
This repo *is* the bridge, so the canonical block above is not documentation about someone else's
|
||||
system: it is the instruction surface this codebase ships. **Every change here must end by asking
|
||||
whether the block still tells the truth.** A code change that silently invalidates it is an
|
||||
incomplete change — the agents reading it have no other source.
|
||||
|
||||
Before you call any work done, check the row that matches what you touched:
|
||||
|
||||
| You changed… | Re-read and update… |
|
||||
|---|---|
|
||||
| a `fleet_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| `Authz` / the role table | invariant 3, and the primary-only vs worker-only claims |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `fleet_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__fleet__*`), and the layering table's top row |
|
||||
| the injector / status gating | invariant 4 |
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
|
||||
That last row is not bookkeeping. Chapters 1–10 answer *how is this built* and *why this way*;
|
||||
none of them has a home for *what can it do and how do I turn it on*, so for twenty tickets a
|
||||
shipped capability landed nowhere and the Roadmap went on claiming the stage was finished. The
|
||||
*why* line is the one that matters — without it a decision gets re-litigated from scratch a month
|
||||
later. Internal contract changes go to `wiki/9-Implementation.md` instead; test and coverage work
|
||||
is a Roadmap line. A change that touches none of the three earns no entry, and that is a normal
|
||||
outcome rather than an omission.
|
||||
|
||||
Then **propagate**: the block in this file and the template in the wiki
|
||||
([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
block*) must stay byte-identical, and other projects carrying the block need the same edit. Verify
|
||||
rather than trust:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
import pathlib
|
||||
c = pathlib.Path("CLAUDE.md").read_text()
|
||||
w = pathlib.Path("wiki/7-Use-Cases.md").read_text()
|
||||
S, E = "## Bridge communication (enforced", "## Project addendum — claude-bridge"
|
||||
block = c[c.index(S):c.index(E)].rstrip() + "\n"
|
||||
i = w.index("```markdown\n") + len("```markdown\n")
|
||||
print("in sync:", w[i:w.index("\n```\n", i) + 1] == block)
|
||||
PY
|
||||
```
|
||||
|
||||
## IDE MCP tools & validation workflow (enforced)
|
||||
|
||||
> **Primary only.** Workers have no IDE MCP mount — if you are a worker, skip this section and
|
||||
> report the build/test output you actually ran (see §Bridge communication → Worker).
|
||||
|
||||
Two IDE MCP servers are connected: **intellij-index** (semantic code intelligence) and
|
||||
**jetbrains** (file problems, reformat, debugger). IntelliJ has multiple projects open; our
|
||||
module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
module is **`fleetd`**. Always pass these to IDE MCP tools:
|
||||
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/bridged`
|
||||
- IDE paths are relative to `bridged/` (e.g. `src/main/java/dev/ltms/bridged/...`)
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/fleetd`
|
||||
- IDE paths are relative to `fleetd/` (e.g. `src/main/java/dev/ltms/fleet/...`)
|
||||
|
||||
### After editing any file — mandatory
|
||||
|
||||
@@ -20,6 +328,15 @@ module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
test run). A per-file-clean file can still break the build or another module. This is the
|
||||
whole-project gate before declaring work done or committing.
|
||||
|
||||
**Whenever dependencies change (or a `pom.xml` edit), validate CVEs with
|
||||
`jetbrains get_file_problems{filePath: "fleetd/pom.xml"}`** — its Mend.io check reflects the
|
||||
dependencies on disk. (Note: `ide_diagnostics` / intellij-index does NOT re-resolve dependencies
|
||||
after a pom edit without a full Maven reimport, so it reports stale CVE results — don't trust it
|
||||
for this.) Treat a CVE warning like any other: bump to a patched version and confirm
|
||||
`mvn clean install` still passes. If the latest available version is still flagged (EOL line,
|
||||
"insufficient information", or config-file-only advisories), document it as accepted in the pom
|
||||
rather than chasing a fix that doesn't exist.
|
||||
|
||||
### Use IDE MCP tools for navigation, refactoring, and diagnostics only
|
||||
|
||||
- **Navigate (prefer over Grep/Read for symbols):** `ide_find_definition`, `ide_find_class`,
|
||||
|
||||
@@ -9,32 +9,33 @@ Sibling of [`crush-bridge`](https://git.ltms.dev/systems/vms) (which drives a he
|
||||
process*, so it inherits `CLAUDE.md`, hooks, skills, and MCP — just pointed at a
|
||||
cheaper/local model.
|
||||
|
||||
## Leading approach — herdr-centric message server (`bridged`)
|
||||
## Leading approach — herdr-centric message server (`fleetd`)
|
||||
|
||||
A small always-on message server, **`bridged`**, controls
|
||||
A small always-on message server, **`fleetd`**, controls
|
||||
[herdr](https://herdr.dev) (an agent multiplexer) over its Unix-socket API and exposes a
|
||||
clean 2-way messaging API as an **MCP server that both the primary and the workers mount** —
|
||||
one unified Claude setup and the **sole communication gateway** (REST/SSE stays for non-Claude
|
||||
clients; any broker is `bridged`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `bridged` owns
|
||||
clients; any broker is `fleetd`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `fleetd` owns
|
||||
policy (subscription boundary, session lifecycle, status-gated delivery) and the client
|
||||
contract. The worker `claude` launches with `ANTHROPIC_BASE_URL=https://ollama.ltms.dev` + a
|
||||
bearer token; the primary Opus stays env-clean and calls `bridged`'s MCP tools.
|
||||
contract. A Claude member launches with `ANTHROPIC_BASE_URL` pointed at the gateway,
|
||||
`https://llm.ltms.dev/anthropic`, plus a bearer token; the lead stays env-clean and calls
|
||||
`fleetd`'s MCP tools. See the wiki's **[13 User Guide](wiki/13-User-Guide.md)** to run it.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["bridged — standalone daemon (not a claude process)"]
|
||||
subgraph BD["fleetd — standalone daemon (not a claude process)"]
|
||||
SRV["SERVER face<br/>MCP · REST/SSE · policy"]
|
||||
CLI["CLIENT face<br/>status-gated injector · herdr socket"]
|
||||
SRV --> CLI
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
M["ollama.ltms.dev<br/>(worker model)"]
|
||||
M["llm.ltms.dev<br/>(the one gateway)"]
|
||||
|
||||
OPUS -->|"MCP bridge_send (blocks)"| SRV
|
||||
W -.->|"MCP bridge_reply"| SRV
|
||||
OPUS -->|"MCP fleet_send (blocks)"| SRV
|
||||
W -.->|"MCP fleet_reply"| SRV
|
||||
CLI -->|"Unix socket<br/>send_text · events.subscribe"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
W -->|"inference"| M
|
||||
@@ -46,23 +47,26 @@ flowchart LR
|
||||
```
|
||||
|
||||
- **Subscription boundary:** the *primary* never sets `ANTHROPIC_BASE_URL` (stays on
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `bridged` itself is a
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `fleetd` itself is a
|
||||
plain daemon (no Anthropic quota), so it may poll/subscribe freely.
|
||||
- **One gateway (unified MCP setup):** `bridged` is the **sole communication path** for every
|
||||
- **One gateway (unified MCP setup):** `fleetd` is the **sole communication path** for every
|
||||
Claude session. Primary and workers each mount it as an MCP server (one `claude mcp add`
|
||||
line, same on both) and talk over MCP tools — `bridge_send` / `bridge_reply` / `bridge_ask` /
|
||||
`bridge_status`. **No Claude session ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `bridged`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`bridge_send`);
|
||||
`bridged` holds it open until the worker calls `bridge_reply` or its turn hits
|
||||
line, same on both) and talk over MCP tools — `fleet_send` / `fleet_reply` /
|
||||
`fleet_status` (with `fleet_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `fleetd`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
**Tool naming:** the tools are `fleet_*` (renamed from `bridge_*` in CB-622). The old
|
||||
`bridge_*` names were removed in CB-634 — only `fleet_*` answers now.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`fleet_send`);
|
||||
`fleetd` holds it open until the worker calls `fleet_reply` or its turn hits
|
||||
`agent_status=done`, then returns the reply as the tool result. No cross-turn busy-poll, so
|
||||
no quota burn. SSE is an optional side-channel for humans/dashboards watching status.
|
||||
- **Worker → primary** rides `bridged`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `bridged` **injects the primary's idle pane** when it's
|
||||
- **Worker → primary** rides `fleetd`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `fleetd` **injects the primary's idle pane** when it's
|
||||
ready), so *no keystroke-into-primary and no broker are involved, even single-host*. The one
|
||||
exception: a split-host primary that isn't a herdr pane wakes via its own `Stop`-hook, which
|
||||
polls **`bridged`** (never a broker). See the wiki for the two topologies.
|
||||
polls **`fleetd`** (never a broker). See the wiki for the two topologies.
|
||||
- **Different model per process** sidesteps Claude Code's lack of per-subagent provider
|
||||
routing — the worker isn't a subagent, it's its own configured process.
|
||||
- **AgentAPI** ([`coder/agentapi`](https://github.com/coder/agentapi)) is retained only as a
|
||||
@@ -71,11 +75,11 @@ flowchart LR
|
||||
|
||||
## Docs
|
||||
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/lms/claude-bridge/wiki)**,
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/fleet/fleetd/wiki)**,
|
||||
vendored here as a submodule under [`wiki/`](./wiki):
|
||||
|
||||
```bash
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/lms/claude-bridge.git
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/fleet/fleetd.git
|
||||
# or, after a plain clone:
|
||||
git submodule update --init
|
||||
```
|
||||
@@ -85,6 +89,38 @@ Gitea wiki.
|
||||
|
||||
## Status
|
||||
|
||||
🟢 Design — herdr-centric **`bridged`** message server selected as the primary approach
|
||||
(2026-07-11), superseding the AgentAPI plan (2026-07-08). AgentAPI retained as fallback
|
||||
injector.
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`fleetd`** message server is built and in
|
||||
real use: an Opus primary delegates tasks to off-subscription workers that reply through the
|
||||
bridge (code reviews delegated this way have produced committed bug fixes). Selected as the
|
||||
primary approach 2026-07-11, superseding the AgentAPI plan (2026-07-08); AgentAPI retained as a
|
||||
fallback injector.
|
||||
|
||||
**Shipped** (Java 25 · Maven · 266 unit/acceptance tests green; the live-herdr and broker contract
|
||||
tests run separately via `mvn test -Pcontract`):
|
||||
|
||||
- **Core gateway** — herdr socket client (contract-tested vs live 0.7.0); guard-checked worker
|
||||
spawn with `ANTHROPIC_BASE_URL` injected only into the worker's env; status-gated injector;
|
||||
blocking `fleet_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `fleet_send` / `fleet_reply` / `fleet_status` (messaging) and `fleet_spawn`
|
||||
/ `fleet_list` / `fleet_stop` / `fleet_profiles` / `fleet_poll` (fleet). Caller identity is
|
||||
connection-based (loopback peer PID → herdr pane), so the same mount serves primary and workers.
|
||||
- **Delivery reliability** — completion fallback (a confirmed `working→idle` turn resolves a
|
||||
send); async fire-and-poll (beats the caller's MCP call timeout for long tasks); and failure
|
||||
detection for wedged (`unknown`), vanished, and never-ready workers so a send never hangs.
|
||||
- **Fleet** — multiple worker profiles, each with an independent base_url guard check; workers
|
||||
inherit the primary's working directory (never `$HOME`); a readiness gate holds delivery until
|
||||
a worker's Claude has connected the bridge MCP (no paste lost into its boot window).
|
||||
- **Blocked-worker path** — `fleet_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
ask the primary and resumes the *same* turn with the answer (CB-205).
|
||||
- **Session lifecycle** — session manager with spawn/reuse/recycle, `idle_ttl` reaper, `context_cap`,
|
||||
and graceful drain on shutdown (CB-301/CB-303); per-worker git worktrees on their own branch with
|
||||
a config-parity overlay, so parallel implementers never stomp each other (CB-301-ext).
|
||||
- **Reliable worker→primary delivery** — a durable `ReplyInbox` (in-memory by default, AMQP/LavinMQ
|
||||
for cross-restart durability) holds a reply that arrives with no open send, and an active
|
||||
status-gated push loop nudges the primary to drain it (CB-307).
|
||||
- **Pluggable peers** — a `PeerLauncher` SPI with two in-tree adapters, `claude-code` and `opencode`,
|
||||
routed by a `kind:` discriminator (CB-401/CB-402).
|
||||
|
||||
**Next** (see the [roadmap](wiki/8-Roadmap.md)) — Stage 5 hardening (auth/TLS, `/metrics`, CI,
|
||||
service supervision, per-session authz + audit), then cross-host: CB-308 multi-host federation and
|
||||
CB-500 multi-tier coordination.
|
||||
|
||||
@@ -1,11 +0,0 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from bridged.example.yaml)
|
||||
bridged.yaml
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -1,33 +0,0 @@
|
||||
# bridged configuration (example). Copy to bridged.yaml and adjust.
|
||||
#
|
||||
# bridged is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback — bridged is same-host in Stage-1.
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8080
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# How a worker session is spawned. Stage-1 uses the existing ccs `ltms-local`
|
||||
# profile, whose .claude.json routes to the gx00 vLLM below.
|
||||
worker:
|
||||
profile: ltms-local
|
||||
baseUrl: http://gx00.gw:8000 # the gx00 vLLM (models: coder / deepseek-v4-flash)
|
||||
model: coder
|
||||
# Placement: each worker lands in its OWN tab inside a dedicated worker space, so it
|
||||
# never splits or clutters your real work spaces. Use `pane` for the legacy behaviour
|
||||
# (split the currently-focused tab).
|
||||
placement: tab # tab | pane
|
||||
workspace: bridged-workers # the dedicated worker space (found-or-created, shared)
|
||||
tabLabel: "worker: {profile} #{n}" # {profile}/{model}/{n} substituted; {n} keeps sibling tabs distinct
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Grounded in ltms-local's real endpoints.
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- ollama.ltms.dev
|
||||
-122
@@ -1,122 +0,0 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project xmlns="http://maven.apache.org/POM/4.0.0"
|
||||
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
|
||||
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
<groupId>dev.ltms</groupId>
|
||||
<artifactId>bridged</artifactId>
|
||||
<version>0.1.0-SNAPSHOT</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<name>bridged</name>
|
||||
<description>claude-bridge message server: sole gateway between primary/worker Claude sessions and herdr</description>
|
||||
|
||||
<properties>
|
||||
<maven.compiler.release>25</maven.compiler.release>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<mainClass>dev.ltms.bridged.Bridged</mainClass>
|
||||
|
||||
<jackson.version>2.18.2</jackson.version>
|
||||
<javalin.version>6.3.0</javalin.version>
|
||||
<slf4j.version>2.0.16</slf4j.version>
|
||||
<logback.version>1.5.15</logback.version>
|
||||
<junit.version>5.11.4</junit.version>
|
||||
</properties>
|
||||
|
||||
<dependencies>
|
||||
<!-- JSON + YAML (config, herdr wire format, REST bodies) -->
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.core</groupId>
|
||||
<artifactId>jackson-databind</artifactId>
|
||||
<version>${jackson.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.dataformat</groupId>
|
||||
<artifactId>jackson-dataformat-yaml</artifactId>
|
||||
<version>${jackson.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- REST: the testability surface. MCP tools are thin adapters over these endpoints. -->
|
||||
<dependency>
|
||||
<groupId>io.javalin</groupId>
|
||||
<artifactId>javalin</artifactId>
|
||||
<version>${javalin.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
<artifactId>slf4j-api</artifactId>
|
||||
<version>${slf4j.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>ch.qos.logback</groupId>
|
||||
<artifactId>logback-classic</artifactId>
|
||||
<version>${logback.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Tests -->
|
||||
<dependency>
|
||||
<groupId>org.junit.jupiter</groupId>
|
||||
<artifactId>junit-jupiter</artifactId>
|
||||
<version>${junit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
<finalName>bridged</finalName>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-compiler-plugin</artifactId>
|
||||
<version>3.14.0</version>
|
||||
</plugin>
|
||||
|
||||
<!-- Unit tests run by default; contract tests (live herdr) are tag-excluded. -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-surefire-plugin</artifactId>
|
||||
<version>3.5.2</version>
|
||||
<configuration>
|
||||
<excludedGroups>${excludedGroups}</excludedGroups>
|
||||
</configuration>
|
||||
</plugin>
|
||||
|
||||
<!-- Runnable fat jar: java -jar target/bridged.jar -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-shade-plugin</artifactId>
|
||||
<version>3.6.0</version>
|
||||
<executions>
|
||||
<execution>
|
||||
<phase>package</phase>
|
||||
<goals><goal>shade</goal></goals>
|
||||
<configuration>
|
||||
<transformers>
|
||||
<transformer implementation="org.apache.maven.plugins.shade.resource.ManifestResourceTransformer">
|
||||
<mainClass>${mainClass}</mainClass>
|
||||
</transformer>
|
||||
<transformer implementation="org.apache.maven.plugins.shade.resource.ServicesResourceTransformer"/>
|
||||
</transformers>
|
||||
</configuration>
|
||||
</execution>
|
||||
</executions>
|
||||
</plugin>
|
||||
</plugins>
|
||||
</build>
|
||||
|
||||
<profiles>
|
||||
<!-- `mvn test` excludes contract tests. `mvn test -Pcontract` runs them against a live herdr. -->
|
||||
<profile>
|
||||
<id>default-excludes</id>
|
||||
<activation><activeByDefault>true</activeByDefault></activation>
|
||||
<properties><excludedGroups>contract</excludedGroups></properties>
|
||||
</profile>
|
||||
<profile>
|
||||
<id>contract</id>
|
||||
<properties><excludedGroups></excludedGroups></properties>
|
||||
</profile>
|
||||
</profiles>
|
||||
</project>
|
||||
@@ -1,64 +0,0 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.inject.StatusPoller;
|
||||
import dev.ltms.bridged.rest.BridgedApp;
|
||||
import dev.ltms.bridged.worker.WorkerService;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
|
||||
/**
|
||||
* {@code bridged} entry point. Wires the real herdr socket client to the REST app and
|
||||
* starts listening. Before anything else it asserts its own environment is clean —
|
||||
* {@code bridged} is not a Claude process and must never carry a base_url.
|
||||
*/
|
||||
public final class Bridged {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Bridged.class);
|
||||
|
||||
/** How often the injector samples a busy worker's status while it has queued work. */
|
||||
private static final long INJECT_POLL_MILLIS = 250;
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = Path.of(args.length > 0 ? args[0] : "bridged.yaml");
|
||||
BridgedConfig cfg = BridgedConfig.load(configPath);
|
||||
|
||||
// The primary/host env that launched bridged must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(herdr::close));
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
WorkerService workers = new WorkerService(agents, spaces, guard, cfg.worker(), System::getenv);
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
Injector injector = new Injector(agents);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, INJECT_POLL_MILLIS);
|
||||
poller.start();
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(poller::stop));
|
||||
|
||||
Javalin app = new BridgedApp(herdr, workers).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("bridged listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
private Bridged() {
|
||||
}
|
||||
}
|
||||
@@ -1,121 +0,0 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import com.fasterxml.jackson.annotation.JsonIgnoreProperties;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.dataformat.yaml.YAMLFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* {@code bridged} configuration, loaded from a YAML file (see
|
||||
* {@code bridged.example.yaml}). Unknown keys are ignored so config can grow ahead
|
||||
* of the code.
|
||||
*
|
||||
* @param bind REST/MCP listen host:port
|
||||
* @param herdrSocket path to herdr's Unix socket ({@code null} → client default)
|
||||
* @param worker worker-spawn settings
|
||||
* @param guard subscription-boundary allowlist
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record BridgedConfig(
|
||||
Bind bind,
|
||||
String herdrSocket,
|
||||
Worker worker,
|
||||
Guard guard) {
|
||||
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Bind(String host, int port) {
|
||||
public Bind {
|
||||
if (host == null || host.isBlank()) host = "127.0.0.1";
|
||||
if (port <= 0) port = 8080;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* @param profile ccs profile a worker is spawned under (Stage-1: {@code ltms-local})
|
||||
* @param baseUrl the off-subscription endpoint injected as {@code ANTHROPIC_BASE_URL}
|
||||
* @param model model alias, injected as {@code ANTHROPIC_MODEL} (may be {@code null})
|
||||
* @param configDir {@code CLAUDE_CONFIG_DIR} so the worker inherits the profile's
|
||||
* skills/MCP/hooks (may be {@code null})
|
||||
* @param tokenEnv name of the host env var holding the worker's auth token; its value
|
||||
* is injected as {@code ANTHROPIC_AUTH_TOKEN} (never stored in config)
|
||||
* @param argv launch command; defaults to {@code ["claude"]}
|
||||
* @param placement where a worker lands: {@code "tab"} (default — its own tab in the
|
||||
* worker space) or {@code "pane"} (legacy — split the focused tab)
|
||||
* @param workspace label of the dedicated worker space; found-or-created on first
|
||||
* spawn (default {@code "bridged-workers"}). A future per-session
|
||||
* layout is just a distinct label here — the shared space is default.
|
||||
* @param tabLabel template for a worker tab's label; {@code {profile}}/{@code {model}}
|
||||
* and {@code {n}} (per-worker number, to keep sibling tabs distinct)
|
||||
* are substituted (default {@code "worker: {profile} #{n}"})
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Worker(String profile, String baseUrl, String model,
|
||||
String configDir, String tokenEnv, List<String> argv,
|
||||
String placement, String workspace, String tabLabel) {
|
||||
public Worker {
|
||||
argv = (argv == null || argv.isEmpty()) ? List.of("claude") : List.copyOf(argv);
|
||||
tokenEnv = (tokenEnv == null || tokenEnv.isBlank()) ? "BRIDGED_WORKER_TOKEN" : tokenEnv;
|
||||
placement = (placement == null || placement.isBlank()) ? "tab" : placement.toLowerCase();
|
||||
workspace = (workspace == null || workspace.isBlank()) ? "bridged-workers" : workspace;
|
||||
tabLabel = (tabLabel == null || tabLabel.isBlank()) ? "worker: {profile} #{n}" : tabLabel;
|
||||
}
|
||||
|
||||
/** True when workers should land in their own tab in the worker space. */
|
||||
public boolean tabPlacement() {
|
||||
return "tab".equals(placement);
|
||||
}
|
||||
|
||||
/**
|
||||
* Render {@link #tabLabel} for the {@code n}-th worker (substitutes
|
||||
* {@code {profile}}/{@code {model}}/{@code {n}}), so sibling worker tabs are distinct.
|
||||
*/
|
||||
public String renderTabLabel(long n) {
|
||||
return tabLabel
|
||||
.replace("{profile}", profile == null ? "" : profile)
|
||||
.replace("{model}", model == null ? "" : model)
|
||||
.replace("{n}", Long.toString(n));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Subscription boundary. Only these hosts may back a worker's
|
||||
* {@code ANTHROPIC_BASE_URL}; the primary must carry none.
|
||||
*
|
||||
* @param offSubscriptionHosts hostnames allowed for worker base_urls
|
||||
*/
|
||||
@JsonIgnoreProperties(ignoreUnknown = true)
|
||||
public record Guard(List<String> offSubscriptionHosts) {
|
||||
public Guard {
|
||||
offSubscriptionHosts = offSubscriptionHosts == null ? List.of() : List.copyOf(offSubscriptionHosts);
|
||||
}
|
||||
|
||||
public Set<String> hostSet() {
|
||||
return Set.copyOf(offSubscriptionHosts);
|
||||
}
|
||||
}
|
||||
|
||||
private static final ObjectMapper YAML = new ObjectMapper(new YAMLFactory());
|
||||
|
||||
/** Load and validate config from {@code path}. */
|
||||
public static BridgedConfig load(Path path) {
|
||||
try {
|
||||
BridgedConfig cfg = YAML.readValue(Files.readString(path), BridgedConfig.class);
|
||||
return cfg.withDefaults();
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot read bridged config at " + path, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** Fill in nested defaults so callers never see nulls for structural fields. */
|
||||
public BridgedConfig withDefaults() {
|
||||
Bind b = bind != null ? bind : new Bind(null, 0);
|
||||
Guard g = guard != null ? guard : new Guard(List.of());
|
||||
return new BridgedConfig(b, herdrSocket, worker, g);
|
||||
}
|
||||
}
|
||||
@@ -1,95 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Domain layer over herdr's native {@code agent.*} namespace — the worker south side.
|
||||
* Chosen in the CB-102 spike over the pane + {@code send_text} fallback because
|
||||
* {@code agent.start} takes a first-class {@code env} map (clean, guard-checked
|
||||
* subscription injection) and herdr tracks each worker's Claude session UUID itself.
|
||||
*
|
||||
* <p>Every method is one herdr call through the injected {@link HerdrClient}, so this
|
||||
* layer is unit-testable with a fake and contract-tested against a live daemon.
|
||||
*/
|
||||
public final class AgentControl {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
public AgentControl(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn an agent. {@code env} is applied to the process environment verbatim — this
|
||||
* is where a worker's {@code ANTHROPIC_BASE_URL} lives, and the ONLY place it should.
|
||||
*
|
||||
* @param name label/kind for herdr status detection (e.g. {@code "claude"})
|
||||
* @param argv launch command, e.g. {@code ["claude"]}
|
||||
* @param env process environment additions ({@code ANTHROPIC_BASE_URL}, token, …)
|
||||
*/
|
||||
public Agent start(String name, List<String> argv, Map<String, String> env) {
|
||||
return start(name, argv, env, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn an agent into a specific tab. With a non-null {@code tabId} the worker lands
|
||||
* in that tab (the placement policy's dedicated worker tab); with {@code null} herdr
|
||||
* splits the currently-focused tab (legacy pane placement).
|
||||
*/
|
||||
public Agent start(String name, List<String> argv, Map<String, String> env, String tabId) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("name", name);
|
||||
params.put("argv", argv);
|
||||
params.put("env", env);
|
||||
if (tabId != null) {
|
||||
params.put("tab_id", tabId);
|
||||
}
|
||||
JsonNode result = herdr.call("agent.start", params);
|
||||
return Agent.from(result.get("agent"));
|
||||
}
|
||||
|
||||
/** Deliver {@code text} to an agent (its next prompt input). */
|
||||
public void send(String target, String text) {
|
||||
herdr.call("agent.send", Map.of("target", target, "text", text));
|
||||
}
|
||||
|
||||
/**
|
||||
* Read an agent's terminal.
|
||||
*
|
||||
* @param source one of {@code visible|recent|recent_unwrapped|detection}
|
||||
*/
|
||||
public String read(String target, String source) {
|
||||
JsonNode result = herdr.call("agent.read", Map.of("target", target, "source", source));
|
||||
return result.path("read").path("text").asText("");
|
||||
}
|
||||
|
||||
/** Current agent record (status, session UUID, pane). */
|
||||
public Agent get(String target) {
|
||||
return Agent.from(herdr.call("agent.get", Map.of("target", target)).get("agent"));
|
||||
}
|
||||
|
||||
/** Just the lifecycle status — what the status-gated injector checks before send. */
|
||||
public AgentStatus status(String target) {
|
||||
return get(target).status();
|
||||
}
|
||||
|
||||
/** All agents herdr tracks — the discovery surface ("what workers exist"). */
|
||||
public List<Agent> list() {
|
||||
JsonNode result = herdr.call("agent.list");
|
||||
List<Agent> out = new ArrayList<>();
|
||||
for (JsonNode a : result.path("agents")) {
|
||||
out.add(Agent.from(a));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Tear a worker down (there is no agent.stop — close its pane). */
|
||||
public void close(String paneId) {
|
||||
herdr.call("pane.close", Map.of("pane_id", paneId));
|
||||
}
|
||||
}
|
||||
@@ -1,29 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
/**
|
||||
* A herdr agent's lifecycle state, as reported by {@code agent_status}. Drives the
|
||||
* status-gated injector: a worker is safe to inject into only when {@link #IDLE} or
|
||||
* {@link #BLOCKED}, never mid-turn ({@link #WORKING}).
|
||||
*/
|
||||
public enum AgentStatus {
|
||||
IDLE,
|
||||
WORKING,
|
||||
BLOCKED,
|
||||
UNKNOWN;
|
||||
|
||||
/** Map herdr's wire string ({@code idle|working|blocked|unknown}) to the enum. */
|
||||
public static AgentStatus fromWire(String s) {
|
||||
if (s == null) return UNKNOWN;
|
||||
return switch (s.toLowerCase()) {
|
||||
case "idle" -> IDLE;
|
||||
case "working" -> WORKING;
|
||||
case "blocked" -> BLOCKED;
|
||||
default -> UNKNOWN;
|
||||
};
|
||||
}
|
||||
|
||||
/** Whether {@code bridged} may inject a message now without stepping on a live turn. */
|
||||
public boolean injectable() {
|
||||
return this == IDLE || this == BLOCKED;
|
||||
}
|
||||
}
|
||||
@@ -1,182 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayDeque;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Deque;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* The status-gated injector (CB-103): the single writer that delivers a message into a
|
||||
* worker only when it is safe — {@code idle} or {@code blocked}, never mid-turn.
|
||||
*
|
||||
* <p>Per target it holds a FIFO queue and delivers <strong>at most one message per turn</strong>:
|
||||
* after a send it waits for the worker to pick the message up (go {@code working}) before
|
||||
* delivering the next, so two rapid deliveries never interleave into one turn. Each
|
||||
* status-check-then-send for a target is serialized on the target's monitor, closing the
|
||||
* TOCTOU window between "is it idle?" and "send" — one writer per worker.
|
||||
*
|
||||
* <p><strong>Polling caveat.</strong> This is driven by {@link #onStatus} sampling (a
|
||||
* {@code StatusPoller}), not by reliable status <em>edges</em>. A turn can begin and end
|
||||
* entirely between two polls, so the {@code working} pickup may never be sampled. To avoid
|
||||
* wedging a queue forever, an awaited pickup is released after {@link #PICKUP_GRACE_POLLS}
|
||||
* consecutive injectable samples (the worker has plainly moved on). Only {@code working} — not
|
||||
* a transient {@code unknown} — counts as a real pickup, so a detection glitch can't prematurely
|
||||
* release the latch. Perfectly reliable turn boundaries require a herdr {@code events.subscribe}
|
||||
* stream; that is the intended upgrade and would replace only the sampling, not this queue.
|
||||
*/
|
||||
public final class Injector {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Injector.class);
|
||||
|
||||
/**
|
||||
* How many consecutive injectable samples (with no {@code working} in between) after a send
|
||||
* before we assume the turn completed unobserved and release the pickup latch. At the
|
||||
* default 250ms poll interval this is a ~2s grace — far longer than a worker takes to start
|
||||
* a turn, so it only fires on a genuinely missed pickup edge.
|
||||
*/
|
||||
private static final int PICKUP_GRACE_POLLS = 8;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
||||
|
||||
public Injector(AgentControl agents) {
|
||||
this.agents = agents;
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, CompletableFuture<Void> delivered) {
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
private static final class Target {
|
||||
final Deque<Pending> queue = new ArrayDeque<>();
|
||||
boolean awaitingPickup; // sent a message, waiting for the worker to pick it up
|
||||
int injectableSincePickup; // consecutive injectable samples while awaitingPickup
|
||||
|
||||
synchronized void add(Pending p) {
|
||||
queue.add(p);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Queue {@code text} for delivery to {@code target} (a herdr {@code terminal_id}). Returns
|
||||
* immediately with a future that completes when the message is actually sent — the worker
|
||||
* may be mid-turn, in which case delivery waits for the next injectable window.
|
||||
*
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
return t;
|
||||
});
|
||||
return delivered;
|
||||
}
|
||||
|
||||
/**
|
||||
* Feed a fresh status observation for {@code target}. Delivers the head of the queue iff the
|
||||
* worker is injectable and no earlier message is still awaiting pickup. Serialized per target
|
||||
* so the check and the send cannot race another delivery to the same worker; the delivered
|
||||
* future is completed <em>after</em> the monitor is released so a caller's continuation never
|
||||
* runs on the poller thread while it holds the lock.
|
||||
*/
|
||||
public void onStatus(String target, AgentStatus status) {
|
||||
Target t = targets.get(target);
|
||||
if (t == null) return;
|
||||
|
||||
Pending sent = null;
|
||||
RuntimeException sendError = null;
|
||||
synchronized (t) {
|
||||
if (status == AgentStatus.WORKING) {
|
||||
// Definitive pickup: the worker is busy on our last message.
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
if (t.awaitingPickup && ++t.injectableSincePickup >= PICKUP_GRACE_POLLS) {
|
||||
// Pickup edge was never sampled (turn faster than the poll, or status lag).
|
||||
// The worker has plainly moved on — release the latch rather than wedge.
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
}
|
||||
if (!t.awaitingPickup) {
|
||||
Pending p = t.queue.peek();
|
||||
if (p != null) {
|
||||
try {
|
||||
agents.send(target, p.text());
|
||||
t.queue.poll();
|
||||
t.awaitingPickup = true;
|
||||
t.injectableSincePickup = 0;
|
||||
sent = p;
|
||||
} catch (RuntimeException e) {
|
||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
||||
// rather than blocking the queue behind it.
|
||||
t.queue.poll();
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// UNKNOWN (and any other non-injectable, non-working): do nothing — neither a safe
|
||||
// window nor a reliable pickup signal, so we must not deliver or release the latch.
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent, so the map cannot grow without
|
||||
// bound across many short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
|
||||
if (sent != null) {
|
||||
if (sendError != null) {
|
||||
log.warn("inject to {} failed, dropped message: {}", target, sendError.getMessage());
|
||||
sent.delivered().completeExceptionally(sendError);
|
||||
} else {
|
||||
sent.delivered().complete(null);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Targets the poller must keep sampling: those with a queued message or an awaited pickup. */
|
||||
public Set<String> activeTargets() {
|
||||
return targets.entrySet().stream()
|
||||
.filter(e -> {
|
||||
synchronized (e.getValue()) {
|
||||
return !e.getValue().queue.isEmpty() || e.getValue().awaitingPickup;
|
||||
}
|
||||
})
|
||||
.map(java.util.Map.Entry::getKey)
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Forget a target whose worker is gone, failing every still-queued message so awaiting
|
||||
* callers unblock instead of hanging forever. Futures are completed after the monitor is
|
||||
* released.
|
||||
*/
|
||||
public void drop(String target, Throwable cause) {
|
||||
Target t = targets.remove(target);
|
||||
if (t == null) return;
|
||||
List<Pending> pending;
|
||||
synchronized (t) {
|
||||
pending = new ArrayList<>(t.queue);
|
||||
t.queue.clear();
|
||||
}
|
||||
for (Pending p : pending) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,112 +0,0 @@
|
||||
package dev.ltms.bridged.rest;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import dev.ltms.bridged.guard.GuardException;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.worker.WorkerService;
|
||||
import io.javalin.Javalin;
|
||||
import io.javalin.http.Context;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* The REST surface — {@code bridged}'s contract and its testability seam. Every
|
||||
* feature is reachable here without Claude or MCP in the loop, so each is an
|
||||
* acceptance test against plain HTTP. MCP tools (later) are thin adapters over these
|
||||
* same endpoints and are validated by parity, not by re-implementing behaviour.
|
||||
*
|
||||
* <p>Built from injected collaborators so tests supply fakes and run on an ephemeral
|
||||
* port; {@code main} supplies the real Unix-socket client and worker service.
|
||||
*/
|
||||
public final class BridgedApp {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final WorkerService workers;
|
||||
|
||||
public BridgedApp(HerdrClient herdr, WorkerService workers) {
|
||||
this.herdr = herdr;
|
||||
this.workers = workers;
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
public Javalin build() {
|
||||
Javalin app = Javalin.create(cfg -> cfg.showJavalinBanner = false);
|
||||
app.get("/healthz", this::healthz);
|
||||
app.get("/sessions", this::sessions);
|
||||
app.get("/agents", this::agents);
|
||||
app.post("/workers", this::spawnWorker);
|
||||
app.delete("/workers/{paneId}", this::stopWorker);
|
||||
return app;
|
||||
}
|
||||
|
||||
/** Liveness + herdr reachability. 200 when herdr answers ping, 503 otherwise. */
|
||||
private void healthz(Context ctx) {
|
||||
try {
|
||||
JsonNode pong = herdr.call("ping");
|
||||
ctx.status(200).json(Map.of(
|
||||
"status", "ok",
|
||||
"herdr", Map.of(
|
||||
"version", pong.path("version").asText(""),
|
||||
"protocol", pong.path("protocol").asInt())));
|
||||
} catch (HerdrException e) {
|
||||
ctx.status(503).json(Map.of(
|
||||
"status", "degraded",
|
||||
"herdr", "unreachable",
|
||||
"detail", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
/** Sessions view derived from herdr {@code workspace.list} (one workspace → one row). */
|
||||
private void sessions(Context ctx) {
|
||||
JsonNode result = herdr.call("workspace.list");
|
||||
List<Map<String, Object>> out = new ArrayList<>();
|
||||
for (JsonNode w : result.path("workspaces")) {
|
||||
out.add(Map.of(
|
||||
"id", w.path("workspace_id").asText(""),
|
||||
"label", w.path("label").asText(""),
|
||||
"focused", w.path("focused").asBoolean(false),
|
||||
"paneCount", w.path("pane_count").asInt(),
|
||||
"agentStatus", w.path("agent_status").asText("unknown")));
|
||||
}
|
||||
ctx.status(200).json(Map.of("sessions", out));
|
||||
}
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
ctx.status(200).json(Map.of("agents", workers.list().stream().map(BridgedApp::view).toList()));
|
||||
}
|
||||
|
||||
/** Spawn a guard-checked worker. 403 if the base_url would breach the subscription boundary. */
|
||||
private void spawnWorker(Context ctx) {
|
||||
try {
|
||||
Agent worker = workers.spawn();
|
||||
ctx.status(201).json(view(worker));
|
||||
} catch (GuardException e) {
|
||||
ctx.status(403).json(Map.of("error", "subscription_boundary", "detail", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id. */
|
||||
private void stopWorker(Context ctx) {
|
||||
workers.stop(ctx.pathParam("paneId"));
|
||||
ctx.status(204);
|
||||
}
|
||||
|
||||
/** Stable JSON projection of an agent (null-safe for the start-time shape). */
|
||||
private static Map<String, Object> view(Agent a) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("terminalId", a.terminalId());
|
||||
m.put("paneId", a.paneId());
|
||||
m.put("workspaceId", a.workspaceId());
|
||||
m.put("tabId", a.tabId());
|
||||
m.put("sessionId", a.sessionId());
|
||||
m.put("agentType", a.agentType());
|
||||
m.put("status", a.status().name().toLowerCase());
|
||||
return m;
|
||||
}
|
||||
}
|
||||
@@ -1,206 +0,0 @@
|
||||
package dev.ltms.bridged.worker;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Function;
|
||||
|
||||
/**
|
||||
* Spawns and lists worker sessions — the safe path from a delegation request to a
|
||||
* running off-subscription Claude.
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em>
|
||||
* touching herdr, and only then {@code agent.start}. A worker's base_url lives in the
|
||||
* env map handed to herdr and nowhere else; {@code bridged}'s own environment is never
|
||||
* mutated.
|
||||
*
|
||||
* <p>Placement: in the default {@code tab} policy a worker lands in its own tab inside a
|
||||
* dedicated worker space (found-or-created once, then shared), so workers never split or
|
||||
* clutter the user's real work spaces. Teardown removes the worker's pane <em>and</em> its
|
||||
* now-empty tab, tolerating an already-gone worker so a repeated DELETE is harmless.
|
||||
*/
|
||||
public final class WorkerService {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(WorkerService.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final SubscriptionGuard guard;
|
||||
private final BridgedConfig.Worker cfg;
|
||||
private final Function<String, String> env; // host env lookup (injectable for tests)
|
||||
private final AtomicLong nameSeq = new AtomicLong(); // per-worker counter (also the tab #)
|
||||
// Per-process token mixed into each worker name so a fresh process (nameSeq back at 0)
|
||||
// cannot collide with same-profile workers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
|
||||
public WorkerService(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
BridgedConfig.Worker cfg, Function<String, String> env) {
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.guard = guard;
|
||||
this.cfg = cfg;
|
||||
this.env = env;
|
||||
}
|
||||
|
||||
/** Spawn a worker for the configured profile. Guard runs before any herdr call. */
|
||||
public Agent spawn() {
|
||||
String baseUrl = cfg.baseUrl();
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
|
||||
Map<String, String> workerEnv = new LinkedHashMap<>();
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
String token = env.apply(cfg.tokenEnv());
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", token);
|
||||
|
||||
return cfg.tabPlacement() ? spawnInTab(workerEnv) : spawnAsPane(workerEnv);
|
||||
}
|
||||
|
||||
/** Dedicated worker space → own tab → drop the placeholder shell so only the worker remains. */
|
||||
private Agent spawnInTab(Map<String, String> workerEnv) {
|
||||
Workspace space = spaces.ensureWorkspace(cfg.workspace());
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId());
|
||||
log.info("spawning worker profile={} base_url={} space={} tab={}",
|
||||
cfg.profile(), cfg.baseUrl(), space.workspaceId(), tab.tab().tabId());
|
||||
|
||||
Started started;
|
||||
try {
|
||||
started = startUniquelyNamed(workerEnv, tab.tab().tabId());
|
||||
} catch (RuntimeException e) {
|
||||
// The worker never started — don't leave the tab we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned tab {} after spawn error: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
|
||||
// The worker is LIVE now. The remaining steps are cosmetic (drop herdr's seed shell
|
||||
// so the tab holds only the worker; label the tab). They must not fail the spawn or
|
||||
// orphan the running worker — on error we log and still return it so the caller gets
|
||||
// its paneId and can tear it down.
|
||||
if (tab.rootPaneId() != null) {
|
||||
tidy("close seed pane " + tab.rootPaneId(), () -> agents.close(tab.rootPaneId()));
|
||||
} else {
|
||||
log.warn("tab {} had no seed pane in the create response; worker tab may hold an extra pane",
|
||||
tab.tab().tabId());
|
||||
}
|
||||
tidy("label tab " + tab.tab().tabId(),
|
||||
() -> spaces.renameTab(tab.tab().tabId(), cfg.renderTabLabel(started.seq())));
|
||||
log.info("worker started pane={} tab={} terminal={}",
|
||||
started.agent().paneId(), started.agent().tabId(), started.agent().terminalId());
|
||||
return started.agent();
|
||||
}
|
||||
|
||||
/** Run a best-effort post-start cleanup step, logging (not throwing) on failure. */
|
||||
private void tidy(String what, Runnable step) {
|
||||
try {
|
||||
step.run();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("post-start step failed ({}) — worker is running regardless: {}", what, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Legacy placement: herdr splits the currently-focused tab. */
|
||||
private Agent spawnAsPane(Map<String, String> workerEnv) {
|
||||
log.info("spawning worker (pane placement) profile={} base_url={} argv={}",
|
||||
cfg.profile(), cfg.baseUrl(), cfg.argv());
|
||||
Agent worker = startUniquelyNamed(workerEnv, null).agent();
|
||||
log.info("worker started pane={} terminal={}", worker.paneId(), worker.terminalId());
|
||||
return worker;
|
||||
}
|
||||
|
||||
/** A started worker together with the sequence its unique name/label used. */
|
||||
private record Started(Agent agent, long seq) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the worker under a unique herdr agent name. herdr requires each running
|
||||
* agent's {@code name} to be distinct (a 2nd {@code name:"claude"} fails
|
||||
* {@code agent_name_taken}) — the exact case that makes multiple workers useful. The name
|
||||
* is {@code claude-<profile>-<nonce>-<seq>}: {@code seq} distinguishes workers within this
|
||||
* process, and the per-process {@code nonce} keeps a fresh process (whose {@code seq}
|
||||
* restarts at 0) from colliding with same-profile workers that outlived a restart. The
|
||||
* retry is a belt-and-braces backstop for the astronomically unlikely nonce+seq clash;
|
||||
* the name is a label only — herdr detects kind and status from terminal output, not it.
|
||||
*/
|
||||
private Started startUniquelyNamed(Map<String, String> workerEnv, String tabId) {
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = "claude-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(agents.start(name, cfg.argv(), workerEnv, tabId), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("worker name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/** All herdr-tracked agents — discovery for "what workers exist". */
|
||||
public List<Agent> list() {
|
||||
return agents.list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Tear a worker down by pane id: close the pane, and close its tab <em>only</em> when the
|
||||
* worker is that tab's sole occupant. The single-pane check is what makes this safe
|
||||
* regardless of how the worker was placed (or a placement-config change across a restart):
|
||||
* a pane-placement worker sitting in one of the user's shared tabs has siblings, so its
|
||||
* tab is never closed — we only ever remove a tab we created to hold one worker.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed worker) is treated as success; any other failure propagates so
|
||||
* a genuinely failed teardown is not reported as done.
|
||||
*/
|
||||
public void stop(String paneId) {
|
||||
WorkspaceControl.PaneLocation loc = cfg.tabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) throw e;
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated worker tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
private static void putIfPresent(Map<String, String> m, String k, String v) {
|
||||
if (v != null && !v.isBlank()) {
|
||||
m.put(k, v);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,15 +0,0 @@
|
||||
<configuration>
|
||||
<appender name="STDOUT" class="ch.qos.logback.core.ConsoleAppender">
|
||||
<encoder>
|
||||
<pattern>%d{HH:mm:ss.SSS} %-5level [%thread] %logger{28} - %msg%n</pattern>
|
||||
</encoder>
|
||||
</appender>
|
||||
|
||||
<logger name="dev.ltms.bridged" level="DEBUG"/>
|
||||
<logger name="io.javalin" level="INFO"/>
|
||||
<logger name="org.eclipse.jetty" level="WARN"/>
|
||||
|
||||
<root level="INFO">
|
||||
<appender-ref ref="STDOUT"/>
|
||||
</root>
|
||||
</configuration>
|
||||
@@ -1,55 +0,0 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
class BridgedConfigTest {
|
||||
|
||||
@Test
|
||||
void loadsFullConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8080
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
worker:
|
||||
profile: ltms-local
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: coder
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- ollama.ltms.dev
|
||||
""");
|
||||
|
||||
BridgedConfig cfg = BridgedConfig.load(f);
|
||||
assertEquals(8080, cfg.bind().port());
|
||||
assertEquals("ltms-local", cfg.worker().profile());
|
||||
assertTrue(cfg.guard().hostSet().contains("gx00.gw"));
|
||||
assertTrue(cfg.guard().hostSet().contains("ollama.ltms.dev"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void appliesDefaultsForMissingSections(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("minimal.yaml");
|
||||
Files.writeString(f, "bind:\n host: 0.0.0.0\n port: 9000\n");
|
||||
|
||||
BridgedConfig cfg = BridgedConfig.load(f);
|
||||
assertEquals(9000, cfg.bind().port());
|
||||
assertNotNull(cfg.guard(), "guard must default to empty, never null");
|
||||
assertTrue(cfg.guard().offSubscriptionHosts().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void ignoresUnknownKeys(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("future.yaml");
|
||||
Files.writeString(f, "bind:\n port: 8080\nfutureFeature:\n enabled: true\n");
|
||||
assertDoesNotThrow(() -> BridgedConfig.load(f));
|
||||
}
|
||||
}
|
||||
@@ -1,63 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import org.junit.jupiter.api.Tag;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
import static org.junit.jupiter.api.Assumptions.assumeTrue;
|
||||
|
||||
/**
|
||||
* Contract test for the {@code agent.*} south side against a REAL herdr, locking in
|
||||
* the CB-102 spike findings. It spawns a HARMLESS probe command (never {@code claude},
|
||||
* so no subscription/token involvement), proves the {@code env} map reaches the process
|
||||
* environment, exercises status/read, and always tears the pane down.
|
||||
*
|
||||
* <p>Tagged {@code contract}; run with {@code mvn test -Pcontract}.
|
||||
*/
|
||||
@Tag("contract")
|
||||
class AgentControlContractTest {
|
||||
|
||||
private boolean noSocket() {
|
||||
return !Files.exists(UnixSocketHerdrClient.defaultSocketPath());
|
||||
}
|
||||
|
||||
@Test
|
||||
void startInjectsEnvThenReadAndClose() throws Exception {
|
||||
assumeTrue(!noSocket(), "no herdr socket — skipping");
|
||||
try (UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect()) {
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
|
||||
Agent probe = agents.start(
|
||||
"__contract__",
|
||||
List.of("bash", "-c", "printf 'PROBE_BASE=[%s]\\n' \"$ANTHROPIC_BASE_URL\"; sleep 20"),
|
||||
Map.of("ANTHROPIC_BASE_URL", "http://gx00.gw:8000"));
|
||||
|
||||
assertNotNull(probe.terminalId());
|
||||
assertNotNull(probe.paneId());
|
||||
try {
|
||||
// Give the shell a moment to print, then confirm env reached the process.
|
||||
Thread.sleep(800);
|
||||
String visible = agents.read(probe.terminalId(), "visible");
|
||||
assertTrue(visible.contains("PROBE_BASE=[http://gx00.gw:8000]"),
|
||||
"env map must reach the process; saw: " + visible);
|
||||
|
||||
// Status is queryable; the probe appears in the agent list.
|
||||
assertNotNull(agents.status(probe.terminalId()));
|
||||
assertTrue(agents.list().stream()
|
||||
.anyMatch(a -> probe.terminalId().equals(a.terminalId())),
|
||||
"spawned probe should appear in agent.list");
|
||||
} finally {
|
||||
agents.close(probe.paneId());
|
||||
}
|
||||
|
||||
// After close the pane is gone.
|
||||
assertFalse(agents.list().stream()
|
||||
.anyMatch(a -> probe.terminalId().equals(a.terminalId())),
|
||||
"closed probe should no longer be listed");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,163 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Recording fake {@link HerdrClient} for unit/acceptance tests. Returns canned frames
|
||||
* captured from the real herdr 0.7.0 daemon and records every call so tests can assert
|
||||
* both behaviour and that guard-blocked paths never reached herdr.
|
||||
*/
|
||||
public final class FakeHerdr implements HerdrClient {
|
||||
|
||||
public record Call(String method, Object params) {
|
||||
}
|
||||
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
public final List<Call> calls = new ArrayList<>();
|
||||
private boolean healthy = true;
|
||||
private final List<String> extraWorkspaces = new ArrayList<>();
|
||||
private int agentNameTakenFor = 0;
|
||||
private int workerTabPaneCount = 1;
|
||||
private String paneCloseErrorCode = null;
|
||||
private String agentSendErrorCode = null;
|
||||
private volatile String agentStatus = "idle"; // what agent.get reports
|
||||
|
||||
public FakeHerdr healthy(boolean h) {
|
||||
this.healthy = h;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Reject the first {@code n} {@code agent.start} calls with {@code agent_name_taken}. */
|
||||
public FakeHerdr agentNameTakenTimes(int n) {
|
||||
this.agentNameTakenFor = n;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make the worker tab (w9:t2) report this many panes in {@code tab.list} (default 1). */
|
||||
public FakeHerdr withWorkerTabPaneCount(int n) {
|
||||
this.workerTabPaneCount = n;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make {@code pane.close} fail with this herdr error code. */
|
||||
public FakeHerdr paneCloseFailsWith(String code) {
|
||||
this.paneCloseErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Set the {@code agent_status} that {@code agent.get} reports (drives the injector). */
|
||||
public FakeHerdr agentStatus(String status) {
|
||||
this.agentStatus = status;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Make {@code agent.send} fail with this herdr error code. */
|
||||
public FakeHerdr agentSendFailsWith(String code) {
|
||||
this.agentSendErrorCode = code;
|
||||
return this;
|
||||
}
|
||||
|
||||
/** Seed an additional workspace into {@code workspace.list} (e.g. a pre-existing worker space). */
|
||||
public FakeHerdr withWorkspace(String id, String label) {
|
||||
extraWorkspaces.add(("{\"workspace_id\":\"%s\",\"label\":\"%s\",\"focused\":false,"
|
||||
+ "\"pane_count\":1,\"active_tab_id\":\"%s:t1\",\"agent_status\":\"unknown\"}")
|
||||
.formatted(id, label, id));
|
||||
return this;
|
||||
}
|
||||
|
||||
public boolean called(String method) {
|
||||
return calls.stream().anyMatch(c -> c.method().equals(method));
|
||||
}
|
||||
|
||||
public Call lastCall(String method) {
|
||||
return calls.stream().filter(c -> c.method().equals(method))
|
||||
.reduce((_, b) -> b).orElseThrow();
|
||||
}
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
calls.add(new Call(method, params));
|
||||
if (!healthy) throw new HerdrException("herdr unreachable (fake)");
|
||||
try {
|
||||
return switch (method) {
|
||||
case "ping" -> mapper.readTree(
|
||||
"{\"type\":\"pong\",\"version\":\"0.7.0\",\"protocol\":14}");
|
||||
case "workspace.list" -> mapper.readTree(("""
|
||||
{"type":"workspace_list","workspaces":[
|
||||
{"workspace_id":"w1","label":"dev-mgnl","focused":true,"pane_count":7,"agent_status":"unknown"},
|
||||
{"workspace_id":"w2","label":"ltms","focused":false,"pane_count":5,"agent_status":"done"}%s]}""")
|
||||
.formatted(extraWorkspaces.isEmpty() ? "" : "," + String.join(",", extraWorkspaces)));
|
||||
case "agent.list" -> mapper.readTree("""
|
||||
{"type":"agent_list","agents":[
|
||||
{"terminal_id":"term_a","agent":"claude","agent_status":"idle",
|
||||
"agent_session":{"kind":"id","value":"sess-1111"},
|
||||
"workspace_id":"w2","tab_id":"w2:t7","pane_id":"w2:p7"}]}""");
|
||||
case "agent.send" -> {
|
||||
if (agentSendErrorCode != null) {
|
||||
throw new HerdrException("herdr error [" + agentSendErrorCode + "]: agent.send failed",
|
||||
agentSendErrorCode, null);
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
case "agent.get" -> mapper.readTree(("""
|
||||
{"type":"agent_info","agent":{"terminal_id":"term_a","agent":"claude",
|
||||
"agent_status":"%s","workspace_id":"w2","tab_id":"w2:t7","pane_id":"w2:p7"}}""")
|
||||
.formatted(agentStatus));
|
||||
case "agent.start" -> {
|
||||
long starts = calls.stream().filter(c -> c.method().equals("agent.start")).count();
|
||||
if (starts <= agentNameTakenFor) {
|
||||
throw new HerdrException(
|
||||
"herdr error [agent_name_taken]: agent name already used",
|
||||
"agent_name_taken", null);
|
||||
}
|
||||
yield mapper.readTree("""
|
||||
{"type":"agent_started","agent":{
|
||||
"terminal_id":"term_new","name":"claude","agent_status":"unknown",
|
||||
"workspace_id":"w9","tab_id":"w9:t2","pane_id":"w9:pW"}}""");
|
||||
}
|
||||
case "workspace.create" -> mapper.readTree("""
|
||||
{"type":"workspace_created",
|
||||
"workspace":{"workspace_id":"w9","label":"bridged-workers","focused":false,
|
||||
"pane_count":1,"tab_count":1,"active_tab_id":"w9:t1","agent_status":"unknown"},
|
||||
"tab":{"tab_id":"w9:t1","workspace_id":"w9","label":"1","pane_count":1},
|
||||
"root_pane":{"pane_id":"w9:p1","workspace_id":"w9","tab_id":"w9:t1"}}""");
|
||||
case "tab.create" -> mapper.readTree("""
|
||||
{"type":"tab_created",
|
||||
"tab":{"tab_id":"w9:t2","workspace_id":"w9","label":"2","pane_count":1},
|
||||
"root_pane":{"pane_id":"w9:pRoot","workspace_id":"w9","tab_id":"w9:t2"}}""");
|
||||
case "tab.rename" -> mapper.readTree("""
|
||||
{"type":"tab_info","tab":{"tab_id":"w9:t2","workspace_id":"w9",
|
||||
"label":"worker: ltms-local","pane_count":1}}""");
|
||||
case "tab.list" -> mapper.readTree(("""
|
||||
{"type":"tab_list","tabs":[
|
||||
{"tab_id":"w9:t1","workspace_id":"w9","label":"1","pane_count":1},
|
||||
{"tab_id":"w9:t2","workspace_id":"w9","label":"worker: ltms-local","pane_count":%d}]}""")
|
||||
.formatted(workerTabPaneCount));
|
||||
case "tab.close" -> mapper.readTree("{\"type\":\"ok\"}");
|
||||
case "pane.get" -> mapper.readTree("""
|
||||
{"type":"pane_info","pane":{"pane_id":"w9:pW","workspace_id":"w9",
|
||||
"tab_id":"w9:t2","agent_status":"idle"}}""");
|
||||
case "pane.close" -> {
|
||||
if (paneCloseErrorCode != null) {
|
||||
throw new HerdrException("herdr error [" + paneCloseErrorCode + "]: pane.close failed",
|
||||
paneCloseErrorCode, null);
|
||||
}
|
||||
yield mapper.readTree("{\"type\":\"ok\"}");
|
||||
}
|
||||
default -> throw new HerdrException("fake has no canned response for " + method);
|
||||
};
|
||||
} catch (HerdrException e) {
|
||||
throw e; // intentional protocol errors (e.g. agent_name_taken) propagate with their code
|
||||
} catch (Exception e) {
|
||||
throw new HerdrException("fake decode failed for " + method, e);
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
}
|
||||
@@ -1,185 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* Golden-transcript tests for the status-gated injector (CB-103). Delivery rules are driven
|
||||
* deterministically by feeding {@code onStatus}, so no timing or real polling is involved.
|
||||
*/
|
||||
class InjectorTest {
|
||||
|
||||
private static final String T = "term_a";
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final Injector injector = new Injector(new AgentControl(herdr));
|
||||
|
||||
/** Text of every agent.send, in order. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private List<String> sent() {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("agent.send"))
|
||||
.map(c -> ((Map<String, Object>) c.params()).get("text").toString())
|
||||
.toList();
|
||||
}
|
||||
|
||||
@Test
|
||||
void deliversWhenIdle() {
|
||||
CompletableFuture<Void> f = injector.enqueue(T, "hello");
|
||||
assertFalse(f.isDone(), "not delivered until an injectable status arrives");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertTrue(f.isDone());
|
||||
assertEquals(List.of("hello"), sent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void holdsWhileWorkingThenDeliversOnIdle() {
|
||||
injector.enqueue(T, "later");
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
assertEquals(List.of(), sent(), "must not inject mid-turn");
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertEquals(List.of("later"), sent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void blockedIsInjectableButUnknownIsNot() {
|
||||
injector.enqueue(T, "answer");
|
||||
injector.onStatus(T, AgentStatus.UNKNOWN);
|
||||
assertEquals(List.of(), sent(), "unknown status is not safe to inject");
|
||||
injector.onStatus(T, AgentStatus.BLOCKED);
|
||||
assertEquals(List.of("answer"), sent(), "blocked worker can be answered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void twoRapidDeliveriesNeverInterleave() {
|
||||
injector.enqueue(T, "m1");
|
||||
injector.enqueue(T, "m2");
|
||||
|
||||
// First idle window delivers only m1, even if idle is observed twice before pickup.
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertEquals(List.of("m1"), sent(), "second message must wait for the turn to complete");
|
||||
|
||||
// Worker picks up m1 (works), then returns idle → m2 delivers.
|
||||
injector.onStatus(T, AgentStatus.WORKING);
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertEquals(List.of("m1", "m2"), sent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void transientUnknownDoesNotReleaseThePickupLatch() {
|
||||
injector.enqueue(T, "m1");
|
||||
injector.enqueue(T, "m2");
|
||||
injector.onStatus(T, AgentStatus.IDLE); // m1 sent, awaiting pickup
|
||||
assertEquals(List.of("m1"), sent());
|
||||
|
||||
injector.onStatus(T, AgentStatus.UNKNOWN); // a detection glitch is NOT a pickup
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertEquals(List.of("m1"), sent(), "unknown must not let the next message interleave the turn");
|
||||
|
||||
injector.onStatus(T, AgentStatus.WORKING); // real pickup
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertEquals(List.of("m1", "m2"), sent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void missedPickupEdgeIsReleasedByGraceSoTheQueueNeverWedges() {
|
||||
injector.enqueue(T, "m1");
|
||||
injector.enqueue(T, "m2");
|
||||
injector.onStatus(T, AgentStatus.IDLE); // m1 sent
|
||||
assertEquals(List.of("m1"), sent());
|
||||
|
||||
// WORKING is never sampled (turn faster than the poll). The latch must release after
|
||||
// the grace window so m2 is delivered rather than wedged forever.
|
||||
for (int i = 0; i < 20; i++) injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertEquals(List.of("m1", "m2"), sent(), "safety valve must eventually deliver m2");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fifoOrderAcrossManyTurns() {
|
||||
injector.enqueue(T, "a");
|
||||
injector.enqueue(T, "b");
|
||||
injector.enqueue(T, "c");
|
||||
for (int i = 0; i < 3; i++) {
|
||||
injector.onStatus(T, AgentStatus.IDLE); // deliver one
|
||||
injector.onStatus(T, AgentStatus.WORKING); // pickup
|
||||
}
|
||||
injector.onStatus(T, AgentStatus.IDLE);
|
||||
assertEquals(List.of("a", "b", "c"), sent());
|
||||
}
|
||||
|
||||
@Test
|
||||
void activeWhileQueuedOrInFlightThenQuietAfterPickup() {
|
||||
assertTrue(injector.activeTargets().isEmpty());
|
||||
injector.enqueue(T, "x");
|
||||
assertEquals(java.util.Set.of(T), injector.activeTargets(), "active while a message is queued");
|
||||
|
||||
injector.onStatus(T, AgentStatus.IDLE); // delivers; still in-flight (awaiting pickup)
|
||||
assertEquals(java.util.Set.of(T), injector.activeTargets(),
|
||||
"stays active so the poller can observe the worker pick the message up");
|
||||
|
||||
injector.onStatus(T, AgentStatus.WORKING); // pickup observed → in-flight cleared
|
||||
assertTrue(injector.activeTargets().isEmpty(), "quiet once queue is empty and pickup is seen");
|
||||
}
|
||||
|
||||
@Test
|
||||
void sendFailureDropsMessageAndFailsItsFuture() {
|
||||
FakeHerdr failing = new FakeHerdr().agentSendFailsWith("send_failed");
|
||||
Injector inj = new Injector(new AgentControl(failing));
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "boom");
|
||||
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
assertTrue(f.isCompletedExceptionally());
|
||||
assertTrue(inj.activeTargets().isEmpty(), "poisoned message is dropped, not left blocking the queue");
|
||||
}
|
||||
|
||||
@Test
|
||||
void dropFailsPendingWaiters() {
|
||||
CompletableFuture<Void> f = injector.enqueue(T, "orphan");
|
||||
injector.drop(T, new HerdrException("worker gone", "pane_not_found", null));
|
||||
assertTrue(f.isCompletedExceptionally(), "queued waiters unblock when the worker vanishes");
|
||||
}
|
||||
|
||||
@Test
|
||||
void pollerDeliversToAnIdleWorker() throws Exception {
|
||||
// End-to-end through the poller: idle worker → message delivered without manual onStatus.
|
||||
FakeHerdr idle = new FakeHerdr().agentStatus("idle");
|
||||
Injector inj = new Injector(new AgentControl(idle));
|
||||
StatusPoller poller = new StatusPoller(new AgentControl(idle), inj, 10);
|
||||
poller.start();
|
||||
try {
|
||||
CompletableFuture<Void> delivered = inj.enqueue(T, "via-poller");
|
||||
delivered.get(2, TimeUnit.SECONDS); // completes when the poller drives the send
|
||||
} finally {
|
||||
poller.stop();
|
||||
}
|
||||
assertEquals(List.of("via-poller"), idle.calls.stream()
|
||||
.filter(c -> c.method().equals("agent.send"))
|
||||
.map(c -> {
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, Object> p = (Map<String, Object>) c.params();
|
||||
return p.get("text").toString();
|
||||
}).toList());
|
||||
}
|
||||
|
||||
@Test
|
||||
void deliveredFutureCarriesSendFailure() {
|
||||
FakeHerdr failing = new FakeHerdr().agentSendFailsWith("send_failed");
|
||||
Injector inj = new Injector(new AgentControl(failing));
|
||||
CompletableFuture<Void> f = inj.enqueue(T, "boom");
|
||||
inj.onStatus(T, AgentStatus.IDLE);
|
||||
ExecutionException ex = assertThrows(ExecutionException.class, f::get);
|
||||
assertInstanceOf(HerdrException.class, ex.getCause());
|
||||
}
|
||||
}
|
||||
@@ -1,255 +0,0 @@
|
||||
package dev.ltms.bridged.rest;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.worker.WorkerService;
|
||||
import io.javalin.Javalin;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.net.URI;
|
||||
import java.net.http.HttpClient;
|
||||
import java.net.http.HttpRequest;
|
||||
import java.net.http.HttpResponse;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* REST acceptance tests — the feature contract over plain HTTP with a fake herdr, no
|
||||
* live daemon and no Claude in the loop. This is the surface later MCP tools match by
|
||||
* parity, and where the subscription boundary and worker placement are proven at the
|
||||
* API edge.
|
||||
*/
|
||||
class BridgedAppTest {
|
||||
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
private final HttpClient http = HttpClient.newHttpClient();
|
||||
private Javalin app;
|
||||
|
||||
@AfterEach
|
||||
void stop() {
|
||||
if (app != null) app.stop();
|
||||
}
|
||||
|
||||
private int start(FakeHerdr herdr, String workerBaseUrl, Set<String> allow) {
|
||||
return start(herdr, workerBaseUrl, allow, "tab");
|
||||
}
|
||||
|
||||
private int start(FakeHerdr herdr, String workerBaseUrl, Set<String> allow, String placement) {
|
||||
BridgedConfig.Worker wcfg = new BridgedConfig.Worker(
|
||||
"ltms-local", workerBaseUrl, "coder", null, "BRIDGED_WORKER_TOKEN", null,
|
||||
placement, "bridged-workers", "worker: {profile} #{n}");
|
||||
WorkerService workers = new WorkerService(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr), new SubscriptionGuard(allow), wcfg,
|
||||
k -> "BRIDGED_WORKER_TOKEN".equals(k) ? "tok-abc" : null);
|
||||
app = new BridgedApp(herdr, workers).build().start("127.0.0.1", 0);
|
||||
return app.port();
|
||||
}
|
||||
|
||||
private int startHealthy() {
|
||||
return start(new FakeHerdr(), "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
}
|
||||
|
||||
private HttpResponse<String> req(int port, String method, String path) throws Exception {
|
||||
HttpRequest.Builder b = HttpRequest.newBuilder(URI.create("http://127.0.0.1:" + port + path));
|
||||
b = switch (method) {
|
||||
case "POST" -> b.POST(HttpRequest.BodyPublishers.noBody());
|
||||
case "DELETE" -> b.DELETE();
|
||||
default -> b.GET();
|
||||
};
|
||||
return http.send(b.build(), HttpResponse.BodyHandlers.ofString());
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private Map<String, Object> params(FakeHerdr herdr, String method) {
|
||||
return (Map<String, Object>) herdr.lastCall(method).params();
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzOkWhenHerdrAnswers() throws Exception {
|
||||
int port = startHealthy();
|
||||
HttpResponse<String> res = req(port, "GET", "/healthz");
|
||||
assertEquals(200, res.statusCode());
|
||||
JsonNode body = mapper.readTree(res.body());
|
||||
assertEquals("ok", body.get("status").asText());
|
||||
assertEquals(14, body.get("herdr").get("protocol").asInt());
|
||||
}
|
||||
|
||||
@Test
|
||||
void healthzDegradedWhenHerdrDown() throws Exception {
|
||||
int port = start(new FakeHerdr().healthy(false), "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
HttpResponse<String> res = req(port, "GET", "/healthz");
|
||||
assertEquals(503, res.statusCode());
|
||||
assertEquals("degraded", mapper.readTree(res.body()).get("status").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void sessionsMapsWorkspaceList() throws Exception {
|
||||
int port = startHealthy();
|
||||
JsonNode sessions = mapper.readTree(req(port, "GET", "/sessions").body()).get("sessions");
|
||||
assertEquals(2, sessions.size());
|
||||
assertEquals("w1", sessions.get(0).get("id").asText());
|
||||
assertEquals("done", sessions.get(1).get("agentStatus").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void agentsExposesSessionUuid() throws Exception {
|
||||
int port = startHealthy();
|
||||
JsonNode agents = mapper.readTree(req(port, "GET", "/agents").body()).get("agents");
|
||||
assertEquals(1, agents.size());
|
||||
assertEquals("sess-1111", agents.get(0).get("sessionId").asText());
|
||||
assertEquals("idle", agents.get(0).get("status").asText());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnWorkerLandsInOwnTabInWorkerSpaceAndInjectsBaseUrl() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
HttpResponse<String> res = req(port, "POST", "/workers");
|
||||
assertEquals(201, res.statusCode());
|
||||
JsonNode body = mapper.readTree(res.body());
|
||||
assertEquals("w9:pW", body.get("paneId").asText());
|
||||
assertEquals("w9:t2", body.get("tabId").asText());
|
||||
|
||||
// Subscription boundary: agent.start carried base_url + token in its env map.
|
||||
Map<String, Object> start = params(herdr, "agent.start");
|
||||
@SuppressWarnings("unchecked")
|
||||
Map<String, String> env = (Map<String, String>) start.get("env");
|
||||
assertEquals("http://gx00.gw:8000", env.get("ANTHROPIC_BASE_URL"));
|
||||
assertEquals("tok-abc", env.get("ANTHROPIC_AUTH_TOKEN"));
|
||||
assertEquals(List.of("claude"), start.get("argv"));
|
||||
|
||||
// Placement: worker space ensured, worker started INTO its own tab, seed shell
|
||||
// dropped, and the tab given a friendly label.
|
||||
assertTrue(herdr.called("workspace.create"), "worker space must be found-or-created");
|
||||
assertEquals("w9:t2", start.get("tab_id"), "worker must start into its dedicated tab");
|
||||
assertEquals("w9:pRoot", params(herdr, "pane.close").get("pane_id"), "seed shell pane dropped");
|
||||
assertEquals("worker: ltms-local #1", params(herdr, "tab.rename").get("label"),
|
||||
"tab label carries the worker number so siblings stay distinct");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnWorkerReusesExistingWorkerSpace() throws Exception {
|
||||
// A space labelled "bridged-workers" already exists → no second workspace.create.
|
||||
FakeHerdr herdr = new FakeHerdr().withWorkspace("w9", "bridged-workers");
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
assertEquals(201, req(port, "POST", "/workers").statusCode());
|
||||
assertFalse(herdr.called("workspace.create"), "existing worker space must be reused, not recreated");
|
||||
assertTrue(herdr.called("tab.create"), "a fresh tab is still created for the worker");
|
||||
}
|
||||
|
||||
@Test
|
||||
@SuppressWarnings("unchecked")
|
||||
void spawnRetriesUnderAFreshNameWhenAgentNameTaken() throws Exception {
|
||||
// herdr rejects a duplicate agent name; the service must bump and retry.
|
||||
FakeHerdr herdr = new FakeHerdr().agentNameTakenTimes(2);
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
assertEquals(201, req(port, "POST", "/workers").statusCode());
|
||||
List<String> names = herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("agent.start"))
|
||||
.map(c -> ((Map<String, Object>) c.params()).get("name").toString())
|
||||
.toList();
|
||||
assertEquals(3, names.size(), "2 rejected + 1 success");
|
||||
assertEquals(3, Set.copyOf(names).size(), "each attempt must use a distinct name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnClosesTheCreatedTabWhenTheWorkerNeverStarts() throws Exception {
|
||||
// Every agent.start attempt is rejected → spawn fails; the tab we created must not leak.
|
||||
FakeHerdr herdr = new FakeHerdr().agentNameTakenTimes(99);
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
assertEquals(500, req(port, "POST", "/workers").statusCode());
|
||||
assertTrue(herdr.called("tab.create"), "a tab was created before the failed start");
|
||||
assertEquals("w9:t2", params(herdr, "tab.close").get("tab_id"), "orphaned tab must be closed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnWorkerRejectsOffAllowlistBaseUrlAndNeverTouchesHerdr() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// base_url points at the subscription — guard must block before any herdr call.
|
||||
int port = start(herdr, "https://api.anthropic.com", Set.of("gx00.gw"));
|
||||
|
||||
HttpResponse<String> res = req(port, "POST", "/workers");
|
||||
assertEquals(403, res.statusCode());
|
||||
assertEquals("subscription_boundary", mapper.readTree(res.body()).get("error").asText());
|
||||
assertFalse(herdr.called("agent.start"), "guard must stop the spawn before herdr");
|
||||
assertFalse(herdr.called("workspace.create"), "guard must stop before provisioning a space");
|
||||
}
|
||||
|
||||
@Test
|
||||
void panePlacementSplitsFocusedTabWithoutADedicatedSpace() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"), "pane");
|
||||
|
||||
assertEquals(201, req(port, "POST", "/workers").statusCode());
|
||||
Map<String, Object> start = params(herdr, "agent.start");
|
||||
assertFalse(start.containsKey("tab_id"), "pane placement must not target a tab");
|
||||
assertFalse(herdr.called("workspace.create"), "pane placement uses no dedicated space");
|
||||
assertFalse(herdr.called("tab.create"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopWorkerClosesPaneAndItsTab() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
assertEquals(204, req(port, "DELETE", "/workers/w9:pW").statusCode());
|
||||
assertTrue(herdr.called("pane.close"));
|
||||
// Tab resolved from the pane (pane.get), then closed.
|
||||
assertEquals("w9:t2", params(herdr, "tab.close").get("tab_id"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopWorkerInPanePlacementClosesOnlyThePane() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"), "pane");
|
||||
|
||||
assertEquals(204, req(port, "DELETE", "/workers/w9:pW").statusCode());
|
||||
assertTrue(herdr.called("pane.close"));
|
||||
assertFalse(herdr.called("tab.close"), "pane placement owns no tab to close");
|
||||
assertFalse(herdr.called("pane.get"), "no tab resolution in pane placement");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopNeverClosesATabThatHoldsOtherPanes() throws Exception {
|
||||
// The worker's tab has 2 panes (e.g. a pane-placement worker sharing a user tab).
|
||||
FakeHerdr herdr = new FakeHerdr().withWorkerTabPaneCount(2);
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
assertEquals(204, req(port, "DELETE", "/workers/w9:pW").statusCode());
|
||||
assertTrue(herdr.called("pane.close"), "the worker's own pane is still closed");
|
||||
assertFalse(herdr.called("tab.close"), "must not close a tab that holds the user's other panes");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopReportsFailureWhenPaneCloseFailsForARealReason() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr().paneCloseFailsWith("herdr_busy");
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
// A genuine teardown failure must surface, not be reported as a successful 204.
|
||||
assertEquals(500, req(port, "DELETE", "/workers/w9:pW").statusCode());
|
||||
assertFalse(herdr.called("tab.close"), "tab is not removed when the pane close failed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopToleratesAnAlreadyGonePane() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr().paneCloseFailsWith("pane_not_found");
|
||||
int port = start(herdr, "http://gx00.gw:8000", Set.of("gx00.gw"));
|
||||
|
||||
// Already-gone is success; the (now-empty) tab is still cleaned up.
|
||||
assertEquals(204, req(port, "DELETE", "/workers/w9:pW").statusCode());
|
||||
assertTrue(herdr.called("tab.close"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<!--
|
||||
CB-504 / CB-594 — launchd agent for fleetd (macOS).
|
||||
|
||||
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
|
||||
no systemd. A systemd unit ships alongside (deploy/fleetd.service) for the Linux gateways
|
||||
CB-308 introduces.
|
||||
|
||||
Install:
|
||||
cp deploy/dev.ltms.fleetd.plist ~/Library/LaunchAgents/
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist
|
||||
launchctl list | grep fleetd
|
||||
|
||||
The paths below are already filled in for this host (resolved 2026-08-16 from
|
||||
`/usr/libexec/java_home`... except that reported the system Applet-plugin JVM, not the jenv-
|
||||
managed JDK 25 actually used to build/run fleetd, so JAVA_HOME here is the real one:
|
||||
`JENV_VERSION=25.0.3 java -XshowSettings:properties -version 2>&1 | grep java.home`; `which mvn`;
|
||||
`echo $HOME`). If this file is copied to a different host, re-resolve all three paths and check
|
||||
no placeholder path is left behind; scripts/redeploy-fleetd.sh's check mode does not (and
|
||||
cannot) check this file for you.
|
||||
|
||||
CB-594 — launchd cannot run a login shell (see the PATH comment on EnvironmentVariables below,
|
||||
and scripts/fleetd-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
|
||||
not java directly, so WORKER_GITEA_TOKEN and AI_GATEWAY_TOKEN still get sourced from
|
||||
${SHARED_ENV}/tools/secrets.sh even though launchd itself never sources anything.
|
||||
|
||||
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
|
||||
does systemd in a way that survives a socket appearing late. fleetd retries the herdr socket
|
||||
on startup instead, so an agent that comes up before herdr converges rather than dying — that
|
||||
retry is the actual fix; KeepAlive below is the backstop.
|
||||
|
||||
CB-594 — KeepAlive vs. scripts/redeploy-fleetd.sh: a bare SIGTERM makes this JVM exit 143 even
|
||||
with its shutdown hook running to completion (measured, see the CB-594 report), which
|
||||
SuccessfulExit:false below reads as a crash and races to restart the OLD jar. The redeploy
|
||||
script now detects a loaded agent and uses `launchctl unload`/`load` instead of a raw kill, so
|
||||
only one supervisor ever touches the process at a time — read that script's own output on a
|
||||
redeploy for the confirmation.
|
||||
-->
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>dev.ltms.fleetd</string>
|
||||
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/target/fleetd.jar</string>
|
||||
<string>fleetd.yaml</string>
|
||||
</array>
|
||||
|
||||
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
|
||||
<key>WorkingDirectory</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd</string>
|
||||
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
<key>JAVA_HOME</key>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home</string>
|
||||
<key>HERDR_SOCKET_PATH</key>
|
||||
<string>/Users/dai.ha/.config/herdr/herdr.sock</string>
|
||||
<!--
|
||||
PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker
|
||||
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
|
||||
source .zprofile/.zshrc, so without this the daemon (and therefore every worker) gets a
|
||||
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
|
||||
-->
|
||||
<key>PATH</key>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin:/Users/dai.ha/Softwares/apache-maven/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
|
||||
<!--
|
||||
Worker/API tokens are NOT set here: this file is committed. CB-594 —
|
||||
scripts/fleetd-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
|
||||
them, by execing a login shell that sources ${SHARED_ENV}/tools/secrets.sh before the
|
||||
daemon itself starts. fleetd also reads the API token from the env var named by
|
||||
auth.tokenEnv (default FLEETD_API_TOKEN) and only in auth.mode: token — the wrapper
|
||||
covers that one too, since it is the same login shell.
|
||||
-->
|
||||
</dict>
|
||||
|
||||
<key>RunAtLoad</key>
|
||||
<true/>
|
||||
|
||||
<!--
|
||||
CB-600 — read this before assuming ThrottleInterval bounds anything. It paces restarts to at
|
||||
most one per 10s; it does NOT cap how many times launchd retries. If fleetd fails fast on
|
||||
every start — a bad fleetd.yaml, for example auth.mode: token with the token env var unset,
|
||||
which throws in main() before the daemon ever binds a port — launchd restarts it forever,
|
||||
once every 10s, until a human intervenes. LaunchAgents have no "give up after N attempts"
|
||||
primitive, so this is not something a config change here can fix.
|
||||
|
||||
That loop stops only two ways: (1) `launchctl unload -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist`,
|
||||
or (2) the underlying cause gets fixed, so the process starts successfully and stays up (no
|
||||
more exits to restart). scripts/redeploy-fleetd.sh does not add a third way — it does not
|
||||
make fleetd self-disable on a config error, on purpose: a fail-fast exit path that
|
||||
sometimes decides "this is unrecoverable, stop trying" is one more thing that can misfire,
|
||||
and a wrongly self-disabled daemon needs the exact same manual `launchctl load -w` recovery
|
||||
this comment already names — so it buys nothing an operator watching for the crash loop
|
||||
doesn't already have, at the cost of a new way to be silently down. Watch for it with
|
||||
`launchctl list dev.ltms.fleetd` (a high restart count) or by tailing fleetd.out for the
|
||||
same startup error repeating every ~10s.
|
||||
-->
|
||||
<key>KeepAlive</key>
|
||||
<dict>
|
||||
<key>SuccessfulExit</key>
|
||||
<false/>
|
||||
</dict>
|
||||
<key>ThrottleInterval</key>
|
||||
<integer>10</integer>
|
||||
|
||||
<!--
|
||||
CB-594 — same file scripts/redeploy-fleetd.sh already tails ($BRIDGED/fleetd.out), and both
|
||||
streams point at it, not two separate log files: the script's fresh-line / ERROR-count checks
|
||||
after a restart read this one path regardless of whether launchd or the script started the
|
||||
process, and a stdout/stderr split would make half of what happens during a launchd-driven
|
||||
restart invisible to it.
|
||||
-->
|
||||
<key>StandardOutPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
|
||||
<key>ProcessType</key>
|
||||
<string>Background</string>
|
||||
</dict>
|
||||
</plist>
|
||||
@@ -0,0 +1,61 @@
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.fleetd.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
#
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/fleetd.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now fleetd
|
||||
# journalctl --user -u fleetd -f
|
||||
|
||||
[Unit]
|
||||
Description=fleetd — claude-bridge message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# fleetd retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down with it.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/fleetd
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/fleetd.jar fleetd.yaml
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit fleetd → [Service] / Environment=FLEETD_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/fleetd/env
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes fleetd fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=fleetd
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,60 @@
|
||||
# LavinMQ — the AMQP broker behind fleetd's durable ReplyInbox (CB-307 Stage 2).
|
||||
#
|
||||
# Why this file exists: the broker was previously run ad hoc and simply vanished from the host,
|
||||
# which takes fleetd down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Fleetd.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# degraded mode. This pins the version, keeps the data, and brings itself back after a reboot.
|
||||
#
|
||||
# Usage:
|
||||
# docker compose -f deploy/lavinmq/compose.yaml up -d
|
||||
# docker compose -f deploy/lavinmq/compose.yaml ps
|
||||
# docker compose -f deploy/lavinmq/compose.yaml logs -f
|
||||
# docker compose -f deploy/lavinmq/compose.yaml down # keeps the volume
|
||||
# docker compose -f deploy/lavinmq/compose.yaml down -v # DESTROYS held replies
|
||||
#
|
||||
# Management UI: http://127.0.0.1:15672 (guest / guest)
|
||||
#
|
||||
# This is fleetd's OWN broker. Do not point fleetd at any other AMQP server on this host —
|
||||
# notably not the `local-rabbitmq` container, which belongs to a different project and would end
|
||||
# up carrying this project's queues.
|
||||
|
||||
name: fleetd-broker
|
||||
|
||||
services:
|
||||
lavinmq:
|
||||
# Pinned deliberately: :latest silently moves the broker under a running daemon.
|
||||
image: cloudamqp/lavinmq:2.9.1
|
||||
container_name: fleetd-lavinmq
|
||||
|
||||
# The failure this deployment exists to prevent — survive reboots and Docker restarts, but
|
||||
# stay down if it was stopped on purpose.
|
||||
restart: unless-stopped
|
||||
|
||||
# Loopback-bound on purpose. LavinMQ ships a default guest/guest account, which is only
|
||||
# acceptable because nothing off-host can reach it. fleetd connects over 127.0.0.1, and
|
||||
# binding 0.0.0.0 here would expose a broker with default credentials to the network.
|
||||
ports:
|
||||
- "127.0.0.1:5672:5672" # AMQP — fleetd.yaml broker.uri points here
|
||||
- "127.0.0.1:15672:15672" # HTTP management API + UI
|
||||
|
||||
# The whole point of Stage 2. Held-but-unacked replies live here; without a named volume a
|
||||
# `docker compose down` would discard exactly what the durable inbox exists to protect.
|
||||
volumes:
|
||||
- lavinmq-data:/var/lib/lavinmq
|
||||
|
||||
healthcheck:
|
||||
test: ["CMD", "lavinmqctl", "status"]
|
||||
interval: 30s
|
||||
timeout: 5s
|
||||
retries: 3
|
||||
start_period: 10s
|
||||
|
||||
logging:
|
||||
driver: json-file
|
||||
options:
|
||||
max-size: "10m"
|
||||
max-file: "3"
|
||||
|
||||
volumes:
|
||||
lavinmq-data:
|
||||
name: fleetd-lavinmq-data
|
||||
@@ -0,0 +1,124 @@
|
||||
# CB-301 — Session Manager (one-shot, no reuse)
|
||||
|
||||
**Status:** design spec for review → delegate implementation.
|
||||
**Grounded in:** `WorkerService`, `Injector`/`StatusPoller`/`TurnListener`, `MessageService`,
|
||||
`FleetMcp`, `FleetApp` (see [wiki 9. Implementation](../wiki/9-Implementation.md)).
|
||||
|
||||
## Problem
|
||||
|
||||
`WorkerService` is **stateless about what it spawned**. Its own Javadoc says it plainly:
|
||||
|
||||
> "there is no registry; `list()` only asks herdr." — `WorkerService.reapOrphanWorkers` (line 288)
|
||||
|
||||
Consequences today:
|
||||
- The daemon cannot answer "which workers did *I* spawn, in what lifecycle state, owned by whom,
|
||||
since when?" without shelling to herdr for a raw agent list (no state, no ownership, no age).
|
||||
- Cleanup of a worker that outlived its owning process depends entirely on the boot-time
|
||||
name-nonce **reaper** (CB-117) — there is no live, authoritative roster during a run.
|
||||
- `fleet_list` (CB-304) can only surface herdr's view, not a bridge-owned roster.
|
||||
- There is no seam for per-session policy (checkpoint on teardown → CB-302; idle_ttl /
|
||||
context_cap / drain → CB-303).
|
||||
|
||||
## Goal & non-goals
|
||||
|
||||
**Goal.** Introduce a `SessionManager` that owns an authoritative in-daemon registry of the worker
|
||||
sessions this daemon process spawned, tracks each one's lifecycle state, and tears each down
|
||||
deterministically. It becomes the single source of truth for the roster and the seam CB-302/303/304
|
||||
build on.
|
||||
|
||||
**Non-goals (explicit — reuse policy chosen: one-shot, no reuse).**
|
||||
- **No pooling / no reuse.** Every delegated task gets a fresh worker; a finished worker is torn
|
||||
down, never handed to a later task. No "warm idle" pool, no `role@profile` keying.
|
||||
- **No auto-teardown *timing*.** *When* a one-shot worker is released (immediately on turn
|
||||
completion vs after an idle grace) is CB-303. CB-301 provides the **mechanism** (`release`) and
|
||||
the registry; CB-303 sets the policy.
|
||||
- **No checkpoint content.** Writing `STATE.md` + commit on teardown is CB-302; CB-301 only exposes
|
||||
the release hook it will attach to.
|
||||
|
||||
Under no-reuse, a released session is terminal. A new `acquire` always creates a fresh session.
|
||||
|
||||
## Design
|
||||
|
||||
`SessionManager` **wraps** `WorkerService` (does not replace it). `WorkerService` keeps doing the
|
||||
subscription-guarded spawn/teardown mechanics; `SessionManager` adds the registry, lifecycle, and
|
||||
ownership on top.
|
||||
|
||||
**Package:** new `dev.ltms.fleet.session` — keeps the registry/lifecycle concern separate from
|
||||
the `worker` spawn mechanics. Holds `SessionManager` + `WorkerSession`.
|
||||
|
||||
### `WorkerSession` (record or small mutable holder)
|
||||
|
||||
| Field | Source | Notes |
|
||||
|---|---|---|
|
||||
| `paneId` | `Agent.paneId()` | registry key |
|
||||
| `terminalId` | `Agent.terminalId()` | for status/identity joins |
|
||||
| `profile` | spawn arg | which profile spawned it |
|
||||
| `cwd` | resolved cwd | the worker's working dir |
|
||||
| `ownerTerminal` | caller identity (nullable) | the primary/turn that requested it; `null` = daemon/anon |
|
||||
| `spawnedAtNanos` | `System.nanoTime()` | age basis for CB-303 (monotonic; no wall clock in tests) |
|
||||
| `state` | lifecycle FSM | see below |
|
||||
|
||||
State is held in a `ConcurrentHashMap<String /*paneId*/, WorkerSession>`.
|
||||
|
||||
### Lifecycle state machine (one-shot)
|
||||
|
||||
```
|
||||
SPAWNING --ready(MCP present)--> READY
|
||||
READY --onDelivered--> BUSY
|
||||
BUSY --onTurnComplete--> DONE
|
||||
BUSY --onTurnFailed--> FAILED
|
||||
READY|DONE|FAILED --release()--> RELEASED (deregistered)
|
||||
SPAWNING|READY|BUSY|DONE --vanished/drop--> FAILED
|
||||
```
|
||||
|
||||
- Transitions are driven by hooks the manager already has access to:
|
||||
`WorkerPresence.markPresent` → `READY`; `TurnListener.onDelivered/onTurnComplete/onTurnFailed`
|
||||
(the manager implements or decorates `TurnListener`) → `BUSY`/`DONE`/`FAILED`.
|
||||
- `RELEASED` sessions are removed from the registry (teardown is terminal).
|
||||
- Any state → `FAILED` on drop (worker vanished / injector `drop`), mirroring `Injector`.
|
||||
|
||||
### API
|
||||
|
||||
```java
|
||||
final class SessionManager {
|
||||
WorkerSession acquire(String profile, String requestedCwd, String callerCwd, String ownerTerminal);
|
||||
void release(String paneId); // deterministic teardown + deregister
|
||||
Optional<WorkerSession> get(String paneId);
|
||||
List<WorkerSession> roster(); // bridge-owned view (CB-304 consumes this)
|
||||
// lifecycle hooks (package-private): onReady/onDelivered/onComplete/onFailed(target)
|
||||
}
|
||||
```
|
||||
|
||||
- `acquire` = `workerService.spawn(profile, requestedCwd, callerCwd)` → register `SPAWNING`.
|
||||
- `release` = `workerService.stop(paneId)` → deregister. Idempotent (already-gone tolerated, matching
|
||||
`WorkerService.stop`).
|
||||
- `roster` joins the registry with live herdr status for a truthful "roster + live" (CB-304).
|
||||
|
||||
### Integration points
|
||||
|
||||
- **`Fleetd.main`** — construct `SessionManager(workerService, ...)`; wire it as/decorating the
|
||||
`TurnListener` alongside `CompletionResolver` so it sees turn boundaries, and give it the
|
||||
`WorkerPresence` signal for `READY`.
|
||||
- **`FleetMcp.spawn` / `FleetApp.spawnWorker`** — route spawn through `SessionManager.acquire`
|
||||
(carry `callerTerminal` as `ownerTerminal`). **`fleet_stop` / `DELETE /workers/{paneId}`** →
|
||||
`SessionManager.release`.
|
||||
- **`fleet_list` / `GET /sessions` (CB-304 later)** — read `SessionManager.roster()`.
|
||||
- **`MessageService`** — no change required for one-shot; a later CB-303 auto-release hook can call
|
||||
`release` from `onTurnComplete` under policy.
|
||||
|
||||
## Acceptance (tests, no live herdr — fakes as elsewhere)
|
||||
|
||||
1. `acquire` registers a `SPAWNING` session with the right owner/profile/cwd; a second `acquire`
|
||||
yields a **distinct** paneId and a **distinct** session (no reuse).
|
||||
2. Presence signal moves `SPAWNING → READY`; a delivered turn moves `READY → BUSY → DONE`.
|
||||
3. `release` tears the worker down via `WorkerService.stop` and removes it from `roster()`;
|
||||
a second `release` on the same paneId is a harmless no-op.
|
||||
4. `onTurnFailed` / drop moves the session to `FAILED` and it is absent from the live roster.
|
||||
5. `roster()` reflects exactly the sessions acquired-minus-released, joined with live status.
|
||||
|
||||
## Seams left open (deliberately)
|
||||
|
||||
- **CB-302** — attach a checkpoint step (`STATE.md` + commit) to the `release` path.
|
||||
- **CB-303** — a policy loop over `roster()` using `spawnedAtNanos`/state to auto-`release` on
|
||||
`idle_ttl`, or drain on `context_cap`.
|
||||
- **CB-304** — `fleet_list` reads `roster()` for a bridge-owned roster + live join.
|
||||
@@ -0,0 +1,183 @@
|
||||
# CB-301-ext — Worktree provisioning + config-parity overlay
|
||||
|
||||
**Status:** ✅ shipped — implemented at commit `97ecc71` (per-worker git worktree + config-parity
|
||||
overlay). As-built: `session/GitWorktrees.java` behind the `Worktrees` port, wired in
|
||||
`Fleetd.main` and configurable via `worktreeRoot` / per-profile `parityOverlay`
|
||||
(see `fleetd.example.yaml`). Branch/worktree surface in `fleet_list` landed with CB-304
|
||||
(`9fe04bf`); the worker-opened-PR checkpoint landed as CB-302 (`64e70ef`).
|
||||
**Extends:** [CB-301 Session Manager](CB-301-Session-Manager.md) (shipped, commit `54d907c`).
|
||||
**Realizes:** the config-parity requirement in [Worker Git Workflow](Worker-Git-Workflow.md).
|
||||
**Grounded in:** `SessionManager`, `WorkerService.spawn/effectiveCwd`, `FleetConfig.Worker`,
|
||||
`inject/…LsofPeerPidLookup` (the `ProcessBuilder` exec pattern).
|
||||
|
||||
## Problem
|
||||
|
||||
CB-301 gives each worker a session record but every worker still runs in the **primary's own
|
||||
working tree** (`cwd = callerCwd`). One worker at a time is safe; two **parallel implementers** would
|
||||
stomp each other. We need each implementer session to get an **isolated git worktree on its own
|
||||
branch** — *without* degrading the worker: a bare worktree checks out **tracked files only**, so it
|
||||
silently drops the untracked/local config (`.claude/settings.local.json`, the locally-modified
|
||||
`.mcp.json`, `.env`) that makes a session a full peer of the primary. **CB-301-ext provisions the
|
||||
worktree AND hydrates it to config parity**, so a worker differs from the primary only in the LLM
|
||||
provider.
|
||||
|
||||
## Decisions (locked)
|
||||
|
||||
1. **Opt-in, not default.** A worktree is provisioned **only** when the caller requests one. Absent a
|
||||
request, `acquire` behaves exactly as it does today (shared primary tree) — auditors, smoke tests,
|
||||
and conversational workers are unaffected. **Backward compatibility is a hard requirement.**
|
||||
2. **Copy-overlay + `--skip-worktree`, not symlink.** Each parity file is **copied** primary→worktree
|
||||
(isolation-friendly, no symlink type-change noise on tracked files). For a *tracked* overlay file
|
||||
(`.mcp.json`) the worktree copy is then marked `git update-index --skip-worktree`, so the worker's
|
||||
commits can **never** include the parity overlay. Ignored files (`settings.local.json`) stay
|
||||
ignored in the worktree (shared `info/exclude`), so no marking is needed.
|
||||
3. **Branch persists; worktree is disposable.** `release` runs `git worktree remove --force` (the
|
||||
working dir is throwaway) but **never deletes the branch** — the branch holds the worker's commits
|
||||
and its PR (CB-302). Teardown of the checkout ≠ teardown of the work.
|
||||
4. **Git behind a seam.** SessionManager depends on a `Worktrees` interface (production impl shells
|
||||
`git` via `ProcessBuilder`; tests use a fake). No live `git` in unit tests — mirrors the
|
||||
`WorkerService`/`FakeHerdr` seam.
|
||||
|
||||
## Design
|
||||
|
||||
### `WorktreeRequest` (new, nullable = "no worktree")
|
||||
|
||||
```java
|
||||
package dev.ltms.fleet.session;
|
||||
/** Ask acquire() to provision an isolated worktree. null ⇒ run in the shared primary tree. */
|
||||
public record WorktreeRequest(String ticketSlug, String baseRef) {
|
||||
// ticketSlug seeds the branch name; baseRef null/blank ⇒ current HEAD of the repo.
|
||||
}
|
||||
```
|
||||
|
||||
### `WorkerSession` — two nullable fields added
|
||||
|
||||
| Field | Notes |
|
||||
|---|---|
|
||||
| `worktree` | absolute path of the provisioned worktree; `null` ⇒ shared tree |
|
||||
| `branch` | the worker's branch (`worker/<slug>-<nonce>`); `null` ⇒ shared tree |
|
||||
|
||||
Add to the record + `withState`. A `null` worktree keeps every existing test and the shared-tree path
|
||||
untouched.
|
||||
|
||||
### `Worktrees` seam (new)
|
||||
|
||||
```java
|
||||
package dev.ltms.fleet.session;
|
||||
public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree add <path> -b <branch> <baseRef|HEAD>. Returns the worktree path. */
|
||||
String add(String repoRoot, String branch, String baseRef);
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
/** Copy each existing overlay path repoRoot→worktree; mark tracked ones --skip-worktree. */
|
||||
void overlayParity(String repoRoot, String worktreePath, List<String> overlay);
|
||||
/** git -C <cwd> rev-parse --show-toplevel — the repo root that owns cwd. */
|
||||
String repoRoot(String cwd);
|
||||
}
|
||||
```
|
||||
|
||||
- **Production impl** `GitWorktrees implements Worktrees` — `ProcessBuilder` per path, `redirectErrorStream(true)`, non-zero exit → a `WorktreeException`. Worktree location = `<worktreeRoot>/<nonce>` where `worktreeRoot` is a daemon setting (default: sibling `../.bridged-worktrees` of the repo root — **outside** the repo, never nested).
|
||||
- `overlayParity` per file: skip if absent in `repoRoot`; else copy into the worktree; if `git -C <wt> ls-files --error-unmatch <path>` succeeds (tracked), run `git -C <wt> update-index --skip-worktree <path>`.
|
||||
|
||||
### `acquire` — extended, old signature preserved
|
||||
|
||||
```java
|
||||
// existing (unchanged): shared tree
|
||||
WorkerSession acquire(String profile, String requestedCwd, String callerCwd, String ownerTerminal);
|
||||
// new overload: provision a worktree when wt != null
|
||||
WorkerSession acquire(String profile, String requestedCwd, String callerCwd, String ownerTerminal,
|
||||
WorktreeRequest wt);
|
||||
```
|
||||
|
||||
When `wt != null`:
|
||||
1. `repoRoot = worktrees.repoRoot(firstNonBlank(requestedCwd, callerCwd))`.
|
||||
2. `branch = "worker/" + slug(wt.ticketSlug()) + "-" + nonce`.
|
||||
3. `path = worktrees.add(repoRoot, branch, wt.baseRef())`.
|
||||
4. `worktrees.overlayParity(repoRoot, path, cfg.parityOverlay())`.
|
||||
5. `spawn(profile, path, callerCwd)` — **the worktree path becomes the worker's cwd** (highest
|
||||
precedence in `WorkerService.resolveCwd`).
|
||||
6. register the session with `worktree=path, branch=branch`.
|
||||
7. **On any failure in 1–5, unwind**: if the worktree was added, `remove` it; do not leave a dangling
|
||||
registry entry. (Guard/spawn already throw before herdr on a bad base_url — unchanged.)
|
||||
|
||||
### `release` — remove the worktree, keep the branch
|
||||
|
||||
```java
|
||||
public void release(String paneId) {
|
||||
WorkerSession s = registry.remove(paneId);
|
||||
workerService.stop(paneId); // existing
|
||||
if (s != null && s.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(s.cwd()), s.worktree()); // branch is NOT deleted
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
### Config — `FleetConfig.Worker.parityOverlay` + a `worktreeRoot`
|
||||
|
||||
- Add `List<String> parityOverlay` to the `Worker` record (12th field). Compact-constructor default
|
||||
when null/empty: `[".mcp.json", ".claude/settings.local.json", ".env", ".envrc"]` (missing paths are
|
||||
silently skipped, so the default is safe across repos). Update `withProfile`.
|
||||
- Add a top-level daemon setting `worktreeRoot` (String, nullable → `<repoParent>/.bridged-worktrees`).
|
||||
- `@JsonIgnoreProperties(ignoreUnknown = true)` already set → additive, no parser breakage.
|
||||
|
||||
### Surface: MCP + REST
|
||||
|
||||
- `fleet_spawn` gains an optional `worktree` arg: `true`, or a ticket slug string. Truthy ⇒ build a
|
||||
`WorktreeRequest(slug, null)` and call the 5-arg `acquire`.
|
||||
- `POST /workers` gains `worktree` (+ optional `ticket`) in the body/query, same mapping.
|
||||
- `workerView`/`view(WorkerSession)` include `worktree` and `branch` **when non-null** (omit for
|
||||
shared-tree sessions, so existing response assertions for shared-tree spawns are unchanged).
|
||||
|
||||
## Flow
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A["acquire(profile, ..., WorktreeRequest?)"] --> B{"worktree<br/>requested?"}
|
||||
B -->|"no (default)"| C["spawn(cwd = callerCwd)<br/>— shared tree, unchanged"]
|
||||
B -->|yes| D["repoRoot = rev-parse --show-toplevel"]
|
||||
D --> E["git worktree add path -b branch base"]
|
||||
E --> F["overlayParity: copy local config in;<br/>--skip-worktree the tracked ones"]
|
||||
F --> G["spawn(cwd = worktree path)"]
|
||||
G --> H["register worktree + branch on the session"]
|
||||
E -.->|"add/overlay/spawn fails"| X["unwind: remove worktree,<br/>no dangling registry entry"]:::warn
|
||||
C --> R["worker is a full peer of the primary"]:::goal
|
||||
H --> R
|
||||
classDef warn fill:#b7791f,stroke:#7b341e,color:#ffffff;
|
||||
classDef goal fill:#2b6cb0,stroke:#2a4365,color:#ffffff;
|
||||
```
|
||||
|
||||
*Green = the invariant: whether shared-tree (parity for free) or worktree (parity via overlay), the
|
||||
worker matches the primary. Amber = the failure-unwind path.*
|
||||
|
||||
## Acceptance (fake `Worktrees`, no live git)
|
||||
|
||||
1. **Backward-compat:** `acquire` with **no** `WorktreeRequest` makes **zero** `Worktrees` calls,
|
||||
spawns with `cwd = callerCwd`, and records `worktree == null` / `branch == null`. (Every CB-301
|
||||
test still passes.)
|
||||
2. **Provision:** `acquire(..., new WorktreeRequest("cb-999", null))` calls `add(repoRoot,
|
||||
"worker/cb-999-<nonce>", null)`, then `spawn` receives the returned worktree path as
|
||||
`requestedCwd`; the session records that path + branch.
|
||||
3. **Overlay:** `overlayParity` is invoked with the profile's `parityOverlay` (default list when
|
||||
unset); the fake asserts tracked paths were `--skip-worktree`'d and missing paths skipped.
|
||||
4. **Release removes worktree, keeps branch:** releasing a worktree session calls
|
||||
`Worktrees.remove(repoRoot, path)` and performs **no** branch-delete; a shared-tree session's
|
||||
release makes no `Worktrees` calls.
|
||||
5. **Failure unwind:** a fake `add` that throws ⇒ `acquire` throws, the session is **not** registered,
|
||||
and no worker is left running (spawn not reached / torn down).
|
||||
6. **Distinct worktrees:** two worktree acquires yield **distinct** branches and paths (no collision).
|
||||
|
||||
## Constraints & exclusions (standing, non-negotiable)
|
||||
|
||||
- **Only a worker sets `ANTHROPIC_BASE_URL`.** Worktrees touch cwd + files only; env path is unchanged
|
||||
— the guard still runs before any herdr call.
|
||||
- **Worker cwd stays inside the primary repo** (the worktree is a checkout of it) — never `$HOME`.
|
||||
- **`.mcp.json` and `wiki/` never enter a worker commit.** `.mcp.json` is overlaid for *reference* but
|
||||
`--skip-worktree`'d so it can't be staged; `wiki/` is a submodule the worker must not touch. The
|
||||
CB-302 commit step (and the implementer skill) exclude both.
|
||||
- **The overlay list stays explicit + minimal** (trust: local secrets flow to an off-subscription
|
||||
worker). No blanket tree copy.
|
||||
|
||||
## Seams left for later
|
||||
|
||||
- **CB-302** — the worker commit → push → PR checkpoint runs *inside* the worktree on its branch.
|
||||
- **CB-304** — `roster()` rows surface `worktree`/`branch` for the fleet view.
|
||||
@@ -0,0 +1,143 @@
|
||||
# CB-306 — Spawn-Readiness Gate (launcher-owned terminal readiness)
|
||||
|
||||
**Status:** design note / delegation spec (branch `worker/cb-306-readiness`)
|
||||
**Issue:** gitea `fleet/fleetd` #4
|
||||
**Owner of the behaviour:** `ClaudeCodeLauncher` (the `PeerLauncher` adapter) — NOT core.
|
||||
|
||||
## 1. Problem
|
||||
|
||||
`fleet_spawn` today returns a session the instant the herdr pane is started. The pane is not
|
||||
yet a usable Claude REPL — it may still be sitting at the folder-trust prompt, or the CLI may
|
||||
never come up at all. Nothing blocks or times out on that. Consequences:
|
||||
|
||||
- A `fleet_send` to a not-yet-ready worker surfaces as a **~60 s MCP-client timeout** (the send
|
||||
blocks waiting for a turn that can't start) instead of a fast, explicit spawn failure.
|
||||
- A worker stuck at the folder-trust prompt lingers in `SPAWNING` forever; nothing fails it.
|
||||
|
||||
We want **fail-fast spawn**: `spawn()` returns only once the peer is genuinely up and usable in
|
||||
its terminal, or throws a clean error (and leaves no orphan pane) within a bounded timeout.
|
||||
|
||||
## 2. What already exists (do NOT rebuild)
|
||||
|
||||
`SessionManager` + `PresenceBridge.markPresent()` already flip a session `SPAWNING → READY` on
|
||||
**any MCP contact from the worker** (`onReady(terminal)` → `transitionByTerminal(SPAWNING, READY)`,
|
||||
SessionManager ~line 249, "worker became available on the bridge MCP"). That is the **delivery
|
||||
lifecycle** and it stays exactly as-is. CB-306 does **not** touch it and does **not** replace it.
|
||||
|
||||
CB-306 adds a *complementary, launcher-side* gate: the launcher guarantees the **terminal** is a
|
||||
live, interactive REPL before it hands a handle back. The two signals are layered:
|
||||
|
||||
| Signal | Owner | Means | CB-306 |
|
||||
|---|---|---|---|
|
||||
| terminal reaches interactive REPL (herdr `IDLE`) | `ClaudeCodeLauncher` (this ticket) | pane is past folder-trust, CLI is up | **NEW — the spawn gate** |
|
||||
| first MCP contact → `SPAWNING→READY` | core (`SessionManager`/`PresenceBridge`) | worker spoke to the bridge | unchanged |
|
||||
|
||||
## 3. The readiness predicate (herdr status)
|
||||
|
||||
`AgentStatus.fromWire` maps herdr's wire strings to `IDLE | WORKING | BLOCKED | DONE | UNKNOWN`.
|
||||
A freshly started pane that has **not** reached an interactive Claude — including one stalled at
|
||||
the folder-trust prompt — reports **`UNKNOWN`** (herdr has not detected a Claude REPL yet). Once
|
||||
the CLI is up and settled at its prompt it reports **`IDLE`**.
|
||||
|
||||
**Predicate:** the pane is *ready* when `AgentControl.status(target)` first returns an
|
||||
**injectable** state (`IDLE`, `BLOCKED`, or `DONE` — reuse `AgentStatus.injectable()`). `UNKNOWN`
|
||||
= not ready. `WORKING` alone is ambiguous this early and should not by itself satisfy readiness;
|
||||
wait for an injectable state. (We do not need to know *why* a pane isn't ready — a trust stall,
|
||||
a crash, and a slow start all present as "never becomes injectable" and all correctly time out.)
|
||||
|
||||
## 4. Contract change on `spawn(SpawnRequest)`
|
||||
|
||||
`ClaudeCodeLauncher.spawn(SpawnRequest)` becomes **block-until-ready-or-throw**:
|
||||
|
||||
1. Start the pane exactly as today (`spawn(profile, cwd, callerCwd) → Agent`, build env + guard +
|
||||
argv, `spawnInTab`/`spawnAsPane`, unique-named).
|
||||
2. **Poll** `agentControl.status(paneId)` every `pollIntervalMs` (~300 ms) until it is `injectable()`
|
||||
or `spawnReadyTimeoutMs` elapses.
|
||||
3. **Ready** → return the `WorkerHandle(paneId, terminalId)` as today.
|
||||
4. **Timeout** → the launcher **closes the pane it started** (and its tab, via the same path
|
||||
`release`/`stop` uses) and throws **`PeerUnreachableException`** (new, in `dev.ltms.fleet.peer`).
|
||||
No orphan pane is left behind — the launcher cleans up its own failed birth.
|
||||
|
||||
`spawnReadyTimeoutMs == 0` (or unset) **disables** the gate = legacy non-blocking behaviour, so the
|
||||
change is opt-in per deployment and existing tests that don't configure it keep their old semantics.
|
||||
|
||||
### Testability seam (required)
|
||||
|
||||
Do **not** call `Thread.sleep` directly in the poll loop against a real clock — unit tests must not
|
||||
real-sleep. Introduce a small injectable seam, mirroring the existing `StatusPoller` style:
|
||||
|
||||
- a `LongSupplier nowMillis` (monotonic clock) **and** a sleep/wait hook (e.g. a
|
||||
`Sleeper`/`Waiter` functional interface, or reuse whatever `StatusPoller` already uses), both
|
||||
defaulting to the real implementations in the production constructor and overridable in tests.
|
||||
|
||||
Unit tests (add to the existing `ClaudeCodeLauncher` test):
|
||||
- fake `AgentControl` returns `UNKNOWN` a few times then `IDLE` → `spawn` returns the handle; assert
|
||||
no `close` was called.
|
||||
- fake `AgentControl` always `UNKNOWN` → `spawn` throws `PeerUnreachableException`; assert the pane
|
||||
**was closed** (verify `close(paneId)` invoked) and the fake clock advanced past the timeout.
|
||||
- `spawnReadyTimeoutMs == 0` → `spawn` returns immediately without polling (legacy path).
|
||||
|
||||
## 5. Config
|
||||
|
||||
Add to the launcher-level config (a fleetd-level knob, not per-profile) in `fleetd.yaml` +
|
||||
`FleetConfig`:
|
||||
|
||||
```yaml
|
||||
spawn_ready_timeout_ms: 20000 # 0 disables the gate (legacy non-blocking spawn)
|
||||
spawn_ready_poll_ms: 300
|
||||
```
|
||||
|
||||
Jackson ignores unknown keys, so omitting them in existing YAML is safe; pick sane defaults in code
|
||||
(`20000` / `300`). Keep the names consistent with existing config field style in `FleetConfig`.
|
||||
|
||||
## 6. Core / MCP propagation
|
||||
|
||||
`SessionManager.acquire(...)` already calls `launcher.spawn(req)`. A thrown
|
||||
`PeerUnreachableException` must propagate out as a **clean spawn failure**:
|
||||
|
||||
- The **worktree** acquire path already has a try/catch that cleans up a provisioned worktree when
|
||||
`spawn` throws — verify the new exception flows through it (worktree removed, nothing registered).
|
||||
- The **non-worktree** path registers the session only *after* `spawn` returns, so a throw means no
|
||||
half-live `SPAWNING` session is ever registered — confirm this and add a test.
|
||||
- `fleet_spawn` (MCP verb) must return an **error result** carrying the exception message, not a
|
||||
success with a dead session. Trace `FleetMcp`/`FleetApp` spawn handlers and make sure the
|
||||
exception becomes a clean tool error, not an uncaught 500 with a stack trace.
|
||||
|
||||
**Out of scope (do NOT do here):** gating `fleet_send` on session `READY` (existing status-gate +
|
||||
this spawn gate already close the window), MCP-handshake-as-readiness signal, the CB-307 broker,
|
||||
any config `kind:` discriminator, any second adapter.
|
||||
|
||||
## 7. Definition of done
|
||||
|
||||
- `ClaudeCodeLauncher.spawn` blocks until injectable or throws `PeerUnreachableException` +
|
||||
self-reaps the pane; gate disabled when timeout is 0.
|
||||
- New `PeerUnreachableException` in `dev.ltms.fleet.peer`.
|
||||
- Config knobs wired (`spawn_ready_timeout_ms`, `spawn_ready_poll_ms`) with safe defaults.
|
||||
- Existing `SPAWNING→READY` MCP-contact transition untouched.
|
||||
- New unit tests (ready / timeout+reap / disabled) green; **all existing tests still pass unchanged**.
|
||||
- Build clean via the worker's own `mvn` (primary re-runs the authoritative IDE + `mvn clean install`
|
||||
gate — self-reports are not verified facts).
|
||||
|
||||
## 8. Sequence
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant SM as SessionManager.acquire
|
||||
participant L as ClaudeCodeLauncher.spawn
|
||||
participant T as herdr (AgentControl)
|
||||
SM->>L: spawn(SpawnRequest)
|
||||
L->>T: start pane (env+guard+argv)
|
||||
loop until injectable or timeout
|
||||
L->>T: status(paneId)
|
||||
T-->>L: UNKNOWN / IDLE
|
||||
end
|
||||
alt reached injectable
|
||||
L-->>SM: PeerHandle(id, terminalId)
|
||||
else timed out
|
||||
L->>T: close(paneId) + tab
|
||||
L-->>SM: throw PeerUnreachableException
|
||||
SM-->>SM: no session registered / worktree cleaned
|
||||
end
|
||||
```
|
||||
|
||||
*Figure — the launcher blocks in `spawn` until the pane is a usable REPL, else self-reaps and throws.*
|
||||
@@ -0,0 +1,201 @@
|
||||
# CB-307 Stage 1 — Reply-Inbox Port + In-Memory Adapter (delegation spec)
|
||||
|
||||
**Ticket:** gitea #5 (CB-307). **Stage:** 1 of 2 (see the issue's "Implementation staging" comment).
|
||||
**Scope of THIS delegation:** the `ReplyInbox` port + the in-memory (soft-state) adapter, wired at the
|
||||
exact drop seam so a worker's terminal reply is **held instead of silently discarded** when no primary
|
||||
send is open. **No broker, no new dependency, no infra** — fully unit-testable and primary-gate-verifiable.
|
||||
Stage 2 (the AMQP/LavinMQ adapter behind the same port) is explicitly **out of scope** here.
|
||||
|
||||
> **Read this whole spec before starting.** The port and the publish seam are prescriptive
|
||||
> (non-negotiable). Where a choice is genuinely open it says "DECISION" with the required default —
|
||||
> follow the default and flag it in your completion message for the primary's review.
|
||||
|
||||
## 1. The bug this fixes (grounded in current code)
|
||||
|
||||
The reverse (worker→primary) path is `Rendezvous` — a `ConcurrentHashMap<session, CompletableFuture<Resolution>>`
|
||||
of **live blocking waiters only**. No queue, no store. When a worker calls `fleet_reply` and **no send
|
||||
is currently open** for that worker:
|
||||
|
||||
- `Rendezvous.resolve(session, content)` → `complete(...)` → `waiters.get(session) == null` →
|
||||
returns `false` (`msg/Rendezvous.java:212-215`).
|
||||
- The `content` string is **never retained** — it is dropped. The worker is told it failed:
|
||||
`FleetMcp.reply` returns `error("no send is awaiting a reply for this worker")` (`mcp/FleetMcp.java:270-272`);
|
||||
REST returns `409 no_pending_send` (`rest/FleetApp.java:339-345`).
|
||||
|
||||
This is the observed "communication break": a worker that finishes just after its `fleet_send` timed
|
||||
out (the ~60s sync window) replies into the void. There is **no message-id, dedup, or ack** anywhere in
|
||||
the message path today.
|
||||
|
||||
## 2. What to build
|
||||
|
||||
### 2.1 The port — `dev.ltms.fleet.msg.ReplyInbox`
|
||||
|
||||
A thin interface owned by the `msg` layer. The in-memory adapter is Stage 1; the AMQP adapter (Stage 2)
|
||||
implements the **same** interface, so keep it broker-agnostic.
|
||||
|
||||
```java
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Holds terminal worker→primary replies that arrive with no live send to resolve, keyed by worker
|
||||
* session (target), until the primary drains them. Soft-state in Stage 1 (in-memory, lost on restart);
|
||||
* the Stage 2 AMQP adapter implements the same contract with cross-restart durability.
|
||||
*/
|
||||
public interface ReplyInbox {
|
||||
|
||||
/** A queued reply: an idempotency id, the worker session it came from, and the reply text. */
|
||||
record InboxMessage(String msgId, String target, String content) {}
|
||||
|
||||
/**
|
||||
* Queue {@code content} from worker {@code target} under {@code msgId}. Idempotent: publishing an
|
||||
* already-present {@code msgId} for {@code target} is a no-op (dedup), so an at-least-once Stage-2
|
||||
* redelivery cannot double-queue.
|
||||
*/
|
||||
void publish(String target, String msgId, String content);
|
||||
|
||||
/** Non-destructive snapshot of pending replies for {@code target} (FIFO), empty list if none. */
|
||||
List<InboxMessage> peek(String target);
|
||||
|
||||
/** Remove the reply {@code msgId} for {@code target} once the primary has taken it. No-op if absent. */
|
||||
void ack(String target, String msgId);
|
||||
}
|
||||
```
|
||||
|
||||
### 2.2 The default adapter — `InMemoryReplyInbox`
|
||||
|
||||
- Backed by a `ConcurrentHashMap<String, ...>` keyed by target session; per-target FIFO ordering.
|
||||
- Dedup by `msgId` within a target (a `LinkedHashMap<msgId, InboxMessage>` per target, or a deque + a
|
||||
seen-set — your call; preserve insertion order).
|
||||
- `peek` returns an immutable copy; `ack` removes by `msgId`. Thread-safe (concurrent publish vs. drain).
|
||||
- **This is soft-state, NOT persistence.** Lost on a `java -jar` bounce — that is correct and consistent
|
||||
with "fleetd stays soft-state." Do **not** add any file/DB backing.
|
||||
|
||||
### 2.3 Publish seam — route reply through the service layer
|
||||
|
||||
Keep `Rendezvous` a pure synchronization primitive (do **not** give it an inbox field). Instead centralize
|
||||
in `MessageService`, which already owns the `Rendezvous` and will own the `ReplyInbox`:
|
||||
|
||||
- Add `MessageService.reply(String session, String content)`:
|
||||
```java
|
||||
/** Route a worker's explicit fleet_reply: resolve an open send, or queue it in the inbox if none. */
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content); // was a silent drop
|
||||
return true; // held, not lost
|
||||
}
|
||||
```
|
||||
- Repoint the two callers off the bare `rendezvous.resolve(...)` onto `messages.reply(...)`:
|
||||
- `FleetMcp.reply` (`mcp/FleetMcp.java:262-273`) — on success return a normal ack; **remove** the
|
||||
`error("no send is awaiting a reply…")` branch (that case is now a successful queue).
|
||||
- `FleetApp.replyMessage` (`rest/FleetApp.java:330-346`) — return `200` (queued) instead of
|
||||
`409 no_pending_send`.
|
||||
|
||||
**DO NOT touch the QUESTION path.** `fleet_ask` / `rendezvous.resolveQuestion` must keep today's
|
||||
`NO_WAITER` behaviour — a mid-turn question is **interactive** (the worker blocks synchronously and cannot
|
||||
consume a late answer), so it must **never** be queued. Only terminal `REPLY`s go to the inbox.
|
||||
|
||||
**DO NOT queue the completion/failure fallbacks** (`resolveCompletion` / `resolveFailure`,
|
||||
`Rendezvous.java:196-210`). They target a *captured* waiter (CB-116); a `false` there means the turn was
|
||||
already resolved or the scrape is a stale late duplicate — queueing it risks double-delivery. Leave them
|
||||
exactly as they are. (Extending durability to completions is a deliberate Stage-2 consideration, not this.)
|
||||
|
||||
### 2.4 Drain seam — how the primary collects a stranded reply
|
||||
|
||||
The primary re-checks a worker it delegated to. Expose a drain keyed by **worker session (target)**:
|
||||
|
||||
- Add `MessageService.drainReplies(String target)`: `peek` the inbox, `ack` each returned `msgId`, hand
|
||||
back the `List<InboxMessage>` (or just the contents). At-least-once: peek→deliver→ack (ack only after
|
||||
the caller has them, so an in-flight failure re-surfaces them).
|
||||
- **DECISION (required default): expose via the existing poll verb, keyed by target.** Extend `fleet_poll`
|
||||
to accept an optional `target` (worker session) and, when present, return that worker's drained replies —
|
||||
alongside a matching REST route `GET /sessions/{id}/replies`. Do **not** change `send`/`answer` semantics
|
||||
(do not drain inside `send` — that conflates "deliver to worker" with "collect its mail"). Keep the
|
||||
existing ticket-based `fleet_poll(ticket)` path working unchanged. If you see a cleaner surface, still
|
||||
ship this default and note the alternative for review.
|
||||
|
||||
## 3. Config
|
||||
|
||||
**None for Stage 1.** The in-memory adapter is the unconditional default — wire `new InMemoryReplyInbox()`
|
||||
into `MessageService` in `Fleetd.main`. Do **not** add a `broker:` config block (that arrives with the
|
||||
Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 4. Acceptance criteria (what the primary will verify)
|
||||
|
||||
1. New `ReplyInbox` + `InboxMessage` + `InMemoryReplyInbox` in `dev.ltms.fleet.msg`.
|
||||
2. `fleet_reply` with **no open send** now **succeeds and queues** (no more `error` / `409`); the reply is
|
||||
later retrievable and identical.
|
||||
3. The queued reply is drainable by the primary keyed by target; draining **acks** it (a second drain
|
||||
returns nothing); dedup by `msgId` (re-publishing the same id does not double-queue).
|
||||
4. **QUESTION path unchanged** — `fleet_ask` with no open send still returns `NO_WAITER` (add/keep a test
|
||||
proving a question is never queued).
|
||||
5. Completion/failure fallbacks unchanged.
|
||||
6. Unit tests covering: `InMemoryReplyInbox` publish/peek/ack/dedup/FIFO/concurrency; `MessageService.reply`
|
||||
queues on no-waiter and resolves-not-queues when a send is open; `drainReplies` returns+acks;
|
||||
the QUESTION-not-queued guard.
|
||||
7. **All pre-existing tests still green** (baseline is **188**; your total must be ≥ 188 + your new tests).
|
||||
|
||||
## 5. Build & verification (worker side)
|
||||
|
||||
- Build with Maven from the worktree's `fleetd/` dir. **Capture the exit code without a masking pipe**
|
||||
(`mvn clean install; echo "MVN_EXIT=$?"` — never `mvn … | tail`, which hides failures).
|
||||
- Read the real test totals from `target/surefire-reports/TEST-*.xml`, not from stdout scroll.
|
||||
- You do **not** have IDE MCP access — do not claim `ide_diagnostics` results. The **primary** runs the
|
||||
authoritative gate (IDE sync + diagnostics 0/0 + `mvn clean install`) before integrating. Your
|
||||
self-reported counts are inputs to that gate, not final facts.
|
||||
|
||||
## 6. Hard constraints (non-negotiable)
|
||||
|
||||
- **`.mcp.json` is `--skip-worktree` in your worktree — never edit, `git add`, or commit it.**
|
||||
- **`wiki/` is a submodule — never run git in it; never touch it.**
|
||||
- Commit only your feature changes (the new port/adapter, the `msg`/`mcp`/`rest` wiring, tests, and if
|
||||
you add config wiring in `Fleetd.java`). Nothing else.
|
||||
- Work only inside your assigned worktree on your feature branch. The primary fast-forwards `main` after
|
||||
re-gating — do not touch `main`.
|
||||
- Java 25 idioms are welcome (unnamed `_` params, records). Keep the diff minimal and match surrounding style.
|
||||
|
||||
## 7. Definition of done (report back over the bridge)
|
||||
|
||||
Commit on your feature branch and reply with: the commit SHA, the surefire total (run/failures/errors), a
|
||||
one-line note on the drain-surface decision (§2.4) you shipped, and confirmation that `.mcp.json`/`wiki/`
|
||||
were untouched. The primary re-gates and integrates.
|
||||
|
||||
## 8. Running the contract tests (CB-521)
|
||||
|
||||
`AmqpReplyInboxContractTest` is the real-broker proof of the `ReplyInbox` port (eventual visibility, ack
|
||||
removal, msgId dedup, cross-restart redelivery). It is `@Tag("contract")`, so the default
|
||||
`mvn test` / `mvn clean install` **skip it** — that hermetic, Docker-free default is deliberate and
|
||||
untouched. Run it explicitly when Docker (or a broker) is available:
|
||||
|
||||
```bash
|
||||
cd fleetd
|
||||
mvn -Pcontract test -Dtest=AmqpReplyInboxContractTest # local: spins a RabbitMQ Testcontainers fixture
|
||||
```
|
||||
|
||||
### Two broker modes
|
||||
|
||||
| Mode | Trigger | Broker | Needs Docker? |
|
||||
|---|---|---|---|
|
||||
| Local | `AMQP_URI` unset | Testcontainers starts `rabbitmq:3.13-management` | Yes |
|
||||
| CI / external | `AMQP_URI` set | the broker at that URI (CI RabbitMQ service container) | **No** — binds straight to the URI, never touches Testcontainers |
|
||||
|
||||
In CI the broker is provided as a RabbitMQ **service container** and `AMQP_URI` points at it, so the
|
||||
contract job runs the same assertions with no Docker on the runner and no skipped test
|
||||
(see `.gitea/workflows/ci.yml` → `contract`). The `build` job stays hermetic and Docker-free — keep
|
||||
that separation.
|
||||
|
||||
### Docker-engine discovery (why the contract profile pins `api.version`)
|
||||
|
||||
Out of the box, Testcontainers 1.20.4's docker-java client defaults to Docker API **1.32** when no
|
||||
version is requested. Modern engines reject that as too old — on this host's OrbStack (`min API 1.40`)
|
||||
testcontainers fails with *"Could not find a valid Docker environment … client version 1.32 is too
|
||||
old"* even though the `docker` CLI works (the CLI negotiates a newer API).
|
||||
|
||||
The `contract` Maven profile sets `api.version=1.43` in surefire, which works on OrbStack and Docker
|
||||
24+, and is overridable per host: `mvn -Pcontract -Dapi.version=1.54 test …`. It only applies under
|
||||
`-Pcontract`, so the default build is unaffected. If your engine differs, set `-Dapi.version` to a
|
||||
version ≥ your engine's minimum API (e.g. `docker version` shows `API version`).
|
||||
|
||||
@@ -0,0 +1,303 @@
|
||||
# CB-308 — Multi-Host Federation (Stage 5)
|
||||
|
||||
**Status:** design note (proposal) — core design decisions resolved 2026-08-10 (§7)
|
||||
**Depends on:** CB-307 (broker-based reliable delivery) — CB-308 is the multi-host layer built *on*
|
||||
CB-307's broker fabric.
|
||||
**Relates to:** CB-401 (`PeerHandle` opaque id), CB-304 (`rosterView`), CB-306 (spawn-readiness),
|
||||
CB-303 (lifecycle limits), CB-117 (orphan reap).
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Let `claude-bridge` coordinate agents that live on **more than one host** — a primary on host A
|
||||
delegating to workers on hosts B, C, … — without any host learning another host's terminals. The
|
||||
bus stays a **provider-neutral communication fabric**; multi-host is an addressing + routing
|
||||
concern, not a new kind of peer.
|
||||
|
||||
The design rests on three pieces (the shape this ticket proposes):
|
||||
|
||||
1. **Dedicated per-agent channels** — every agent has its own addressable inbox on the broker.
|
||||
2. **A federated agent directory** — a global "who/where/status" lookup, assembled from per-host
|
||||
presence, not a central database.
|
||||
3. **A per-host gateway** — each host runs a `fleetd` that owns its local herdr, registers/manages
|
||||
its own sessions, and proxies messages to/from other hosts over the broker.
|
||||
|
||||
## 2. What is single-host today (the assumptions to break)
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph host["Single host (today)"]
|
||||
primary["primary<br/>(MCP client)"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765"]
|
||||
reg["in-process registry<br/>keyed by PeerHandle.id() == paneId"]
|
||||
herdr["herdr<br/>(local unix-socket PTY mux)"]
|
||||
w1["worker pane wQ:p1"]
|
||||
w2["worker pane wQ:p2"]
|
||||
primary --> daemon
|
||||
daemon --> reg
|
||||
daemon --> herdr
|
||||
herdr --> w1
|
||||
herdr --> w2
|
||||
end
|
||||
```
|
||||
|
||||
*Figure 1 — everything is co-located and loopback.*
|
||||
|
||||
Three concrete bake-ins assume one host:
|
||||
|
||||
| Assumption | Where | Why it blocks multi-host |
|
||||
|---|---|---|
|
||||
| **herdr is local** | `herdr/` unix socket `~/.config/herdr/herdr.sock` | You cannot drive another host's PTYs → each host **must** own its herdr. This is why a per-host gateway is mandatory. |
|
||||
| **registry is in-process, keyed by `paneId`** | `session/SessionManager` | `paneId` (e.g. `wQ:p2B`) is a herdr-local coordinate — meaningless off-host. Routing needs a host-unique id. |
|
||||
| **loopback, no authn** | `rest/FleetApp` binds `127.0.0.1:8765` | Fine on one host; the moment a second host can talk to a gateway, that link is a trust boundary. |
|
||||
|
||||
## 3. Target architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A"]
|
||||
gA["gateway = fleetd A"]
|
||||
regA["local registry + herdr"]
|
||||
primary["primary (MCP client)"]
|
||||
gA --- regA
|
||||
primary --- gA
|
||||
end
|
||||
subgraph hostB["HOST B"]
|
||||
gB["gateway = fleetd B"]
|
||||
regB["local registry + herdr"]
|
||||
wb["worker panes"]
|
||||
gB --- regB
|
||||
gB --- wb
|
||||
end
|
||||
subgraph broker["BROKER (LavinMQ / AMQP) — CB-307 fabric"]
|
||||
inbox["agent.<id>.inbox queues"]
|
||||
roster["roster.* presence topic"]
|
||||
dlq["DLQ · delayed-retry (remind)"]
|
||||
end
|
||||
gA -->|"publish to agent.<id>.inbox"| inbox
|
||||
gB -->|"publish to agent.<id>.inbox"| inbox
|
||||
inbox -->|"owning gateway consumes"| gA
|
||||
inbox -->|"owning gateway consumes"| gB
|
||||
gA -->|"announce local agents"| roster
|
||||
gB -->|"announce local agents"| roster
|
||||
roster -->|"union view"| gA
|
||||
roster -->|"union view"| gB
|
||||
```
|
||||
|
||||
*Figure 2 — each gateway owns its local herdr + registry, consumes only its own agents' inboxes,
|
||||
and announces its agents onto a shared presence topic. The broker routes; no host sees another
|
||||
host's terminals.*
|
||||
|
||||
### 3.1 Component mapping (the three pieces)
|
||||
|
||||
- **Dedicated per-agent channels** = a per-agent AMQP routing key / queue, e.g.
|
||||
`agent.<globalId>.inbox`. The agent's **owning gateway is the only consumer** of its inbox.
|
||||
Senders publish to `agent.<id>.inbox` and never need to know the agent's host — the broker
|
||||
routes to whichever gateway holds it. LavinMQ additionally gives durability, DLX, and a native
|
||||
delayed-message exchange (the remind/backoff loop for free) — the same reasons CB-307 picked it.
|
||||
|
||||
- **Federated agent directory** = a **soft-state, bridge-owned** roster, *not* a broker-stored
|
||||
database. Per the persistence-boundary decision (fleetd is soft-state; the broker owns *message*
|
||||
durability, not *who/where/status*), each gateway announces its local agents `(globalId, host,
|
||||
status, capabilities)` on a `roster.*` presence topic with periodic heartbeats. Every gateway
|
||||
builds an eventually-consistent **union view** — literally CB-304's `rosterView`, federated. A
|
||||
stale entry expires by missed heartbeat (reuses CB-303's idle/TTL thinking).
|
||||
|
||||
- **Per-host gateway** = today's `fleetd` daemon, evolved. It already registers/manages sessions
|
||||
and controls its local herdr; multi-host adds exactly two responsibilities: (a) a broker client
|
||||
that consumes its agents' inboxes and injects into local herdr, and (b) presence announce +
|
||||
union-roster assembly. Evolution, not rewrite.
|
||||
|
||||
### 3.2 Routing rule
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
send["fleet_send(globalId, msg)"] --> lookup{"directory:<br/>is globalId local?"}
|
||||
lookup -->|"yes"| local["inject via local herdr<br/>(today's Injector path)"]
|
||||
lookup -->|"no"| pub["publish agent.<id>.inbox<br/>(broker routes to owning gateway)"]
|
||||
pub --> consume["owning gateway consumes<br/>→ injects into its local herdr"]
|
||||
```
|
||||
|
||||
*Figure 3 — one fork: local agents keep today's in-process inject path; remote agents go over the
|
||||
broker. A sender is oblivious to which branch it took.*
|
||||
|
||||
## 4. What CB-307 already provides vs. what is net-new
|
||||
|
||||
**CB-307 delivers the transport half** and is independently valuable on a single host: the AMQP
|
||||
broker fabric, the `fleetd → broker` client/adapter, at-least-once + idempotent (dedup-by-id)
|
||||
delivery, DLQ, and delayed-retry (remind). That *is* the "proxy cross-host message" backbone;
|
||||
extending the same broker from "worker→primary reliability" to "gateway↔gateway" is incremental.
|
||||
|
||||
**Net-new for CB-308 (multi-host), five items:**
|
||||
|
||||
1. **Global agent id** — decouple the routing key from `paneId`. CB-401's `PeerHandle` already
|
||||
abstracts the routing id; make it host-unique (e.g. `<host>/<paneId>` or a UUID minted at spawn).
|
||||
The registry and all verbs route on the global id.
|
||||
2. **Federated directory** — presence announce + heartbeat + union roster over `roster.*`
|
||||
(§3.1).
|
||||
3. **Gateway routing** — the `local ? inject : publish` fork (§3.2), plus each gateway consuming
|
||||
its own agents' inbox queues and injecting into local herdr.
|
||||
4. **Cross-host spawn** — `spawn on host B` = publish a control request to B's control channel →
|
||||
gateway B runs `ClaudeCodeLauncher.spawn` **locally** (CB-306's readiness gate becomes *more*
|
||||
valuable here: the far side wants a positive "agent ready" before anyone sends) → announces the
|
||||
new agent into the federated roster.
|
||||
5. **Trust** — the broker connection is now the security boundary. A gateway injects env/tokens at
|
||||
daemon privilege (the CB-401 Stage-C concern), so a **remote-triggered spawn/send** needs
|
||||
authn/authz: who may act on which host, and which control channels a gateway will honour.
|
||||
*Authenticity* is resolved — signed messages, §7.1; *authorization* (who may do what) remains
|
||||
open — §8.
|
||||
|
||||
## 5. The one thing the broker does NOT dissolve
|
||||
|
||||
The MCP asymmetry survives the network. The primary is an MCP **client** to its **local** gateway;
|
||||
it cannot be called into. A worker on B replying to a primary on A flows:
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant W as worker (host B)
|
||||
participant GB as gateway B
|
||||
participant BR as broker
|
||||
participant GA as gateway A
|
||||
participant P as primary (host A, MCP client)
|
||||
W->>GB: fleet_reply
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route to A's primary inbox
|
||||
Note over GA: held durably until the primary pulls
|
||||
P->>GA: blocking fleet_send resolves / fleet_poll
|
||||
GA-->>P: reply (then ACK to broker)
|
||||
```
|
||||
|
||||
*Figure 4 — the broker makes the middle hop lossless, ordered, and idempotent; the **final** hop
|
||||
into the primary is still a **pull** (gateway A holds the message until the primary's blocking
|
||||
`fleet_send` or `fleet_poll`). Cross-host neither improves nor worsens this — it just spans hosts.
|
||||
This is precisely the gap CB-307 closes on one host and CB-308 stretches across hosts.*
|
||||
|
||||
## 6. Staging & dependencies
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
cb307["CB-307<br/>broker-based reliable delivery<br/>(single host first)"] --> cb308["CB-308<br/>multi-host federation<br/>(this note)"]
|
||||
cb308 --> a["global agent id"]
|
||||
cb308 --> b["federated directory"]
|
||||
cb308 --> c["gateway routing"]
|
||||
cb308 --> d["cross-host spawn"]
|
||||
cb308 --> e["cross-host trust model"]
|
||||
classDef gate fill:#b7791f,stroke:#7b341e,color:#ffffff;
|
||||
class e gate
|
||||
```
|
||||
|
||||
*Figure 5 — CB-307 is the foundation; CB-308's five items build on it. The trust model (e) is the
|
||||
gating concern before any host accepts remote control.*
|
||||
|
||||
**Recommendation:** keep CB-307 scoped to single-host broker reliability (foundation, independently
|
||||
useful), and build CB-308's items on top once the broker fabric exists. Choose CB-307's broker /
|
||||
channel naming **multi-host-ready** now (per-agent routing keys, a `roster.*` topic namespace) so
|
||||
CB-308 doesn't have to repaint the topology.
|
||||
|
||||
## 7. Resolved design decisions (2026-08-10)
|
||||
|
||||
Settled in a design review of this note + wiki chapter 10. The broker-level operational rules
|
||||
(inbox caps, TLS + private broker, schema versioning, trace id, exclusive consumers, U8 broadcast)
|
||||
are recorded in wiki 10 §10; the CB-308-side decisions are below. Entries 1–6 are the first-pass
|
||||
decisions; 7–10 came out of the adversarial second-pass review (same day) and supersede 1–6 where
|
||||
they overlap (notably: the envelope is no longer optional, and dedup is split by path).
|
||||
|
||||
1. **Sender authenticity — sign every message.** Each gateway holds its own signing key and signs
|
||||
what it publishes (sender gid, `msgId`, timestamp). The receiving gateway verifies the
|
||||
signature **and** checks against the roster that the claimed sender lives on the signing
|
||||
gateway's host. This extends the single-host invariant — *identity comes from the connection,
|
||||
never an argument* — across the broker: cross-host, identity comes from the key. Complements
|
||||
(not replaces) per-gateway broker logins over TLS.
|
||||
2. **Profiles are owned by the worker's host.** `fleet_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `fleetd.yaml`. Gateways advertise their profile names in presence
|
||||
heartbeats, so a leader sees what each host offers before spawning; an unknown name is a clear
|
||||
error from the target. Secrets (base URLs, tokens) never leave the host that uses them.
|
||||
3. **Repo provisioning — clone from the forge, pinned.** A cross-host spawn names the repo URL and
|
||||
the exact commit. The target gateway clones from the forge into a local cache (first spawn
|
||||
only), then cuts a per-worker worktree — the CB-301-ext flow with a clone step in front,
|
||||
covered by the same repo-scoped forge token (CB-302). Git stays the only channel code moves
|
||||
through.
|
||||
4. **Asks are live-only, with expiry.** `ASK`/`ANSWER` (U2) traverse the broker as short-lived
|
||||
(TTL'd) messages carrying the `turn_id`, and are never held durably — the single-host rule
|
||||
kept. An answer arriving after its turn ended is **not** injected; it is dropped and the leader
|
||||
gets a `TOO_LATE` notice, so the one failure case is loud rather than weird. Only terminal
|
||||
replies are durable. Walkthrough: wiki 10 §7.4.
|
||||
5. **Spawn dedup — a spawn id, remembered on the target.** The control queue redelivers like any
|
||||
queue; a replayed `SpawnRequest` must not double-spawn. Requests carry a unique spawn id; the
|
||||
target gateway keeps a short memory of handled ids and answers a redelivery with the existing
|
||||
`PeerHandle`. CB-117's orphan reap stays as the backstop.
|
||||
6. **Broker down — local unaffected, remote fails fast.** The routing fork (§3.2) means same-host
|
||||
traffic never touches the broker; that is now a written promise. A send to a remote agent while
|
||||
the broker is unreachable **fails immediately** with a clear error — the gateway never buffers
|
||||
on the broker's behalf (it stays soft-state, so a crash cannot lose messages it claimed to
|
||||
deliver). Gateways auto-reconnect; remote hosts read as unknown in the roster meanwhile. Broker
|
||||
HA is a later ops choice, not a design requirement.
|
||||
|
||||
7. **Turn state — split by where the signals are.** The *worker's* gateway owns the turn record
|
||||
(turnId minting, ask coalescing, STALE_TURN, the completion/failure fallbacks, CB-516 abandon):
|
||||
every input to those decisions — pane status, injection, teardown — is local to it. The
|
||||
*sender's* gateway owns only the waiter. The two are stitched by terminal-outcome envelope
|
||||
kinds (`REPLY` / `FAILED` / `ABANDONED`) published to the sender's inbox: a worker dying on B
|
||||
fails A's waiter fast because gateway B sees the death synchronously and says so.
|
||||
**`ABANDONED` is belt-and-braces over the waiter's own timeout and roster expiry, never a
|
||||
replacement** — the case where the waiter hangs longest is gateway B itself dying, which is
|
||||
exactly when B can publish nothing.
|
||||
8. **Dual ack model + spawn idempotence by construction.** Forward path (a brief into a worker):
|
||||
ack **before** the inject — at-most-once, duplicates structurally impossible; the loss window
|
||||
is closed by an `INJECTED` confirmation published after the inject lands (no `INJECTED` within
|
||||
a bound = loud fast failure at the sender, not a silent send-timeout). Reply/pull path keeps
|
||||
ack-after-drain — a duplicate reply is benign, deduped by `msgId`. Spawn: the requester mints
|
||||
**spawn id = the new worker's gid**; the target checks it against the **live pane registry**,
|
||||
and the gid is **stored in the herdr pane itself** (label/env, readable back), so a restarted
|
||||
gateway rebuilds gid↔pane from herdr and the check survives restarts with *no persisted
|
||||
ledger* — this storage point is the load-bearing detail of the no-ledger position. An
|
||||
**in-flight reservation set**, entered before the launcher call, absorbs a redelivery arriving
|
||||
while the first spawn is still inside CB-306's readiness gate; a crash mid-spawn leaves a
|
||||
half-built pane, which is exactly what CB-117 reaps.
|
||||
9. **Publish is enforced, not fire-and-forget.** Publisher confirms + the `mandatory` flag + a
|
||||
return listener, on a **publish channel separate from the consume/ack channel** — synchronous
|
||||
confirms on the single shared channel would hold its lock across a broker round trip and
|
||||
serialize acks fleet-wide. Ordering caveat: a *return* (unroutable) arrives **before** the
|
||||
confirm, so "confirmed" ≠ "routed"; the sender checks the returned-set at confirm time.
|
||||
`mandatory` is false only for `BROADCAST`, where an empty group is legal silence.
|
||||
10. **Queue lifecycle is session lifecycle.** `fleet_stop`/reap deletes the worker's inbox queue
|
||||
(its `broadcast.*` bindings die with it — no broadcasts to the dead); `x-expires` collects
|
||||
queues orphaned by a crashed gateway (long for main/orchestrator inboxes, short for workers).
|
||||
Queue names carry a version suffix (`.v2`): AMQP refuses to redeclare an existing durable
|
||||
queue with new arguments (`PRECONDITION_FAILED` — a crash loop on an in-place upgrade from
|
||||
v1.0.0), and the suffix keeps old sender-keyed and new recipient-keyed queues apart during
|
||||
the keying migration (wiki 10 §3 footnote).
|
||||
|
||||
## 8. Still open
|
||||
|
||||
- **Directory ground-truth:** pure soft-state presence (heartbeats) vs. also treating broker queue
|
||||
existence as authoritative. Lean soft-state to preserve the persistence boundary; revisit if
|
||||
split-brain roster views cause mis-routing.
|
||||
- **Gateway discovery:** how gateways find the broker and each other (static config vs. discovery).
|
||||
- **Control authorization — THE GATE ON U4.** Signing (§7.1) settles *who sent it*; authorization
|
||||
is *who may do what*. **Cross-host spawn must not land before the minimal version exists**: a
|
||||
per-host allowlist in `fleetd.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
publish control to this host, checked against the verified signature. A few lines of config and
|
||||
check; without them, any principal holding broker credentials can start processes on every host
|
||||
in the fleet.
|
||||
- **Key distribution & rotation:** static config (host → public key in each `fleetd.yaml`) is
|
||||
fine at the current 2–3 host scale; rotation is manual. A refinement, not a blocker.
|
||||
- **Gateway death mid-turn:** the roster reaps it by missed heartbeat, and in-flight primary-bound
|
||||
messages survive by broker durability; still open is reconciling *worker* state when the dead
|
||||
gateway's host comes back (orphaned panes vs. still-valid sessions).
|
||||
|
||||
*(Resolved and moved up: the global id scheme — an opaque UUID minted by the spawn requester as
|
||||
the spawn id, host carried as roster metadata; §7.8.)*
|
||||
|
||||
## 9. Implementation order (each step verifiable single-host)
|
||||
|
||||
1. **As-built fixes, independent of CB-308** (v1.0.x tickets): `basicQos` prefetch on the AMQP
|
||||
consumer (today the queue drains into gateway heap, so any cap would guard an empty queue);
|
||||
publisher confirms + `mandatory` (§7.9); the `drainReplies` javadoc that claims "the ack is
|
||||
local" — false for the AMQP adapter.
|
||||
2. Envelope + signing (wiki 10 §2.1) — testable against the single-host broker.
|
||||
3. Recipient-keyed queue migration (`.v2` names, drain-by-`from`, `ReplyPushLoop` rekeyed).
|
||||
4. Global id + queue lifecycle (§7.8, §7.10).
|
||||
5. Roster: host-level heartbeat + signed presence; then the routing fork (§3.2).
|
||||
6. U2 cross-host with the terminal-outcome kinds (§7.4, §7.7).
|
||||
7. U4 cross-host spawn — **gated on the control allowlist (§8)**.
|
||||
8. U8 broadcast **last** — it is the feature that punishes an unfinished queue lifecycle.
|
||||
@@ -0,0 +1,200 @@
|
||||
# CB-401 — Peer Launcher SPI (Stage 4: pluggable peers)
|
||||
|
||||
**Status:** design note (feature branch `feature/peer-launcher-spi`)
|
||||
**Stage:** 4 — makes the bridge scalable to heterogeneous peers (Claude Code, Codex, …) without
|
||||
the core learning any one peer's environment.
|
||||
|
||||
## 1. Why
|
||||
|
||||
`claude-bridge` is a **communication bus between heterogeneous AI agents** — its stable surface is
|
||||
the protocol (`fleet_spawn / send / poll / reply / ask / list / stop / status`), and that surface
|
||||
should stay provider-neutral. Today the daemon can only materialize one kind of peer: an
|
||||
off-subscription Claude Code CLI over herdr. Everything specific to *how that peer is set up*
|
||||
(`ANTHROPIC_BASE_URL`, the subscription guard, `--mcp-config`/system-prompt flags, `claude-*`
|
||||
naming, git-token injection) is baked directly into the core spawn path.
|
||||
|
||||
The goal of CB-401 is a seam — a **`PeerLauncher` SPI** — so that "how to bring a peer of kind X to
|
||||
life" lives in a swappable adapter the bus *delegates to*, while the bus itself owns only transport,
|
||||
session/turn lifecycle, and routing. This both unlocks a second peer kind (Codex, a human, another
|
||||
Claude) and retroactively gives the CB-301-ext / CB-302 environment features a principled home (an
|
||||
adapter) instead of sitting in core.
|
||||
|
||||
> **Non-goal for CB-401:** dynamic/external plugin loading (arbitrary jars). That is Stage C and
|
||||
> carries a security model of its own (§7). CB-401 delivers a *first-party, in-tree, config-selected*
|
||||
> SPI with exactly one implementation, proving the seam.
|
||||
|
||||
## 2. The boundary
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph core["BRIDGE CORE — provider-neutral"]
|
||||
proto["Protocol verbs<br/>spawn / send / poll / reply / ask / list / stop"]
|
||||
sm["SessionManager<br/>FSM · registry · roster · lifecycle"]
|
||||
msg["Message store / routing"]
|
||||
life["Lifecycle limits<br/>idle_ttl · context_cap · drain"]
|
||||
end
|
||||
subgraph adapters["PEER ADAPTERS — env-specific"]
|
||||
cc["ClaudeCodeLauncher<br/>(today's WorkerService)"]
|
||||
cx["CodexLauncher<br/>(future, CB-402)"]
|
||||
hu["HumanLauncher<br/>(future)"]
|
||||
end
|
||||
sm -->|"delegates spawn/release"| spi{{"PeerLauncher SPI"}}
|
||||
spi --> cc
|
||||
spi --> cx
|
||||
spi --> hu
|
||||
cc -.->|"herdr transport"| herdr["herdr (terminal multiplexer)"]
|
||||
```
|
||||
|
||||
*Figure 1 — the core delegates peer materialization to a launcher chosen by profile; the core never
|
||||
learns a peer's env.*
|
||||
|
||||
## 3. Coupling audit (as-built, main @ `0efb65c`)
|
||||
|
||||
Where Claude/herdr specifics actually live today:
|
||||
|
||||
| Concern | Location | Verdict |
|
||||
|---|---|---|
|
||||
| `ANTHROPIC_BASE_URL` / `ANTHROPIC_MODEL` / `CLAUDE_CONFIG_DIR` / `ANTHROPIC_AUTH_TOKEN` env | `WorkerService.spawn` | **→ adapter** |
|
||||
| `SubscriptionGuard.assertWorker(baseUrl)` (billing boundary) | `WorkerService.spawn` → `guard` | **→ adapter** (it guards an `ANTHROPIC_*` concept) |
|
||||
| `--mcp-config` + `--append-system-prompt REPLY_CHARTER` (Claude Code CLI flags) | `WorkerService.argvWithBridge` | **→ adapter** |
|
||||
| `claude-<profile>-<nonce>-<seq>` naming, `WORKER_NAME` regex, orphan reap (CB-117) | `WorkerService` | **→ adapter** (naming is a herdr-label detail) |
|
||||
| `GITEA_TOKEN` / `GITEA_HOST` injection (CB-302 checkpoint) | `WorkerService.spawn` | **→ adapter** + a **capability** (§6) |
|
||||
| tab/pane placement, worker space, tab labels | `WorkerService.spawnInTab/spawnAsPane` via herdr `WorkspaceControl` | **→ adapter** (herdr transport detail) |
|
||||
| `FleetConfig.Worker` profile shape (`baseUrl`, `model`, `configDir`, …) | `config` | **mostly adapter-shaped** — see §5 |
|
||||
| FSM, registry, roster, `reapIdle`/`drainAll`/`contextCap`, `rosterView` | `SessionManager` | **stays core** |
|
||||
| turn/completion detection (`TurnListener`, `CompletionResolver`, `StatusPoller`, `WorkerPresence`) | `inject/` | **stays core**, but reads herdr terminal output → transport-coupled (§4b) |
|
||||
| message store & routing | `msg/` | **stays core** |
|
||||
| `Agent` / `paneId` handle | `herdr/` | **generalize** — see §4a |
|
||||
|
||||
**Finding:** the extraction is tractable because ~90% of the coupling is already funnelled through
|
||||
one class (`WorkerService`). Renaming/adapting it to `ClaudeCodeLauncher implements PeerLauncher` and
|
||||
having `SessionManager` depend on the interface is the bulk of Stage A.
|
||||
|
||||
## 4. Two friction points
|
||||
|
||||
### 4a. `paneId` is a herdr handle, not a peer-neutral id
|
||||
|
||||
`SessionManager` keys its registry by `paneId`, MCP/REST route by `paneId`, and `WorkerSession`
|
||||
stores it. `paneId` is a herdr pane handle — meaningless for a peer that isn't a herdr pane.
|
||||
|
||||
**Decision:** introduce an opaque `PeerHandle` the launcher returns. It carries a launcher-assigned
|
||||
**`id`** (the registry/routing key) plus launcher-private coordinates (for herdr: paneId, tabId,
|
||||
terminalId). Stage A keeps `id == paneId` for the Claude adapter so nothing downstream changes value,
|
||||
but the *type* stops being "a herdr pane" — the core routes on `PeerHandle.id()`.
|
||||
|
||||
```mermaid
|
||||
classDiagram
|
||||
class PeerLauncher {
|
||||
<<interface>>
|
||||
+Set~Capability~ capabilities()
|
||||
+PeerHandle spawn(SpawnRequest req)
|
||||
+void release(PeerHandle h)
|
||||
+String effectiveCwd(SpawnRequest req)
|
||||
+List~String~ parityOverlay(String profile)
|
||||
+int reapOrphans()
|
||||
}
|
||||
class PeerHandle {
|
||||
<<interface>>
|
||||
+String id()
|
||||
}
|
||||
class ClaudeCodeLauncher {
|
||||
herdr AgentControl/WorkspaceControl
|
||||
SubscriptionGuard
|
||||
}
|
||||
PeerLauncher <|.. ClaudeCodeLauncher
|
||||
ClaudeCodeLauncher ..> PeerHandle : returns
|
||||
```
|
||||
|
||||
*Figure 2 — the SPI the core sees. `ClaudeCodeLauncher` is today's `WorkerService`, adapted.*
|
||||
|
||||
### 4b. Turn/completion detection reads herdr output
|
||||
|
||||
`inject/` (turn listener, completion resolver, status poller, presence) infers turn boundaries from
|
||||
herdr terminal scraping. That is genuinely peer-transport-specific — a Codex peer would signal turns
|
||||
differently. For CB-401 this stays in core (it's the *Claude/herdr* transport's detector), but §6's
|
||||
capability model is what lets a future non-herdr peer bring its own turn-signalling without the core
|
||||
assuming terminal scraping. **Out of scope for Stage A**; noted so the SPI doesn't accidentally
|
||||
hard-wire "turns come from herdr".
|
||||
|
||||
## 5. Config shape
|
||||
|
||||
`FleetConfig.Worker` is Claude-shaped (`baseUrl`, `model`, `configDir`, `tokenEnv`). Rather than
|
||||
break existing YAML, CB-401 keeps `workers:` exactly as-is and treats those fields as the
|
||||
**ClaudeCodeLauncher's** profile schema. A future peer kind adds a `kind:` discriminator
|
||||
(default `"claude-code"`) selecting the launcher; unknown-kind → clear config error. No migration of
|
||||
existing configs. (Jackson already ignores unknown keys, so adding `kind` is backward-safe.)
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MCP as fleet_spawn (MCP/REST)
|
||||
participant SM as SessionManager
|
||||
participant L as PeerLauncher (by profile.kind)
|
||||
participant T as transport (herdr)
|
||||
MCP->>SM: acquire(profile, cwd, owner)
|
||||
SM->>L: spawn(SpawnRequest)
|
||||
L->>L: build env + guard + argv (adapter-private)
|
||||
L->>T: start(name, argv, env, cwd)
|
||||
T-->>L: handle (paneId…)
|
||||
L-->>SM: PeerHandle(id)
|
||||
SM->>SM: register session keyed by handle.id()
|
||||
SM-->>MCP: session view
|
||||
```
|
||||
|
||||
*Figure 3 — spawn delegation. The core's `acquire` is unchanged in shape; only the thing it calls
|
||||
becomes an interface.*
|
||||
|
||||
## 6. Capabilities
|
||||
|
||||
Peers are not uniform. Let each launcher declare a capability set; the protocol is the union and
|
||||
degrades gracefully when a launcher lacks one:
|
||||
|
||||
| Capability | Meaning | Claude Code | Codex (likely) | Human |
|
||||
|---|---|---|---|---|
|
||||
| `MID_TURN_ASK` | supports `fleet_ask` rendezvous | ✓ | ? | ✗ |
|
||||
| `SELF_PR` | can open its own PR at checkpoint (CB-302) | ✓ (opt-in token) | ? | ✗ |
|
||||
| `WORKTREE` | can run in a provisioned git worktree | ✓ | ✓ | ✗ |
|
||||
| `ORPHAN_REAP` | spawner can reconcile orphaned peers on boot | ✓ | ? | ✗ |
|
||||
|
||||
A verb invoked against a peer that lacks the capability returns a clean "unsupported for this peer"
|
||||
rather than a crash. This keeps the protocol honest as peers diversify and prevents the core from
|
||||
assuming "every peer is a Claude in a worktree" (the drift signal from the identity note).
|
||||
|
||||
## 7. Staging & the Stage-C security gate
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A["Stage A — CB-401<br/>extract PeerLauncher SPI in-tree<br/>ClaudeCodeLauncher = adapted WorkerService<br/>one impl, config-selected"] --> B["Stage B — CB-402+<br/>2nd in-tree adapter (Codex/human)<br/>proves the SPI held"]
|
||||
B --> C["Stage C<br/>dynamic external plugin loading<br/>ServiceLoader / jar discovery"]
|
||||
C -.requires.-> G["Trust & capability model<br/>what env/tokens a plugin may inject"]
|
||||
classDef gate fill:#b7791f,stroke:#7b341e,color:#ffffff;
|
||||
class G gate
|
||||
```
|
||||
|
||||
*Figure 4 — deliver A now; B when a real second peer exists; C only if third parties must ship
|
||||
adapters, and only behind a trust model.*
|
||||
|
||||
**Security note (Stage C, not now):** a launcher runs at daemon privilege and touches process
|
||||
spawning **and env/token injection into peers** — the most sensitive path in the system. "Extra
|
||||
plugins" must mean *first-party, in-tree, config-selected* for the foreseeable future. A third-party
|
||||
plugin that can inject env into a peer needs a genuine trust/capability model before it may exist.
|
||||
Regardless of stage, the **primary remains the merge/verify gate** — self-reports over the bus are
|
||||
messages, not verified facts.
|
||||
|
||||
## 8. Stage A scope (this branch, delegation-ready)
|
||||
|
||||
Deliverable for CB-401 Stage A — mechanical, behaviour-preserving:
|
||||
|
||||
1. `PeerLauncher` interface + `PeerHandle` (opaque id) + `SpawnRequest` (profile, requestedCwd,
|
||||
callerCwd) + `Capability` enum, new package `dev.ltms.fleet.peer`.
|
||||
2. `ClaudeCodeLauncher implements PeerLauncher` = today's `WorkerService`, adapted: `spawn(...)`
|
||||
returns a `PeerHandle` (id = paneId), `capabilities()` declares
|
||||
`MID_TURN_ASK, SELF_PR(when token), WORKTREE, ORPHAN_REAP`.
|
||||
3. `SessionManager` depends on `PeerLauncher`, not `WorkerService` concretely; routing keys on
|
||||
`PeerHandle.id()` (== paneId today, so zero value change).
|
||||
4. `Fleetd.main` wires the concrete `ClaudeCodeLauncher` behind the interface.
|
||||
5. **No behaviour change, no config change.** Full green gate: `ide_sync` → `ide_diagnostics`
|
||||
(0 errors/0 warnings) → `mvn clean install` with `MVN_EXIT` captured (no masking pipe). All
|
||||
existing tests pass unchanged; add tests only for the new `PeerHandle` indirection.
|
||||
|
||||
Explicitly **out of scope** for Stage A: `kind:` config discriminator, any second adapter, capability
|
||||
*enforcement* at the verb layer (declare only), touching `inject/` turn detection, dynamic loading.
|
||||
@@ -0,0 +1,305 @@
|
||||
# CB-402 — Second peer adapter: opencode (Stage B of the Peer Launcher SPI)
|
||||
|
||||
**Status:** ✅ **complete — implemented, merged (`ded226a`), and live-dogfooded 2026-07-29.**
|
||||
All five increments of §4 are done, including increment 5 (the §5 live checklist). See
|
||||
[§8 As-built](#8-as-built--live-dogfood-2026-07-29) for the run. Gitea issue #7 closed.
|
||||
**Depends on:** CB-401 Stage A (`PeerLauncher` SPI, merged `3aa69a9`)
|
||||
**Stage:** 4 (Pluggable peers) · Stage B
|
||||
**Owner action:** design-note → file issue → delegate → primary-verify (per CB-401/306/307)
|
||||
|
||||
---
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Prove the [`PeerLauncher`](../fleetd/src/main/java/dev/ltms/fleet/peer/PeerLauncher.java) SPI
|
||||
actually holds for a **non-Claude** coding agent by shipping a second, first-class in-tree
|
||||
adapter: **opencode** (`opencode` 1.1.31, a provider-agnostic terminal coding agent).
|
||||
|
||||
The product direction is *heterogeneous coding agents, Claude Code first-class* — not a
|
||||
human/mock peer. opencode is the right proof precisely because it differs from Claude Code on
|
||||
every seam the SPI is meant to hide:
|
||||
|
||||
| Seam | Claude Code | opencode | ⇒ SPI proof |
|
||||
|------|-------------|----------|-------------|
|
||||
| Subscription boundary | `ANTHROPIC_BASE_URL` + `SubscriptionGuard.assertWorker()` before any herdr call | none — provider-agnostic, off-subscription by nature | the guard is **Claude-private**, not core |
|
||||
| MCP mount | inline `--mcp-config '{…}'` launch flag | `opencode mcp add` / config file (`OPENCODE_CONFIG`) — **no inline flag** | "mount the bridge MCP" is adapter-private |
|
||||
| Instruction injection | `--append-system-prompt "<REPLY_CHARTER>"` | config `instructions` / `AGENTS.md` / `--agent` — **no append flag** | the reply-charter mount is adapter-private |
|
||||
| Model selection | `ANTHROPIC_MODEL` env | `-m provider/model` flag | env-vs-flag is adapter-private |
|
||||
| Name / reap scheme | `claude-<profile>-<nonce>-<seq>` | `opencode-<profile>-<nonce>-<seq>` | each adapter reaps only its own kind |
|
||||
|
||||
Everything *else* — herdr tab/pane placement, the CB-306 spawn-readiness gate, CB-301-ext
|
||||
worktree provisioning, CB-117 orphan reap, teardown, `list()`, cwd resolution — is transport
|
||||
machinery that is **identical** for both. That split is the whole design.
|
||||
|
||||
> Out of scope (deferred to Stage C / later): dynamic external plugin loading behind a
|
||||
> trust/capability model, capability *enforcement* at the verb layer (Stage A only *declares*
|
||||
> caps), and a `human`/mock peer.
|
||||
|
||||
---
|
||||
|
||||
## 2. Current state — one launcher, two concerns mixed
|
||||
|
||||
`ClaudeCodeLauncher` (584 LOC) is the sole `PeerLauncher`. It interleaves two concerns:
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph CCL["ClaudeCodeLauncher (584 LOC) — today"]
|
||||
direction TB
|
||||
T["herdr transport (GENERIC / reusable)<br/>tab-pane placement · spawn-ready gate · worktree<br/>orphan reap · teardown · list · cwd resolution · unique naming"]
|
||||
C["Claude-specific (per-agent)<br/>ANTHROPIC_BASE_URL + SubscriptionGuard · ANTHROPIC_MODEL<br/>argv --mcp-config · --append-system-prompt REPLY_CHARTER · 'claude-' name prefix"]
|
||||
end
|
||||
classDef generic fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
classDef specific fill:#b7791f,stroke:#7b341e,color:#ffffff;
|
||||
class T generic
|
||||
class C specific
|
||||
```
|
||||
|
||||
*Figure 1 — the two concerns tangled inside today's single launcher; CB-402 splits them.*
|
||||
|
||||
There is also a **Stage-A deferral** to finish: `Fleetd.main` still casts
|
||||
`(ClaudeCodeLauncher) workers` at the `FleetMcp` and `FleetApp` constructors. Those two
|
||||
callers only invoke `profiles()`, `defaultProfile()`, and `list()` — **all already on the
|
||||
`PeerLauncher` interface**. The cast survives for one reason only: `PeerLauncher.list()`
|
||||
returns `List<?>` (element type erased) while the callers use `Agent` element methods in their
|
||||
roster join. Finishing the migration is therefore small and contained (§4.D).
|
||||
|
||||
---
|
||||
|
||||
## 3. Target design
|
||||
|
||||
Template-Method base + two thin adapters + a routing composite that keeps the Stage-A seam
|
||||
(one `PeerLauncher` reference held by `SessionManager` / `FleetMcp` / `FleetApp`) intact.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
IFACE["«interface»<br/>PeerLauncher"]
|
||||
COMP["CompositePeerLauncher<br/>routes by profile kind; fans out list/reap/caps"]
|
||||
BASE["«abstract»<br/>HerdrPeerLauncher<br/>transport: placement · ready-gate · reap · stop · cwd · naming"]
|
||||
CCL2["ClaudeCodeLauncher<br/>hooks: guard+ANTHROPIC_* env · --mcp-config · charter flag · prefix 'claude'"]
|
||||
OCL["OpenCodeLauncher<br/>hooks: provider env · OPENCODE_CONFIG file · AGENTS charter · prefix 'opencode'"]
|
||||
|
||||
IFACE -.implemented by.-> COMP
|
||||
IFACE -.implemented by.-> BASE
|
||||
BASE --> CCL2
|
||||
BASE --> OCL
|
||||
COMP -->|"kind=claude-code"| CCL2
|
||||
COMP -->|"kind=opencode"| OCL
|
||||
|
||||
classDef iface fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef base fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
classDef leaf fill:#6b46c1,stroke:#44337a,color:#ffffff;
|
||||
class IFACE,COMP iface
|
||||
class BASE base
|
||||
class CCL2,OCL leaf
|
||||
```
|
||||
|
||||
*Figure 2 — extracted base, two adapters, and a routing composite behind the unchanged SPI.*
|
||||
|
||||
### A. Extract `HerdrPeerLauncher` (abstract base)
|
||||
|
||||
Move all transport machinery down from `ClaudeCodeLauncher`. What stays generic:
|
||||
|
||||
- fields `agents`, `spaces`, `profiles`, `defaultProfile`, `env`, `nameSeq`, the CB-306 gate
|
||||
knobs (`spawnReadyTimeoutMs`/`spawnReadyPollMs`/`nowMillis`/`sleeper`), and `nameNonce`;
|
||||
- `profiles()`, `defaultProfile()`, `parityOverlay()`, `effectiveCwd(…)`, `resolveCwd`;
|
||||
- the `spawn(SpawnRequest)` **skeleton**: resolve profile → cfg → *hook* → placement → gate → `WorkerHandle`;
|
||||
- `spawnInTab` / `spawnAsPane` / `tidy` / `startUniquelyNamed` (name built from a *hook* prefix);
|
||||
- `list()`, `reapOrphanWorkers()` / `isForeignWorker` / `workerNonce` (pattern built from the prefix hook), `stop()` / `usesTabPlacement` / `isAlreadyGone`;
|
||||
- `waitUntilInjectableOrThrow`, the `WorkerHandle` record, `putIfPresent`, `resolveEnv`, `sleepUninterruptibly`.
|
||||
|
||||
Two adapter **hooks** (abstract):
|
||||
|
||||
```java
|
||||
/** Label prefix for this peer kind; drives unique naming AND the orphan-reap pattern. */
|
||||
protected abstract String namePrefix(); // "claude" | "opencode"
|
||||
|
||||
/** Build the peer-specific launch: env map + argv. Runs any pre-spawn guard here. */
|
||||
protected abstract Launch buildLaunch(FleetConfig.Worker cfg, SpawnRequest req);
|
||||
record Launch(Map<String,String> env, List<String> argv) {}
|
||||
```
|
||||
|
||||
`capabilities()` stays abstract/per-adapter (it already is). The subscription guard is **not**
|
||||
a base field — it is a constructor dependency of `ClaudeCodeLauncher` alone.
|
||||
|
||||
**Reap isolation:** the reap pattern becomes `Pattern.compile(namePrefix() + "-.*-([0-9a-f]{6})-\\d+")`,
|
||||
so the opencode adapter never reaps a `claude-*` pane and vice-versa. The composite sums both.
|
||||
|
||||
### B. `kind:` config discriminator
|
||||
|
||||
Add one field to `FleetConfig.Worker`:
|
||||
|
||||
```java
|
||||
String kind // "claude-code" (default) | "opencode"
|
||||
```
|
||||
|
||||
- Compact-ctor default: `kind = blank ? "claude-code" : kind.toLowerCase()`.
|
||||
- `argv` default is currently `List.of("claude")`; when `kind=opencode` and the operator left
|
||||
`argv` unset, default it to `List.of("opencode")`. (Handle in normalization, keyed off `kind`,
|
||||
so the record stays declarative.)
|
||||
- Keep the existing back-compat constructors; `kind` is additive and optional.
|
||||
|
||||
`fleetd.example.yaml` documents a two-kind `workers:` block.
|
||||
|
||||
### C. `OpenCodeLauncher` — the adapter hooks for opencode
|
||||
|
||||
`namePrefix()` → `"opencode"`. `buildLaunch(cfg, req)`:
|
||||
|
||||
- **Env:** *no* `ANTHROPIC_BASE_URL`, *no* `SubscriptionGuard` call. Pass through provider
|
||||
credentials the operator names (reuse the existing `tokenEnv` indirection; opencode reads
|
||||
provider keys from env / `opencode auth`). CB-302 git-token injection is reused unchanged
|
||||
(it is peer-neutral: `GITEA_TOKEN`/`GITEA_HOST`).
|
||||
- **MCP mount (non-invasive):** opencode has no inline `--mcp-config`. The adapter writes a
|
||||
throwaway config file and points `OPENCODE_CONFIG=<tempfile>` in the worker env, containing
|
||||
the bridge MCP server block (opencode HTTP MCP schema, `type: "remote"`) — the opencode analog
|
||||
of Claude Code's inline flag. Nothing is written into the worker's real project or profile.
|
||||
- **Reply-charter:** carry `REPLY_CHARTER` as an `instructions` entry in that same generated
|
||||
config (or an `AGENTS.md` written into the per-worker worktree, which is already a throwaway
|
||||
isolated checkout under CB-301-ext). Recommend the config-file route to keep the "touch
|
||||
nothing the user owns" invariant.
|
||||
- **argv:** `opencode <project-or-cwd>` (interactive TUI, the mode a herdr pane drives), plus
|
||||
`-m <provider/model>` when the profile sets a model.
|
||||
|
||||
> `REPLY_CHARTER` is peer-neutral text — hoist it to a shared constant (base or a small
|
||||
> `PeerCharter`), consumed by each adapter through its own injection mechanism.
|
||||
|
||||
### D. `CompositePeerLauncher` + finish the Stage-A migration
|
||||
|
||||
- `Fleetd.main` groups configured profiles by `kind`, instantiates one launcher per kind
|
||||
present, and wraps them in `CompositePeerLauncher implements PeerLauncher`.
|
||||
- Routing methods (`spawn(req)`, `effectiveCwd(req)`, `parityOverlay(name)`) dispatch by the
|
||||
profile's kind. Fan-out methods (`list()`, `reapOrphanWorkers()`, `capabilities()`,
|
||||
`profiles()`, `defaultProfile()`) merge across sub-launchers. `stop(id)` tries each (teardown
|
||||
only knows the pane id) — already best-effort/idempotent.
|
||||
- **Migrate `FleetMcp` + `FleetApp` to the `PeerLauncher` interface**, dropping both
|
||||
`(ClaudeCodeLauncher)` casts. Only friction is `list()`'s `List<?>`; resolve by giving the SPI
|
||||
a typed roster element (small neutral `PeerAgent` view exposing `id()`/`name()`/status) that
|
||||
the CB-304 roster join consumes — or, minimally, narrow at the callsite. Prefer the typed view.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant P as Primary
|
||||
participant M as FleetMcp / REST
|
||||
participant C as CompositePeerLauncher
|
||||
participant O as OpenCodeLauncher
|
||||
participant B as HerdrPeerLauncher (base)
|
||||
participant H as herdr
|
||||
P->>M: fleet_spawn(profile="oc-impl")
|
||||
M->>C: spawn(SpawnRequest)
|
||||
C->>C: kind(profile)=="opencode"
|
||||
C->>O: spawn(req)
|
||||
O->>O: buildLaunch → provider env + OPENCODE_CONFIG file + argv
|
||||
O->>B: placement + startUniquelyNamed("opencode-…")
|
||||
B->>H: agent.start(name, argv, env, tab, cwd)
|
||||
B->>H: poll status until injectable (CB-306 gate)
|
||||
B-->>O: Agent
|
||||
O-->>C: PeerHandle(paneId, terminalId)
|
||||
C-->>M: PeerHandle
|
||||
M-->>P: session id
|
||||
```
|
||||
|
||||
*Figure 3 — an opencode spawn: composite routes by kind, adapter builds the peer-specific launch, shared base drives herdr + the readiness gate.*
|
||||
|
||||
---
|
||||
|
||||
## 4. Increment plan (delegate-then-verify friendly)
|
||||
|
||||
1. **Extract base, no behaviour change.** Introduce `HerdrPeerLauncher`; make
|
||||
`ClaudeCodeLauncher` extend it with `namePrefix()="claude"` and `buildLaunch()` wrapping
|
||||
today's guard+env+argv logic. Green build, identical tests — pure refactor. *(IDE
|
||||
`refactor` where possible; the primary re-runs the gate workers can't.)*
|
||||
2. **`kind:` discriminator.** Add the field + normalization + `fleetd.example.yaml`. Default
|
||||
path unchanged (`kind=claude-code`).
|
||||
3. **`OpenCodeLauncher`.** Implement the three hooks; unit-test `buildLaunch` (env has no
|
||||
`ANTHROPIC_BASE_URL`; `OPENCODE_CONFIG` points at a file carrying the bridge MCP block +
|
||||
charter; argv shape).
|
||||
4. **`CompositePeerLauncher` + wiring + finish Stage-A migration** (drop the two casts).
|
||||
5. **Live dogfood** (§5) + wiki as-built (primary-gated submodule commit).
|
||||
|
||||
Each increment is independently buildable/mergeable; the feature branch stays **unmerged**
|
||||
until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
|
||||
---
|
||||
|
||||
## 5. Risks & validation (live dogfood, not assumed)
|
||||
|
||||
- **opencode TUI ⇄ herdr injection.** herdr drives a pane by typing into a TUI. Must confirm
|
||||
opencode's TUI accepts injected keystrokes/submit the way `claude` does, and reaches an
|
||||
`injectable` status the CB-306 gate recognizes. *Validation:* spawn one opencode worker,
|
||||
watch the readiness gate pass, `fleet_send` a trivial task.
|
||||
- **Bridge MCP visibility in opencode.** Confirm `OPENCODE_CONFIG` (or `opencode mcp add`)
|
||||
actually surfaces the `fleet_*` tools inside the opencode session, and that `fleet_reply`
|
||||
is callable — the reply-charter is worthless if the tool isn't mounted. *Validation:* the
|
||||
worker completes a task by calling `fleet_reply`; the reply lands via the CB-307 path.
|
||||
- **opencode MCP/config schema drift.** opencode is fast-moving (1.1.31 today). Pin the config
|
||||
schema we generate against the installed version; treat the exact keys (`type: "remote"` vs
|
||||
`"http"`, `instructions` shape) as a dogfood-verified fact, not an assumption.
|
||||
- **Provider credentials.** opencode needs a configured provider (env key or `opencode auth`).
|
||||
The dogfood profile must name a provider the host actually has, distinct from the primary's
|
||||
subscription.
|
||||
|
||||
---
|
||||
|
||||
## 6. Test plan
|
||||
|
||||
- **Unit (hermetic):** base-extraction regression (existing `ClaudeCodeLauncher` tests pass
|
||||
unchanged); `OpenCodeLauncher.buildLaunch` env/argv/config assertions; `kind` normalization
|
||||
in `FleetConfigTest`; `CompositePeerLauncher` routing + fan-out (merge of `profiles()`,
|
||||
summed `reapOrphanWorkers()`, per-kind reap isolation) with fake sub-launchers.
|
||||
- **Live (dogfood, manual):** the §5 checklist on the running daemon.
|
||||
- **Gate (primary):** IDE diagnostics 0/0 on every changed file, `mvn clean install` green with
|
||||
the surefire summary captured (not `| tail`), manual diff review — the authoritative checks a
|
||||
worker cannot self-run.
|
||||
|
||||
---
|
||||
|
||||
## 7. Open questions for the lead
|
||||
|
||||
1. ✅ **Provider — resolved 2026-07-29.** None was needed. opencode's own gateway serves
|
||||
**free-tier models with zero credentials** (`opencode auth list` → *0 credentials*, yet
|
||||
`opencode run -m opencode/north-mini-code-free` answers). The dogfood profile uses
|
||||
`opencode/north-mini-code-free`. It is distinct from the primary's Anthropic subscription by
|
||||
construction, and needs no `guard` entry — opencode carries no `ANTHROPIC_BASE_URL`, so the
|
||||
`SubscriptionGuard` does not apply to it at all.
|
||||
2. ✅ **Charter carrier — confirmed as designed:** generated `OPENCODE_CONFIG` `instructions`.
|
||||
Verified working against the installed version.
|
||||
3. ✅ **Merge cadence — resolved as it happened:** Stage B landed as one unit (`ded226a`).
|
||||
|
||||
---
|
||||
|
||||
## 8. As-built — live dogfood (2026-07-29)
|
||||
|
||||
Run against `fleetd` on `127.0.0.1:8766` at main `19cdf8d`, with opencode **1.18.5** installed
|
||||
via Homebrew. Every §5 risk is now a verified fact rather than an assumption.
|
||||
|
||||
**The version-drift risk was the real one, and it did not bite.** This adapter was designed against
|
||||
opencode **1.1.31**; the installed version is **1.18.5**. The generated config schema still
|
||||
validates unchanged — `mcp.<name>.type: "remote"`, `url`, `enabled`, and `instructions: [path]` are
|
||||
all accepted, and `OPENCODE_CONFIG=… opencode mcp list` reports `✓ bridge connected`. Pinned here
|
||||
as a dogfood-verified fact for 1.18.5.
|
||||
|
||||
| §5 risk | Result |
|
||||
|---|---|
|
||||
| opencode TUI ⇄ herdr injection; CB-306 gate | ✅ `peer pane=wD:p3 reached injectable state` ~0.6s after `agent.start` |
|
||||
| Bridge MCP visible + `fleet_reply` callable | ✅ MCP `initialize` from `Implementation[name=opencode, version=1.18.5]`; worker replied through the tool |
|
||||
| Config schema drift (1.1.31 → 1.18.5) | ✅ unchanged, see above |
|
||||
| Provider credentials | ✅ free tier, zero credentials |
|
||||
|
||||
Full lifecycle exercised through the REST surface:
|
||||
|
||||
1. `POST /workers?profile=opencode-free` → `201`, routed by `kind:` through `CompositePeerLauncher`
|
||||
to `OpenCodeLauncher` (`spawning opencode profile=opencode-free`), pane `wD:p3`.
|
||||
2. Readiness: `{"ready":true,"status":"idle"}`, roster state `ready`.
|
||||
3. `POST /sessions/{id}/message` → **`{"replySource":"reply","reply":"391"}`** — a *structured*
|
||||
`fleet_reply`, not the CB-115 completion-fallback transcript scrape. The clean path.
|
||||
4. `DELETE /workers/wD:p3` → `204`, roster empty, tolerant teardown (`tab_not_found` ignored —
|
||||
opencode had already closed its own tab).
|
||||
|
||||
**Unplanned cross-validation with CB-501.** The audit trail recorded the worker's reply as
|
||||
`role: WORKER, actor: worker:term_657c1dad2b9731e, action: REPLY, outcome: allowed`. Connection-based
|
||||
identity (loopback peer PID → herdr pane) classified an **opencode** process as a worker with no
|
||||
opencode-specific handling — confirming the identity model is peer-kind-agnostic, which is exactly
|
||||
what CB-308 needs when it stretches the roster across hosts.
|
||||
|
||||
CB-502 counters for the same run: `fleet_sends_total{outcome="replied"} 1`,
|
||||
`fleet_replies_total{path="rendezvous"} 1`, `fleet_inbox_depth{...} 0`.
|
||||
@@ -0,0 +1,482 @@
|
||||
# CB-500 — Multi-Tier Coordination (Stage 6)
|
||||
|
||||
**Status:** design note. Developments A/B remain proposals; Development C (§6 and Figures 7–8) is
|
||||
**SUPERSEDED** by the advisory-architect design in Gitea issue #16 and the `architects:` configuration
|
||||
block (CB-548).
|
||||
**Depends on:** CB-401/402 (Peer Launcher SPI + composite router — placement-neutral spawn),
|
||||
CB-308 (per-agent broker channels + global id + federated roster — the addressing substrate),
|
||||
CB-307 (durable inbox + push loop), CB-301/303 (session FSM + context-cap/idle-ttl), CB-304
|
||||
(`rosterView`).
|
||||
**Relates to:** the bus-identity boundary — see §7. This note stays a **proposal**; no code until the
|
||||
staging in §6 is reviewed and the arc is split into tickets.
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Grow `claude-bridge` from a **single-tier** coordinator (one human-driven primary → a flat pool of
|
||||
workers) into a **multi-tier** one, along three axes the lead has asked for:
|
||||
|
||||
1. **Sandboxed workers** — each worker runs in a **separated, peer-owned sandbox** carrying its own
|
||||
toolchain (Claude routed via `ANTHROPIC_BASE_URL`, a headless IDE, git, MCP, dev-tools), with
|
||||
**per-role** sandboxes (a backend-agent image, a frontend-agent image).
|
||||
2. **Main-agent pairs** — **SUPERSEDED.** The considered model made the "main" tier a pair
|
||||
(on-subscription Opus + one cloud module). The actual fleet is one human-driven lead plus two
|
||||
short-lived advisory architects on different model families.
|
||||
3. **An orchestrator tier** — a supervisor **above** the mains that owns their **session identity**
|
||||
(naming, resume) and **curates context**, so every main→worker delegation carries the *exact*
|
||||
slice of context it needs and nothing else.
|
||||
|
||||
The through-line: **this is not a new pillar.** It is the existing `PeerLauncher` and
|
||||
`SessionManager` patterns extended one tier up, riding the **same CB-308 substrate** that multi-host
|
||||
already needs. Sandbox = a placement-neutral spawn target (CB-402 pattern). Pair + orchestrator =
|
||||
per-agent channels + a recursive session manager (CB-308 pattern). The bus stays a
|
||||
**provider-neutral communication fabric**; every addition is addressing, launch, or session scoping —
|
||||
never toolchain ownership (§7).
|
||||
|
||||
## 2. Single-tier today (the assumptions to break)
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
human["human (types)"]
|
||||
primary["PRIMARY (Opus)<br/>MCP client — pull-only"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765 (single host)"]
|
||||
comp["CompositePeerLauncher<br/>routes by kind"]
|
||||
cc["ClaudeCodeLauncher"]
|
||||
oc["OpenCodeLauncher"]
|
||||
w1["worker pane (gx00 vLLM)"]
|
||||
w2["worker pane (ollama)"]
|
||||
human --> primary
|
||||
primary -->|"fleet_send / spawn / ask"| daemon
|
||||
daemon --> comp
|
||||
comp --> cc
|
||||
comp --> oc
|
||||
cc --> w1
|
||||
oc --> w2
|
||||
```
|
||||
|
||||
*Figure 1 — one human-driven primary, one daemon, a flat pool of bare herdr-pane workers.*
|
||||
|
||||
Four concrete bake-ins assume a single tier:
|
||||
|
||||
| Assumption | Where (verified) | Why it blocks the direction |
|
||||
|---|---|---|
|
||||
| **Exactly one primary** | `mcp/PrimaryRegistry` — an `AtomicReference<String>`, "single-slot registry for the primary's terminal" | A *pair* needs N addressable mains, each with its own pull inbox. |
|
||||
| **Workers are bare panes** | `worker/*Launcher` spawn a herdr pane via `argv:["ccs", …]` into a pre-existing env | A *sandbox* is a richer launch target (container/devcontainer) — a new placement, not a new provider. |
|
||||
| **`SpawnRequest` is flat** | `peer/SpawnRequest(profileName, requestedCwd, callerCwd)` | A sandbox/role selection needs a spawn-target dimension the record does not carry. |
|
||||
| **No tier above the primary** | there is no manager of the *primary's own* session — `SessionManager` manages *workers* only | An orchestrator that names/resumes/scopes the mains is a wholly new (but pattern-reusable) tier. |
|
||||
|
||||
## 3. Target multi-tier architecture
|
||||
|
||||
> **SUPERSEDED fleet sketch.** Figure 2 records the former two-main model. The actual fleet is one
|
||||
> lead, two independent advisory architects, and N workers; architects are sideways peers, not leads
|
||||
> and not a tier above the lead. See Gitea issue #16 and the `architects:` block.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
human["human"]
|
||||
subgraph orch["TIER 0 — orchestrator"]
|
||||
osm["OrchestratorSessionManager<br/>(SessionManager, recursed up)<br/>names · resumes · scopes context"]
|
||||
end
|
||||
subgraph mains["TIER 1 — main pair"]
|
||||
m1["main A: Opus<br/>MCP client"]
|
||||
m2["main B: cloud module<br/>MCP client"]
|
||||
end
|
||||
subgraph bus["fleetd fabric (CB-307/308 substrate)"]
|
||||
chan["per-agent inbox channels<br/>agent.<globalId>.inbox"]
|
||||
roster["federated roster (union view)"]
|
||||
end
|
||||
subgraph workers["TIER 2 — sandboxed workers"]
|
||||
sbBE["backend sandbox<br/>Claude via ANTHROPIC_BASE_URL<br/>+ headless IDE · git · MCP · dev-tools"]
|
||||
sbFE["frontend sandbox<br/>(role-specific image)"]
|
||||
end
|
||||
human --> osm
|
||||
osm -->|"spawn / name / resume"| m1
|
||||
osm -->|"spawn / name / resume"| m2
|
||||
m1 <-->|"pull inbox"| chan
|
||||
m2 <-->|"pull inbox"| chan
|
||||
m1 -->|"scoped delegation"| bus
|
||||
m2 -->|"scoped delegation"| bus
|
||||
bus --> sbBE
|
||||
bus --> sbFE
|
||||
chan --- roster
|
||||
```
|
||||
|
||||
*Figure 2 — **SUPERSEDED historical fleet sketch.** It proposed a collaborating pair of managed mains.
|
||||
The actual fleet keeps one human-driven lead and uses two independent, short-lived advisory architects
|
||||
on different model families, so agreement is evidence rather than correlated echo.*
|
||||
|
||||
The recursion is the key idea: **`orchestrator : mains :: main : workers`** — the same
|
||||
spawn/name/resume/scope verbs at two levels.
|
||||
|
||||
## 4. Development A — Sandboxed, role-specific workers
|
||||
|
||||
A "sandbox" is a **placement**, not a provider — so it slots into the CB-401 SPI exactly the way
|
||||
CB-402's opencode adapter slotted in as a new *provider*. CB-402 proved the SPI is
|
||||
provider-neutral; a `SandboxLauncher` proves it is **placement-neutral**.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
req["SpawnRequest<br/>(profileName, cwd, + sandbox/role)"]
|
||||
comp["CompositePeerLauncher<br/>routes by kind"]
|
||||
cc["ClaudeCodeLauncher<br/>kind: claude-code"]
|
||||
oc["OpenCodeLauncher<br/>kind: opencode"]
|
||||
sb["SandboxLauncher (NEW)<br/>kind: sandbox"]
|
||||
subgraph owned["bridge OWNS (launch + inject boundary)"]
|
||||
launch["run sandbox entrypoint<br/>(docker/devcontainer up → agent)"]
|
||||
inject["inject + guard baseUrl,<br/>mount bridge MCP + charter"]
|
||||
end
|
||||
subgraph peer["peer OWNS (inside the sandbox)"]
|
||||
img["image = backend|frontend role<br/>headless IDE · git · dev-tools · MCP"]
|
||||
end
|
||||
req --> comp
|
||||
comp --> cc
|
||||
comp --> oc
|
||||
comp --> sb
|
||||
sb --> launch --> inject
|
||||
inject -.->|"launches into, never builds"| img
|
||||
classDef line fill:#b7791f,stroke:#7b341e,color:#ffffff;
|
||||
class inject line
|
||||
```
|
||||
|
||||
*Figure 3 — the ownership line (amber). The bridge runs the sandbox entrypoint and injects the same
|
||||
boundary it owns today (guarded `baseUrl`, mounted MCP + reply charter). Everything inside the image
|
||||
— the IDE, git, dev-tools — is the peer's. The bridge references the image/role; it never provisions
|
||||
tools. This is what keeps "give the worker a headless IDE" on the right side of the "bus, not
|
||||
env-manager" rule (§7).*
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant M as main (delegator)
|
||||
participant D as fleetd
|
||||
participant SL as SandboxLauncher
|
||||
participant SB as sandbox (peer-owned)
|
||||
participant A as agent in sandbox
|
||||
M->>D: fleet_spawn(profile=backend, role=backend)
|
||||
D->>SL: spawn(SpawnRequest)
|
||||
SL->>SB: start entrypoint (image = backend role)
|
||||
Note over SL,SB: bridge injects guarded ANTHROPIC_BASE_URL,<br/>mounts bridge MCP url + reply charter
|
||||
SB->>A: launch Claude (headless IDE, git, MCP ready — peer's own)
|
||||
A-->>SL: MCP connects → readiness gate (CB-306) passes
|
||||
SL-->>D: PeerHandle(globalId)
|
||||
D-->>M: spawned, injectable
|
||||
```
|
||||
|
||||
*Figure 4 — spawn into a peer-owned sandbox. Identical control flow to today's pane spawn (incl. the
|
||||
CB-306 readiness gate); only the launcher's `buildLaunch` differs — exactly the CB-402 seam.*
|
||||
|
||||
**Deltas:** a new `kind: sandbox` adapter (extends the same `HerdrPeerLauncher`/`PeerLauncher` base);
|
||||
a spawn-target/role dimension on `SpawnRequest` and the `Worker` profile; optionally a `SANDBOX`
|
||||
`Capability`. Per-role = two profiles → two images; `CompositePeerLauncher` already routes them. If a
|
||||
sandbox is a *separate host/container*, it reuses CB-308's global id + per-host gateway wholesale —
|
||||
**the distributed case is resolved in §11: a sandbox on another host is one spawned by that host's
|
||||
gateway, because herdr keystroke-injection needs a locally-owned PTY.**
|
||||
|
||||
## 5. Development B — Main-agent pairs
|
||||
|
||||
> **SUPERSEDED — do not implement this model.** The two-main fleet was replaced by one human-driven
|
||||
> lead and two independent advisory architects. They are deliberately different model families (Claude
|
||||
> Sonnet 5 and GPT-5.6 through opencode), receive the same brief, and work independently so agreement
|
||||
> is evidence rather than correlated echo. See Gitea issue #16 and `architects:`.
|
||||
|
||||
Both mains are MCP **clients**, so **neither can be called into** — each needs a **pull-based
|
||||
per-agent inbox**, which is precisely CB-308 item #1 (per-agent AMQP channels). The primary machinery
|
||||
that is singular today (single-slot `PrimaryRegistry`, a push-loop aimed at one terminal, "these
|
||||
tools only the primary calls") generalizes from a singleton to a **set**.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph pair["TIER 1 — collaborating pair"]
|
||||
m1["main A: Opus<br/>MCP client (pull-only)"]
|
||||
m2["main B: cloud module<br/>MCP client (pull-only)"]
|
||||
end
|
||||
reg["PrimaryRegistry → multi-slot<br/>(terminal per main)"]
|
||||
subgraph fabric["fleetd"]
|
||||
ca["agent.A.inbox"]
|
||||
cb["agent.B.inbox"]
|
||||
push["ReplyPushLoop → N terminals"]
|
||||
end
|
||||
m1 <-->|"peer-to-peer message"| m2
|
||||
m1 -->|"register terminal"| reg
|
||||
m2 -->|"register terminal"| reg
|
||||
ca -->|"pull / nudge"| m1
|
||||
cb -->|"pull / nudge"| m2
|
||||
reg --> push
|
||||
push --> ca
|
||||
push --> cb
|
||||
```
|
||||
|
||||
*Figure 5 — the pair. Each main owns an addressable inbox; `PrimaryRegistry` becomes multi-slot; the
|
||||
push loop nudges each main's terminal. Mains message each other as equals over the same bus (the
|
||||
transport is already peer-neutral — what was missing is N pull endpoints).*
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MA as main A (Opus)
|
||||
participant BR as fleetd / broker
|
||||
participant MB as main B (cloud)
|
||||
MA->>BR: fleet_send(to = main B, msg)
|
||||
BR->>BR: publish agent.B.inbox (durable, msg id)
|
||||
Note over BR: held until B pulls (B is a client too)
|
||||
MB->>BR: blocking fleet_send / poll resolves
|
||||
BR-->>MB: msg (then ACK)
|
||||
MB->>BR: fleet_reply(to = main A)
|
||||
BR->>BR: publish agent.A.inbox
|
||||
MA->>BR: poll resolves
|
||||
BR-->>MA: reply
|
||||
```
|
||||
|
||||
*Figure 6 — main↔main is the CB-307 asymmetry applied on both ends: two clients, so both hops are
|
||||
pull. This is why Part B **depends on** the per-agent-channel substrate, not just a config flag.*
|
||||
|
||||
**Deltas:** `PrimaryRegistry` single-slot → keyed-by-main; per-main inbox routing (CB-308 #1);
|
||||
push-loop fan-out; relax "orchestration tools only the primary calls" to "any registered main."
|
||||
|
||||
## 6. Development C — Orchestrator tier
|
||||
|
||||
> **SUPERSEDED — do not implement this model.** The operator rejected a supervisor above the lead.
|
||||
> The human continues to drive the pre-existing lead directly; fleetd neither spawns nor resumes that
|
||||
> lead. What replaced this proposal is **one lead, two short-lived advisory architects, and N workers**:
|
||||
> the lead engages architects sideways for a strong-model assessment, then discards them. Architect
|
||||
> slots are declared in `architects:` (see Gitea issue #16), rather than making leads managed sessions.
|
||||
> The two architects deliberately use different model families — Claude Sonnet 5 and GPT-5.6 through
|
||||
> opencode — and receive the same brief independently. Agreement is evidence, not correlated echo
|
||||
> from one provider or one conversation.
|
||||
|
||||
> **Historical alternative retained.** The text and figures below record the considered model and why it
|
||||
> was rejected: it re-rooted the human-facing session above the lead, violating the still-true premise
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by fleetd.
|
||||
|
||||
The orchestrator is **`SessionManager` recursed one tier up**: today it spawns/names/reaps *worker*
|
||||
sessions; the orchestrator does the same for *main* sessions, and adds **context scoping**.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
human["human"]
|
||||
subgraph t0["TIER 0 — orchestrator (new top MCP client)"]
|
||||
osm["OrchestratorSessionManager<br/>= SessionManager pattern"]
|
||||
idm["session identity<br/>name · resume · idle-ttl (CB-303)"]
|
||||
ctx["context scoper<br/>(CB-303 context-cap + turn_id)"]
|
||||
end
|
||||
subgraph t1["TIER 1 — mains (now MANAGED sessions)"]
|
||||
m1["main A"]
|
||||
m2["main B"]
|
||||
end
|
||||
subgraph t2["TIER 2 — workers"]
|
||||
w["sandboxed workers"]
|
||||
end
|
||||
human --> osm
|
||||
osm --> idm
|
||||
osm --> ctx
|
||||
idm -->|"spawn / name / resume"| m1
|
||||
idm -->|"spawn / name / resume"| m2
|
||||
ctx -->|"inject exact context slice"| m1
|
||||
m1 -->|"scoped delegation"| w
|
||||
m2 -->|"scoped delegation"| w
|
||||
```
|
||||
|
||||
*Figure 7 — **SUPERSEDED historical alternative.** The recursion re-rooted the human-facing session:
|
||||
the human drove an orchestrator and the mains became managed, resumable sessions. The operator rejected
|
||||
that re-root. The replacement keeps the human-driven, pre-existing lead and engages architects sideways
|
||||
as short-lived advisory peers; see Gitea issue #16 and `architects:`.*
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant H as human
|
||||
participant O as orchestrator
|
||||
participant MA as main A
|
||||
participant W as worker
|
||||
H->>O: high-level goal (large context)
|
||||
O->>O: name/resume main A session
|
||||
O->>MA: task + SCOPED context slice (not the whole history)
|
||||
MA->>W: fleet_send(delegation, carrying only the relevant slice)
|
||||
W-->>MA: result
|
||||
MA-->>O: rollup
|
||||
O->>O: fold into orchestrator context, pick next main/turn
|
||||
```
|
||||
|
||||
*Figure 8 — **SUPERSEDED historical alternative.** This proposed an orchestrator holding global context
|
||||
and slicing it for managed mains. The replacement has the human-driven lead send the same advisory brief
|
||||
issue #16 and `architects:`.*
|
||||
|
||||
**Deltas:** a second, higher `SessionManager` instance whose "peers" are mains; the orchestrator
|
||||
becomes the top MCP client; context-slice selection (new) layered on CB-303's `context_cap` +
|
||||
`turn_id` scoping; mains gain a resumable session id in the federated roster.
|
||||
|
||||
## 7. The identity boundary — the one clause to hold
|
||||
|
||||
The bus is a **communication fabric, not an env/toolchain manager**. This direction is compatible
|
||||
**only** with the ownership split below; the amber line in Figure 3 is where it must hold.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph ok["STAYS A BUS (owned)"]
|
||||
a["launch INTO a sandbox<br/>(opaque image/role reference)"]
|
||||
b["inject + guard baseUrl,<br/>mount MCP + charter"]
|
||||
c["session identity + context scope<br/>(name/resume/turn_id)"]
|
||||
d["per-agent addressing + roster"]
|
||||
end
|
||||
subgraph drift["BECOMES ENV-MANAGER (forbidden)"]
|
||||
e["build images / install IDE<br/>or dev-tools"]
|
||||
f["wire the bridge's OWN IDE MCP<br/>into a worker"]
|
||||
g["enumerate 'what a frontend<br/>agent needs'"]
|
||||
end
|
||||
ok -.->|"red flag: any feature that only<br/>makes sense for ONE kind of peer"| drift
|
||||
classDef bad fill:#9b2c2c,stroke:#742a2a,color:#ffffff;
|
||||
class e,f,g bad
|
||||
```
|
||||
|
||||
*Figure 9 — the guardrail. A worker having a headless IDE **inside its own sandbox** is the peer
|
||||
owning its toolchain (left) — the opposite of the bridge reaching into the peer (right). Sandbox
|
||||
specs are peer-owned references (like `argv`/image id); the instant the bridge builds or installs
|
||||
them, it has drifted. This resolves the apparent contradiction between "do NOT give workers IDE MCP
|
||||
access" and "give workers a sandboxed IDE" — different owners.*
|
||||
|
||||
## 8. Staging & dependencies
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
cb402["CB-401/402<br/>Peer Launcher SPI + composite<br/>(DONE / in-flight)"]
|
||||
A["A · SandboxLauncher<br/>(placement-neutral, independent)"]
|
||||
cb308["CB-308 substrate<br/>per-agent channels + global id<br/>+ federated roster"]
|
||||
B["B · main-agent pair (SUPERSEDED)<br/>(multi-slot PrimaryRegistry)"]
|
||||
C["C · orchestrator tier<br/>(SessionManager recursed up)"]
|
||||
cb402 --> A
|
||||
cb402 --> cb308
|
||||
cb308 --> B
|
||||
cb308 --> C
|
||||
B --> C
|
||||
A -.->|"if sandbox = separate host/container,<br/>reuses CB-308 global id"| cb308
|
||||
classDef gate fill:#b7791f,stroke:#7b341e,color:#ffffff;
|
||||
class cb308 gate
|
||||
```
|
||||
|
||||
*Figure 10 — the substrate (amber) is the shared enabler for B and C. Recommended order:*
|
||||
|
||||
1. **Finish CB-402** (merge the opencode adapter branch).
|
||||
2. **A · SandboxLauncher** — independent; a second proof of the SPI (placement-neutral). Ships anytime.
|
||||
3. **CB-308 substrate** — per-agent channels + global id + federated roster (the multi-host work,
|
||||
promoted from host-to-host to tier-to-tier).
|
||||
4. **B · main-agent pair** — **SUPERSEDED** by lead + two advisory architects.
|
||||
5. **C · orchestrator tier** — the capstone; the recursive session manager + context scoping.
|
||||
|
||||
## 9. Open questions (to resolve at ticket-split)
|
||||
|
||||
- **Sandbox mechanism:** container (`docker exec`) vs devcontainer — how the role→image mapping is
|
||||
expressed on the profile. *(Topology **resolved** in §11: distributed = gateway-per-host × local
|
||||
sandboxes; the remaining choice is only the local launch mechanism, not the shape.)*
|
||||
- **Pair semantics:** **SUPERSEDED.** The two-main question is replaced by the architect role's
|
||||
least-privilege boundary: advisory architects can send/reply/ask/read but cannot spawn/stop/drain.
|
||||
- **Orchestrator drivenness:** the mains become programmatically spawned/resumed — does the human
|
||||
still ever type directly into a main, or only into the orchestrator? (The re-root caveat, Fig 7.)
|
||||
- **Context-slice selection:** who decides the slice — orchestrator heuristics, explicit tool args,
|
||||
or the main pulling on demand? This is the genuinely new responsibility; keep it *scoping*, not
|
||||
content authorship, to stay inside the boundary.
|
||||
- **Trust:** every new tier boundary that accepts spawn/send is a trust edge (the CB-308 item #5 /
|
||||
CB-401 Stage-C concern) — orchestrator→main and main→sandbox both need authz.
|
||||
|
||||
## 10. Ticket-split guidance (deferred)
|
||||
|
||||
This note is deliberately one arc; when split, the natural tickets are **A** (SandboxLauncher +
|
||||
role/spawn-target), **the CB-308 substrate** (likely already its own ticket), **B** (multi-primary
|
||||
pair), and **C** (orchestrator tier + context scoping) — with the identity clause (§7) as an
|
||||
acceptance criterion on **A** specifically. Sequence per §8; nothing here is a new pillar, so each
|
||||
ticket is an extension of an existing pattern (CB-402 for A, CB-308 for B/C).
|
||||
|
||||
## 11. Distributed sandboxes — the resolved topology
|
||||
|
||||
The follow-up question — *"clarify the architecture when we have distributed agents in sandboxes"* —
|
||||
resolves the fork left open in §4 and §9. **Decision: Development A (sandbox launcher) and CB-308
|
||||
(per-host federation) *compose*, not compete — each host runs a `fleetd` gateway whose launcher
|
||||
spawns agents into that host's *local* sandboxes.** A sandbox is never reached across the network; it
|
||||
is reached by the gateway sitting next to it.
|
||||
|
||||
### 11.1 The one fact that fixes the shape
|
||||
|
||||
The bus delivers a turn by **herdr keystroke-injection** — `Injector → AgentControl.send` writes into
|
||||
a PTY that its **local** herdr owns. The broker moves *messages and presence*, **never keystrokes**.
|
||||
So an agent's PTY must live in a herdr that *some* `fleetd` instance drives locally: a remote
|
||||
container with no local herdr **cannot be injected into**. That rules out a central daemon reaching
|
||||
remote PTYs, and collapses the design to a single identity:
|
||||
|
||||
> **"a sandboxed agent on another host" ≡ "a sandbox spawned by that host's gateway."**
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A — gateway"]
|
||||
mA["main / orchestrator<br/>MCP client → LOCAL gateway"]
|
||||
gA["fleetd A<br/>herdr + CompositePeerLauncher<br/>(incl. SandboxLauncher)"]
|
||||
cBEa["sandbox: backend<br/>(local container)"]
|
||||
cFEa["sandbox: frontend<br/>(local container)"]
|
||||
mA --- gA
|
||||
gA -->|"spawn (docker/devcontainer)<br/>→ PTY in A's herdr"| cBEa
|
||||
gA --> cFEa
|
||||
end
|
||||
subgraph broker["BROKER (AMQP) — CB-307/308 fabric"]
|
||||
inbox["agent.ID.inbox queues"]
|
||||
roster["roster.* (federated presence)"]
|
||||
end
|
||||
subgraph hostB["HOST B — gateway"]
|
||||
gB["fleetd B<br/>herdr + SandboxLauncher"]
|
||||
cBEb["sandbox: backend<br/>(local container)"]
|
||||
gB -->|"spawn → PTY in B's herdr"| cBEb
|
||||
end
|
||||
gA <-->|"messages + presence<br/>(NOT keystrokes)"| inbox
|
||||
gB <-->|"messages + presence"| inbox
|
||||
gA --- roster
|
||||
gB --- roster
|
||||
classDef line fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
class inbox,roster line
|
||||
```
|
||||
|
||||
*Figure 11 — the composed topology. Each gateway owns its local herdr and runs a `SandboxLauncher`
|
||||
(the §4 adapter) that spawns role-specific containers **on its own host**; the broker (blue) carries
|
||||
only messages + roster between gateways. Keystroke-injection stays strictly local to each gateway.*
|
||||
|
||||
### 11.2 How a delegation reaches a sandboxed agent on another host
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MA as main (host A)
|
||||
participant GA as gateway A
|
||||
participant BR as broker
|
||||
participant GB as gateway B
|
||||
participant SB as sandbox agent (host B, container)
|
||||
MA->>GA: fleet_send(globalId on B, msg)
|
||||
GA->>GA: directory lookup - is globalId local? NO
|
||||
GA->>BR: publish agent.ID.inbox (durable)
|
||||
BR->>GB: route to the owning gateway
|
||||
GB->>SB: inject via B's LOCAL herdr (keystrokes)
|
||||
Note over GB,SB: SandboxLauncher already spawned the container -<br/>its PTY is in B's herdr, CB-306 readiness passed
|
||||
SB-->>GB: fleet_reply (to B's LOCAL MCP endpoint)
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route back to A
|
||||
Note over GA: held until the main pulls (the main is a client)
|
||||
MA->>GA: poll / blocking send resolves
|
||||
GA-->>MA: reply
|
||||
```
|
||||
|
||||
*Figure 12 — the `local ? inject : publish` fork (CB-308 §3.2) with a sandboxed far side. Only the
|
||||
**middle** hop crosses the network via the broker; **both** injection points (into the sandbox on B,
|
||||
and the drain-nudge back into the main on A) are local herdr writes. This is CB-308's routing rule
|
||||
unchanged — the sandbox is transparent to it.*
|
||||
|
||||
### 11.3 Two reachability changes any sandbox forces
|
||||
|
||||
| Change | Today | Under sandboxes |
|
||||
|---|---|---|
|
||||
| **`mcpUrl`** | `http://127.0.0.1:8765/mcp` (loopback) | must be **host-routable from inside the container** (e.g. `host.docker.internal` or the gateway's LAN IP) — the worker connects to **its own gateway's** MCP, never a remote one. |
|
||||
| **PTY ownership** | pane in the daemon's herdr | pane is the **container's** attached PTY, in the **local** gateway's herdr (via `docker exec`/devcontainer) — non-negotiable per §11.1. |
|
||||
|
||||
### 11.4 Everything maps to an existing seam (nothing new invented)
|
||||
|
||||
| Concern | Provided by |
|
||||
|---|---|
|
||||
| Per-host gateway (owns local herdr + sessions) | **CB-308** (today's `fleetd`, evolved) |
|
||||
| Spawn into a local sandbox / role→image | **Development A** `SandboxLauncher` (§4), routed by `CompositePeerLauncher` |
|
||||
| Addressing a remote sandboxed agent | **CB-308** global id + federated roster (host + role as metadata) |
|
||||
| Orphan reap after a gateway restart | **CB-117** per-gateway, summed by the composite — each reaps only its **local** herdr |
|
||||
| Container up/down | tied to **CB-303** session lifecycle — `SandboxLauncher.stop` tears the container down with the pane |
|
||||
| Cross-gateway spawn/send trust | **CB-308 item #5** / CB-401 Stage-C — each gateway edge is a trust boundary |
|
||||
|
||||
*The net: distributed sandboxes add **zero** new pillars — they are `CB-308 gateway × Development-A
|
||||
launcher` at every host, with the §7 ownership line (peer owns the image; the bridge only launches
|
||||
into it) holding at each gateway.*
|
||||
@@ -0,0 +1,467 @@
|
||||
# CB-591 — move the fleet onto the LLM and MCP gateway
|
||||
|
||||
**Status: DONE — the fleet is on the gateway as of 2026-08-15.** `local` runs on `/anthropic` and
|
||||
`gx` on `/v1`, both at `weight: 100`; `local-direct` stays at `weight: 0` as the escape hatch. Getting
|
||||
here took a revert and two upstream fixes — see §7.1, which is the useful part of this document. One
|
||||
risk is **accepted rather than solved**: a stream cut by any mid-response timer arrives as HTTP 200
|
||||
with no terminator, and our third-party members cannot detect it (§7.2).
|
||||
· **Upstream:** [systems/vms wiki → LLM and MCP Gateway](https://git.ltms.dev/systems/vms/wiki/LLM-and-MCP-Gateway)
|
||||
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
||||
|
||||
The gateway went live on 2026-08-15 and replaced Bifrost. This plan says what that means for a
|
||||
**member definition** in `fleetd.yaml`, because that is the part of this repo the change actually
|
||||
touches.
|
||||
|
||||
---
|
||||
|
||||
## 1. What changed upstream
|
||||
|
||||
One front door for every LLM and MCP client: `https://llm.ltms.dev`, one token per consumer.
|
||||
|
||||
| Surface | URL |
|
||||
|---|---|
|
||||
| OpenAI chat | `https://llm.ltms.dev/v1/chat/completions` |
|
||||
| OpenAI models | `https://llm.ltms.dev/v1/models` |
|
||||
| **Anthropic messages** | `https://llm.ltms.dev/anthropic/v1/messages` |
|
||||
| MCP, all servers multiplexed | `https://llm.ltms.dev/mcp` |
|
||||
|
||||
Anything outside that list returns **404 before any token is checked**, on purpose — the gateway must
|
||||
never become a blanket proxy.
|
||||
|
||||
The model backend is unchanged: GX10 vLLM at `10.10.10.26:8000` (`gx00.gw`), model name exactly
|
||||
`deepseek-v4-flash`. The direct LAN path stays open on purpose as an escape hatch.
|
||||
|
||||
---
|
||||
|
||||
## 2. Where claude-bridge sits today
|
||||
|
||||
We do **not** use the gateway. The `local` profile talks straight to the vLLM:
|
||||
|
||||
```yaml
|
||||
local:
|
||||
kind: claude-code
|
||||
baseUrl: http://gx00.gw:8000 # direct vLLM — no auth, LAN only
|
||||
model: deepseek-v4-flash
|
||||
configDir: /Users/dai.ha/.ccs/instances/gx10
|
||||
```
|
||||
|
||||
Three facts about our side that decide the shape of this work:
|
||||
|
||||
1. **`baseUrl` becomes `ANTHROPIC_BASE_URL`** in the member's environment, and `tokenEnv` becomes
|
||||
`ANTHROPIC_AUTH_TOKEN` (the value is read from a host env var and never stored in config).
|
||||
`local` sets no `tokenEnv` today, because a direct vLLM needs no token.
|
||||
2. **`SubscriptionGuard` refuses any host not on an allowlist**, and that allowlist is
|
||||
`guard.offSubscriptionHosts: [gx00.gw]`. It is built once in `Fleetd.java:93` and handed to the
|
||||
launcher, so **it is a restart-required key**, not a hot one. Changing `baseUrl` without changing
|
||||
this makes every `local` spawn throw.
|
||||
3. **The wiki names us as a blocker.** Under *Not done yet*: retiring the shared `legacy` token is
|
||||
blocked because "kb, brain, **claude-bridge** and the workstation still share it. Each needs its
|
||||
own consumer first."
|
||||
|
||||
Context7 is mounted twice today, both times straight at `https://ct7.ltms.dev/mcp` — once in
|
||||
`.mcp.json` (the primary) and once in `opencode.json` (the `sol` and `terra` members).
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
subgraph now["Today"]
|
||||
M1["local member<br/>claude-code"] -->|"ANTHROPIC_BASE_URL"| V1["vLLM gx00.gw:8000<br/>no auth, LAN only"]
|
||||
M2["sol / terra<br/>opencode"] --> CT1["ct7.ltms.dev/mcp"]
|
||||
P1["primary"] --> CT1
|
||||
end
|
||||
subgraph after["Proposed"]
|
||||
M3["local member"] -->|"ANTHROPIC_BASE_URL<br/>+ ANTHROPIC_AUTH_TOKEN"| G["llm.ltms.dev/anthropic<br/>consumer: claude-bridge"]
|
||||
G --> V2["vLLM gx00.gw:8000"]
|
||||
M4["local-direct<br/>weight 0, escape hatch"] --> V2
|
||||
end
|
||||
```
|
||||
|
||||
*The member definition is the only thing that moves. The model behind it does not.*
|
||||
|
||||
---
|
||||
|
||||
## 3. The member definition change
|
||||
|
||||
The gateway serves an Anthropic surface *and* an OpenAI surface, so **both member kinds can point at
|
||||
it**. That is the main opportunity here, and it is bigger than the `local` profile alone.
|
||||
|
||||
### 3a. `local` — claude-code, on `/anthropic`
|
||||
|
||||
| Key | Today | After | Note |
|
||||
|---|---|---|---|
|
||||
| `baseUrl` | `http://gx00.gw:8000` | `https://llm.ltms.dev/anthropic` | see the schema warning below |
|
||||
| `tokenEnv` | *(unset)* | `AI_GATEWAY_TOKEN` | new consumer token, `llmk-claude-bridge-<32 hex>` |
|
||||
| `model` | `deepseek-v4-flash` | unchanged | must stay **exact**; a regex match returns an empty `/v1/models` while completions keep working |
|
||||
| `guard.offSubscriptionHosts` | `[gx00.gw]` | `[gx00.gw, llm.ltms.dev]` | **restart required** |
|
||||
|
||||
### 3b. A new opencode profile on `/v1` — no code needed
|
||||
|
||||
`OpenCodeLauncher` already supports a pinned OpenAI-compatible endpoint (CB-508). Given `baseUrl` it
|
||||
writes a custom provider block into the worker's opencode config:
|
||||
|
||||
- `baseUrl` → `options.baseURL`. `openAiBaseUrl` appends `/v1` to a bare host, and takes a URL that
|
||||
already has a path **as-is** — so `https://llm.ltms.dev/v1` works unchanged.
|
||||
- `tokenEnv` → `options.apiKey` (falls back to a placeholder when unset, since a local vLLM ignores it).
|
||||
- `model:` **must** be `<provider>/<model>` when `baseUrl` is set — a bare name is rejected loudly
|
||||
rather than silently falling back to opencode's default gateway.
|
||||
|
||||
So the profile is pure config:
|
||||
|
||||
```yaml
|
||||
gx:
|
||||
kind: opencode
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
model: gx/deepseek-v4-flash # provider id is ours to choose; the half after / is the model
|
||||
argv: ["opencode"]
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
weight: 100 # same tier as `local` — free
|
||||
maxLoad: 2
|
||||
# deliberately NO credentialId — this is our own box, not the shared OpenAI account
|
||||
```
|
||||
|
||||
**Why this matters more than it looks.** Today every opencode member is `sol` or `terra`, and those
|
||||
are two models on **one** OpenAI account sharing `credentialId: openai-shared` — so an exhaustion on
|
||||
either locks out both, and half the fleet's opencode capacity dies at once. A gateway-backed opencode
|
||||
profile is free, is not on that credential, and therefore is not in that quarantine pair. It removes
|
||||
a single point of failure rather than just adding capacity.
|
||||
|
||||
**Note the asymmetry, it is deliberate:** `SubscriptionGuard` does not apply to opencode at all — the
|
||||
guard exists to stop a *Claude* worker borrowing the operator's subscription, and opencode reads its
|
||||
own provider credentials. So 3b needs **no allowlist change**; only 3a does.
|
||||
|
||||
**Both still need a restart, for a different reason.** `tokenEnv` is resolved by
|
||||
`HerdrPeerLauncher.resolveEnv` → `env.apply(name)`, which reads the **daemon's own process
|
||||
environment**. The running `fleetd` inherited its environment when it started, so a variable added to
|
||||
`secrets.sh` afterwards is simply not there — the launcher would inject an empty token and the
|
||||
gateway would answer 401. This is the same failure as trap 1 in `scripts/redeploy-fleetd.sh`
|
||||
(`WORKER_GITEA_TOKEN`), and it has the same fix: **restart from a login shell**, and use
|
||||
`scripts/redeploy-fleetd.sh --check` to confirm the name resolves before restarting anything.
|
||||
|
||||
### 3c. What this does to `ccs`
|
||||
|
||||
Once a profile carries `baseUrl`, `tokenEnv` and `model` itself, the ccs instance stops being what
|
||||
routes a member. Be precise about what is left, though: `configDir` still supplies **folder trust**
|
||||
and `settings.json`, and dropping it is what produced the trust dialog and the wrong-model error
|
||||
recorded in `fleetd.yaml`. So ccs goes from *deciding where the tokens go* to *holding client-side
|
||||
state*. Less load-bearing, not removable.
|
||||
|
||||
### Why `/anthropic` and never `/v1/chat/completions`
|
||||
|
||||
The gateway declares its Anthropic backend as `schema.name: Anthropic`, which means **no
|
||||
translation** — streaming, tool use and thinking blocks pass through exactly as they do against vLLM
|
||||
directly.
|
||||
|
||||
Declared as `OpenAI`, Envoy's translator looks for a `thinking_blocks` field that our vLLM does not
|
||||
send (it sends `reasoning_content`), and **every thinking delta disappears silently**. Claude Code
|
||||
speaks the Anthropic protocol, so `/anthropic` is both correct and the only safe choice.
|
||||
|
||||
This is the exact failure shape this repo keeps hitting: it compiles, it answers, it looks healthy,
|
||||
and a capability is quietly off. Treat it as a `silent-default` risk, not a config preference.
|
||||
|
||||
**Open question for 3b — ANSWERED, 2026-08-15.** The worry was that the OpenAI surface might drop
|
||||
reasoning the way the wiki documents for a mis-declared Anthropic backend. It does not. Checked at
|
||||
the API before any profile was switched:
|
||||
|
||||
| surface | request | result |
|
||||
|---|---|---|
|
||||
| `/anthropic/v1/messages` | `deepseek-v4-flash`, 64 tokens | 200, response carries a real `"type":"thinking"` block |
|
||||
| `/v1/chat/completions` | same | 200, message carries a populated `reasoning_content` (and a `reasoning` field) |
|
||||
| `/v1/models` | — | 200, exactly `["deepseek-v4-flash"]` — the exact-name trap is clear |
|
||||
| `/v1/models`, **no token** | — | **401** — Caddy is gating, as designed |
|
||||
|
||||
So reasoning survives on **both** surfaces, and the `/anthropic` choice for `local` is about protocol
|
||||
correctness rather than a repair for a known loss. The last row matters on its own: the wiki warns
|
||||
the gateway's own `SecurityPolicy` fails open, so it is worth knowing the proxy in front really does
|
||||
refuse an unauthenticated request here.
|
||||
|
||||
---
|
||||
|
||||
## 4. Decisions
|
||||
|
||||
### D1 — switch, but keep the direct path as an explicit profile · **recommended**
|
||||
|
||||
Switching buys four things we do not have:
|
||||
|
||||
- **Free opencode capacity, off the shared credential.** The largest single win. See §3b — it retires
|
||||
a real single point of failure, not just a cost line.
|
||||
- **Per-consumer usage figures.** The cockpit counts requests per consumer. That is the first real
|
||||
measurement of what the fleet consumes, and it feeds [CB-589](https://git.ltms.dev/fleet/fleetd/issues/74) Gap 2 directly.
|
||||
- **Our own revocable token.** One consumer to revoke if a worker ever leaks it, instead of a shared
|
||||
`legacy` token used by four systems.
|
||||
- **It works off-LAN.** `gx00.gw` resolves on the LAN only.
|
||||
|
||||
The cost is honest and worth stating: we add a TLS edge, an auth proxy and a gateway to the path of
|
||||
every member spawn. The wiki keeps the direct route open precisely because "if the gateway breaks,
|
||||
nothing that matters is blocked."
|
||||
|
||||
So keep it. Add a second profile `local-direct` pointing at `http://gx00.gw:8000` with **`weight: 0`**
|
||||
— never auto-selected, still spawnable with an explicit `fleet_spawn{profile: "local-direct"}`.
|
||||
That is exactly what CB-554 made `weight: 0` mean, and it turns the escape hatch into something the
|
||||
lead can actually reach during an incident.
|
||||
|
||||
### D2 — do members also mount the gateway's `/mcp`? · **OPEN, operator's call**
|
||||
|
||||
Not a detail. `CLAUDE.md` states in two places that a member mounts **only** the bridge MCP, and a
|
||||
worker's honesty rule leans on it ("never claim the result of a check you had no way to run").
|
||||
|
||||
- **Keep bridge-only.** The invariant stays true and simple. Workers stay cheap and narrow.
|
||||
- **Add the gateway MCP.** Implementers get context7 documentation lookups, which is genuinely useful
|
||||
for library work. But `mcpUrl` in `FleetConfig.Profile` is a **single `String`**, so a
|
||||
claude-code member can mount exactly one MCP — this needs a code change, not a config edit.
|
||||
|
||||
Note the invariant is **already inaccurate**: `opencode.json` gives `sol` and `terra` both context7
|
||||
and gitea. So the choice is really "make the rule true" or "make the rule match reality". Either is
|
||||
defensible; picking one is not mine to do.
|
||||
|
||||
### D3 — token scope
|
||||
|
||||
One consumer, `claude-bridge`, its token in `${SHARED_ENV}/tools/secrets.sh` as `AI_GATEWAY_TOKEN`,
|
||||
referenced by name only. Never the literal value in `fleetd.yaml` — `tokenEnv` exists for this.
|
||||
|
||||
---
|
||||
|
||||
## 5. Units of work
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
U1["U1 · consumer token<br/>issue via cockpit, add to secrets.sh"]
|
||||
U2["U2 · profile + guard<br/>fleetd.yaml, restart"]
|
||||
U3["U3 · verify live<br/>spawn, prove thinking survives"]
|
||||
U4["U4 · context7 via gateway<br/>.mcp.json + opencode.json"]
|
||||
U5["U5 · docs<br/>CLAUDE.md, wiki 11-Features"]
|
||||
U1 --> U2 --> U3
|
||||
U4 --> U5
|
||||
U3 --> U5
|
||||
```
|
||||
|
||||
| # | Scope | Who | Why |
|
||||
|---|---|---|---|
|
||||
| U1 | Issue the `claude-bridge` consumer at `auth.ltms.dev`; store as `AI_GATEWAY_TOKEN` | **operator** | touches secrets and a host we do not own |
|
||||
| U2a | New `gx` opencode profile on `/v1` — pure config, no guard change | **lead** | `fleetd.yaml` is gitignored, so a worker cannot see or edit it |
|
||||
| U2b | `local` → `/anthropic`; add `local-direct` weight 0; add `llm.ltms.dev` to the guard allowlist | **lead** | same |
|
||||
| U2c | One restart from a **login shell**, after U2a and U2b | **lead** | picks up `AI_GATEWAY_TOKEN` into the daemon env *and* the guard allowlist, in one stop |
|
||||
| U3 | Live spawn on both new profiles; confirm reasoning survives on each surface | **lead** | needs real spawns and the running daemon |
|
||||
| U4 | Point `.mcp.json` and `opencode.json` context7 at the gateway `/mcp`; rename pinned tools | delegatable | tracked files, self-contained |
|
||||
| U5 | Fix the "members mount only the bridge" claim; add a `wiki/11-Features.md` entry | delegatable | writing, clear criteria |
|
||||
|
||||
U1 blocks U2a, U2b and U3. U4 and U5 do not depend on it.
|
||||
|
||||
**Write U2a and U2b, then restart once (U2c), then verify `gx` before `local`.** Since both profiles
|
||||
need the same restart there is no reason to do two, but there is still a reason to *verify* in order:
|
||||
`gx` exercises the token and the gateway with no guard involved, so if it fails the cause is upstream.
|
||||
`local` adds the guard allowlist on top, so a failure there points at our config instead. Testing them
|
||||
in that order separates the two causes instead of confusing them.
|
||||
|
||||
> **U1 status, 2026-08-15:** the operator issued the consumer and exported it as `AI_GATEWAY_TOKEN`
|
||||
> (one key for every agent and MCP client behind `llm.ltms.dev`). Confirmed: it resolves in a login
|
||||
> shell, is 48 characters and carries the documented `llmk-` prefix. The value was never printed.
|
||||
|
||||
---
|
||||
|
||||
## 6. Traps carried over from the wiki
|
||||
|
||||
Each of these cost someone real debugging time upstream. They apply to us.
|
||||
|
||||
1. **Rotating a token restarts the auth proxy, which drops in-flight streaming responses.** For us
|
||||
that means rotating `AI_GATEWAY_TOKEN` kills every live member mid-turn, and an async ticket's
|
||||
report goes with it. This is the same rule as a daemon redeploy: **drain the fleet first**
|
||||
(`fleet_list` → `fleet_poll` anything wanted → `fleet_stop`), then rotate.
|
||||
2. **The gateway's own `SecurityPolicy` fails open.** Standalone `aigw run` accepts it and silently
|
||||
ignores it — an unauthenticated request returned **200**. Auth is the Caddy proxy in front, and
|
||||
nothing else. Never reason as if the gateway authenticates.
|
||||
3. **Exact model name.** A regex match routes fine but returns an **empty** `/v1/models` list while
|
||||
completions keep working. A wrong name returns a bare 404 that reads exactly like a dead gateway.
|
||||
4. **MCP tool names changed prefix separator.** Bifrost used one dash (`ct7-resolve-library-id`); the
|
||||
gateway uses **two underscores** (`ct7__resolve-library-id`). Relevant only if U4 is done.
|
||||
5. **`/v1/models` 404 vs empty list are different faults.** 404 means no route loaded at all; empty
|
||||
means the model match is a regex. Do not conflate them when diagnosing.
|
||||
|
||||
---
|
||||
|
||||
## 7. Verification — what would prove this works
|
||||
|
||||
Merging config is not proving it. The checks, in order:
|
||||
|
||||
1. `fleet_spawn{profile: "gx"}` succeeds and the member completes a real turn ending in
|
||||
`fleet_reply`. This is the first proof of the token, the URL and the model name, and it risks
|
||||
nothing the fleet depends on.
|
||||
2. `fleet_spawn{profile: "local"}` succeeds. If the guard allowlist was missed, this **throws** — a
|
||||
loud, self-correcting failure, which is the good kind. If the restart was missed, it also throws,
|
||||
for the same reason.
|
||||
3. A `local` member completes a turn. That exercises streaming through two TLS edges, the auth proxy
|
||||
and the gateway.
|
||||
4. **Reasoning survives, checked separately on each surface.** For `local` on `/anthropic` this is
|
||||
the check that catches the `/v1` versus `/anthropic` mistake, and it is the only one that does —
|
||||
nothing else distinguishes a working passthrough from a translator quietly dropping thinking
|
||||
deltas. For `gx` on `/v1`, this answers the open question in §3 rather than assuming it.
|
||||
5. The cockpit at `auth.ltms.dev` shows requests counted against the `claude-bridge` consumer, not
|
||||
`legacy`. That is the whole point of taking our own token.
|
||||
6. `fleet_spawn{profile: "local-direct"}` still works, so the escape hatch is real rather than
|
||||
theoretical.
|
||||
7. `fleet_list` shows `gx` carrying no `credentialId`, so a `sol`/`terra` exhaustion cannot
|
||||
quarantine it. This is the single-point-of-failure claim in §3b, checked rather than asserted.
|
||||
|
||||
---
|
||||
|
||||
## 7.1 What the live run actually found — 2026-08-15
|
||||
|
||||
U1–U2c were done, the daemon restarted onto them, and both new profiles were spawned for real. The
|
||||
migration was then **reverted**. This section is the result, so none of it has to be re-derived.
|
||||
|
||||
### The blocker
|
||||
|
||||
`llm.ltms.dev` answers **HTTP 413 Request Entity Too Large** above **32 KiB (32768 bytes)**, on both
|
||||
surfaces:
|
||||
|
||||
```
|
||||
/v1 32695 bytes -> 200 /anthropic 32095 bytes -> 200
|
||||
/v1 32795 bytes -> 413 /anthropic 32855 bytes -> 413
|
||||
```
|
||||
|
||||
32 KiB is far below one real agent turn.
|
||||
|
||||
**Root cause — confirmed by the systems/vms side, 2026-08-15.** My guess that it was a Caddy
|
||||
`request_body max_size` was **wrong**. It is Envoy, inside `aigw` on `llm.vm`. Envoy Gateway defaults
|
||||
a listener's `per_connection_buffer_limit_bytes` to **32768**, and the AI Gateway buffers the *whole*
|
||||
request body before it can route on the model name — so that default is not a network tuning knob
|
||||
here, it is a hard ceiling on prompt size. Read out of the live Envoy `config_dump`:
|
||||
|
||||
```
|
||||
listener default/llm/http per_connection_buffer_limit_bytes: 32768
|
||||
```
|
||||
|
||||
Nobody chose 32 KiB; it was inherited from the default. Both TLS edges are innocent: the same
|
||||
boundary reproduces on the LAN path and the internet path, and both 413s carry an `x-llm-consumer`
|
||||
header their auth proxy sets only *after* authenticating — so the body cleared both edges and the
|
||||
auth. Directly on `llm.vm`, `aigw` 413s at 39 KB while the vLLM backend accepts the same 39 KB and
|
||||
answers 200.
|
||||
|
||||
**Do not plan around 32 KiB.** The intended ceiling is far higher. Their fix — a `ClientTrafficPolicy`
|
||||
setting `bufferLimit: 8Mi` — is written but **not deployed** as of this note, pending their operator's
|
||||
approval. I have not re-tested and will not until they confirm, so as not to measure a half-changed
|
||||
system. Fixed in **systems/vms**, not here.
|
||||
|
||||
### The part worth remembering
|
||||
|
||||
Two members were spawned at the same moment with the same message:
|
||||
|
||||
| | `local` (claude-code, `/anthropic`) | `gx` (opencode, `/v1`) |
|
||||
|---|---|---|
|
||||
| READY → BUSY | 19:07:26 | 19:07:45 |
|
||||
| BUSY → DONE | **19:08:51 (66s)** | **never — 10+ min, ticket FAILED** |
|
||||
|
||||
**`local` passed.** It passed only because the probe was three trivial questions in a fresh session,
|
||||
so the request fit under 32 KiB. The profile looked healthy and was a landmine set to fire on the
|
||||
first turn that reads a file.
|
||||
|
||||
So §7's checklist was not wrong, it was **too easy**. Any future run of it must use a task that reads
|
||||
a real file. A liveness probe proves the token and the URL; it does not prove the path.
|
||||
|
||||
`gx` did not fail loudly either. Reproduced outside the bridge by running `opencode` by hand with the
|
||||
launcher's own generated config:
|
||||
|
||||
```
|
||||
Error: Request Entity Too Large
|
||||
...compacts context, retries...
|
||||
Error: Request Entity Too Large
|
||||
```
|
||||
|
||||
opencode **catches the 413, compacts, and retries — indefinitely**. A member that fails loudly costs
|
||||
one turn; this one costs the whole task and is indistinguishable from a slow worker.
|
||||
|
||||
> **Diagnosing a stuck opencode member.** Do not read its pane. The launcher writes its config to a
|
||||
> temp dir and passes it as `OPENCODE_CONFIG` — find it with
|
||||
> `ls -dt /var/folders/*/*/T/fleetd-opencode-* | head -1`, check the provider block and the key's
|
||||
> length and prefix (never its value), then reproduce with `opencode run --auto -m <provider>/<model>`
|
||||
> using the same `OPENCODE_CONFIG`. That is what turned "it hangs" into a one-line error.
|
||||
|
||||
### What checked out, and needs no re-testing
|
||||
|
||||
- Token accepted on both surfaces. **Unauthenticated → 401**, so the Caddy proxy really does gate —
|
||||
the wiki's "SecurityPolicy fails open" warning is about the gateway itself, not the edge.
|
||||
- `/v1/models` returns exactly `["deepseek-v4-flash"]`, so trap 3 is clear.
|
||||
- **Reasoning survives both surfaces** — see §3b above.
|
||||
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
||||
key rather than the `fleetd-local-noauth` placeholder.
|
||||
- `SubscriptionGuard` accepted `llm.ltms.dev` after the allowlist edit and the restart: `local`
|
||||
spawned without throwing, which is the check that catches a missed restart.
|
||||
|
||||
### Resolution — both ceilings fixed, migration completed
|
||||
|
||||
systems/vms fixed both, and each was re-checked from this side rather than taken on trust:
|
||||
|
||||
| ceiling | was | now | our own check |
|
||||
|---|---|---|---|
|
||||
| listener buffer | 32 KiB | 32 Mi | 1.2 MB body → **200** (was 413) |
|
||||
| LLM route timeout | 60s | 86400s | the request that truncated: **101s, `message_stop` present, 4000/4000** |
|
||||
|
||||
The timeout moved in two steps on 2026-08-15: 60s → 1800s, then 1800s → **86400s (24 hours)** after
|
||||
the truncation risk below was discussed. They tried `request: 0s` first, which removes the
|
||||
total-duration timer completely. It works, but on an `AIGatewayRoute` the **idle timeout is derived
|
||||
from the request timeout**, so `0s` also removed any bound on a stalled connection. 86400s keeps a
|
||||
reaper for dead connections while putting the truncation timer out of practical reach.
|
||||
|
||||
Neither was deliberate. The 32 KiB was Envoy Gateway's default `per_connection_buffer_limit_bytes`;
|
||||
the 60s was Envoy AI Gateway's own documented default. The 60s bounded **generation** as well as
|
||||
prompt size — a tiny prompt with a long answer returned 504 at 60.05s.
|
||||
|
||||
Two configuration facts worth keeping, from their bisection:
|
||||
|
||||
- **`ClientTrafficPolicy` is honoured in standalone `aigw run`; `BackendTrafficPolicy` is NOT.** A
|
||||
`BackendTrafficPolicy` setting `requestTimeout` is accepted, logs nothing, and leaves the routes
|
||||
unchanged (upstream `envoyproxy/gateway#9513`). What works is `timeouts: {request: …}` on each
|
||||
`AIGatewayRoute` rule. Nothing from the outside distinguishes the two — the same silent-default
|
||||
shape as their `SecurityPolicy` caveat.
|
||||
- In that stack, "the config was accepted" proves nothing. Read the live `config_dump`.
|
||||
|
||||
## 7.2 The risk we accepted, and why we could not remove it
|
||||
|
||||
Raising the timeout made the failure **rare, not impossible**, and the residual failure is silent.
|
||||
|
||||
On a mid-response timeout over chunked HTTP/1.1, Envoy ends the chunked encoding *cleanly* instead of
|
||||
resetting the connection, so the client receives what looks like a complete transfer
|
||||
(`envoyproxy/envoy#17186` — acknowledged as a bug in 2021, closed by a stale bot, never fixed). The
|
||||
December 2025 fix `envoyproxy/envoy#42269` changes locally-originated resets from `NO_ERROR` to
|
||||
`INTERNAL_ERROR`, but it is **HTTP/2 only** and SSE clients here speak HTTP/1.1.
|
||||
|
||||
Measured on our side while the timeout was still 60s:
|
||||
|
||||
```
|
||||
HTTP 200 61.07s 141992 bytes
|
||||
message_stop 0 message_delta 0 error events 0
|
||||
emitted 2473 of 4000, ending on a WELL-FORMED SSE frame
|
||||
```
|
||||
|
||||
A syntactically valid stream that simply stops. Any timer firing mid-stream — route timeout, idle
|
||||
timeout, `max_stream_duration` — fails this same way.
|
||||
|
||||
**The recommended defence does not transfer to us.** The right fix is to treat a stream with no
|
||||
`message_stop` / `[DONE]` / `finish_reason` as failed. We cannot: our members are Claude Code and
|
||||
opencode, third-party clients whose SSE parsing we do not own, and there is no seam to insert the
|
||||
check. Whether either detects a missing terminator is unverified — and opencode's handling of the 413
|
||||
(swallow, compact, retry forever, never surface an error) does not suggest it is strict.
|
||||
|
||||
So the honest statement of our position:
|
||||
|
||||
> Gateway traffic is acceptable at 86400s because a single request would have to run for 24 hours to
|
||||
> trip the bug — **not** because we could detect it if it did.
|
||||
|
||||
At 86400s our **own** limit binds first, which is the ordering we want. `MessageService.ASYNC_TIMEOUT_MS`
|
||||
caps a turn at 30 minutes, so a runaway request ends as a clean `FAILED` ticket that we raised, rather
|
||||
than as a silently truncated `200` that we cannot see. While the gateway sat at 1800s the two numbers
|
||||
were equal and did not nest, so a gateway-side stall could have been misread as a bug in our own ticket
|
||||
handling. That ambiguity is now gone.
|
||||
|
||||
**If a member ever returns a confident but truncated answer, suspect this before anything in our own
|
||||
code.** That is the whole reason this section exists.
|
||||
|
||||
---
|
||||
|
||||
## 8. Related
|
||||
|
||||
- [CB-589 / #74](https://git.ltms.dev/fleet/fleetd/issues/74) — cost-first placement and a
|
||||
gateway that reports live capacity. The per-consumer figures this migration unlocks are the first
|
||||
input that ticket actually needs.
|
||||
- `docs/CB-500-Multi-Tier-Coordination.md` §11 — the distributed-sandbox topology this gateway is
|
||||
part of.
|
||||
@@ -0,0 +1,242 @@
|
||||
# CB-5xx — Stage 5 Hardening (auth · metrics · CI · supervision · authz+audit)
|
||||
|
||||
**Status:** design note (pre-implementation) — the single-host close-out before cross-host work.
|
||||
**Covers:** CB-501 (bearer auth + TLS) · CB-502 (`/metrics`) · CB-503 (mock-socket CI) ·
|
||||
CB-504 (service supervision) · CB-505 (per-session authz + audit log).
|
||||
**Depends on:** everything shipped through CB-402. Nothing here changes messaging semantics.
|
||||
**Blocks:** CB-308. The cross-host trust model is CB-308's own gating concern, and it inherits
|
||||
whatever identity/authz shape lands here — so this stage is deliberately *before* federation,
|
||||
not after it.
|
||||
|
||||
---
|
||||
|
||||
## 1. Why this stage is not optional bookkeeping
|
||||
|
||||
`fleetd` today has **exactly one security control: the loopback bind**. Every other guarantee
|
||||
rests on it.
|
||||
|
||||
The identity model (`mcp/ConnectionIdentity.java`) resolves a caller from the connection alone —
|
||||
the OS reports the connecting PID, herdr owns the PID→pane map, so a worker cannot forge another
|
||||
worker. Its own javadoc is explicit: *"Single-host only (the herd shares the `fleetd` host); the
|
||||
token path is the split-host fallback."* The token path does not exist yet.
|
||||
|
||||
That leaves a seam that is **latent today and load-bearing the moment the bind moves**:
|
||||
|
||||
```java
|
||||
// ConnectionIdentity.resolve — non-loopback callers get a null terminal
|
||||
if (!isLoopback(remoteAddr)) return new Caller(null, -1);
|
||||
```
|
||||
|
||||
…and `null` terminal is interpreted downstream as **"this caller is the primary"**. Combined:
|
||||
|
||||
> Any caller that is not a recognised on-host worker pane is treated as the primary — including,
|
||||
> if `bind.host` is ever widened, an arbitrary remote client.
|
||||
|
||||
Today `bind` defaults to `127.0.0.1` so this is unreachable. But CB-308 exists precisely to widen
|
||||
the boundary, and the primary is the *most* privileged role on the bus (it spawns, stops, sends to
|
||||
any session, and drains any inbox). Shipping federation on top of "unauthenticated ⇒ primary"
|
||||
would be building the security boundary backwards.
|
||||
|
||||
**So CB-501 is not "add a token header". It is: make identity explicit, and make the absence of
|
||||
identity mean *nothing*, not *everything*.**
|
||||
|
||||
---
|
||||
|
||||
## 2. Decisions (locked)
|
||||
|
||||
### D1 — Three caller roles, one resolution path
|
||||
|
||||
Introduce `Role { PRIMARY, WORKER, ANONYMOUS }` resolved by a single `CallerResolver` that both
|
||||
REST and MCP go through. Resolution order:
|
||||
|
||||
1. **Connection identity wins where it applies.** A loopback peer PID that maps to a herdr worker
|
||||
pane ⇒ `WORKER` with that terminal. Unforgeable, unchanged from today, zero config.
|
||||
2. **Token, if presented.** A valid bearer token ⇒ the role that token is provisioned for.
|
||||
3. **Otherwise `ANONYMOUS`** — *not* `PRIMARY`.
|
||||
|
||||
This inverts today's default. `PRIMARY` becomes something you must *prove* (by being a loopback
|
||||
non-worker process when auth is disabled, or by presenting a primary-scoped token when it is
|
||||
enabled), rather than something you get by failing every other check.
|
||||
|
||||
### D2 — Auth is opt-in by config, but the *default* must stay zero-friction on loopback
|
||||
|
||||
The daemon is dogfooded constantly on one machine. If enabling hardening breaks the local setup,
|
||||
it will be disabled and the stage is wasted. So:
|
||||
|
||||
```yaml
|
||||
auth:
|
||||
mode: loopback-trust # default — behaves exactly like today: loopback ⇒ PRIMARY, no token needed
|
||||
# mode: token # every non-worker caller must present a valid bearer token
|
||||
# tokenEnv: FLEETD_API_TOKEN # host env var holding the token; never the literal value
|
||||
```
|
||||
|
||||
`mode: loopback-trust` is the current behaviour, named honestly and now *chosen* rather than
|
||||
implied. `mode: token` is what a non-loopback bind requires. **A non-loopback `bind.host` with
|
||||
`mode: loopback-trust` must fail fast at startup** — that check is the single highest-value line
|
||||
in this stage, because it makes the dangerous configuration unrepresentable rather than merely
|
||||
discouraged.
|
||||
|
||||
### D3 — TLS terminates *outside* the JVM
|
||||
|
||||
Do **not** add TLS config to Javalin/Jetty. The deployment story for a cross-host gateway is a
|
||||
reverse proxy (or an SSH/WireGuard tunnel) in front of the daemon; the AMQP link has its own TLS
|
||||
via the broker URI (`amqps://`). Adding keystore handling here would mean certificate lifecycle
|
||||
code in a daemon whose whole value is being small, and would duplicate what the proxy does better.
|
||||
|
||||
**CB-501 therefore ships bearer auth + the fail-fast bind check, and documents TLS as a
|
||||
deployment concern with a worked reverse-proxy example.** This is a deliberate narrowing of the
|
||||
roadmap's "auth/TLS" wording — flagged in §6 for the lead.
|
||||
|
||||
### D4 — Metrics without a new dependency
|
||||
|
||||
The roadmap's tech-stack table says Micrometer→Prometheus. Recommend **not** taking that dep:
|
||||
|
||||
- This pom already carries an unusually heavy dependency-reconciliation burden (a hand-pinned
|
||||
`jackson-annotations` 3.0-rc5 to reconcile the MCP SDK's Jackson 3 with our Jackson 2.19, a
|
||||
Jetty BOM import to stop version skew, plus four documented accepted-CVE advisories). Every new
|
||||
transitive tree is a real cost here, not a hypothetical one.
|
||||
- The CVE gate that CLAUDE.md mandates for dependency changes (`jetbrains get_file_problems` →
|
||||
Mend.io) **cannot currently be run** — no JetBrains MCP server is connected. Adding a dependency
|
||||
tree we cannot scan violates the project's own stated policy.
|
||||
- The metric set is small and fully known (§4). Prometheus text exposition is a trivial,
|
||||
stable, well-specified format.
|
||||
|
||||
So: a ~120-line `metrics/Metrics.java` holding `LongAdder` counters and gauge suppliers, rendered
|
||||
to the Prometheus text format at `GET /metrics`. If Micrometer is wanted later for its
|
||||
registry/push ecosystem, this stays a drop-in swap behind the same endpoint. **Flagged in §6 —
|
||||
this deviates from a documented tech-stack choice.**
|
||||
|
||||
### D5 — Supervision targets launchd first, systemd second
|
||||
|
||||
The roadmap says "systemd unit". **This host is macOS — there is no systemd on it** (`systemctl`
|
||||
not found), and the daemon that has been dogfooded for weeks runs as a bare foreground
|
||||
`java -jar`. Ship **both**:
|
||||
|
||||
- `deploy/dev.ltms.fleet.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
ordered start after herdr.
|
||||
- `deploy/fleetd.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
|
||||
Ordering after herdr is advisory in both: the herdr socket may not exist at boot, so the daemon
|
||||
must **retry the socket rather than exit** — supervision ordering is a nicety, socket-retry is the
|
||||
actual fix. That retry behaviour is part of CB-504, not a separate ticket.
|
||||
|
||||
### D6 — Audit log is a separate append-only stream, not the app log
|
||||
|
||||
Privileged actions (spawn, stop, send, reply-drain, ack) emit a structured JSON line to a
|
||||
dedicated `audit` SLF4J logger with its own appender, carrying `{ts, role, terminal, pid, action,
|
||||
target, outcome}`. Keeping it off the chatty app logger is what makes it greppable and, later,
|
||||
shippable. **No message *content* in the audit record** — the bridge carries the user's source
|
||||
code and prompts; an audit trail that quietly becomes a transcript archive is a liability, not a
|
||||
control. Content stays out; correlation ids go in.
|
||||
|
||||
---
|
||||
|
||||
## 3. Authorization model (CB-505)
|
||||
|
||||
With D1's roles, the rules are small enough to state completely:
|
||||
|
||||
| Action | REST | PRIMARY | WORKER | ANONYMOUS |
|
||||
|---|---|---|---|---|
|
||||
| spawn worker | `POST /workers` | ✅ | ❌ | ❌ |
|
||||
| stop worker | `DELETE /workers/{paneId}` | ✅ | ❌ | ❌ |
|
||||
| send to a session | `POST /sessions/{id}/message` | ✅ | ❌ | ❌ |
|
||||
| reply | `POST /sessions/{id}/reply` | ❌ | ✅ **own session only** | ❌ |
|
||||
| ask | `POST /sessions/{id}/ask` | ❌ | ✅ **own session only** | ❌ |
|
||||
| drain replies | `GET /sessions/{id}/replies` | ✅ | ❌ | ❌ |
|
||||
| status / list / profiles | `GET …` | ✅ | ✅ | ❌ |
|
||||
| health | `GET /healthz` | ✅ | ✅ | ✅ (unauthenticated by design) |
|
||||
| metrics | `GET /metrics` | ✅ | ✅ | ❌ |
|
||||
|
||||
The load-bearing row is **"own session only"**: a worker may only reply or ask *as itself*. That is
|
||||
already true de facto — `ConnectionIdentity` derives the terminal rather than reading it from the
|
||||
body — so CB-505 mostly **asserts an existing invariant explicitly** and adds the test that pins
|
||||
it. The one real change is rejecting a worker that names a *different* session id in the path.
|
||||
|
||||
`/healthz` stays open: it must answer for a load balancer or supervisor before any credential is
|
||||
configured. It already leaks nothing but herdr's version and up/down.
|
||||
|
||||
### 3.1 There are TWO entry paths, and only one of them has identity today
|
||||
|
||||
The wiki describes MCP as "a thin adapter over the REST core". **At the code level that is not
|
||||
literally true, and the difference is security-relevant.** `FleetMcp` calls `MessageService` /
|
||||
`SessionManager` *directly*; it never issues an HTTP request against a Javalin route. And `/mcp` is
|
||||
mounted as a raw servlet on Jetty's `ServletContextHandler`
|
||||
(`FleetApp.build → cfg.jetty.modifyServletContextHandler`), so it does **not** pass through
|
||||
Javalin's `before` filters at all.
|
||||
|
||||
The current split is the mirror image of what you'd expect:
|
||||
|
||||
| Path | Caller identity today | Authz today |
|
||||
|---|---|---|
|
||||
| MCP `/mcp` | ✅ resolved per call (`ConnectionIdentity` via the transport-context extractor) | ❌ none |
|
||||
| REST routes | ❌ **none at all** — the session id is taken from the URL path and trusted | ❌ none |
|
||||
|
||||
So REST is the *more* exposed surface: `POST /sessions/{id}/reply` accepts any `{id}` from the
|
||||
path, whereas the MCP `fleet_reply` derives the worker from the connection and refuses to read it
|
||||
from an argument. Loopback-only bind is what makes this safe today.
|
||||
|
||||
**Therefore CB-505 must enforce on both paths against one shared resolver** — not at a single
|
||||
choke point. Concretely: a Javalin `before` filter for REST, and the existing transport-context
|
||||
extractor for MCP, both delegating to `auth.CallerResolver`. Any authz check that lives in only
|
||||
one of the two is not a control.
|
||||
|
||||
---
|
||||
|
||||
## 4. Metric set (CB-502)
|
||||
|
||||
Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
|
||||
| Metric | Type | Why it exists |
|
||||
|---|---|---|
|
||||
| `fleet_sends_total{outcome}` | counter | outcome ∈ replied\|completion_fallback\|timeout\|failed — the completion-fallback rate is the health signal for turn detection (CB-115/116/118) |
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
| `fleet_auth_failures_total{reason}` | counter | only meaningful once CB-501 lands; catches misconfigured workers |
|
||||
|
||||
---
|
||||
|
||||
## 5. Increment plan
|
||||
|
||||
Ordered so each step is independently mergeable and the risky one lands first.
|
||||
|
||||
1. **CB-501a — `CallerResolver` + `Role`.** Pure refactor: route today's connection identity through
|
||||
the new type, `ANONYMOUS` not yet reachable (loopback-trust default preserves behaviour).
|
||||
Green build, no behaviour change.
|
||||
2. **CB-501b — token mode + fail-fast bind check.** Config block, bearer parsing, the
|
||||
non-loopback-bind guard. This is the security-relevant commit; keep it small and reviewable.
|
||||
3. **CB-505 — authz table + audit logger.** Enforce §3 on **both** entry paths (see §3.1); add the
|
||||
audit appender.
|
||||
4. **CB-502 — `Metrics` + `/metrics`.** Instrument the paths in §4.
|
||||
5. **CB-503 — CI.** Runs `mvn -B clean install` with `-Dgroups='!contract'` so the live-herdr and
|
||||
RabbitMQ contract tests are excluded; the mock-UDS suite is the CI surface, exactly as the
|
||||
roadmap's testability section intends.
|
||||
6. **CB-504 — launchd plist + systemd unit + herdr-socket retry.**
|
||||
|
||||
---
|
||||
|
||||
## 6. Open questions for the lead
|
||||
|
||||
*All four resolved 2026-07-29 — the lead confirmed D3, D4, and the D6 sink; the CI runner question
|
||||
was answered from the forge itself. Kept here as the decision record.*
|
||||
|
||||
1. ✅ **TLS scope (D3) — confirmed.** Bearer auth + the fail-fast bind guard ship in the daemon;
|
||||
TLS terminates at a reverse proxy, documented with a worked example. AMQP gets TLS via an
|
||||
`amqps://` URI. No keystore handling in `fleetd`.
|
||||
2. ✅ **Micrometer (D4) — confirmed dropped.** Zero-dependency Prometheus text renderer, for the
|
||||
reasons in D4 (pom reconciliation burden + the mandated CVE gate being un-runnable this
|
||||
session). Revisit if a push-gateway or JVM-metrics requirement appears; the endpoint is the
|
||||
swap seam.
|
||||
3. ~~**CI runner (CB-503).**~~ ✅ **Resolved during design** — a Gitea Actions runner *is*
|
||||
registered and healthy (`lms/alms-memory` has 28 completed runs; `lms/alms` runs on push and
|
||||
pull_request). CB-503 targets `.gitea/workflows/ci.yml` with `runs-on: ubuntu-latest`, matching
|
||||
the sibling repo's convention. Note the runner's image ships an older `default-jdk`, so the
|
||||
workflow must provision **JDK 25** explicitly rather than apt-installing the default.
|
||||
Contract-test exclusion needs no CI flag: the pom's `default-excludes` profile already sets
|
||||
`excludedGroups=contract`, so a plain `mvn -B clean install` *is* the mock-socket surface.
|
||||
4. ✅ **Audit sink — confirmed dedicated file.** Its own logback appender writing JSON lines beside
|
||||
the daemon, separate from the app log, per D6. Content still never enters the record.
|
||||
@@ -0,0 +1,972 @@
|
||||
# M4 - Fleet health, recovery, routing, and capacity
|
||||
|
||||
**Status:** Design accepted on 2026-08-15. CB-573 part 1 has shipped the classification model and
|
||||
the `fleet_list` capacity view; the remaining M4 units are not yet shipped. See
|
||||
[Unit 2 — what has landed so far](#unit-2---what-has-landed-so-far) before planning Unit 2 work:
|
||||
some of its criteria were met by separate CB tickets, and one of them contradicts the unit text.
|
||||
**Scope:** Fleet evidence, safe mechanical repair, lead routing, capacity reporting, and optional
|
||||
human notification.
|
||||
**Grounded in:** `health/FleetHealth`, `health/PaneBudget`, `inject/StatusPoller`,
|
||||
`inject/StatusRefiner`, `inject/CompletionResolver`, `inject/Injector`, `session/SessionManager`,
|
||||
`msg/MessageService`, `msg/ReplyInbox`, `msg/ReplyPushLoop`, `msg/LeadHeartbeatLoop`,
|
||||
`mcp/PrimaryRegistry`, and `herdr/AgentControl`.
|
||||
|
||||
## 1. Problem and decision boundary
|
||||
|
||||
The operator asked the bridge to detect idle agents, exceptions, stopped work, and broken
|
||||
communication. The bridge may read an agent pane from time to time. It must notify a person when
|
||||
the fleet cannot move forward.
|
||||
|
||||
The four operator terms are not four equal health states. `IDLE` is a normal mode. An exception is
|
||||
sometimes visible only as pane text. Stopped work may look the same as slow work. Broken
|
||||
communication can occur on several links.
|
||||
|
||||
M4 uses this boundary:
|
||||
|
||||
- The bridge detects facts and joins evidence.
|
||||
- The bridge repairs only mechanical failures with no judgement.
|
||||
- The lead decides whether to stop, retry, replace, or reassign a member.
|
||||
- A human is notified only when no healthy lead can act.
|
||||
- n8n may route an outbound incident. It never classifies state or chooses recovery.
|
||||
|
||||
An inbound n8n decider would need bridge authority. No narrow machine-decider role exists. Giving a
|
||||
workflow engine lead authority is unsafe, while adding a new role is a separate authorization
|
||||
design. An outbound sink needs no bridge role.
|
||||
|
||||
The bridge must never replay a delivered task. That task may already have changed files, pushed a
|
||||
branch, opened a pull request, or changed external state. A replay can run those side effects twice.
|
||||
This rule must remain true even if later code stores delivered prompt text.
|
||||
|
||||
## 2. Evidence model
|
||||
|
||||
A health state is mainly a comparison between two views:
|
||||
|
||||
- **herdr view:** current agents and raw live status from one `AgentControl.list()` call.
|
||||
- **bridge view:** session FSM, MCP presence, accepted turns, tasks, inbox state, and lead ownership.
|
||||
|
||||
A strong fault often appears as a disagreement between those views. For example, `BUSY` in the
|
||||
session FSM and `DONE` in herdr means the bridge missed a turn boundary. Pane reads support this
|
||||
model, but they are not the main monitor.
|
||||
|
||||
`SessionManager.rosterView` already joins session state and live status. `AgentControl.list()`
|
||||
already gets the whole live fleet in one call. M4 makes that join persistent and adds timers,
|
||||
accepted-turn state, and incident state.
|
||||
|
||||
### 2.1 Real traces behind the design
|
||||
|
||||
The first trace was an architect that stopped making progress:
|
||||
|
||||
```text
|
||||
profile=opus role=architect state=busy liveStatus=done
|
||||
```
|
||||
|
||||
The session moved from `DONE` to `BUSY` for turn 2. Eighteen minutes later, the session still said
|
||||
`BUSY`, herdr still said `DONE`, the async task still said `PENDING`, and no completion fallback had
|
||||
run. This is `TURN_BOUNDARY_LOST`, not a general slow-turn guess.
|
||||
|
||||
The second trace had two async sends to the same pane, one second apart. The pane was then stopped.
|
||||
One ticket became failed. The other stayed `pending - worker unknown`. Current
|
||||
`MessageService.abandon` resolves only `Rendezvous.currentWaiter(target)`, while async tasks live in
|
||||
a separate ticket map. CB-568 is intended to fix that bug. M4 still keeps an independent
|
||||
post-teardown invariant so a later regression becomes `DELEGATION_ORPHANED`.
|
||||
|
||||
### 2.2 Corrections made during design
|
||||
|
||||
The first state table missed `BUSY` in fleetd plus `IDLE` or `DONE` in herdr. It would have found
|
||||
the real trace only through a late, weak stall timer. The final model adds
|
||||
`TURN_BOUNDARY_LOST` as a strong disagreement state.
|
||||
|
||||
The first notification design also required a webhook before `health.enabled` could turn on. That
|
||||
removed useful local detection to avoid a narrower human-notification gap. The final design splits
|
||||
detection from notification. Missing human escalation is shown as partial coverage instead of
|
||||
disabling health.
|
||||
|
||||
## 3. Classification precedence
|
||||
|
||||
Evidence is applied in this order. A lower rule cannot hide a higher one.
|
||||
|
||||
1. **Control link:** failed fleet list plus failed ping becomes `CONTROL_LINK_DOWN`.
|
||||
2. **Definitive target loss:** `_not_found` becomes `GONE` or `LEAD_UNREACHABLE` when the control
|
||||
link is healthy.
|
||||
3. **Startup and teardown invariants:** readiness expiry becomes `NEVER_READY`; surviving tasks
|
||||
after teardown become `DELEGATION_ORPHANED`.
|
||||
4. **Bridge/live disagreement:** `BUSY` plus stable raw `IDLE` or `DONE` becomes
|
||||
`TURN_BOUNDARY_LOST`.
|
||||
5. **Known screen evidence:** a tested fatal signature becomes `ERROR_ON_SCREEN`.
|
||||
6. **Timed suspicion:** unchanged sparse pane probes may become `STALL_SUSPECTED`.
|
||||
7. **Communication quality:** completion fallback becomes `MUTE`; an old inbox entry becomes
|
||||
`REPLY_STRANDED`.
|
||||
8. **Normal mode:** `STARTING`, `IDLE`, `WORKING`, `WORK_PENDING`, or `BLOCKED_AMBIGUOUS`.
|
||||
|
||||
The member flow in Figure 1 shows lifecycle states and the main fault exits. Fault states are
|
||||
reported beside the session FSM; most are not new FSM values.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Registered["Member registered"] --> Starting["STARTING"]
|
||||
Starting -->|"MCP presence"| Idle["IDLE"]
|
||||
Starting -->|"Readiness grace expires"| NeverReady["NEVER_READY"]
|
||||
Idle -->|"Accepted delivery"| Working["WORKING"]
|
||||
Working -->|"Trusted turn boundary"| Idle
|
||||
Working -->|"Bridge BUSY and herdr IDLE or DONE"| Lost["TURN_BOUNDARY_LOST"]
|
||||
Working -->|"Known fatal screen"| Error["ERROR_ON_SCREEN"]
|
||||
Working -->|"Long age and unchanged sparse probes"| Stall["STALL_SUSPECTED"]
|
||||
Working -->|"Target not found"| Gone["GONE"]
|
||||
Idle -->|"Inbox or queued delivery exists"| Pending["WORK_PENDING"]
|
||||
Pending -->|"Delivery or collection finishes"| Idle
|
||||
Idle -->|"Raw BLOCKED with an open turn"| Blocked["BLOCKED_AMBIGUOUS"]
|
||||
Lost -->|"Strict guarded repair"| Repaired["DONE with reconciled completion"]
|
||||
Lost -->|"Repair refused"| LeadDecision["Lead decision required"]
|
||||
```
|
||||
|
||||
*Figure 1. The member lifecycle and the main health exits. Pane-based states never authorise an
|
||||
automatic retry of the task.*
|
||||
|
||||
## 4. State model
|
||||
|
||||
### 4.1 Normal and transitional member states
|
||||
|
||||
| State | Exact evidence | Meaning and certainty |
|
||||
|---|---|---|
|
||||
| `STARTING` | Session is `SPAWNING`; MCP presence is absent | Normal inside the startup grace. MCP contact is the readiness signal. |
|
||||
| `IDLE` | Session is `READY` or `DONE`; live status is `IDLE` or `DONE`; no open turn or inbox item exists | Normal. Idle is not a fault. |
|
||||
| `WORKING` | Session is `BUSY`; raw live status is `WORKING`; the accepted turn is open | Certain that herdr sees work. It does not prove useful progress. |
|
||||
| `WORK_PENDING` | Queued delivery or inbox content exists while the target is injectable | Transitional. Existing injector or push logic should move it. |
|
||||
| `BLOCKED_AMBIGUOUS` | An open turn exists and raw live status is `BLOCKED` | The bridge cannot tell whether this is permission, input, or a settled screen. |
|
||||
|
||||
Idle may drive configured resource cleanup. It never opens an incident and never pages a person.
|
||||
|
||||
### 4.2 Member fault and quality states
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `NEVER_READY` | `SPAWNING`, no MCP presence, and an accepted delivery waits through the existing readiness grace | Delivery never became possible. The exact cause is unknown. Fail the send, stop the process, and preserve a provisioned worktree. |
|
||||
| `GONE` | Per-target herdr call returns `_not_found` while fleet list or ping works | Certain target loss. Fail all target work. Do not replay it. |
|
||||
| `TURN_BOUNDARY_LOST` | Same session turn stays `BUSY`; same accepted task stays open; two raw snapshots show `IDLE` or `DONE` | Strong disagreement. Strict reconciliation may repair it. |
|
||||
| `ERROR_ON_SCREEN` | Suspicious non-working state survives grace; `detection` matches a tested adapter-specific fatal signature | Certain only for the matched signature. A bare word such as `Exception` is not enough. |
|
||||
| `STALL_SUSPECTED` | Open turn is older than the configured threshold; two normalised `recent_unwrapped` digests are unchanged; no boundary or reply occurs | Not certain. A long valid API call can look the same. Lead decides. |
|
||||
| `MUTE` | Turn resolves through completion fallback instead of `fleet_reply` | Certain that no structured reply won. It does not prove an MCP failure. A single event is a metric, not an incident. |
|
||||
| `REPLY_STRANDED` | Typed reply or health message remains after owning-lead push reaches its cap | Collection failed. This does not explain whether the lead is busy, dead, or ignoring the nudge. |
|
||||
| `DELEGATION_ORPHANED` | Target is gone, failed, or released, but one or more tasks remain `PENDING` after reconciliation grace | Certain bridge invariant failure. This is not an inbox-drain fault. |
|
||||
| `WORK_PRODUCT_AT_RISK` | Provisioned branch has commits after its recorded base; member is `DONE`, `FAILED`, or preserved after release; no turn or inbox item remains; long-idle threshold passed | A warning, not proof of loss. Work may already have an open pull request or a squash merge. |
|
||||
|
||||
`MUTE` opens an incident only after a small fixed rate threshold for one target or profile, or when
|
||||
it appears with another fault.
|
||||
|
||||
`WORK_PRODUCT_AT_RISK` must not become `WORK_PRODUCT_UNCOLLECTED`. The bridge does not know pull
|
||||
request or merge state. If committed work appears with `REPLY_STRANDED` or
|
||||
`DELEGATION_ORPHANED`, the existing incident gains `committedWorkAtRisk: true`.
|
||||
|
||||
### 4.3 Control-link state
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `CONTROL_LINK_DOWN` | Two full-fleet `agent.list` calls fail across the grace, and herdr `ping` also fails | Certain for the fleetd-to-herdr link. Retry calls, record the incident, and use human escalation if no lead can be reached. |
|
||||
|
||||
A failed fleet list alone is not a dead-member claim. A single `_not_found` with a healthy global
|
||||
link is a target fault, not a control-link fault.
|
||||
|
||||
### 4.4 Lead states
|
||||
|
||||
| State | Exact evidence | Meaning and action |
|
||||
|---|---|---|
|
||||
| `LEAD_IDLE` | Expected lead is present with raw injectable status; no actionable state waits | Normal. Existing heartbeat may run under its own policy. |
|
||||
| `LEAD_WORKING` | Expected lead is present with raw `WORKING`; stall threshold is not met | Reachable and busy. Never inject into the live turn. |
|
||||
| `LEAD_STATUS_UNKNOWN` | Expected lead is present with raw `UNKNOWN` | Neither dead nor a healthy routing target. Retain evidence and retry. |
|
||||
| `LEAD_UNREACHABLE` | Expected lead is absent from two successful live-agent snapshots while ping works, or targeted lookup returns `_not_found` with a healthy control link | Route to a healthy peer. If none exists, use human escalation. |
|
||||
| `LEAD_UNRESPONSIVE` | Actionable state waits; lead stays injectable; bounded nudges exhaust; inbox remains uncollected | Route to a healthy peer or a person. |
|
||||
| `LEAD_STALL_SUSPECTED` | Lead stays `WORKING` past threshold; two sparse pane probes show no progress | Not certain. Never kill or restart automatically. Route to peer or person. |
|
||||
|
||||
The monitor retains the lead name and terminal, last successful sighting, raw status and age,
|
||||
consecutive list absences, targeted errors, pane-probe facts, pending incident age, and nudge
|
||||
outcomes. Current heartbeat and push loops discard much of this history.
|
||||
|
||||
Expected lead identity comes from the same supplier used by `CallerResolver`. It is not liveness
|
||||
evidence. `LeadTabScanner` keeps cached identity after a failed scan, so the health monitor compares
|
||||
that identity with a fresh successful agent list. A dynamic identity also survives a two-successful-
|
||||
snapshot retirement grace. This stops a dead lead from escaping health by disappearing from one map.
|
||||
|
||||
### 4.5 Evidence limits
|
||||
|
||||
M4 cannot tell these cases apart with current evidence:
|
||||
|
||||
- A valid long call and a hung call may have the same status and pane digest.
|
||||
- `BLOCKED` does not explain which input is needed.
|
||||
- An idle prompt after failure may look like an idle prompt after success.
|
||||
- A missing structured reply does not prove a broken MCP connection.
|
||||
- An undrained inbox does not explain why the lead did not collect it.
|
||||
- Arbitrary pane text cannot safely classify arbitrary exceptions.
|
||||
- A branch ahead of its base does not prove that work was not collected.
|
||||
|
||||
Logs are outputs, not classifier inputs. The monitor never parses its own logs.
|
||||
|
||||
## 5. Automatic action and lead action
|
||||
|
||||
### 5.1 Actions the bridge may take
|
||||
|
||||
The bridge may:
|
||||
|
||||
- retry transient herdr status, list, ping, and pane-read failures with bounded backoff;
|
||||
- re-submit Enter after the existing paste/submit race;
|
||||
- fail queued delivery after `NEVER_READY`;
|
||||
- stop a never-ready process while preserving its provisioned worktree;
|
||||
- fail all queued, accepted, and async tasks for a gone or released target;
|
||||
- reconcile one lost boundary when every strict gate in Section 8 passes;
|
||||
- hold typed messages, nudge the owning lead, and stop at the configured cap;
|
||||
- use the existing bounded idle-lead heartbeat;
|
||||
- deduplicate, route, update, and resolve incidents.
|
||||
|
||||
These actions do not choose new work and do not replay old work.
|
||||
|
||||
### 5.2 Decisions reserved for the lead
|
||||
|
||||
Only the lead may:
|
||||
|
||||
- stop or continue `BLOCKED_AMBIGUOUS`;
|
||||
- stop, inspect, or wait on `ERROR_ON_SCREEN`;
|
||||
- kill or continue `STALL_SUSPECTED`;
|
||||
- spawn a replacement or reassign work;
|
||||
- retry a delivered task;
|
||||
- choose how to use partial work in a worktree;
|
||||
- restart herdr or change network, model, credentials, backend, or configuration.
|
||||
|
||||
Reports include literal safe tool calls such as `fleet_status(sessionId="...")`,
|
||||
`fleet_poll(ticket="...")`, `fleet_list()`, and optional `fleet_stop(paneId="...")`. A judgement
|
||||
state never presents stop as the only action.
|
||||
|
||||
### 5.3 Release causes and worktree safety
|
||||
|
||||
| Release cause | Process action | Provisioned worktree |
|
||||
|---|---|---|
|
||||
| `SPAWN_ROLLBACK` before registration or delivery | Stop and clean up | Remove |
|
||||
| `COMPLETED` for `READY` or `DONE` without pending work, idle TTL, or successful context-cap completion | Stop | Remove only if clean; preserve a dirty worktree (CB-576) |
|
||||
| `NEVER_READY` | Stop | Preserve |
|
||||
| `GONE` | Best-effort stop | Preserve |
|
||||
| `TURN_FAILED` or lead abort while `BUSY` or `FAILED` | Stop | Preserve |
|
||||
| `RELEASE_WITH_PENDING_TASKS` | Stop | Preserve |
|
||||
| `SHUTDOWN` | Stop | Preserve |
|
||||
|
||||
Explicit stop is state-aware. `SPAWNING`, `BUSY`, `FAILED`, or any target with pending tasks uses a
|
||||
preserving cause.
|
||||
|
||||
Before abnormal release removes the live session, M4 writes an atomic manifest under the worktree
|
||||
root. It records session identity, owner, role, profile, repository, path, branch, base commit,
|
||||
release cause, release time, state, and pending task ids. `fleet_list.preservedWorktrees` loads these
|
||||
manifests after restart. Stop output and WARN logs also name the path and cause. M4 never
|
||||
auto-deletes a preserved worktree.
|
||||
|
||||
## 6. Fleet health monitor
|
||||
|
||||
Add `FleetHealthMonitor`. Do not widen `StatusPoller` into a policy loop.
|
||||
|
||||
`StatusPoller` has a 250 ms delivery cadence and samples only injector targets with outstanding
|
||||
work. Health needs all sessions, all leads, task state, inbox age, and global control evidence. One
|
||||
loop cannot serve both cadences safely.
|
||||
|
||||
Build the monitor like `LeadHeartbeatLoop`:
|
||||
|
||||
- pure `decide(snapshot, priorState, now)` logic;
|
||||
- a thin scheduler;
|
||||
- an injected clock;
|
||||
- edge-triggered state changes;
|
||||
- no network work in the pure function;
|
||||
- no sleeping in tests.
|
||||
|
||||
Each enabled fleet tick reads:
|
||||
|
||||
- one `AgentControl.list()` result for the whole fleet;
|
||||
- one in-memory `SessionManager.roster()` snapshot;
|
||||
- accepted turns and async task state;
|
||||
- typed inbox depth, kind, and age;
|
||||
- push and heartbeat outcomes;
|
||||
- configured and discovered leads.
|
||||
|
||||
Existing failure paths publish structured evidence to the monitor. The monitor does not infer events
|
||||
from log text.
|
||||
|
||||
### 6.1 Pane budget
|
||||
|
||||
Healthy idle members, recent working members, and quiet leads cause no pane reads.
|
||||
|
||||
A pane is eligible only for a stable lost boundary, sustained `BLOCKED` or `UNKNOWN`, work older
|
||||
than the suspect threshold, or one final evidence read for a confirmed fault when the pane exists.
|
||||
|
||||
Compiled brakes apply even if config asks for more:
|
||||
|
||||
- per-target pane cooldown is at least 60 seconds;
|
||||
- working age before the first progress probe is at least 300 seconds;
|
||||
- at most two pane reads occur in one fleet tick;
|
||||
- targets rotate fairly;
|
||||
- only a normalised digest and optional clipped local excerpt are stored;
|
||||
- no pane excerpt leaves fleetd in a human webhook.
|
||||
|
||||
Use `detection` for tested screen signatures. Use normalised `recent_unwrapped` only for progress
|
||||
comparison.
|
||||
|
||||
## 7. Typed inbox and routing
|
||||
|
||||
### 7.1 Semantic record
|
||||
|
||||
The typed inbox record carries:
|
||||
|
||||
```text
|
||||
schemaVersion
|
||||
kind: reply | health
|
||||
msgId, target, subjectTerminal, recipientLead
|
||||
severity, state, evidence
|
||||
createdAtEpochMillis, firstSeenEpochMillis, lastSeenEpochMillis
|
||||
recoveryTried, suggestedToolCalls, content
|
||||
```
|
||||
|
||||
A health message never calls `Rendezvous.resolve`. It cannot look like the member's task result.
|
||||
|
||||
Both inbox adapters share field preservation, first-id-wins dedup, FIFO among decoded messages,
|
||||
explicit ownership, ack, and release rules. The in-memory adapter stores typed records directly. It
|
||||
does not copy AMQP migration logic.
|
||||
|
||||
### 7.2 AMQP migration
|
||||
|
||||
The reader uses AMQP `content_type`, never body sniffing:
|
||||
|
||||
```text
|
||||
Legacy v0: text/plain
|
||||
Typed family: application/vnd.ltms.fleet.inbox-message+json
|
||||
```
|
||||
|
||||
A legacy reply may begin with `{`. It remains plain text because its media type is `text/plain`.
|
||||
Legacy text becomes `kind=reply` with exact UTF-8 content and absent typed metadata.
|
||||
|
||||
Typed JSON has required integer `schemaVersion: 1`. Version 1 ignores unknown optional fields.
|
||||
Missing required fields, invalid enums, malformed UTF-8 or JSON, and property/body identity mismatch
|
||||
are invalid data.
|
||||
|
||||
An unknown schema version is not partly decoded. It remains unacknowledged on the original queue and
|
||||
creates one operator-visible `unsupported_version` failure. A newer daemon may read it later.
|
||||
|
||||
Invalid known-format data is copied byte-for-byte to durable queue
|
||||
`agent.<target>.inbox.quarantine`. A dedicated confirm-mode publisher confirms the persistent copy
|
||||
before the original is acknowledged. A failed quarantine handoff leaves the original unacknowledged.
|
||||
The raw body never enters logs.
|
||||
|
||||
Decode failure creates a redacted WARN, metric, `fleet_list` summary, and routed health incident.
|
||||
One bad entry never escapes the consumer callback and never stops later valid messages.
|
||||
|
||||
Safe downgrade is not supported. The previous build ignores `content_type` and would show typed JSON
|
||||
as ordinary reply text. If drained, it would acknowledge the message and lose typed meaning. Typed
|
||||
queues must be drained or preserved before an old jar runs.
|
||||
|
||||
The existing contract suite uses RabbitMQ. Production uses LavinMQ. The migration and lead-key
|
||||
ownership cases must run once against production LavinMQ before release, or the release must state
|
||||
that LavinMQ was not checked.
|
||||
|
||||
### 7.3 Member routing
|
||||
|
||||
A member incident first goes to the exact lead that owns its accepted delegation.
|
||||
`PrimaryRegistry` needs a no-fallback `delegatingLeadFor(memberTarget)` query. Health routing must not
|
||||
use the old singular-primary fallback when several leads exist.
|
||||
|
||||
Publish the incident under the affected member target. Trigger the existing bounded push route. The
|
||||
push waits until the owning lead is injectable, so it does not interrupt a live lead turn.
|
||||
|
||||
### 7.4 Peer lead routing
|
||||
|
||||
Figure 2 shows the route from incident to lead, peer, or person.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Incident["Open incident"] --> Member{"Member incident?"}
|
||||
Member -->|"yes"| Known{"Exact delegation owner known?"}
|
||||
Known -->|"no"| Sink{"Human webhook enabled and healthy?"}
|
||||
Known -->|"yes"| Owner{"Owner lead healthy?"}
|
||||
Owner -->|"yes"| OwnerInbox["Publish to owner lead path"]
|
||||
Owner -->|"no"| PeerSet["Build healthy peer candidate set"]
|
||||
Member -->|"no, lead incident"| PeerSet
|
||||
PeerSet --> Peer{"Healthy peer exists?"}
|
||||
Peer -->|"yes"| Select["Choose fewest assigned incidents<br/>then stable name and terminal id"]
|
||||
Select --> PeerInbox["Publish to peer lead inbox<br/>and status-gated push"]
|
||||
Peer -->|"no"| Sink
|
||||
Sink -->|"yes"| Webhook["Send classified outbound incident"]
|
||||
Sink -->|"no"| Passive["Keep incident open<br/>show partial coverage on local surfaces"]
|
||||
```
|
||||
|
||||
*Figure 2. Routing keeps delegation ownership separate from temporary peer fallback.*
|
||||
|
||||
Peer candidates exclude the incident subject, failed owner, absent leads, raw-unknown leads, and
|
||||
leads with an open unhealthy state. A reachable `WORKING` peer may be selected; its push waits for an
|
||||
injectable window.
|
||||
|
||||
Choose the candidate with the fewest assigned foreign incidents. Break ties by stable lead name,
|
||||
then terminal id. Pin the recipient. Reassign only if that peer becomes unhealthy or retires. A
|
||||
routing generation marks a reassignment, and old pending assignments become superseded.
|
||||
|
||||
`fleet_list` lead rows show health, health age, assigned foreign incident count, and a bounded list
|
||||
of incident id, subject, state, severity, age, and routing generation. The top-level view also shows
|
||||
owner, recipient, and routing reason.
|
||||
|
||||
A peer incident is published under the recipient lead's inbox key, not the failed subject's key. Its
|
||||
status-gated nudge names the failed lead and gives the exact
|
||||
`fleet_poll(target="<recipient-terminal>")` call.
|
||||
|
||||
### 7.5 Lead inbox ownership
|
||||
|
||||
Add `LeadInboxRegistry`, driven by the same expected-lead supplier as `CallerResolver`.
|
||||
|
||||
It calls `replyInbox.own(leadTerminal)` at startup for configured leads, after successful discovery,
|
||||
after config adds a lead, and before publication. Ownership is not an authorization side effect.
|
||||
|
||||
A missing lead keeps its key owned. Release happens only after confirmed retirement, all incidents
|
||||
are reassigned or resolved, typed health messages move or ack, and the queue is empty. Own a
|
||||
replacement terminal before moving messages from the old key. Never release a non-empty in-memory
|
||||
lead key, because in-memory release clears local data.
|
||||
|
||||
### 7.6 Single-lead deployment
|
||||
|
||||
One lead and no peer is a normal mode, not an edge case.
|
||||
|
||||
An idle, reachable lead may receive the existing bounded nudge. An unreachable or stalled sole lead
|
||||
has no safe in-loop recovery. The bridge must not restart or replace it. A new lead would not have the
|
||||
failed lead's plan or context, and an uncertain relaunch could create two orchestrators.
|
||||
|
||||
With no webhook, only `fleet_list`, `/healthz`, metrics, WARN logs, and the incident journal remain.
|
||||
These are passive surfaces. They are not a human notification.
|
||||
|
||||
## 8. Lost-boundary reconciliation
|
||||
|
||||
This is the only M4 path that reconstructs a result. It must prefer a visible stall over a fabricated
|
||||
reply.
|
||||
|
||||
### 8.1 Why normal completion rules are not enough
|
||||
|
||||
Current `CompletionResolver.resolve` has two fail-open rules. It resolves when the delivery baseline
|
||||
is missing. It also resolves an empty completion when the pane read fails. Those choices are valid
|
||||
after a trusted `WORKING -> IDLE` boundary because the bridge knows the turn ran. They are unsafe
|
||||
when health only guesses that a boundary was lost.
|
||||
|
||||
M4 gives each accepted send an internal `TurnToken`. It ties target, exact waiter, session turn,
|
||||
delivery baseline, and task outcome together.
|
||||
|
||||
### 8.2 Delivery baseline
|
||||
|
||||
Capture the baseline immediately after prompt send and before the delivery future completes. Store:
|
||||
|
||||
```text
|
||||
TurnToken
|
||||
exact waiter identity
|
||||
capture time and pane source
|
||||
normalised assistant block clipped to MAX_SCRAPE_CHARS
|
||||
whether a supported assistant marker was recognised
|
||||
capture result: PRESENT | READ_FAILED | UNRECOGNISED
|
||||
```
|
||||
|
||||
A failed or missing baseline never authorises repair. A late baseline is not valid evidence. After a
|
||||
daemon restart, the old waiter, task, token, and baseline are gone, so the old turn cannot be
|
||||
repaired.
|
||||
|
||||
Automatic repair is enabled only for agent kinds with tested assistant-block fixtures. Current
|
||||
extraction is Claude Code-specific and falls back to arbitrary raw text without `⏺`. That raw fallback
|
||||
cannot authorise repair. OpenCode repair stays disabled until live pane fixtures exist.
|
||||
|
||||
### 8.3 Strict gates and resolver result
|
||||
|
||||
Figure 3 shows the repair gates. Any failed gate keeps the waiter unchanged.
|
||||
|
||||
The two raw snapshots must describe the same `TurnToken` and session turn. No `WORKING`,
|
||||
`BLOCKED`, `UNKNOWN`, missing-agent, reply, failure, or new-delivery observation may occur between
|
||||
them.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Candidate["TURN_BOUNDARY_LOST candidate"] --> Stable{"Same TurnToken and BUSY turn<br/>across two raw IDLE or DONE snapshots?"}
|
||||
Stable -->|"no"| Resnapshot["Take a fresh snapshot"]
|
||||
Stable -->|"yes"| Waiter{"Exact captured waiter<br/>still open by identity?"}
|
||||
Waiter -->|"no"| Stale["STALE_TURN or ALREADY_RESOLVED"]
|
||||
Waiter -->|"yes"| Baseline{"Successful recognised<br/>delivery baseline exists?"}
|
||||
Baseline -->|"no"| Refuse["Refuse repair<br/>leave ticket pending"]
|
||||
Baseline -->|"yes"| Read{"Fresh pane read succeeds?"}
|
||||
Read -->|"no"| Refuse
|
||||
Read -->|"yes"| Output{"Recognised non-blank assistant block<br/>differs from clipped baseline?"}
|
||||
Output -->|"no"| Refuse
|
||||
Output -->|"yes"| Resolve["Shared CompletionResolver guard core<br/>resolves exact waiter"]
|
||||
Resolve -->|"won race"| Repaired["RECONCILED_COMPLETION<br/>same turn becomes DONE"]
|
||||
Resolve -->|"lost race"| Resnapshot
|
||||
```
|
||||
|
||||
*Figure 3. Repair needs stronger evidence than a normal observed turn boundary.*
|
||||
|
||||
Refactor the current resolver into one guard core with two policies:
|
||||
|
||||
```text
|
||||
resolveCaptured(target, inFlight, OBSERVED_BOUNDARY)
|
||||
resolveCaptured(target, inFlight, LOST_BOUNDARY_REPAIR)
|
||||
```
|
||||
|
||||
The health monitor calls only:
|
||||
|
||||
```text
|
||||
CompletionResolver.reconcileLostBoundary(target, expectedTurnToken)
|
||||
```
|
||||
|
||||
It returns `REPAIRED`, `ALREADY_RESOLVED`, `REFUSED_NO_CAPTURE`, `REFUSED_NO_BASELINE`,
|
||||
`REFUSED_UNREADABLE`, `REFUSED_UNCHANGED`, `REFUSED_AMBIGUOUS_OUTPUT`, `STALE_TURN`, or
|
||||
`RACE_LOST`.
|
||||
|
||||
Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move that turn from `BUSY` to `DONE`. A
|
||||
per-target reconciliation gate stops a queued second send from being accepted between waiter
|
||||
resolution and the FSM transition.
|
||||
|
||||
### 8.4 Lead-visible marker and refusal
|
||||
|
||||
A repaired result uses distinct `RECONCILED_COMPLETION` values in `Rendezvous`, `MessageService`,
|
||||
task poll source, and metrics. The lead sees:
|
||||
|
||||
```text
|
||||
[repaired completion - fleetd detected a lost turn boundary. The member did not call
|
||||
fleet_reply; pane-derived text follows and may be partial]
|
||||
```
|
||||
|
||||
Clipped text also keeps the existing clipped-tail marker.
|
||||
|
||||
A refused repair leaves `TURN_BOUNDARY_LOST` open and the ticket pending. The report states that no
|
||||
reply was reconstructed and no task was replayed. `UNCHANGED`, `UNREADABLE`, and
|
||||
`AMBIGUOUS_OUTPUT` get at most one delayed retry for the same token. Missing capture or baseline gets
|
||||
no retry. After two refused scrapes, automatic repair stops for that token.
|
||||
|
||||
### 8.5 Target-wide teardown invariant
|
||||
|
||||
CB-568 owns the multi-ticket cancellation mechanism. M4 routes every terminal cause through that one
|
||||
idempotent operation and checks this independent invariant after teardown:
|
||||
|
||||
- no injector entry exists for the target;
|
||||
- no accepted turn or completion record exists;
|
||||
- no rendezvous waiter or ask exists;
|
||||
- every async task is terminal or was already terminal;
|
||||
- no thread waiting for the target send lock can later accept it;
|
||||
- new sends fail immediately;
|
||||
- each old task has one terminal outcome and one metric count.
|
||||
|
||||
A violation becomes `DELEGATION_ORPHANED`. The monitor may call the same idempotent target-wide
|
||||
failure operation once. It never recreates the task.
|
||||
|
||||
## 9. Capacity and utilisation
|
||||
|
||||
Capacity is a view, not a health state.
|
||||
|
||||
`fleet_list` adds one block per profile:
|
||||
|
||||
```text
|
||||
profile, maxLoad, live, free, reclaimable
|
||||
```
|
||||
|
||||
For an unlimited profile, `maxLoad` and `free` are null. `free` is
|
||||
`max(0, maxLoad - live)` for a capped profile.
|
||||
|
||||
The view must use the exact live-count function used by placement. A second calculation could show a
|
||||
free slot that placement then refuses. Member rows add `idleForSeconds` only when state is `READY` or
|
||||
`DONE`, no accepted turn exists, and the inbox is empty. `reclaimable` means only that the member
|
||||
holds capacity without open bridge work.
|
||||
|
||||
The existing idle-lead nudge gains a bounded capacity summary. It lists per-profile live, cap, free,
|
||||
and reclaimable counts, plus at most three long-idle members. Capacity does not make
|
||||
`FleetState.hasPending()` true. A changed capacity fingerprint may re-arm one capped heartbeat
|
||||
sequence. The fingerprint excludes changing idle durations, so a static idle fleet cannot reset the
|
||||
cap forever. Reply-push stand-down remains first.
|
||||
|
||||
The bridge must never:
|
||||
|
||||
- spawn a member because a slot is free;
|
||||
- generate a task or acceptance criteria;
|
||||
- move queued work to another member or profile;
|
||||
- treat a free slot or idle member as an incident;
|
||||
- stop an idle member only to improve utilisation.
|
||||
|
||||
The bridge knows capacity facts but has no work list. Only the lead has the plan, task context,
|
||||
side-effect history, and acceptance criteria.
|
||||
|
||||
Capacity calculation is in memory and adds no pane reads. Work-product checks run on a terminal
|
||||
session edge, not every fleet tick.
|
||||
|
||||
This capacity design adds no automatic stop. The accepted `NEVER_READY` cleanup can still stop a
|
||||
very slow startup after the existing grace, which is a known risk. Free capacity and long idle time
|
||||
never trigger that path.
|
||||
|
||||
## 10. Human escalation and notification
|
||||
|
||||
### 10.1 Escalation rule
|
||||
|
||||
Notify a person only when no healthy lead can act:
|
||||
|
||||
- `CONTROL_LINK_DOWN` survives grace;
|
||||
- a lead is unhealthy and no healthy peer can receive the incident;
|
||||
- a member incident has no known owning lead;
|
||||
- the only owning lead becomes unreachable, unresponsive, or stalled;
|
||||
- incident publication or routing itself fails.
|
||||
|
||||
Do not page a person for a member fault while a healthy owning lead exists. An uncollected member
|
||||
incident feeds lead-health evidence. If the lead then becomes unhealthy, peer or human routing starts.
|
||||
|
||||
### 10.2 Detection and notification switches
|
||||
|
||||
`health.enabled` controls detection and bridge-local reporting. It does not require a webhook.
|
||||
|
||||
`health.notifications.mode` is `disabled` or `webhook`. Disabled is valid and is the default.
|
||||
Webhook mode requires a resolved environment variable. Turning notification off stops outbound
|
||||
attempts but keeps incidents. Turning it back on resumes still-open human incidents.
|
||||
|
||||
Without a sink, `fleet_list.healthCoverage` states that human escalation is unavailable. `/healthz`
|
||||
keeps its existing HTTP liveness result and adds a nested `fleetHealth.status=partial` component.
|
||||
Metrics and one startup or reload WARN expose the same limit.
|
||||
|
||||
### 10.3 Incident and delivery deduplication
|
||||
|
||||
One open incident uses this key:
|
||||
|
||||
```text
|
||||
(scope, subjectStableId, state, causeFingerprint)
|
||||
```
|
||||
|
||||
The cause fingerprint includes stable error codes, dependency names, signature ids, or invariant
|
||||
names. It excludes times, ages, retry counts, pane text, and changing digests. A later recurrence
|
||||
after resolution gets a new generation and incident id.
|
||||
|
||||
Each outbound event uses:
|
||||
|
||||
```text
|
||||
Idempotency-Key = hash(incidentId, eventType, eventRevision)
|
||||
```
|
||||
|
||||
Event types are `open`, `severity_changed`, `reminder`, and `resolved`. Transport retries keep the
|
||||
same key.
|
||||
|
||||
An atomic owner-only journal beside the active config stores open incidents, routing, delivered
|
||||
revisions, retry state, and resolution state. It stores no pane or task content. Journal failure does
|
||||
not stop detection, but notification coverage becomes degraded.
|
||||
|
||||
### 10.4 Retry, reminder, and resolve
|
||||
|
||||
Send the first event immediately. Retry network errors, timeouts, HTTP 408, HTTP 429, and HTTP 5xx
|
||||
with full-jitter exponential backoff:
|
||||
|
||||
```text
|
||||
base: 5 seconds
|
||||
factor: 3
|
||||
maximum delay: 15 minutes
|
||||
one outstanding attempt per event
|
||||
```
|
||||
|
||||
Respect `Retry-After` up to 15 minutes. Other HTTP 4xx responses are permanent for that event until
|
||||
config changes or a person requests replay.
|
||||
|
||||
Transport retry is not an incident reminder. `humanRepeatSeconds` creates a new reminder revision
|
||||
for an unresolved critical incident after the last successful human event. Disabled mode does not
|
||||
build an unbounded reminder queue.
|
||||
|
||||
Send `resolved` only if at least one human event for that incident was delivered. If an incident
|
||||
resolves before its first successful delivery, cancel the pending open event and record local
|
||||
resolution.
|
||||
|
||||
### 10.5 Outbound payload boundary
|
||||
|
||||
An outbound payload may contain incident id and event type, severity, state, scope, stable bridge
|
||||
ids, role or profile, times, duration, structured evidence type and counts, recovery attempted,
|
||||
routing reason, coverage, and safe tool calls.
|
||||
|
||||
It must never contain:
|
||||
|
||||
- raw pane text, pane excerpts, or pane digests;
|
||||
- task briefs, prompts, or member reply content;
|
||||
- source files, diffs, or worktree file content;
|
||||
- worktree paths;
|
||||
- environment values, tokens, credentials, headers, or webhook URL;
|
||||
- raw exception messages or stack traces;
|
||||
- arbitrary model output.
|
||||
|
||||
The sink response body is ignored. A webhook cannot direct recovery. n8n remains outbound-only.
|
||||
|
||||
### 10.6 Metrics
|
||||
|
||||
M4 adds bounded-label series:
|
||||
|
||||
```text
|
||||
fleet_health_incidents{scope,state,severity}
|
||||
fleet_health_incidents_total{event}
|
||||
fleet_health_notifications_total{event,outcome}
|
||||
fleet_health_notification_queue_depth
|
||||
fleet_health_notification_last_success_seconds
|
||||
fleet_health_notification_capability{mode,status}
|
||||
fleet_lead_health{lead,state}
|
||||
fleet_lead_assigned_incidents{lead}
|
||||
```
|
||||
|
||||
Metric labels never include terminal ids, incident ids, URLs, or error text.
|
||||
|
||||
## 11. Configuration
|
||||
|
||||
The optional `health:` block is absent or disabled by default. The dormant monitor scheduler does no
|
||||
herdr or pane work while disabled. Every listed key is hot because the monitor reads `ConfigRef` on
|
||||
each tick or notification.
|
||||
|
||||
| Key | Class | Default and hard bound | Purpose |
|
||||
|---|---|---|---|
|
||||
| `health.enabled` | Hot | `false` | Enable detection and bridge-local reporting. |
|
||||
| `health.snapshotIntervalSeconds` | Hot | default 30, minimum 15 | Whole-fleet comparison cadence. |
|
||||
| `health.workingSuspectAfterSeconds` | Hot | default 600, minimum 300 | Age before working-pane probes. |
|
||||
| `health.paneProbeIntervalSeconds` | Hot | default 60, minimum 60 | Per-target pane cooldown. |
|
||||
| `health.leadUnresponsiveAfterSeconds` | Hot | default 300, minimum 120 | Delay after exhausted actionable nudges before lead fault. |
|
||||
| `health.humanRepeatSeconds` | Hot | default 3600, minimum 900 | Minimum repeat period for one open human incident. |
|
||||
| `health.capacityLongIdleAfterSeconds` | Hot | default 900, minimum 300 | Long-idle threshold for capacity summaries. |
|
||||
| `health.includePaneExcerpt` | Hot | `false` | Allow a clipped excerpt in local lead reports only. Human payloads still exclude it. |
|
||||
| `health.notifications.mode` | Hot | `disabled` | Select `disabled` or `webhook`. |
|
||||
| `health.notifications.webhookUrlEnv` | Hot | required in webhook mode | Name of the environment variable that holds the sink URL. |
|
||||
| `health.notifications.requestTimeoutMs` | Hot | default 10000, range 1000-30000 | Whole webhook request limit. |
|
||||
|
||||
Two consecutive snapshots are compiled floors for lost boundary, lead disappearance, and control
|
||||
link failure. The two-pane-reads-per-tick limit is also compiled and cannot be weakened by config.
|
||||
|
||||
## 12. Delivery units and acceptance
|
||||
|
||||
### Unit 1 - Evidence model and fleet snapshot
|
||||
|
||||
Scope: health state model, fleet join, clocks, evidence retention, and pane budget.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. One `agent.list` call covers one enabled fleet tick.
|
||||
2. Pure decision tests cover every state and every evidence limit in Section 4.
|
||||
3. `BUSY` plus stable raw `DONE` opens `TURN_BOUNDARY_LOST` after two snapshots.
|
||||
4. Healthy fleet snapshots perform zero pane reads.
|
||||
5. Pane cooldown, two-read fleet budget, and fair rotation cannot be disabled by config.
|
||||
6. Logs are outputs only; no log parsing exists.
|
||||
7. Fleet snapshots expose the same profile live-count calculation that placement uses.
|
||||
8. Capacity rows report cap, live, free, and reclaimable values without opening incidents.
|
||||
|
||||
### Unit 2 - Lost boundary and task reconciliation
|
||||
|
||||
Scope: accepted-turn identity, guarded repair, target-wide teardown, release causes, and preserved
|
||||
worktree discovery.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Every accepted send receives a stable `TurnToken` tied to target, exact waiter, and delivery
|
||||
baseline.
|
||||
|
||||
**Corrected during implementation (2026-08-15).** This criterion first also required the session
|
||||
turn number and the task outcome. That is not implementable at this layer, and the implementer
|
||||
refused it three times rather than fabricate a value — correctly. The reason is an ordering fact
|
||||
that is invisible from any single class: `MessageService` owns acceptance and holds the waiter and
|
||||
the async `Task`, but it learns nothing about delivery, because the delivery event goes to
|
||||
`CompletionResolver` through `TurnListener.onDelivered`. And `CompletionResolver.onDelivered` runs
|
||||
*before* `SessionManager.onDelivered`, so the session turn number does not exist yet at the only
|
||||
point where the token could capture it.
|
||||
|
||||
Two ways out were rejected. A shared registry keyed by target reintroduces exactly the "whichever
|
||||
send happens to be waiting" ambiguity the token exists to remove — the same weak claim
|
||||
`Rendezvous.currentWaiter` warns about. Injecting a turn counter into `MessageService` adds a
|
||||
required cross-layer dependency to populate a field that nothing in this slice reads, which is
|
||||
speculative coupling across a boundary already shown to be fragile.
|
||||
|
||||
So the token identifies the **accepted send**, and `SessionManager` keeps verifying its own
|
||||
delivery separately. Repair (criterion 2) does need the session turn; binding it means resolving
|
||||
that acceptance-versus-delivery ordering first, and that work belongs to the repair unit, not
|
||||
here. The token record carries a comment saying the field is deliberately absent.
|
||||
2. Repair requires the same `BUSY` token, two raw `IDLE` or `DONE` snapshots, no conflicting
|
||||
observation, exact open waiter, successful baseline, and new recognised assistant output.
|
||||
3. Missing, failed, late, or post-restart baseline never authorises repair.
|
||||
4. Repair is enabled only for agent kinds with tested assistant-block extraction. Raw-text fallback
|
||||
without a recognised marker refuses repair.
|
||||
5. Normal completion and repair use one resolver guard core. Waiter, scrape, clipping, unchanged, and
|
||||
exact-turn guards are not duplicated.
|
||||
6. `reconcileLostBoundary` returns every typed result named in Section 8.3.
|
||||
7. Only `REPAIRED` and same-turn `ALREADY_RESOLVED` may move the same turn to `DONE`.
|
||||
8. A per-target reconciliation gate blocks a queued second send during repair and FSM update.
|
||||
9. Repaired completion has distinct rendezvous kind, message outcome, poll source, lead marker, and
|
||||
metric. Clipping keeps its extra marker.
|
||||
10. Unchanged, unreadable, or ambiguous evidence gets at most one delayed retry. Missing capture or
|
||||
baseline gets none.
|
||||
11. Refusal leaves the ticket pending and tells the lead that no result was rebuilt or replayed.
|
||||
12. Release, gone, never-ready, and abnormal stop use CB-568's one idempotent target-wide failure
|
||||
operation.
|
||||
13. The post-teardown invariant in Section 8.5 is tested independently of CB-568 internals.
|
||||
14. A violated teardown invariant creates `DELEGATION_ORPHANED` and retries only the idempotent
|
||||
failure operation.
|
||||
15. `SPAWN_ROLLBACK` and normal `COMPLETED` remove worktrees. Abnormal and shutdown causes preserve
|
||||
them.
|
||||
16. Explicit stop is state-aware. Any pending task or non-terminal state preserves the worktree.
|
||||
17. Atomic preserved-worktree manifests reload after restart and appear in lead-only
|
||||
`fleet_list.preservedWorktrees`.
|
||||
18. Manifest failure preserves the worktree and opens an operator-visible health failure.
|
||||
19. Provision records the base commit. Terminal, long-idle worktrees report
|
||||
`WORK_PRODUCT_AT_RISK` only under the evidence in Section 4.2 and never auto-delete work.
|
||||
20. No path replays a delivered task, rebuilds its brief, or retargets it, even when prompt text is
|
||||
available.
|
||||
21. Tests cover both real traces, all repair refusals, clipping, explicit-reply and next-turn races,
|
||||
restart without capture, concurrent send and release, and preserved discovery after restart.
|
||||
|
||||
#### Unit 2 - what has landed so far
|
||||
|
||||
Checked against `main` at `e09cac6` on 2026-08-15. Unit 2 was written as one block, but parts of it
|
||||
have since been built by separate CB tickets. Read this before planning the rest, or that work gets
|
||||
done twice.
|
||||
|
||||
The check was a symbol survey of `fleetd/src/main/java` plus the merge history. It tells you whether
|
||||
the machinery exists at all. It is **not** a line-by-line audit of whether each criterion is fully
|
||||
met, and I did not run one.
|
||||
|
||||
| Criterion | Marker searched for | Found in main source | Reading |
|
||||
|---|---|---|---|
|
||||
| 1 | `TurnToken` | 8 files | **Done** — unit 2a, merged as `fec284e`. Criterion 1 was corrected first; see the note under it. |
|
||||
| 2-5, 9 | `REPAIRED` | 0 files | Not started. The whole guarded-repair path is absent. |
|
||||
| 6, 7, 10 | `reconcileLostBoundary` | 0 files | Not started. |
|
||||
| 12 | CB-568 failure operation | via CB-580 | **Partial.** CB-580 (`0af902e`) routes `GONE` and `NEVER_READY` into the one idempotent target-wide failure. I did not check that release and abnormal stop go through the same call. |
|
||||
| 14 | `DELEGATION_ORPHANED` | 3 files | **Partial.** The health state exists. The teardown-invariant check that creates it, and the retry rule, do not. |
|
||||
| 15 | `SPAWN_ROLLBACK` | 0 files | **Contradicted — see below.** |
|
||||
| 16 | — | — | Partial at best. CB-576 made release preserve a dirty worktree; whether explicit stop is state-aware is not checked. |
|
||||
| 17, 18 | `preservedWorktrees` | 0 files | Not started. No manifest, and no lead-only `fleet_list` field. |
|
||||
| 19 | `WORK_PRODUCT_AT_RISK` | 0 files | Not started. |
|
||||
|
||||
**Criterion 15 no longer matches the code, and the code is right.** It says "normal `COMPLETED`
|
||||
remove worktrees". Since CB-576 (`500bfa2`) that is false on purpose: a `COMPLETED` release now
|
||||
preserves the worktree when it still holds uncommitted work, because deleting it destroys work
|
||||
nobody can get back. CB-576 was filed after exactly that loss. CB-581 goes further — if the
|
||||
dirty-check itself fails, the worktree is preserved rather than removed, since "we could not tell"
|
||||
must not be treated as "it is clean".
|
||||
|
||||
So criterion 15 should be rewritten as: `SPAWN_ROLLBACK` and a `COMPLETED` release with a **clean**
|
||||
worktree remove it; abnormal causes, shutdown, a dirty worktree, and a failed dirty-check all
|
||||
preserve it. `SPAWN_ROLLBACK` itself does not exist yet.
|
||||
|
||||
### Unit 3 - Typed inbox and member routing
|
||||
|
||||
Scope: semantic record, AMQP migration, both adapters, member routing, polling, and member health in
|
||||
`fleet_list`.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. AMQP selects legacy or typed decoding only from `content_type`; it never sniffs the body.
|
||||
2. Persistent `text/plain` from the old build becomes `kind=reply` with exact UTF-8 content,
|
||||
including content beginning with `{`.
|
||||
3. New entries use the vendor media type, `schemaVersion: 1`, UTF-8, persistent delivery, and AMQP
|
||||
message ids.
|
||||
4. Version 1 ignores unknown optional fields but rejects missing fields and identity mismatch.
|
||||
5. Unknown versions are not decoded or acked. They remain on the original queue and create one
|
||||
deduplicated failure.
|
||||
6. Invalid known data never escapes the callback, appears as a reply, or blocks later valid messages.
|
||||
7. Invalid data reaches durable per-target quarantine before original ack. Failed handoff leaves the
|
||||
original unacked.
|
||||
8. Decode failures create redacted WARN, metric, `fleet_list` summary, and routed incident without
|
||||
raw content.
|
||||
9. Both adapters pass one semantic contract for fields, FIFO, dedup, ownership, ack, and release.
|
||||
10. Lead keys require explicit ownership. Publication never claims a queue.
|
||||
11. Unit codec tests cover legacy `{`, Unicode, malformed UTF-8, typed round trip, additive fields,
|
||||
malformed JSON, missing fields, identity mismatch, media type, version, and dedup.
|
||||
12. A live broker contract writes old wire data and reads it with the new adapter after reconnect.
|
||||
13. Live contract tests cover mixed entries, quarantine confirm-before-ack, unsupported redelivery,
|
||||
later progress past poison, property persistence, lead ownership, and ack removal.
|
||||
14. Safe downgrade is documented as unsupported.
|
||||
15. RabbitMQ contract tests pass with `mvn test -Pcontract`. The same cases run once on production
|
||||
LavinMQ, or the release states that LavinMQ was not checked.
|
||||
16. Member incidents route to the exact delegating lead and never resolve a task rendezvous.
|
||||
17. `fleet_list` shows compact member health and capacity without pane content. Member
|
||||
`idleForSeconds` is present only when no accepted turn or inbox item exists.
|
||||
|
||||
### Unit 4 - Lead health and peer routing
|
||||
|
||||
Scope: lead evidence, exact ownership, peer selection, explicit-recipient push, and lead inbox
|
||||
lifecycle.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. Lead identity uses the `CallerResolver` supplier. Liveness uses successful current agent data.
|
||||
2. Two successful-list absences with healthy ping become `LEAD_UNREACHABLE`; global link failure does
|
||||
not.
|
||||
3. Raw `WORKING`, raw `UNKNOWN`, first-seen time, failures, last success, and error class persist
|
||||
across ticks.
|
||||
4. Heartbeat and push publish status and nudge outcomes before safe no-injection decisions.
|
||||
5. Dynamic lead identity survives a two-successful-snapshot retirement grace.
|
||||
6. Member incidents first use exact delegation ownership with no singular-primary fallback.
|
||||
7. Peer selection follows the exclusions, load rule, and stable tie break in Section 7.4.
|
||||
8. A selected working peer is not interrupted. Its push waits for an injectable window.
|
||||
9. Recipient assignment stays pinned. Reassignment increments generation and supersedes old pending
|
||||
assignment.
|
||||
10. `fleet_list` shows bounded foreign assignments, recipient, reason, and generation without pane
|
||||
content.
|
||||
11. `LeadInboxRegistry` owns configured and discovered lead keys before publication.
|
||||
12. Missing leads keep ownership. Retirement needs an empty queue and handled incidents.
|
||||
13. Replacement owns the new key before messages move. Non-empty in-memory keys are not released.
|
||||
14. Tests cover dead versus busy, unknown, global failure, stale scan cache, disappearance, one peer,
|
||||
several peers, reassignment, and no peer.
|
||||
15. Adapter tests cover lead ownership, restart re-ownership, retirement, and terminal replacement.
|
||||
LavinMQ is checked or named as unchecked.
|
||||
16. A sole unreachable or stalled lead is never restarted or replaced. Without a sink, only passive
|
||||
evidence remains and every coverage surface says so.
|
||||
|
||||
### Unit 5 - Human sink, hot config, metrics, and operator coverage
|
||||
|
||||
Scope: generic webhook, config split, incident journal, retry, resolve, metrics, example config, and
|
||||
operator documentation.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
1. `health.enabled` works without a human sink.
|
||||
2. Notification mode is hot, defaults to disabled, and supports disabled or webhook.
|
||||
3. Webhook mode requires a resolved environment value. Bad notification config does not disable an
|
||||
already valid detector.
|
||||
4. Mode changes keep open incidents. Re-enable resumes eligible incidents.
|
||||
5. `fleet_list`, `/healthz`, metrics, and one WARN show partial coverage without a sink. HTTP
|
||||
liveness behavior stays unchanged.
|
||||
6. One-lead, no-sink coverage states that lead failure has no active notification or recovery.
|
||||
7. Incident and outbound dedupe use the stable keys in Section 10.3.
|
||||
8. The owner-only local journal survives restart and contains no pane or task content.
|
||||
9. Journal failure keeps detection running but marks notification coverage degraded.
|
||||
10. Retry tests cover network failure, timeout, 408, 429, `Retry-After`, 5xx, permanent 4xx, jitter,
|
||||
delay cap, config re-arm, and one outstanding attempt.
|
||||
11. Reminders and transport retries remain separate. Disabled mode does not build an unbounded queue.
|
||||
12. Resolve sends only after an earlier human event succeeded. Resolve-before-delivery cancels stale
|
||||
open delivery.
|
||||
13. Metrics use bounded labels and exclude ids, URLs, and error text.
|
||||
14. Payload tests reject every content type forbidden in Section 10.5.
|
||||
15. Webhook response bodies are ignored and cannot direct recovery.
|
||||
16. Tests cover disabled mode, one lead without sink, open/update/reminder/resolve, restart, dedup,
|
||||
reassignment, disable/re-enable, and sink failure while local health continues.
|
||||
17. `fleetd.example.yaml` documents all hot keys and compiled floors.
|
||||
18. The operator Features wiki is updated separately. The portable `CLAUDE.md` block is checked and
|
||||
changed only if shipped tool or inbox semantics make it untrue.
|
||||
19. `mvn clean install` passes.
|
||||
|
||||
## 13. Not checked and release gates
|
||||
|
||||
These limits are part of the design, not optional follow-up notes.
|
||||
|
||||
- **OpenCode pane status and assistant markers were not checked.** OpenCode lost-boundary repair is
|
||||
disabled until live fixtures exist.
|
||||
- **Permission-prompt status was not checked** for Claude Code or OpenCode. `BLOCKED` remains
|
||||
ambiguous and has no automatic action.
|
||||
- **`recent_unwrapped` stability was not checked** across all supported agent kinds. If normalisation
|
||||
is not stable, `STALL_SUSPECTED` must say its evidence is weaker.
|
||||
- **The real `BUSY + DONE` trace was not replayed against live herdr.** The design uses the observed
|
||||
production trace and current poller behavior.
|
||||
- **CB-568 was not present when Unit 2 was designed.** Unit 2 must inspect the landed API and keep its
|
||||
independent teardown invariant.
|
||||
- **Production LavinMQ was not checked.** Existing durable-inbox contracts use RabbitMQ. Migration,
|
||||
quarantine, redelivery, lead ownership, and reassignment must run on LavinMQ before release or be
|
||||
recorded as unchecked.
|
||||
- **Live multi-lead routing was not checked.** Peer choice and reassignment are design rules backed by
|
||||
fake-clock and adapter tests until a live exercise runs.
|
||||
- **A live sole-lead failure with a webhook was not checked.** The no-peer path is a design result,
|
||||
not a tested recovery.
|
||||
- **No n8n, Slack, PagerDuty, or other receiver was checked.** The webhook remains generic and
|
||||
outbound-only.
|
||||
- **Deployment supervisor behavior for nested `/healthz` fields was not checked.** HTTP liveness
|
||||
status stays unchanged to reduce this risk.
|
||||
- **Incident-journal crash behavior was not checked** because the journal does not exist yet. Unit 5
|
||||
must test atomic replacement and restart recovery.
|
||||
- **Worktree merge state cannot be checked reliably** without forge or explicit collection evidence.
|
||||
`WORK_PRODUCT_AT_RISK` stays a warning.
|
||||
|
||||
## 14. Locked exclusions
|
||||
|
||||
M4 does not expose `agent.read` as a bridge tool. It does not add a workflow engine, inbound n8n
|
||||
authority, automatic task assignment, task replay, automatic lead replacement, or automatic member
|
||||
spawn for free capacity.
|
||||
|
||||
The bridge remains a message bus with evidence and bounded mechanical repair. The lead remains the
|
||||
place where judgement and work planning happen.
|
||||
@@ -0,0 +1,389 @@
|
||||
# MCP Contract — `fleetd`'s unified gateway
|
||||
|
||||
> **Status: 🔴 HISTORICAL DESIGN — do NOT use as the tool reference.** Written 2026-07-14, before
|
||||
> any MCP code existed. The system shipped and this page never caught up, so **its tool names,
|
||||
> parameter names and REST paths are wrong today**. Audited 2026-08-17; the specific drift:
|
||||
>
|
||||
> - **Tools it names that do not exist:** `fleet_read`, `fleet_cancel`.
|
||||
> - **Shipped tools it omits:** `fleet_poll`, `fleet_ack`, `fleet_profiles`, `fleet_whoami`.
|
||||
> - **Parameter names are wrong nearly everywhere** — it says `message`/`target`/`timeout_seconds`/
|
||||
> `block` where the code takes `content`/`sessionId`/`timeoutMs`/`wait`; `text` where
|
||||
> `fleet_reply` takes `content`; `target` where `fleet_stop` takes `paneId`.
|
||||
> - **REST paths are wrong:** it says `POST /workers` and `DELETE /workers/{paneId}`; the daemon
|
||||
> serves `POST /members` and `DELETE /members/{paneId}`.
|
||||
>
|
||||
> **The authoritative tool surface is the live MCP schema** (each tool's own description and
|
||||
> parameters, as mounted), with the intent→tool table in `CLAUDE.md` as the short form. Both were
|
||||
> checked against `mcp/FleetMcp.java` on 2026-08-17 and are accurate.
|
||||
>
|
||||
> What is still worth reading here is **§6 — the flows and the error model** (rendezvous,
|
||||
> `fleet_ask`, detached delivery, the turn-done fallback). The shapes it describes are the ones
|
||||
> that shipped; only the names around them drifted. Rewriting this page is tracked as **CB-609**.
|
||||
|
||||
`fleetd` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `fleetd` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `fleetd` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`fleetd` owns policy; herdr owns PTYs.** MCP tools express *intent*; `fleetd`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["fleetd — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>fleet_send · fleet_reply<br/>fleet_ask · fleet_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"fleet_send (blocks)"| MCP
|
||||
W -.->|"fleet_reply / fleet_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `fleetd` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `fleetd` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `fleet_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `fleet_reply` / `fleet_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`fleetd` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http fleetd http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`fleet_send`](#fleet_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`fleet_reply`](#fleet_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`fleet_ask`](#fleet_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`fleet_status`](#fleet_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`fleet_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`fleet_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`fleet_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`fleet_read`](#fleet_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`fleet_cancel`](#fleet_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `fleet_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `fleet_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `fleet_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `fleet_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `fleet_status` on a split-host primary.
|
||||
|
||||
#### `fleet_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `fleetd` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `fleet_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `fleet_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `fleet_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`fleet_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`fleet_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`fleet_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `fleet_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `fleet_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `fleet_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as fleetd (MCP + Injector)
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: fleet_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`fleet_ask`)
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: fleet_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: fleet_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve fleet_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: fleet_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
|
||||
P->>B: fleet_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: fleet_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
|
||||
A worker that never calls `fleet_reply` still returns a result: `fleetd` reads its terminal
|
||||
tail when the turn completes.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls fleet_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `fleet_send` is simply its producer.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
end note
|
||||
note right of BLOCKED
|
||||
injectable — deliver head of FIFO
|
||||
end note
|
||||
note right of WORKING
|
||||
NOT injectable — counts as pickup
|
||||
end note
|
||||
note right of UNKNOWN
|
||||
NOT injectable, NOT a pickup — wait
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
|
||||
---
|
||||
|
||||
## 8. Error model
|
||||
|
||||
| Condition | `fleet_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `fleet_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`fleet_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `fleet_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `fleet_reply` / `fleet_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `fleet_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `fleet_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `fleet_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `FleetApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`fleet_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `fleet_send` (keeps the catalog
|
||||
small) vs. a separate `fleet_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `fleet_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `fleet_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `fleet_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `fleet_reply` / `fleet_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`fleet_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `fleet_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
@@ -0,0 +1,119 @@
|
||||
# v1.0.0 — One leader, one host, complete
|
||||
|
||||
This is the first release of **`fleetd`**.
|
||||
|
||||
`fleetd` lets one main Claude Code session (the **leader**, on your Pro/Max subscription)
|
||||
run a team of **workers** — extra Claude Code sessions on a cheaper or local model, and
|
||||
non-Claude agents too. The leader's own session is never touched: it stays on subscription,
|
||||
with a clean environment.
|
||||
|
||||
**This release finishes a full scope — it does not stop halfway.** The scope is *one leader
|
||||
on one machine*, running many workers. Everything that setup needs is now built, tested, and
|
||||
used daily: who-is-who, the subscription line, messages in both directions, worker start and
|
||||
stop, security, and monitoring. Nothing on the single-machine path is left as a known gap.
|
||||
|
||||
This is also how the code is shaped: `PrimaryRegistry` holds exactly **one** leader. Running
|
||||
across many machines is the next big step (see *What comes next* below) — not a missing piece
|
||||
of this one.
|
||||
|
||||
## One gateway for all messages
|
||||
|
||||
- **Everyone talks through the same door.** The leader and every worker connect to the same
|
||||
MCP server and use only its tools: `fleet_whoami` · `fleet_profiles` · `fleet_spawn` ·
|
||||
`fleet_list` · `fleet_status` · `fleet_send` · `fleet_reply` · `fleet_ask` ·
|
||||
`fleet_poll` · `fleet_ack` · `fleet_stop`.
|
||||
- **You are who your connection says you are.** The bridge finds out who is calling from the
|
||||
connection itself, never from a name the caller sends. So a worker cannot pretend to be
|
||||
someone else, and `fleet_whoami` tells each agent its own role — no guessing.
|
||||
- **The subscription line cannot be crossed.** Only a spawned worker gets
|
||||
`ANTHROPIC_BASE_URL`; the leader never does. Each worker profile has a list of allowed
|
||||
model hosts, checked before anything starts.
|
||||
- **Messages wait for the right moment.** The bridge sends one message per turn, only when
|
||||
the other side is ready — no hammering a busy agent.
|
||||
|
||||
## Worker lifecycle
|
||||
|
||||
- **Start → work → stop.** Each worker gets its own git worktree (its own copy of the repo)
|
||||
with the same setup as the leader — `CLAUDE.md`, skills, hooks — so it commits on its own
|
||||
branch and opens its own PR. Ready-made playbooks ship in the repo:
|
||||
`.claude/skills/implementer` and `.claude/skills/reviewer`.
|
||||
- **Fast failure, not a silent hang.** Starting a worker waits until it is really connected.
|
||||
If it never connects, you get a clear error (`PeerUnreachableException`) instead of a stuck
|
||||
send.
|
||||
- **Workers don't live forever.** Idle workers are cleaned up (`idle_ttl`), long sessions have
|
||||
a turn limit (`context_cap`), shutdown drains work first, and workers left behind by an old
|
||||
daemon are found and removed at startup.
|
||||
- **Predictable placement.** Each worker gets its own tab in a shared worker space, in the
|
||||
same order every time.
|
||||
|
||||
## No reply gets lost
|
||||
|
||||
MCP only lets the client call the server, so the bridge could push to a worker but the leader
|
||||
had to ask for its replies. That gap is now closed on a single machine:
|
||||
|
||||
- **Replies are kept, never dropped.** If a reply arrives and nobody is waiting, the bridge
|
||||
holds it until the leader picks it up.
|
||||
- **Replies can survive a restart.** With a broker (LavinMQ or RabbitMQ) set up, held replies
|
||||
live on the broker, so a daemon restart does not lose them — they come back, and repeats are
|
||||
filtered out by `msgId`. No `broker:` in the config → replies are held in memory instead.
|
||||
- **The leader gets a tap on the shoulder.** When a reply lands, the bridge nudges the
|
||||
leader's own pane — only when the leader is free, and only a few times. If the leader is on
|
||||
another machine, this quietly falls back to pick-up mode; the reply still waits.
|
||||
- **Workers can ask questions.** With `fleet_ask`, a worker can pause mid-task, ask the
|
||||
leader something, and continue the *same* task with the answer.
|
||||
|
||||
## More than one kind of worker
|
||||
|
||||
Workers are started through a small plug-in interface (`PeerLauncher`). Two plug-ins ship:
|
||||
one for Claude Code and one for **opencode** (tested live against opencode 1.18.5). The
|
||||
opencode one proves the interface is neutral — it uses nothing Claude-specific. Each worker
|
||||
only sees the tools its own launcher gives it.
|
||||
|
||||
## Security & operations
|
||||
|
||||
- **Auth.** Default is `loopback-trust`: only same-machine callers are trusted. Or set a
|
||||
bearer `token`. Unknown callers count as `ANONYMOUS` — nobody is trusted by accident. If
|
||||
the config would expose the daemon to the network without a token, it **refuses to start**.
|
||||
Workers never need the token, so turning auth on cannot lock them out.
|
||||
- **Rules + audit log.** The role rules are checked on both doors (REST and MCP). A worker
|
||||
may only act as itself. The audit log is JSON and never contains message text.
|
||||
- **Monitoring.** `/healthz` for liveness, `/metrics` for Prometheus — no extra libraries.
|
||||
- **Runs as a service.** launchd (macOS) and systemd (Linux) files are included. If herdr
|
||||
isn't up yet at boot, the daemon waits up to 30 seconds and then runs in a reduced mode
|
||||
instead of crash-looping.
|
||||
- **CI.** Every push builds and tests on the Gitea runner, including the broker test against
|
||||
a real broker. Only the live-herdr test stays local (`-Pcontract`).
|
||||
|
||||
## What you need
|
||||
|
||||
Java 25 · Maven · **herdr 0.8.0 (protocol 19)** · optionally LavinMQ or RabbitMQ for
|
||||
restart-proof replies · macOS (launchd) or Linux (systemd). For TLS, put a reverse proxy in
|
||||
front — the daemon does not do TLS itself, by design.
|
||||
|
||||
## Tested
|
||||
|
||||
`mvn clean install` is green at `84081b2`: **399 tests**, coverage **75.6%** of instructions /
|
||||
**64.5%** of branches. Live end-to-end runs under `e2e/`: a worker asking the leader a
|
||||
question, one leader running several workers at once on a bug hunt, and a 30-turn
|
||||
back-and-forth conversation. The bridge is used on itself — worker-run code reviews have led
|
||||
to real committed fixes in this repo.
|
||||
|
||||
## What comes next
|
||||
|
||||
The single-machine story is done. The next big step stretches the same rules across machines:
|
||||
|
||||
- **Many machines (CB-308).** A leader on machine A with workers on machines B and C — built
|
||||
on this release's broker layer, with one gateway per machine. The design is written
|
||||
(`docs/CB-308-Multi-Host-Federation.md`); the open question is trust between machines.
|
||||
- **Many leaders.** Today the bridge holds exactly one leader. The next idea is a small
|
||||
council — for example one Claude and one Codex — that can discuss a problem together,
|
||||
compare answers, and agree on a decision before the work is handed to workers. This needs
|
||||
leader-to-leader messages and a simple way to settle disagreement, neither of which exists
|
||||
yet.
|
||||
- **Third-party launcher plug-ins.** Loading launcher plug-ins from outside the project needs
|
||||
a trust model first, because a launcher runs with daemon rights and can hand secrets to
|
||||
workers.
|
||||
|
||||
One thing we chose **not** to build, so it isn't read as a gap: the originally planned heavy
|
||||
message envelope (CB-201). Connection identity already routes every reply to the right place,
|
||||
so only a small `QUESTION` message kind and a `turn_id` were added.
|
||||
+154
@@ -0,0 +1,154 @@
|
||||
# Team — lead orchestrating a mixed Claude + local-LLM fleet
|
||||
|
||||
The message server (`fleetd`) delivers **one turn into one worker**. A **team** is the
|
||||
layer above it: a **Claude team-lead** that fans a job out across a **mixed fleet** of
|
||||
workers — some on Claude, some on the remote local LLM — and reduces their replies. Same
|
||||
`fleetd` delivery, same subscription boundary; this doc is only about **orchestration** —
|
||||
who the workers are, how the lead picks one, and how it runs many at once.
|
||||
|
||||
> Delivery mechanics (blocking `POST /message`, status-gated reply envelope) live in the
|
||||
> Message-Server design. Transport rationale is in Approaches. This doc assumes both.
|
||||
|
||||
## The team
|
||||
|
||||
- **Team-lead** — the primary **Opus** (Claude Code, env **CLEAN**, on Pro/Max). Not a
|
||||
worker; a **thin client of `fleetd`**. It plans, routes, dispatches, and integrates, and
|
||||
never sets `ANTHROPIC_BASE_URL`.
|
||||
- **Workers** — a herd of `claude` panes in herdr, each an addressable `fleetd` session
|
||||
with its **own model/env**:
|
||||
- **Claude workers** (clean env, e.g. Sonnet) — reasoning-heavy or high-accuracy subtasks.
|
||||
- **Local workers** (`ANTHROPIC_BASE_URL=https://llm.ltms.dev/anthropic`) — bulk, cheap, or
|
||||
embarrassingly parallel subtasks.
|
||||
|
||||
Every worker is still a *real Claude Code process* (inherits `CLAUDE.md`, hooks, skills,
|
||||
MCP) — only its model differs. Scale each kind horizontally by adding panes.
|
||||
|
||||
### Topology
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
LEAD["lead — Opus<br/>(Claude Code, env CLEAN)"]
|
||||
BD["fleetd<br/>message server + router"]
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
WC1["w-claude-1<br/>Sonnet · CLEAN"]
|
||||
WC2["w-claude-2<br/>Sonnet · CLEAN"]
|
||||
WL1["w-local-1<br/>ANTHROPIC_BASE_URL set"]
|
||||
WL2["w-local-2<br/>ANTHROPIC_BASE_URL set"]
|
||||
ANT["api.anthropic.com<br/>(Pro/Max)"]
|
||||
OLL["llm.ltms.dev<br/>(gateway to the local model)"]
|
||||
|
||||
LEAD -->|"blocking POST /message (target role)"| BD
|
||||
BD -->|"Unix socket · send_text · events.subscribe"| HERDR
|
||||
HERDR --> WC1 & WC2 & WL1 & WL2
|
||||
WC1 --> ANT
|
||||
WC2 --> ANT
|
||||
WL1 --> OLL
|
||||
WL2 --> OLL
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
classDef local fill:#6b46c1,stroke:#44337a,color:#ffffff;
|
||||
class LEAD,WC1,WC2 ext
|
||||
class BD,HERDR core
|
||||
class WL1,WL2 local
|
||||
```
|
||||
|
||||
### Roles & routing
|
||||
|
||||
| Role | Env | Model | Route here when… |
|
||||
|---|---|---|---|
|
||||
| `lead` | clean | Opus (sub) | always — it does the routing |
|
||||
| `w-claude-*` | clean | Sonnet (sub) | task needs Claude-grade reasoning / careful edits |
|
||||
| `w-local-*` | `ANTHROPIC_BASE_URL` set | local LLM | task is bulk / cheap / embarrassingly parallel |
|
||||
|
||||
The lead applies this rubric itself, guided by its `CLAUDE.md` team charter (below). Worker
|
||||
selection is **policy in the lead**, not a `fleetd` concern — `fleetd` just delivers to
|
||||
the session the lead names.
|
||||
|
||||
## Subscription boundary in a team
|
||||
|
||||
Unchanged from the base architecture, and it scales with the fleet: **only local-worker
|
||||
panes** launch with `ANTHROPIC_BASE_URL`. The lead and every Claude worker stay env-clean on
|
||||
the subscription. `fleetd` enforces which panes may carry the off-subscription env, so
|
||||
adding workers never widens the boundary.
|
||||
|
||||
## Parallel fan-out (map / reduce)
|
||||
|
||||
The lead's advantage over a single bridge is **concurrency**: independent subtasks go to
|
||||
different workers at once, then results are gathered.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant L as lead (Opus)
|
||||
participant B as fleetd
|
||||
participant WC as w-claude-1
|
||||
participant WL as w-local-1
|
||||
|
||||
Note over L: split job → subtask A (reasoning), subtask B (bulk)
|
||||
par A → Claude worker
|
||||
L->>B: POST /message {role: w-claude, prompt: A}
|
||||
B->>WC: send_text into running pane
|
||||
WC-->>B: status working → idle + Stop-hook envelope
|
||||
B-->>L: 200 reply A
|
||||
and B → local worker
|
||||
L->>B: POST /message {role: w-local, prompt: B}
|
||||
B->>WL: send_text into running pane
|
||||
WL-->>B: status working → idle + Stop-hook envelope
|
||||
B-->>L: 200 reply B
|
||||
end
|
||||
Note over L: reduce → integrate A + B into final answer
|
||||
```
|
||||
|
||||
- **Map:** the lead issues N concurrent blocking `POST /message` calls (one per subtask → its
|
||||
chosen worker). Each call blocks only *that* request; `fleetd` holds it open until the
|
||||
worker's turn completes (status-gated) and returns the reply envelope.
|
||||
- **Reduce:** the lead collects the N envelopes and integrates. A slow local worker never
|
||||
blocks a fast Claude worker — wall-clock ≈ the slowest single subtask, not the sum.
|
||||
- **Detached / long jobs** use the async broker path instead of a held request (Channel 2 in
|
||||
the base architecture), so the lead never busy-polls across turns.
|
||||
|
||||
Fan-out is bounded by the herd size (pane count) and `fleetd`'s concurrency policy, not by
|
||||
the lead.
|
||||
|
||||
## Knowing the roster
|
||||
|
||||
The lead discovers its team from `fleetd` (session list / roles) rather than hard-coding
|
||||
pane ids, so workers can be added or restarted without editing the lead. A minimal charter
|
||||
in the lead's `CLAUDE.md` turns Opus into the orchestrator:
|
||||
|
||||
```markdown
|
||||
## Your team (via fleetd)
|
||||
You are the team-lead. Delegate through the fleetd client — never launch workers yourself.
|
||||
Roster: ask fleetd for current sessions/roles.
|
||||
- w-claude-* — Claude Sonnet. Reasoning-heavy / high-accuracy subtasks.
|
||||
- w-local-* — remote local LLM. Bulk, cheap, or parallelizable subtasks.
|
||||
|
||||
Route each subtask by the rubric in the Team design. For independent subtasks, DISPATCH ALL
|
||||
of them (concurrent blocking sends), THEN gather — never serialize independent work.
|
||||
Integrate the reply envelopes; you own the final answer.
|
||||
```
|
||||
|
||||
Wrap the send as a Claude Code skill (`/delegate <role> "<task>"`) so the lead calls one
|
||||
tool instead of hand-rolling the HTTP request.
|
||||
|
||||
## What this layer does NOT change
|
||||
|
||||
- **Delivery** is still `fleetd` → herdr `pane.send_text` + status events (Message-Server).
|
||||
- **Completion timing** is still the worker status event; **reply content** still rides the
|
||||
worker `Stop`-hook envelope.
|
||||
- **Single-host** still applies: herdr's socket is local, so the whole herd lives on the
|
||||
`fleetd` host. The lead may be remote — it only needs HTTP to `fleetd`.
|
||||
|
||||
## Open questions
|
||||
|
||||
- **Routing intelligence:** rubric-in-`CLAUDE.md` (lead decides) vs. a `fleetd` role-router
|
||||
(label-based). Start with the former; promote to the latter if routing logic grows.
|
||||
- **Backpressure:** per-role concurrency caps in `fleetd` so a fan-out can't exhaust the
|
||||
local gateway.
|
||||
- **Result schema:** whether reply envelopes should carry structured metadata (worker, model,
|
||||
tokens) to help the lead's reduce step.
|
||||
|
||||
## Status
|
||||
|
||||
🟡 Design (2026-07-11). Orchestration layer over the selected `fleetd` server; inherits
|
||||
herdr (chosen) + AgentAPI (fallback). Delivery unchanged — see the Message-Server design.
|
||||
@@ -0,0 +1,182 @@
|
||||
# Worker Git Workflow — worktree · branch · PR
|
||||
|
||||
**Status:** design (defining the fleet's working model). Builds on the
|
||||
[CB-301 session manager](CB-301-Session-Manager.md) and the one-shot/no-reuse decision.
|
||||
|
||||
## Guiding principle — a worker is a full peer of the primary
|
||||
|
||||
The whole point of the bridge is **the same Claude Code agent running against a different LLM
|
||||
provider**. A worker must be **functionally identical to the primary in context and knowledge** —
|
||||
same project + user `CLAUDE.md`, same skills, same memory, same MCP tools, **same local/private
|
||||
settings** — and differ **only** in `ANTHROPIC_BASE_URL`/`ANTHROPIC_MODEL`. The worktree exists
|
||||
*solely* for git isolation (a branchable checkout for commits + code reference). It must **never**
|
||||
strip the worker of the configuration a main-tree session has. **Config parity is a hard
|
||||
requirement, not a nice-to-have.**
|
||||
|
||||
## Decisions
|
||||
|
||||
- **One-shot, no reuse** (CB-301) — each task gets a fresh worker, torn down after. The **PR is the
|
||||
durable artifact**; no context is carried across workers.
|
||||
- **Worktree provisioned by the daemon** — `SessionManager` creates a dedicated git worktree +
|
||||
branch per session, **hydrates it to full config parity** (below), and tears it down on release.
|
||||
- **Worker opens its own PR** — the worker commits, pushes its branch, and opens the PR/MR itself,
|
||||
returning the PR URL in its `fleet_reply`.
|
||||
|
||||
## Why worktrees (the hazard being fixed)
|
||||
|
||||
Today every bridge worker inherits the **primary's own working tree** as its cwd
|
||||
(`WorkerService` cwd resolution → caller cwd). A single worker editing at a time is safe, but two
|
||||
**parallel implementers** would stomp each other's files. A worktree per session gives each worker
|
||||
an isolated checkout on its own branch — the precondition for fanning out implementation work.
|
||||
|
||||
## Config parity — the worktree is code-only; it must NOT strip the worker
|
||||
|
||||
**The trap:** `git worktree add` creates a checkout of **tracked files only**. Untracked / gitignored
|
||||
files do **not** come along. So a worker moved from the primary's main tree into a bare worktree
|
||||
**silently loses** exactly the local/private configuration that makes it a full peer — while the
|
||||
shared-tree it runs in *today* gives it all of this for free. Moving to worktrees must **preserve**
|
||||
that, not regress it.
|
||||
|
||||
Classify every source of "what a main session knows", by whether a worktree keeps it:
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
subgraph keeps["Inherited automatically — no action"]
|
||||
A["User-global config<br/>~/.claude/CLAUDE.md, ~/.ccs memory"]:::ok
|
||||
B["Tracked project config<br/>CLAUDE.md, committed .claude/skills, committed .mcp.json"]:::ok
|
||||
C["CLAUDE_CONFIG_DIR<br/>(daemon already injects per profile)"]:::ok
|
||||
D["Bridge MCP<br/>(injected via --mcp-config launch flag)"]:::ok
|
||||
end
|
||||
subgraph gap["LOST by a bare worktree — must be hydrated"]
|
||||
E["settings.local.json<br/>local .claude/* overrides"]:::warn
|
||||
F["local .mcp.json mods<br/>(the M .mcp.json in git status)"]:::warn
|
||||
G[".env / .envrc / direnv<br/>local secrets + tokens"]:::warn
|
||||
H["any other gitignored local config"]:::warn
|
||||
end
|
||||
keeps --> R["Worker = full peer of primary"]:::goal
|
||||
gap -->|"overlay step at provision"| R
|
||||
classDef ok fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
classDef warn fill:#b7791f,stroke:#7b341e,color:#ffffff;
|
||||
classDef goal fill:#2b6cb0,stroke:#2a4365,color:#ffffff;
|
||||
```
|
||||
|
||||
*Green is inherited by path (home dir / `CLAUDE_CONFIG_DIR`) or lives in tracked files that the
|
||||
worktree checks out anyway. Amber is the real gap — untracked local config the worktree drops.*
|
||||
|
||||
**Mechanism — worktree hydration (part of `SessionManager.acquire`, after `git worktree add`):**
|
||||
|
||||
1. **Inherit by path, don't copy** — keep the worker's `$HOME`, `CLAUDE_CONFIG_DIR`, and memory dir
|
||||
identical to the primary's. Everything home-scoped (user `CLAUDE.md`, memory, auth, global
|
||||
skills) is already parity for free; only *cwd-relative* project-local files are the gap.
|
||||
2. **Overlay the untracked project-local set** from the primary tree into the worktree — a defined,
|
||||
configurable list: `settings.local.json` (+ any local `.claude/*`), the locally-modified
|
||||
`.mcp.json`, `.env`/`.envrc`, and any other gitignored config the primary depends on. **Symlink**
|
||||
(read-only parity, stays live, nothing to go stale) rather than copy where possible; copy only
|
||||
what a worker may write.
|
||||
3. **Never overlay the git plumbing** — the worktree's own `.git` file/branch is what gives
|
||||
isolation; that's the *one* thing that must differ from the main tree.
|
||||
|
||||
The overlay set lives in config (`FleetConfig.Worker.parityOverlay` — a list of repo-relative
|
||||
paths, with sane defaults) so it's auditable and per-repo tunable.
|
||||
|
||||
> **Trust note (deliberate).** Hydrating local config means the primary's local secrets/tokens
|
||||
> (`.env`, `.mcp.json` auth, gitea token) become visible to an **off-subscription** worker running
|
||||
> against a third-party model endpoint. That is the accepted consequence of "workers must be full
|
||||
> peers" — but it is a real trust expansion over a worker that only sees tracked code. Keep the
|
||||
> overlay list **explicit and minimal**; don't blanket-symlink the whole tree. `.mcp.json` and
|
||||
> `wiki/` remain **excluded from all worker commits** regardless of being present for reference.
|
||||
|
||||
## Lifecycle
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant P as Primary
|
||||
participant SM as SessionManager (daemon)
|
||||
participant G as git / gitea
|
||||
participant W as Worker
|
||||
|
||||
P->>SM: acquire(ticket, profile)
|
||||
SM->>G: git worktree add wt -b worker/ticket-nonce main
|
||||
SM->>SM: overlay parity config into wt
|
||||
SM->>W: spawn (cwd = wt, on its branch)
|
||||
Note over W: implement in the isolated worktree
|
||||
W->>G: git commit + git push (SSH, same user)
|
||||
W->>G: open PR (branch to main)
|
||||
W-->>P: fleet_reply (prUrl, branch, summary, tests)
|
||||
P->>SM: release(paneId)
|
||||
SM->>G: git worktree remove wt
|
||||
Note over G: branch + PR persist for review/merge
|
||||
P->>G: review PR, merge on green
|
||||
```
|
||||
|
||||
*The worker's "checkpoint" (CB-302) is exactly steps 6–8: commit → push → open PR. This replaces the
|
||||
earlier `STATE.md` idea — a PR is reviewable, mergeable, and self-describing.*
|
||||
|
||||
## Infra facts (verified this session)
|
||||
|
||||
- **Remote:** `ssh://git@git.ltms.dev:2224/fleet/fleetd.git` (gitea). Push is over **SSH** —
|
||||
a worker running as the same user with the same keys can `git push` **with no extra credential**.
|
||||
- **gitea is NOT in the project `.mcp.json`** (only `jetbrains`, `intellij-index`, `fleetd`). The
|
||||
primary's gitea MCP comes from a global/user config, so **workers do not inherit it**. A worker
|
||||
gets only the `bridge` MCP mounted (via `--mcp-config` launch flag).
|
||||
- **No gitea CLI** (`tea`) installed; `glab` is present but is the GitLab CLI (wrong backend).
|
||||
|
||||
**Implication:** `git push` is free for workers; only **PR creation** needs a new mechanism.
|
||||
|
||||
## Open decision — how the worker creates the PR
|
||||
|
||||
| Option | Mechanism | Trade-off |
|
||||
|---|---|---|
|
||||
| **A. gitea REST + token** | Worker `curl`s `POST /api/v1/repos/fleet/fleetd/pulls` with a scoped token injected by the daemon into the worker env | Minimal, no new server; token lives in the off-subscription worker's env (scope it tightly) |
|
||||
| **B. mount gitea MCP into workers** | Add the gitea MCP to the worker's `--mcp-config` alongside `bridge` | Clean tool call, but the gitea MCP's own auth/token must be provisioned per worker; more moving parts |
|
||||
| **C. install `tea` CLI** | Worker runs `tea pr create` with a token | Another dependency to install + configure; same token question as A |
|
||||
|
||||
**Recommendation: A (gitea REST + a repo-scoped token).** Smallest surface, reuses SSH for push,
|
||||
and the token is a single scoped secret the daemon injects like it already injects
|
||||
`ANTHROPIC_AUTH_TOKEN`. The implementer skill wraps the `curl` in one documented step.
|
||||
|
||||
### Trust / token scope (the real cost of "worker opens its own PR")
|
||||
|
||||
- Off-subscription workers already *could* push (SSH, same user). The **incremental grant is
|
||||
PR-create**, i.e. a gitea API token.
|
||||
- Scope the token **minimally**: the `fleet/fleetd` repo, `write:repository` (create branch +
|
||||
PR), **not** merge/admin/org. A leaked token can open PRs, not merge them — the primary/human is
|
||||
still the merge gate.
|
||||
- Inject via the daemon (env var, e.g. `GITEA_TOKEN`), never written to the worker's config dir —
|
||||
same non-invasive pattern as the ANTHROPIC token. Guard is unaffected (it concerns
|
||||
`ANTHROPIC_BASE_URL`, not git).
|
||||
|
||||
## Implementation plan
|
||||
|
||||
| Piece | Where | Notes |
|
||||
|---|---|---|
|
||||
| Worktree provision/teardown | **CB-301 ext** — `SessionManager.acquire`/`release`; `WorkerSession` gains `worktree`, `branch` | daemon shells out to `git worktree add/remove` |
|
||||
| **Config-parity overlay** | **CB-301 ext** — `SessionManager.acquire`, after `git worktree add` | symlink/copy the `parityOverlay` set into the worktree so the worker is a full peer; **this is what makes worktrees viable, not a dead-end** |
|
||||
| Overlay config | `FleetConfig.Worker.parityOverlay` — repo-relative paths, sane defaults | auditable, per-repo tunable; keep explicit + minimal (trust) |
|
||||
| Branch naming | `worker/<ticket-slug>-<nonce>` off `main` (or a configured base) | one branch per session |
|
||||
| Commit + push + PR handoff | **CB-302** — worker-driven, guided by the skill | push = SSH; PR = option A |
|
||||
| Implementer skill | `.claude/skills/implementer/SKILL.md` | worktree-aware playbook (see below); mounts automatically since workers inherit repo cwd |
|
||||
| gitea token injection | `WorkerService` env + `FleetConfig` | repo-scoped, minimal perms |
|
||||
| PR review + merge | Primary (has gitea MCP + judgment) | merge on green; the human/primary gate stays |
|
||||
|
||||
## Implementer skill (outline)
|
||||
|
||||
A worker-facing playbook (sibling to the existing `reviewer` skill):
|
||||
|
||||
1. **You are in a git worktree on a dedicated branch** — check `git status`/`git branch`; do all
|
||||
work here, never on `main`.
|
||||
2. **Implement the task**; keep commits focused and message them clearly.
|
||||
3. **Push** your branch (`git push -u origin HEAD`).
|
||||
4. **Open a PR** to `main` (option A `curl`, or the decided mechanism) with a title/body describing
|
||||
the change and referencing the ticket.
|
||||
5. **Reply** via `fleet_reply` with the **PR URL**, branch name, files changed, and test names —
|
||||
that reply is the whole handoff.
|
||||
6. Do **not** merge; do **not** touch `.mcp.json` or `wiki/`.
|
||||
|
||||
## Sequencing
|
||||
|
||||
1. Finish + verify **CB-301 core** (in flight) — registry/FSM.
|
||||
2. Extend CB-301 with **worktree provisioning** (this doc) once the PR mechanism is chosen.
|
||||
3. Add the **implementer skill** + **token injection**.
|
||||
4. **CB-302** = wire the worker checkpoint (commit/push/PR) as the release-time handoff.
|
||||
@@ -0,0 +1,137 @@
|
||||
# Worker startup: working directory & the folder-trust prompt
|
||||
|
||||
When `fleetd` spawns a worker, the worker CLI may show an **interactive startup prompt** before it
|
||||
is ready to accept a task — most importantly a *"Do you trust the files in this folder?"* dialog. An
|
||||
unattended worker parked on that prompt never becomes injectable: the status-gated injector waits for
|
||||
`idle`/`blocked`, the task is never delivered, and (worst case) a stray Enter answers the dialog
|
||||
wrong. How this is handled depends on two things:
|
||||
|
||||
1. **The worker's working directory** — which folder the CLI is asked to trust.
|
||||
2. **Which CLI launches the worker** — each has its own first-run/trust behaviour.
|
||||
|
||||
This doc pins the current assumption (**`ccs`** is the launcher), how its trust prompt works, the
|
||||
rule that **a worker inherits the primary's directory** (never `$HOME`), and how other CLIs differ.
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A["fleet_spawn / POST /workers"] --> B{"explicit cwd?<br/>(profile cwd or spawn arg)"}
|
||||
B -->|"yes — told otherwise"| C["use that cwd"]
|
||||
B -->|"no"| D{"caller PID resolvable?<br/>(MCP peer PID)"}
|
||||
D -->|"yes"| E["cwd = the primary's cwd<br/>lsof -a -p PID -d cwd"]
|
||||
D -->|"no (REST / off-host)"| F["cwd = fleetd daemon cwd<br/>(never $HOME by assumption)"]
|
||||
C --> G["ensureWorkspace → tab.create → agent.start {cwd}"]
|
||||
E --> G
|
||||
F --> G
|
||||
G --> H{"does the CLI trust this folder?"}
|
||||
H -->|"yes"| I["worker reaches its prompt → injectable"]
|
||||
H -->|"no"| J["worker BLOCKS on the trust dialog<br/>never injectable"]
|
||||
classDef good fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
classDef bad fill:#9b2c2c,stroke:#63171b,color:#ffffff;
|
||||
class I good
|
||||
class J bad
|
||||
```
|
||||
|
||||
*Figure 1 — spawn resolves a working directory, then the CLI's trust check gates readiness.*
|
||||
|
||||
## Working directory: inherit the primary's path
|
||||
|
||||
**Rule: a worker opens the same directory the primary (main) session is working in, unless told
|
||||
otherwise. Never assume `$HOME`.** If the primary is in `/Users/you/LTMS/claude-bridge`, its workers
|
||||
open there too — so delegated tasks share the same relative paths and the same (already-trusted)
|
||||
project folder.
|
||||
|
||||
**The herdr seam.** An `agent.start` pane does **not** inherit its tab's or workspace's cwd — it
|
||||
starts in `$HOME` unless told otherwise. herdr's `agent.start` honours an (undocumented) **`cwd`**
|
||||
param, verified live: setting it roots the worker process at that directory. So the worker's cwd is
|
||||
threaded onto `agent.start {…, cwd}`, not the placement step (`workspace.create`/`tab.create` cwd
|
||||
only affect the seed shell, which the bridge closes).
|
||||
|
||||
**Resolution order** (first match wins):
|
||||
|
||||
| # | Source | When |
|
||||
|---|--------|------|
|
||||
| 1 | Explicit `cwd` — a per-profile `cwd:` in config, or a spawn argument | "told otherwise" — pin a fixed workdir |
|
||||
| 2 | The **primary's cwd**, auto-detected from the `fleet_spawn` caller | normal MCP spawn from the primary |
|
||||
| 3 | The `fleetd` daemon's own cwd | REST spawn / off-host caller — **never `$HOME`** |
|
||||
|
||||
The primary's cwd (source 2) is discoverable with no new plumbing: `fleetd` already resolves the MCP
|
||||
caller's loopback **peer PID** for connection identity (`ConnectionIdentity` → `LsofPeerPidLookup`);
|
||||
the same PID yields its cwd via `lsof -a -p <pid> -d cwd -Fn` (the `n…` line). The primary maps to no
|
||||
worker pane (it is not a worker), but its PID and cwd are still readable.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as "Primary (main)"
|
||||
participant B as "fleetd"
|
||||
participant O as "OS (lsof)"
|
||||
participant H as "herdr"
|
||||
P->>B: "fleet_spawn {profile} (no cwd)"
|
||||
B->>O: "peer PID for this connection's port"
|
||||
O-->>B: "pid"
|
||||
B->>O: "cwd of pid (lsof -d cwd)"
|
||||
O-->>B: "/Users/you/LTMS/claude-bridge"
|
||||
B->>H: "tab.create (placement)"
|
||||
B->>H: "agent.start {argv, env, tab_id, cwd}"
|
||||
H-->>B: "worker in the primary's directory"
|
||||
```
|
||||
|
||||
*Figure 2 — a no-cwd spawn inherits the primary's directory from the caller's PID.*
|
||||
|
||||
> **Status:** implemented (CB-112). `fleetd` threads the resolved `cwd` onto **`agent.start {cwd}`**
|
||||
> (verified: the worker process is rooted there), keeping the single shared worker space. On an MCP
|
||||
> `fleet_spawn` the primary's cwd is auto-detected from the caller's PID; over REST (no MCP caller)
|
||||
> it is the explicit `cwd` param else the daemon's cwd. Both placements (`tab` and legacy `pane`)
|
||||
> carry it, since it rides `agent.start`.
|
||||
|
||||
## Assumed launcher: `ccs` (Claude Code)
|
||||
|
||||
For now the fleet assumes **`ccs`** (Claude Code under the hood) as the worker CLI — `argv: ["ccs",
|
||||
"<profile>"]`. Its startup gate is the **folder-trust dialog**.
|
||||
|
||||
### How `ccs`/Claude Code decides whether to prompt
|
||||
|
||||
Trust is recorded **per-directory, per config dir**. Each `ccs` profile is an isolated instance with
|
||||
its own config dir (`~/.ccs/instances/<profile>/`) and its own `.claude.json`:
|
||||
|
||||
```jsonc
|
||||
// ~/.ccs/instances/<profile>/.claude.json
|
||||
"projects": {
|
||||
"/Users/you/LTMS/claude-bridge": { "hasTrustDialogAccepted": true }, // trusted → no prompt
|
||||
"/Users/you": { "hasTrustDialogAccepted": false } // untrusted → prompts
|
||||
}
|
||||
```
|
||||
|
||||
The worker prompts **iff** its cwd is not marked `hasTrustDialogAccepted: true` for *that instance*.
|
||||
This is why the directory rule above matters: land workers in the primary's project folder and you
|
||||
grant trust **once per profile**, instead of scattering trust across `$HOME` and ad-hoc dirs.
|
||||
|
||||
### Clearing the prompt (ranked)
|
||||
|
||||
1. **Inherit the primary's project dir** (the rule above) and trust that folder once per profile.
|
||||
2. **Pre-trust interactively:** run `ccs <profile>` in the target folder and accept — persists
|
||||
`hasTrustDialogAccepted: true` for that path in the instance's `.claude.json`.
|
||||
3. **Set the flag directly** (scriptable, no interaction): set
|
||||
`projects["<cwd>"].hasTrustDialogAccepted = true` in `~/.ccs/instances/<profile>/.claude.json`.
|
||||
4. **Do not** reach for `--dangerously-skip-permissions` — it disables *all* permission gating, not
|
||||
just this dialog, which defeats running off-subscription workers autonomously.
|
||||
|
||||
## Other CLIs: different launchers, different prompts
|
||||
|
||||
`ccs` is the current assumption, not a hard dependency — a worker profile's `argv` can be any CLI.
|
||||
Each launcher has its **own** first-run/trust gate, so the "clear the prompt" step is CLI-specific
|
||||
and belongs with the profile, not hard-coded:
|
||||
|
||||
| Launcher (`argv`) | Startup gate | How to clear it |
|
||||
|---|---|---|
|
||||
| `ccs <profile>` (Claude Code) | Folder-trust dialog | `hasTrustDialogAccepted` per project in the instance's `.claude.json` (above) |
|
||||
| Other Claude-compatible runtimes via `ccs` (codex, gemini, cursor, …) | Each has its own first-run / trust / login prompt | Per-runtime; document per launcher as it is adopted |
|
||||
| A bare command (`bash -c …`, mechanics probe) | None | n/a — used for non-interactive smoke tests |
|
||||
|
||||
When adding a new launcher, capture its startup-prompt behaviour here (what blocks, and the
|
||||
non-interactive way to satisfy it) so a spawned worker of that kind reaches an injectable prompt
|
||||
unattended.
|
||||
|
||||
## See also
|
||||
|
||||
- `docs/MCP-Contract.md` — the tool surface (`fleet_spawn`, `fleet_profiles`, …).
|
||||
- `wiki/2-Message-Server.md` — the herdr `agent.*` / `workspace.*` schema (`workspace.create {cwd}`).
|
||||
@@ -0,0 +1,2 @@
|
||||
__pycache__/
|
||||
*.pyc
|
||||
@@ -0,0 +1,77 @@
|
||||
# Bridge conversation test (`e2e/`)
|
||||
|
||||
A standard, repeatable **live** end-to-end test of the two-way channel: it drives a real
|
||||
multi-turn conversation between a primary and an off-subscription worker **through the
|
||||
running `fleetd` daemon**, captures the full transcript, and grades the channel.
|
||||
|
||||
This is the committed form of the ad-hoc channel test that discovered the CB-115 gaps
|
||||
(herdr `unknown` misclassification wedging delivery, dirty completion scrapes, and workers
|
||||
never calling `fleet_reply` in conversation). Run it after any change to the injector,
|
||||
status handling, completion/failure paths, or the worker reply charter.
|
||||
|
||||
## What it exercises
|
||||
|
||||
Each turn goes through the whole gateway exactly as a primary Opus session would — async
|
||||
fire-and-poll (`POST /sessions/{id}/message {"wait":false}` → `GET /tasks/{ticket}`), so it
|
||||
also validates the path that beats the caller's MCP timeout. It never sets
|
||||
`ANTHROPIC_BASE_URL` and never talks to herdr directly, so it is **subscription-safe by
|
||||
construction** — it only calls the bridge's loopback REST face.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant T as conversation_test.py
|
||||
participant B as fleetd (REST)
|
||||
participant W as worker (off-sub)
|
||||
T->>B: POST /workers (spawn)
|
||||
T->>B: GET /sessions/{id}/status (await ready)
|
||||
loop each turn
|
||||
T->>B: POST /sessions/{id}/message {wait:false}
|
||||
B-->>T: ticket
|
||||
B->>W: inject prompt (status-gated)
|
||||
W-->>B: fleet_reply
|
||||
T->>B: GET /tasks/{ticket} (poll)
|
||||
B-->>T: done + reply
|
||||
end
|
||||
T->>B: DELETE /workers/{pane} (stop)
|
||||
```
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- `fleetd` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
profile configured and its backend reachable.
|
||||
- herdr is up (the daemon needs it).
|
||||
- Python 3 (standard library only — no pip installs).
|
||||
|
||||
## Run
|
||||
|
||||
```bash
|
||||
# spawn the default-profile worker, run the built-in 5-turn conversation, grade, clean up
|
||||
python3 e2e/conversation_test.py
|
||||
|
||||
# pick a profile / reuse a live worker / use your own prompts
|
||||
python3 e2e/conversation_test.py --profile ollama
|
||||
python3 e2e/conversation_test.py --tid term_abc123 --keep-worker
|
||||
python3 e2e/conversation_test.py --prompts my_prompts.txt --out /tmp/run1
|
||||
```
|
||||
|
||||
A prompts file is one prompt per line; blank lines and `#` comments are ignored.
|
||||
|
||||
## Output & grading
|
||||
|
||||
- Writes `transcript.md` (in `--out`, default `e2e/`) — every turn's prompt, worker reply,
|
||||
latency, resolution source, and observed status transitions, with an inline `> **GAP**`
|
||||
note on any non-clean turn.
|
||||
- Prints a per-turn line and an overall summary, and **exits non-zero** if any turn wedged,
|
||||
failed, or returned empty — so it is CI-usable.
|
||||
|
||||
Per-turn grade:
|
||||
|
||||
| Grade | Meaning |
|
||||
|------------|---------------------------------------------------------------------|
|
||||
| `OK` | delivered and resolved by an explicit `fleet_reply` (`source=reply`) |
|
||||
| `DEGRADED` | delivered and answered, but resolved via completion-scrape fallback |
|
||||
| `EMPTY` | turn completed but the reply was empty |
|
||||
| `FAILED` | the worker's turn ended in failure (`phase=failed`) |
|
||||
| `WEDGE` | never resolved within the poll window (delivery wedge / lost turn) |
|
||||
|
||||
`PASS` requires every turn to be `OK` or `DEGRADED`; a clean run is every turn `OK`.
|
||||
@@ -0,0 +1,170 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Sustained back-and-forth bridge test — ONE primary, ONE worker, many dependent turns
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `fleetd` daemon.
|
||||
|
||||
Where conversation_test.py proves a handful of turns work and issue_hunt_test.py proves
|
||||
fan-out isolation, this proves the channel stays healthy under a *sustained, stateful*
|
||||
conversation: a running-total game the worker must keep in its head across turns. Turn N's
|
||||
prompt does NOT restate the total — the worker has to remember it from turn N-1 — so a
|
||||
correct answer is evidence of genuine multi-turn continuity, not just per-turn liveness.
|
||||
Each reply is machine-checked against the primary's own expected total; on a drift the
|
||||
primary re-anchors (states the correct total once) and keeps going, and drift is reported.
|
||||
|
||||
It talks ONLY to the bridge's REST face on loopback — it never sets ANTHROPIC_BASE_URL and
|
||||
never touches herdr directly, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 conversation_sustained_test.py [--base URL] [--profile NAME] [--repo DIR]
|
||||
[--duration SECS] [--turn-timeout SECS] [--keep-worker]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
--profile worker profile (default: the daemon's default)
|
||||
--repo cwd handed to the worker (default: the bridge repo root)
|
||||
--duration wall-clock window, seconds (default 300 = 5 minutes)
|
||||
--turn-timeout per-turn max wait, seconds (default 150)
|
||||
--keep-worker do not stop the worker at the end
|
||||
|
||||
Exit code: 0 if every turn in the window resolved via a clean fleet_reply with no channel
|
||||
break; 1 otherwise. A live per-turn log streams to stdout so the run can be watched.
|
||||
"""
|
||||
import argparse
|
||||
import pathlib
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
|
||||
# Reuse the exact REST primitives the other harnesses use (same daemon contract).
|
||||
sys.path.insert(0, str(pathlib.Path(__file__).parent))
|
||||
from issue_hunt_test import http, now, spawn_worker, await_ready, fire, poll # noqa: E402
|
||||
|
||||
HERE = pathlib.Path(__file__).parent
|
||||
REPO_ROOT = HERE.parent
|
||||
|
||||
# The per-turn increments, cycled. Non-trivial and varied so the running total isn't a
|
||||
# predictable multiple the worker could pattern-match without actually tracking it.
|
||||
STEPS = [7, 3, 11, 5, 9, 4, 13, 6, 8, 2]
|
||||
|
||||
RULES = (
|
||||
"Let's play a running-total game across several messages. The total starts at 0. "
|
||||
"In each message I'll tell you to add a number; keep the running total yourself and "
|
||||
"reply via fleet_reply with ONLY the current total as a plain integer — no words, no "
|
||||
"punctuation, just the number. Do not restate the arithmetic. First move: add {step}."
|
||||
)
|
||||
NEXT = ("Add {step}. Reply via fleet_reply with only the new running total.")
|
||||
REANCHOR = ("Let's re-sync — the running total is {total}. Now add {step}. Reply via "
|
||||
"fleet_reply with only the new running total.")
|
||||
|
||||
|
||||
def parse_int(reply):
|
||||
"""Pull the worker's answer integer from its reply (last integer token wins)."""
|
||||
if not reply:
|
||||
return None
|
||||
nums = re.findall(r"-?\d+", reply.replace(",", ""))
|
||||
return int(nums[-1]) if nums else None
|
||||
|
||||
|
||||
def one_turn(base, tid, prompt, turn_timeout):
|
||||
"""Fire one prompt and block on its reply. Returns the poll record."""
|
||||
t0 = time.time()
|
||||
ticket = fire(base, tid, prompt)
|
||||
rec = poll(base, ticket, "worker", t0, turn_timeout)
|
||||
rec["ticket"] = ticket
|
||||
return rec
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Sustained back-and-forth bridge test (1 primary, 1 worker)")
|
||||
ap.add_argument("--base", default="http://127.0.0.1:8765")
|
||||
ap.add_argument("--profile", default=None)
|
||||
ap.add_argument("--repo", default=str(REPO_ROOT))
|
||||
ap.add_argument("--duration", type=int, default=300)
|
||||
ap.add_argument("--turn-timeout", type=int, default=150)
|
||||
ap.add_argument("--keep-worker", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
mins = args.duration / 60
|
||||
print(f"[{now()}] sustained back-and-forth: 1 primary <-> 1 worker for {args.duration}s "
|
||||
f"(~{mins:.1f} min) profile={args.profile or 'default'} repo={args.repo}", flush=True)
|
||||
|
||||
tid, pane = spawn_worker(args.base, args.profile, args.repo, "worker")
|
||||
await_ready(args.base, tid, "worker")
|
||||
|
||||
start = time.time()
|
||||
expected = 0 # the primary's authoritative running total
|
||||
reanchor = False # re-state the total next turn after a drift
|
||||
turns, oks, drifts, breaks = 0, 0, 0, 0
|
||||
latencies = []
|
||||
print(f"[{now()}] --- conversation start (worker must keep the total in its head) ---\n", flush=True)
|
||||
|
||||
while time.time() - start < args.duration:
|
||||
turns += 1
|
||||
step = STEPS[(turns - 1) % len(STEPS)]
|
||||
if turns == 1:
|
||||
prompt = RULES.format(step=step)
|
||||
elif reanchor:
|
||||
prompt = REANCHOR.format(total=expected, step=step)
|
||||
reanchor = False
|
||||
else:
|
||||
prompt = NEXT.format(step=step)
|
||||
expected += step
|
||||
|
||||
el = round(time.time() - start)
|
||||
print(f"[{now()}] turn {turns:>2} (t+{el}s) PRIMARY → add {step} (expect total {expected})", flush=True)
|
||||
|
||||
rec = one_turn(args.base, tid, prompt, args.turn_timeout)
|
||||
got = parse_int(rec.get("reply"))
|
||||
lat = rec.get("latency")
|
||||
latencies.append(lat)
|
||||
|
||||
if rec.get("phase") != "done" or not (rec.get("reply") or "").strip():
|
||||
breaks += 1
|
||||
print(f"[{now()}] WORKER ✗ CHANNEL BREAK — phase={rec.get('phase')} "
|
||||
f"source={rec.get('source')} detail={str(rec.get('detail'))[:100]} ({lat}s)\n", flush=True)
|
||||
reanchor = True
|
||||
continue
|
||||
|
||||
src = rec.get("source")
|
||||
badge = "OK " if got == expected else "DRIFT"
|
||||
if got == expected:
|
||||
oks += 1
|
||||
else:
|
||||
drifts += 1
|
||||
reanchor = True # re-sync the worker next turn
|
||||
print(f"[{now()}] WORKER → {str(rec.get('reply')).strip()[:60]!r} = {got} "
|
||||
f"[{badge}] via {src} ({lat}s)", flush=True)
|
||||
if got != expected:
|
||||
print(f"[{now()}] (expected {expected}; will re-anchor next turn)", flush=True)
|
||||
print(flush=True)
|
||||
|
||||
dur = round(time.time() - start)
|
||||
clean = sum(1 for lat in latencies if lat)
|
||||
avg = round(sum(latencies) / len(latencies), 1) if latencies else 0
|
||||
print("=" * 72)
|
||||
print(f"SUSTAINED CONVERSATION SUMMARY — 1 primary <-> 1 worker over {dur}s (~{dur/60:.1f} min)")
|
||||
print(f" turns: {turns}")
|
||||
print(f" clean fleet_reply exchanges: {oks + drifts}/{turns} (channel breaks: {breaks})")
|
||||
print(f" arithmetic correct (continuity held): {oks}/{turns} (drifts: {drifts})")
|
||||
print(f" latency: avg {avg}s over {turns} turns")
|
||||
ok = breaks == 0 and turns >= 2
|
||||
if ok and drifts == 0:
|
||||
print(" RESULT: PASS — every turn resolved via fleet_reply and the worker held the "
|
||||
"running total across the whole window.")
|
||||
elif ok:
|
||||
print(f" RESULT: PASS (channel) — every turn resolved via fleet_reply for the full "
|
||||
f"window; {drifts} arithmetic drift(s) (worker recovered after re-anchor).")
|
||||
else:
|
||||
print(" RESULT: FAIL — the channel broke on at least one turn (see CHANNEL BREAK above).")
|
||||
print("=" * 72, flush=True)
|
||||
|
||||
if not args.keep_worker and pane:
|
||||
try:
|
||||
http(args.base, "DELETE", f"/workers/{pane}")
|
||||
print(f"[{now()}] worker stopped (pane {pane})", flush=True)
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f"[{now()}] worker stop failed (ignore): {e}", flush=True)
|
||||
|
||||
sys.exit(0 if ok else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,227 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge conversation test — a multi-turn primary↔worker exchange through
|
||||
the running `fleetd` daemon, fully captured, with automatic gap analysis.
|
||||
|
||||
This is the repeatable form of the ad-hoc channel test that surfaced the CB-115 gaps
|
||||
(herdr `unknown` misclassification, dirty completion scrape, workers not calling
|
||||
fleet_reply). It drives a real off-subscription worker over the live gateway exactly
|
||||
as a primary Opus session would (async fire-and-poll), records every turn, and grades
|
||||
the channel.
|
||||
|
||||
It talks ONLY to the bridge's REST face on loopback — it never sets ANTHROPIC_BASE_URL
|
||||
and never touches herdr directly, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 conversation_test.py [--base URL] [--profile NAME] [--tid TERMINAL_ID]
|
||||
[--prompts FILE] [--out DIR] [--poll-timeout SECS]
|
||||
[--keep-worker]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
--profile worker profile to spawn (default: the daemon's default)
|
||||
--tid reuse an existing worker (skips spawn + readiness wait)
|
||||
--prompts newline-separated prompt file (default: the built-in script)
|
||||
--out output dir for transcript.md (default: alongside this file)
|
||||
--poll-timeout per-turn max wait, seconds (default 300)
|
||||
--keep-worker do not stop a spawned worker at the end
|
||||
|
||||
Exit code: 0 if every turn delivered AND produced a usable reply; 1 otherwise (so it
|
||||
is CI-usable). A per-turn and overall gap report is printed to stdout.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from datetime import datetime
|
||||
|
||||
HERE = pathlib.Path(__file__).parent
|
||||
|
||||
# A default conversation: short, varied turns (a question, a follow-up that needs the
|
||||
# prior context, a tiny reasoning task, a meta-question, a close) — enough to exercise
|
||||
# multi-turn delivery + reply on the channel without being a real coding workload.
|
||||
DEFAULT_PROMPTS = [
|
||||
"Hi! Quick check that our channel works. In one sentence, what are you and what model are you running?",
|
||||
"Thanks. Now a small task: what is 17 * 23? Show just the number.",
|
||||
"Good. Remembering that result, is it a prime number? Answer yes or no with a one-line reason.",
|
||||
"Switching topic: name one thing that would make this bridge conversation feel more reliable to you as the worker.",
|
||||
"That's all — please acknowledge and we'll wrap up.",
|
||||
]
|
||||
|
||||
|
||||
def http(base, method, path, body=None, timeout=20):
|
||||
data = json.dumps(body).encode() if body is not None else None
|
||||
req = urllib.request.Request(base + path, data=data, method=method,
|
||||
headers={"Content-Type": "application/json"})
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
return json.loads(r.read().decode())
|
||||
|
||||
|
||||
def now():
|
||||
return datetime.now().strftime("%H:%M:%S")
|
||||
|
||||
|
||||
def spawn_worker(base, profile):
|
||||
q = f"?profile={profile}" if profile else ""
|
||||
res = http(base, "POST", f"/workers{q}")
|
||||
tid = res.get("terminalId") or res.get("sessionId")
|
||||
if not tid:
|
||||
sys.exit(f"spawn failed: {res}")
|
||||
pane = res.get("paneId")
|
||||
print(f"[{now()}] spawned worker {tid} (profile={profile or 'default'}, pane={pane})")
|
||||
return tid, pane
|
||||
|
||||
|
||||
def await_ready(base, tid, timeout=120):
|
||||
print(f"[{now()}] waiting for worker readiness (bridge MCP connect)…")
|
||||
deadline = time.time() + timeout
|
||||
while time.time() < deadline:
|
||||
try:
|
||||
st = http(base, "GET", f"/sessions/{tid}/status")
|
||||
except urllib.error.URLError:
|
||||
st = {}
|
||||
if st.get("ready"):
|
||||
print(f"[{now()}] worker ready (status={st.get('status')})")
|
||||
return True
|
||||
time.sleep(3)
|
||||
print(f"[{now()}] WARNING: worker never reported ready within {timeout}s — running anyway")
|
||||
return False
|
||||
|
||||
|
||||
def run_turn(base, tid, turn, prompt, poll_timeout):
|
||||
"""Fire one prompt async, poll the ticket to resolution, return a structured record."""
|
||||
t0 = time.time()
|
||||
sent = http(base, "POST", f"/sessions/{tid}/message", {"wait": False, "content": prompt})
|
||||
ticket = sent.get("ticket")
|
||||
samples = [] # (elapsed, phase, live_status)
|
||||
reply = detail = source = phase = None
|
||||
deadline = time.time() + poll_timeout
|
||||
while time.time() < deadline:
|
||||
time.sleep(3)
|
||||
task = http(base, "GET", f"/tasks/{ticket}")
|
||||
phase = task.get("phase")
|
||||
live = (task.get("detail") or "").replace("worker ", "") if phase == "pending" else ""
|
||||
samples.append((round(time.time() - t0, 1), phase, live))
|
||||
if phase in ("done", "failed"):
|
||||
reply = task.get("reply")
|
||||
detail = task.get("detail")
|
||||
source = task.get("replySource")
|
||||
break
|
||||
latency = round(time.time() - t0, 1)
|
||||
|
||||
# compress status samples into a transition string
|
||||
trans, last = [], None
|
||||
for el, ph, live in samples:
|
||||
tag = live if ph == "pending" else ph
|
||||
if tag != last:
|
||||
trans.append(f"{tag}@{el}s")
|
||||
last = tag
|
||||
|
||||
return {
|
||||
"turn": turn, "prompt": prompt, "ticket": ticket, "phase": phase,
|
||||
"source": source, "reply": reply, "detail": detail, "latency": latency,
|
||||
"transitions": " → ".join(trans), "time": now(),
|
||||
}
|
||||
|
||||
|
||||
def grade(rec):
|
||||
"""Classify a turn's outcome. Returns (grade, note)."""
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
return "FAILED", f"worker turn failed: {(rec['detail'] or '').strip()[:120]}"
|
||||
return "WEDGE", "never resolved within the poll window (delivery wedge or lost turn)"
|
||||
|
||||
|
||||
def write_transcript(out_dir, records):
|
||||
path = out_dir / "transcript.md"
|
||||
with path.open("w") as f:
|
||||
f.write(f"# Bridge conversation test — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
for rec in records:
|
||||
g, note = grade(rec)
|
||||
f.write(f"### Turn {rec['turn']} — {rec['time']} "
|
||||
f"(`{g}`, {rec['latency']}s, phase={rec['phase']}, source={rec['source']})\n\n")
|
||||
f.write(f"**PRIMARY:** {rec['prompt']}\n\n")
|
||||
f.write(f"**WORKER:** {rec['reply'] if rec['reply'] else '_(no reply)_ ' + str(rec['detail'])}\n\n")
|
||||
f.write(f"_status: {rec['transitions']}_\n")
|
||||
if g != "OK":
|
||||
f.write(f"\n> **GAP — {g}:** {note}\n")
|
||||
f.write("\n")
|
||||
return path
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Standard bridge conversation test")
|
||||
ap.add_argument("--base", default="http://127.0.0.1:8765")
|
||||
ap.add_argument("--profile", default=None)
|
||||
ap.add_argument("--tid", default=None)
|
||||
ap.add_argument("--prompts", default=None)
|
||||
ap.add_argument("--out", default=str(HERE))
|
||||
ap.add_argument("--poll-timeout", type=int, default=300)
|
||||
ap.add_argument("--keep-worker", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
prompts = DEFAULT_PROMPTS
|
||||
if args.prompts:
|
||||
prompts = [ln.strip() for ln in pathlib.Path(args.prompts).read_text().splitlines()
|
||||
if ln.strip() and not ln.startswith("#")]
|
||||
|
||||
spawned = False
|
||||
tid = args.tid
|
||||
pane = None
|
||||
if not tid:
|
||||
tid, pane = spawn_worker(args.base, args.profile)
|
||||
spawned = True
|
||||
await_ready(args.base, tid)
|
||||
|
||||
print(f"[{now()}] running {len(prompts)}-turn conversation on {tid}\n")
|
||||
records = []
|
||||
for i, prompt in enumerate(prompts, 1):
|
||||
rec = run_turn(args.base, tid, i, prompt, args.poll_timeout)
|
||||
g, note = grade(rec)
|
||||
records.append(rec)
|
||||
print(f"[turn {i}] {g:8} {rec['latency']:6}s phase={rec['phase']} source={rec['source']}")
|
||||
print(f" status: {rec['transitions']}")
|
||||
print(f" reply: {(rec['reply'] or '(none) ' + str(rec['detail'])).strip()[:200]}")
|
||||
print(f" note: {note}\n")
|
||||
|
||||
out_dir = pathlib.Path(args.out)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
path = write_transcript(out_dir, records)
|
||||
|
||||
# ---- gap report -------------------------------------------------------
|
||||
grades = [grade(r)[0] for r in records]
|
||||
counts = {g: grades.count(g) for g in ("OK", "DEGRADED", "EMPTY", "FAILED", "WEDGE") if grades.count(g)}
|
||||
print("=" * 68)
|
||||
print(f"CONVERSATION TEST SUMMARY — {len(records)} turns")
|
||||
print(" " + " ".join(f"{g}:{n}" for g, n in counts.items()))
|
||||
print(f" transcript: {path}")
|
||||
ok = all(g in ("OK", "DEGRADED") for g in grades)
|
||||
reply_clean = all(g == "OK" for g in grades)
|
||||
if reply_clean:
|
||||
print(" RESULT: PASS — every turn delivered and got a clean fleet_reply.")
|
||||
elif ok:
|
||||
print(" RESULT: PASS (with notes) — every turn delivered & replied, but some via fallback.")
|
||||
else:
|
||||
print(" RESULT: FAIL — one or more turns wedged, failed, or returned empty (see GAP notes).")
|
||||
print("=" * 68)
|
||||
|
||||
if spawned and pane and not args.keep_worker:
|
||||
try:
|
||||
http(args.base, "DELETE", f"/workers/{pane}")
|
||||
print(f"[{now()}] stopped worker {tid} (pane {pane})")
|
||||
except Exception as e: # noqa: BLE001 - best-effort cleanup
|
||||
print(f"[{now()}] worker stop failed (ignore): {e}")
|
||||
|
||||
sys.exit(0 if ok else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,277 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Live fleet_ask test — the REVERSE rendezvous (CB-205), watched end to end.
|
||||
|
||||
Every other harness drives the forward path: primary `fleet_send` → worker `fleet_reply`.
|
||||
This drives the one that runs the other way. A worker is told to pause its delegated turn,
|
||||
ask the primary a question via `fleet_ask`, and only finish once it has the answer — so the
|
||||
turn round-trips primary→worker→primary→worker inside a SINGLE delegation.
|
||||
|
||||
The mechanics that only this path exercises:
|
||||
|
||||
• a worker's mid-turn question surfacing on the primary's *own* blocked send (Outcome.QUESTION),
|
||||
• the `turnId` correlation that lets the primary answer the exact paused turn,
|
||||
• the answer resuming that same turn and the worker's final `fleet_reply` landing on the
|
||||
re-opened forward waiter (never a stale or cross-wired one).
|
||||
|
||||
It is two blocking REST calls, no polling:
|
||||
|
||||
1. POST /sessions/{id}/message {content: TASK} → blocks, returns 202 {status:"question",
|
||||
question, turnId} (the worker asked)
|
||||
2. POST /sessions/{id}/message {content: ANSWER, turnId} → blocks, returns 200 {reply, replySource}
|
||||
(the worker resumed and replied)
|
||||
|
||||
Like the rest of the suite it talks ONLY to the bridge's REST face on loopback — it never sets
|
||||
ANTHROPIC_BASE_URL and never touches herdr, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 fleet_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
[--send-timeout SECS] [--answer-timeout SECS] [--keep-worker]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
--profile profile for the worker (default: the daemon's default)
|
||||
--repo cwd handed to the worker (default: the bridge repo root)
|
||||
--out output dir for the transcript (default: alongside this file)
|
||||
--send-timeout max wait for the worker to ASK (default 110)
|
||||
--answer-timeout max wait for the worker to REPLY (default 90)
|
||||
--keep-worker do not stop the spawned worker at the end
|
||||
|
||||
Exit code: 0 if the worker asked, the answer resumed the turn, and the final reply reflected
|
||||
the answer; 1 otherwise (CI-usable).
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from datetime import datetime
|
||||
|
||||
HERE = pathlib.Path(__file__).parent
|
||||
REPO_ROOT = HERE.parent
|
||||
|
||||
# The color the primary will hand back when the worker asks. The final reply must reflect it,
|
||||
# uppercased — proof the answer actually reached the resumed turn (not a value the worker could
|
||||
# have guessed: it is told to ask, and only the primary knows which of red/blue is chosen).
|
||||
ANSWER_COLOR = "blue"
|
||||
|
||||
# A task that CANNOT be completed without asking: the worker is not told which color to choose,
|
||||
# only that the primary will name one when asked. So a correct final reply is only reachable by
|
||||
# actually calling fleet_ask and using the answer.
|
||||
TASK_PROMPT = (
|
||||
"You are a bridge worker in a quick coordination game. You do NOT know which color to pick — "
|
||||
"only the primary does. Do exactly this, in order:\n"
|
||||
"1. Call the `fleet_ask` tool with EXACTLY this question: \"PICK A COLOR: red or blue?\"\n"
|
||||
"2. The primary will answer with one color word. Take that color and uppercase it.\n"
|
||||
"3. Call `fleet_reply` with EXACTLY one line: CHOSEN=<COLOR> (e.g. CHOSEN=GREEN if told green).\n"
|
||||
"Do not guess a color. Do not call fleet_reply before fleet_ask has returned an answer. "
|
||||
"Do nothing else — no file reads, no other tools."
|
||||
)
|
||||
|
||||
|
||||
def http(base, method, path, body=None, timeout=20):
|
||||
"""JSON request. Tolerates an empty body (e.g. 204) → {}. Raises on non-2xx via urllib."""
|
||||
data = json.dumps(body).encode() if body is not None else None
|
||||
req = urllib.request.Request(base + path, data=data, method=method,
|
||||
headers={"Content-Type": "application/json"})
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
raw = r.read().decode().strip()
|
||||
return json.loads(raw) if raw else {}
|
||||
|
||||
|
||||
def now():
|
||||
return datetime.now().strftime("%H:%M:%S")
|
||||
|
||||
|
||||
def spawn_worker(base, profile, repo):
|
||||
res = http(base, "POST", "/workers", {"profile": profile, "cwd": repo} if profile
|
||||
else {"cwd": repo})
|
||||
tid = res.get("terminalId") or res.get("sessionId")
|
||||
if not tid:
|
||||
raise RuntimeError(f"spawn failed: {res}")
|
||||
pane = res.get("paneId")
|
||||
print(f"[{now()}] spawned {tid} (profile={profile or 'default'}, pane={pane})")
|
||||
return tid, pane
|
||||
|
||||
|
||||
def await_ready(base, tid, timeout=150):
|
||||
deadline = time.time() + timeout
|
||||
while time.time() < deadline:
|
||||
try:
|
||||
st = http(base, "GET", f"/sessions/{tid}/status")
|
||||
except urllib.error.URLError:
|
||||
st = {}
|
||||
if st.get("ready"):
|
||||
print(f"[{now()}] worker ready (status={st.get('status')})")
|
||||
return True
|
||||
time.sleep(3)
|
||||
print(f"[{now()}] WARNING: worker never reported ready within {timeout}s — sending anyway")
|
||||
return False
|
||||
|
||||
|
||||
def post_message(base, tid, body, timeout):
|
||||
"""One BLOCKING send. Returns (http_status, parsed_json). urllib raises on 4xx/5xx, so a
|
||||
stale-turn 409 is surfaced here rather than swallowed."""
|
||||
data = json.dumps(body).encode()
|
||||
req = urllib.request.Request(f"{base}/sessions/{tid}/message", data=data, method="POST",
|
||||
headers={"Content-Type": "application/json"})
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
raw = r.read().decode().strip()
|
||||
return r.status, (json.loads(raw) if raw else {})
|
||||
except urllib.error.HTTPError as e:
|
||||
raw = e.read().decode().strip()
|
||||
return e.code, (json.loads(raw) if raw else {})
|
||||
|
||||
|
||||
def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
"""Drive the full reverse rendezvous. Returns a result record for grading + transcript."""
|
||||
rec = {"spawned": False, "tid": None, "pane": None, "phase": "spawn",
|
||||
"question": None, "turnId": None, "reply": None, "replySource": None,
|
||||
"detail": None, "ask_latency": None, "answer_latency": None}
|
||||
|
||||
tid, pane = spawn_worker(base, profile, repo)
|
||||
rec.update(tid=tid, pane=pane, spawned=True)
|
||||
await_ready(base, tid)
|
||||
|
||||
# 1) Delegate the ask-forcing task. This blocks until the worker calls fleet_ask, at which
|
||||
# point our own send unblocks carrying the question and the turnId to answer on.
|
||||
print(f"[{now()}] delegating task (blocks until the worker asks; up to {send_timeout}s)…")
|
||||
t0 = time.time()
|
||||
rec["phase"] = "awaiting_question"
|
||||
try:
|
||||
code, resp = post_message(base, tid, {"content": TASK_PROMPT, "timeoutMs": send_timeout * 1000},
|
||||
timeout=send_timeout + 15)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec.update(phase="send_error", detail=f"delegating send failed: {e}")
|
||||
return rec
|
||||
rec["ask_latency"] = round(time.time() - t0, 1)
|
||||
|
||||
if resp.get("status") != "question":
|
||||
# The worker finished (or stalled) without asking — the whole point didn't happen.
|
||||
rec.update(phase="no_question", detail=f"HTTP {code}: {json.dumps(resp)[:300]}",
|
||||
reply=resp.get("reply"), replySource=resp.get("replySource"))
|
||||
return rec
|
||||
rec.update(phase="question", question=resp.get("question"), turnId=resp.get("turnId"))
|
||||
print(f"[{now()}] worker ASKED ({rec['ask_latency']}s): {rec['question']!r} turnId={rec['turnId']}")
|
||||
|
||||
if not rec["turnId"]:
|
||||
rec.update(phase="no_turnid", detail="question surfaced without a turnId to answer on")
|
||||
return rec
|
||||
|
||||
# 2) Answer on that exact turn. This blocks again until the resumed worker calls fleet_reply.
|
||||
print(f"[{now()}] answering '{ANSWER_COLOR}' on turn {rec['turnId']} (blocks until reply; up to {answer_timeout}s)…")
|
||||
t1 = time.time()
|
||||
rec["phase"] = "awaiting_reply"
|
||||
try:
|
||||
code, resp = post_message(base, tid,
|
||||
{"content": ANSWER_COLOR, "turnId": rec["turnId"],
|
||||
"timeoutMs": answer_timeout * 1000},
|
||||
timeout=answer_timeout + 15)
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec.update(phase="answer_error", detail=f"answer send failed: {e}")
|
||||
return rec
|
||||
rec["answer_latency"] = round(time.time() - t1, 1)
|
||||
|
||||
if code == 409 or resp.get("error") == "stale_turn":
|
||||
rec.update(phase="stale_turn", detail=f"HTTP {code}: {json.dumps(resp)[:300]}")
|
||||
return rec
|
||||
if resp.get("reply") is None:
|
||||
rec.update(phase="no_reply", detail=f"HTTP {code}: {json.dumps(resp)[:300]}")
|
||||
return rec
|
||||
rec.update(phase="replied", reply=resp.get("reply"), replySource=resp.get("replySource"))
|
||||
print(f"[{now()}] worker RESUMED and replied ({rec['answer_latency']}s): {rec['reply']!r} "
|
||||
f"(source={rec['replySource']})")
|
||||
return rec
|
||||
|
||||
|
||||
def grade(rec):
|
||||
"""PASS only if the worker asked, the turn resumed, and the reply reflects the answer."""
|
||||
if rec["phase"] == "no_question":
|
||||
return "NO_ASK", "the worker finished/stalled without ever calling fleet_ask"
|
||||
if rec["phase"] in ("send_error", "answer_error", "spawn"):
|
||||
return "ERROR", rec.get("detail") or "transport error before the round-trip completed"
|
||||
if rec["phase"] == "no_turnid":
|
||||
return "NO_TURNID", "the question surfaced without a turnId — the primary could not answer"
|
||||
if rec["phase"] == "stale_turn":
|
||||
return "STALE", "answering the turn was rejected as stale (it lapsed or was already answered)"
|
||||
if rec["phase"] in ("awaiting_reply", "no_reply"):
|
||||
return "NO_RESUME", "the worker asked but never resumed to a final reply within the window"
|
||||
if rec["phase"] == "replied":
|
||||
reflected = ANSWER_COLOR.upper() in (rec["reply"] or "").upper()
|
||||
if reflected and rec["replySource"] == "reply":
|
||||
return "OK", "asked, resumed the same turn, and the reply reflected the primary's answer"
|
||||
if reflected:
|
||||
return "DEGRADED", f"reply reflected the answer but resolved via {rec['replySource']} " \
|
||||
"(worker did not call fleet_reply cleanly)"
|
||||
return "WRONG_ANSWER", f"the worker replied but did not reflect '{ANSWER_COLOR}' — " \
|
||||
f"the answer may not have reached the resumed turn: {rec['reply']!r}"
|
||||
return "WEDGE", f"unexpected terminal phase {rec['phase']}: {rec.get('detail')}"
|
||||
|
||||
|
||||
def write_transcript(out_dir, rec, meta):
|
||||
path = out_dir / "fleet_ask_transcript.md"
|
||||
g, note = grade(rec)
|
||||
with path.open("w") as f:
|
||||
f.write(f"# Live fleet_ask — reverse rendezvous — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"One worker paused its delegated turn to ask the primary, then resumed with the "
|
||||
f"answer (profile `{meta['profile']}`). Result: **`{g}`**.\n\n")
|
||||
f.write("## Round-trip\n\n")
|
||||
f.write(f"1. **primary → worker** (delegation): the ask-forcing task.\n")
|
||||
f.write(f"2. **worker → primary** (`fleet_ask`, {rec.get('ask_latency')}s): "
|
||||
f"{rec.get('question')!r} — surfaced on the primary's blocked send as a "
|
||||
f"`question` with `turnId={rec.get('turnId')}`.\n")
|
||||
f.write(f"3. **primary → worker** (answer on that turn): `{ANSWER_COLOR}`.\n")
|
||||
f.write(f"4. **worker → primary** (`fleet_reply`, {rec.get('answer_latency')}s, "
|
||||
f"source={rec.get('replySource')}): {rec.get('reply')!r}\n\n")
|
||||
f.write(f"> **{g}:** {note}\n")
|
||||
if rec.get("detail"):
|
||||
f.write(f">\n> detail: {rec['detail']}\n")
|
||||
return path
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Live fleet_ask reverse-rendezvous test (CB-205)")
|
||||
ap.add_argument("--base", default="http://127.0.0.1:8765")
|
||||
ap.add_argument("--profile", default=None)
|
||||
ap.add_argument("--repo", default=str(REPO_ROOT))
|
||||
ap.add_argument("--out", default=str(HERE))
|
||||
ap.add_argument("--send-timeout", type=int, default=110, help="max wait for the worker to ASK")
|
||||
ap.add_argument("--answer-timeout", type=int, default=90, help="max wait for the worker to REPLY")
|
||||
ap.add_argument("--keep-worker", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
print(f"[{now()}] live fleet_ask: 1 primary, 1 worker "
|
||||
f"(profile={args.profile or 'default'}, repo={args.repo})\n")
|
||||
|
||||
rec = {"spawned": False, "pane": None}
|
||||
try:
|
||||
rec = run(args.base, args.profile, args.repo, args.send_timeout, args.answer_timeout)
|
||||
finally:
|
||||
if not args.keep_worker and rec.get("spawned") and rec.get("pane"):
|
||||
try:
|
||||
http(args.base, "DELETE", f"/workers/{rec['pane']}")
|
||||
print(f"[{now()}] stopped worker (pane {rec['pane']})")
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f"[{now()}] stop failed (ignore): {e}")
|
||||
|
||||
g, note = grade(rec)
|
||||
out_dir = pathlib.Path(args.out)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
path = write_transcript(out_dir, rec, {"profile": args.profile or "default"})
|
||||
|
||||
print()
|
||||
print("=" * 72)
|
||||
print("LIVE fleet_ask SUMMARY — reverse rendezvous (CB-205)")
|
||||
print(f" asked: {rec.get('question')!r} (turnId={rec.get('turnId')}, {rec.get('ask_latency')}s)")
|
||||
print(f" answered: {ANSWER_COLOR!r}")
|
||||
print(f" replied: {rec.get('reply')!r} (source={rec.get('replySource')}, {rec.get('answer_latency')}s)")
|
||||
print(f" transcript: {path}")
|
||||
print(f" RESULT: {g} — {note}")
|
||||
print("=" * 72)
|
||||
|
||||
sys.exit(0 if g == "OK" else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,12 @@
|
||||
# Live fleet_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
|
||||
One worker paused its delegated turn to ask the primary, then resumed with the answer (profile `default`). Result: **`OK`**.
|
||||
|
||||
## Round-trip
|
||||
|
||||
1. **primary → worker** (delegation): the ask-forcing task.
|
||||
2. **worker → primary** (`fleet_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
3. **primary → worker** (answer on that turn): `blue`.
|
||||
4. **worker → primary** (`fleet_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
|
||||
> **OK:** asked, resumed the same turn, and the reply reflected the primary's answer
|
||||
@@ -0,0 +1,330 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge fan-out test — ONE primary vs MANY workers, concurrently, for
|
||||
issue hunting through the running `fleetd` daemon, fully captured, with gap analysis.
|
||||
|
||||
Where conversation_test.py exercises a single worker over multiple turns, this drives
|
||||
the path that only appears under fan-out: the primary spawns N workers, sends each a
|
||||
distinct issue-hunting assignment on a slice of the codebase, fires them all at once,
|
||||
and collects every reply concurrently. That stresses what a single worker never can —
|
||||
|
||||
• simultaneous delivery to many panes (the injector's per-worker, not global, writer),
|
||||
• per-session rendezvous isolation (N blocked sends resolving independently),
|
||||
• reply routing under concurrency (worker A's answer must never resolve worker B's send),
|
||||
|
||||
and, as the payload, whether a fleet of off-subscription workers can actually surface
|
||||
real issues in the repo and report them back structurally via fleet_reply.
|
||||
|
||||
It talks ONLY to the bridge's REST face on loopback — it never sets ANTHROPIC_BASE_URL
|
||||
and never touches herdr directly, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 issue_hunt_test.py [--base URL] [--profile NAME] [--repo DIR]
|
||||
[--out DIR] [--poll-timeout SECS] [--keep-workers]
|
||||
[--assignments FILE]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
--profile profile for every worker (default: the daemon's default)
|
||||
--repo cwd handed to each worker (default: the bridge repo root)
|
||||
--out output dir for the transcript (default: alongside this file)
|
||||
--poll-timeout per-worker max wait, seconds (default 300)
|
||||
--keep-workers do not stop spawned workers at the end
|
||||
--assignments JSON file overriding the built-in assignment list
|
||||
|
||||
Exit code: 0 if every worker delivered AND produced a usable reply with no cross-talk;
|
||||
1 otherwise (CI-usable). A per-worker and fleet-level gap report is printed to stdout.
|
||||
"""
|
||||
import argparse
|
||||
import json
|
||||
import pathlib
|
||||
import sys
|
||||
import threading
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
from concurrent.futures import ThreadPoolExecutor
|
||||
from datetime import datetime
|
||||
|
||||
HERE = pathlib.Path(__file__).parent
|
||||
REPO_ROOT = HERE.parent
|
||||
|
||||
# Each worker gets a distinct source file to hunt in, plus a `probe` — a token its reply
|
||||
# should mention if it actually addressed ITS assignment (a soft cross-talk detector: a
|
||||
# reply that references only another worker's file is a routing red flag). Targets are the
|
||||
# hot files this project has been iterating on, so a real issue is plausible to find.
|
||||
DEFAULT_ASSIGNMENTS = [
|
||||
{"id": "completion", "probe": "CompletionResolver",
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java"},
|
||||
{"id": "worker", "probe": "WorkerService",
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/worker/WorkerService.java"},
|
||||
{"id": "rendezvous", "probe": "Rendezvous",
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/msg/Rendezvous.java"},
|
||||
]
|
||||
|
||||
PROMPT_TMPL = (
|
||||
"You are one of several issue-hunting workers in the claude-bridge repo (it is your "
|
||||
"current working directory). Your assignment: inspect the file `{target}` and find the "
|
||||
"SINGLE most important real bug, correctness gap, or risk in it. Read the file before "
|
||||
"answering. Reply via fleet_reply with EXACTLY these four lines:\n"
|
||||
"1. {target}:<line>\n"
|
||||
"2. issue: <one sentence>\n"
|
||||
"3. fix: <one line>\n"
|
||||
"4. severity: high|medium|low\n"
|
||||
"Keep it under 90 words. If after reading you find nothing real, reply 'NO ISSUE' and one "
|
||||
"line why. Do NOT hunt in any other file — only `{target}`."
|
||||
)
|
||||
|
||||
|
||||
def http(base, method, path, body=None, timeout=20):
|
||||
"""JSON request. Tolerates an empty body (e.g. 204 No Content on DELETE) → returns {}."""
|
||||
data = json.dumps(body).encode() if body is not None else None
|
||||
req = urllib.request.Request(base + path, data=data, method=method,
|
||||
headers={"Content-Type": "application/json"})
|
||||
with urllib.request.urlopen(req, timeout=timeout) as r:
|
||||
raw = r.read().decode().strip()
|
||||
return json.loads(raw) if raw else {}
|
||||
|
||||
|
||||
def now():
|
||||
return datetime.now().strftime("%H:%M:%S")
|
||||
|
||||
|
||||
def spawn_worker(base, profile, repo, wid):
|
||||
res = http(base, "POST", "/workers", {"profile": profile, "cwd": repo} if profile
|
||||
else {"cwd": repo})
|
||||
tid = res.get("terminalId") or res.get("sessionId")
|
||||
if not tid:
|
||||
raise RuntimeError(f"spawn failed for {wid}: {res}")
|
||||
pane = res.get("paneId")
|
||||
print(f"[{now()}] [{wid}] spawned {tid} (profile={profile or 'default'}, pane={pane})")
|
||||
return tid, pane
|
||||
|
||||
|
||||
def await_ready(base, tid, wid, timeout=150):
|
||||
deadline = time.time() + timeout
|
||||
while time.time() < deadline:
|
||||
try:
|
||||
st = http(base, "GET", f"/sessions/{tid}/status")
|
||||
except urllib.error.URLError:
|
||||
st = {}
|
||||
if st.get("ready"):
|
||||
print(f"[{now()}] [{wid}] ready (status={st.get('status')})")
|
||||
return True
|
||||
time.sleep(3)
|
||||
print(f"[{now()}] [{wid}] WARNING: never reported ready within {timeout}s — sending anyway")
|
||||
return False
|
||||
|
||||
|
||||
def fire(base, tid, prompt):
|
||||
"""Fire one async send; return its ticket (delivery is confirmed by a ticket coming back)."""
|
||||
sent = http(base, "POST", f"/sessions/{tid}/message", {"wait": False, "content": prompt})
|
||||
return sent.get("ticket")
|
||||
|
||||
|
||||
def poll(base, ticket, wid, t0, poll_timeout):
|
||||
"""Poll a ticket to resolution; return (record fields) mirroring conversation_test."""
|
||||
samples, reply, detail, source, phase = [], None, None, None, None
|
||||
deadline = time.time() + poll_timeout
|
||||
while time.time() < deadline:
|
||||
time.sleep(3)
|
||||
task = http(base, "GET", f"/tasks/{ticket}")
|
||||
phase = task.get("phase")
|
||||
live = (task.get("detail") or "").replace("worker ", "") if phase == "pending" else ""
|
||||
samples.append((round(time.time() - t0, 1), phase, live))
|
||||
if phase in ("done", "failed"):
|
||||
reply, detail, source = task.get("reply"), task.get("detail"), task.get("replySource")
|
||||
break
|
||||
trans, last = [], None
|
||||
for el, ph, live in samples:
|
||||
tag = live if ph == "pending" else ph
|
||||
if tag != last:
|
||||
trans.append(f"{tag}@{el}s")
|
||||
last = tag
|
||||
return {"phase": phase, "reply": reply, "detail": detail, "source": source,
|
||||
"latency": round(time.time() - t0, 1), "transitions": " → ".join(trans)}
|
||||
|
||||
|
||||
def grade(rec):
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
return "FAILED", f"worker turn failed: {(rec['detail'] or '').strip()[:120]}"
|
||||
return "WEDGE", "never resolved within the poll window (delivery wedge or lost turn)"
|
||||
|
||||
|
||||
def worker_lifecycle(base, profile, repo, poll_timeout, a, barrier):
|
||||
"""Full per-worker path: spawn → ready → (barrier) → fire → poll. Spawn+ready run
|
||||
concurrently across workers; the send waits on the shared `barrier` so every ready
|
||||
worker fires within the same instant — the real simultaneous-delivery stress. A worker
|
||||
that fails to spawn aborts the barrier so the rest don't block forever."""
|
||||
wid = a["id"]
|
||||
rec = {"id": wid, "target": a["target"], "probe": a["probe"], "spawned": False,
|
||||
"tid": None, "pane": None, "ticket": None}
|
||||
try:
|
||||
tid, pane = spawn_worker(base, profile, repo, wid)
|
||||
rec.update(tid=tid, pane=pane, spawned=True)
|
||||
await_ready(base, tid, wid)
|
||||
except Exception as e: # noqa: BLE001
|
||||
barrier.abort() # release peers waiting on the barrier
|
||||
rec.update(phase="failed", detail=f"spawn/ready error: {e}", reply=None,
|
||||
source=None, latency=0.0, transitions="")
|
||||
return rec
|
||||
|
||||
try:
|
||||
barrier.wait(timeout=210) # all ready workers proceed together
|
||||
except (threading.BrokenBarrierError, Exception): # noqa: BLE001
|
||||
pass # a peer died or timed out — fire anyway rather than hang
|
||||
t0 = time.time()
|
||||
prompt = PROMPT_TMPL.format(target=a["target"])
|
||||
try:
|
||||
ticket = fire(base, tid, prompt)
|
||||
rec["ticket"] = ticket
|
||||
print(f"[{now()}] [{wid}] fired (ticket={ticket})")
|
||||
rec.update(poll(base, ticket, wid, t0, poll_timeout))
|
||||
except Exception as e: # noqa: BLE001
|
||||
rec.update(phase="failed", detail=f"send error: {e}", reply=None,
|
||||
source=None, latency=round(time.time() - t0, 1), transitions="")
|
||||
return rec
|
||||
|
||||
|
||||
def crosstalk_report(records):
|
||||
"""Fleet-level isolation checks: distinct tickets, distinct non-empty replies, and each
|
||||
reply addressing its OWN assigned file (probe token present). Returns (list_of_gaps)."""
|
||||
gaps = []
|
||||
tickets = [r.get("ticket") for r in records if r.get("ticket")]
|
||||
if len(tickets) != len(set(tickets)):
|
||||
gaps.append("ticket collision: two workers were handed the same ticket id")
|
||||
replies = {r["id"]: (r.get("reply") or "").strip() for r in records}
|
||||
# identical non-empty replies from distinct assignments ⇒ suspected reply misrouting
|
||||
seen = {}
|
||||
for wid, text in replies.items():
|
||||
if text and text in seen:
|
||||
gaps.append(f"identical reply from '{seen[text]}' and '{wid}' "
|
||||
f"(distinct assignments should not yield byte-identical answers)")
|
||||
elif text:
|
||||
seen[text] = wid
|
||||
# a reply that names ANOTHER worker's file but not its own ⇒ likely cross-routing
|
||||
for r in records:
|
||||
text = (r.get("reply") or "")
|
||||
if not text.strip():
|
||||
continue
|
||||
own = r["probe"] in text or pathlib.Path(r["target"]).name in text
|
||||
others = [o["probe"] for o in records if o["id"] != r["id"] and o["probe"] in text]
|
||||
if not own and others:
|
||||
gaps.append(f"worker '{r['id']}' (assigned {r['probe']}) replied about "
|
||||
f"{others} but not its own file — possible cross-routing")
|
||||
return gaps
|
||||
|
||||
|
||||
def write_transcript(out_dir, records, meta):
|
||||
path = out_dir / "issue_hunt_transcript.md"
|
||||
with path.open("w") as f:
|
||||
f.write(f"# Bridge fan-out issue-hunt — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"One primary vs **{len(records)} concurrent workers** "
|
||||
f"(profile `{meta['profile']}`), each hunting a distinct file.\n\n")
|
||||
for r in records:
|
||||
g, note = grade(r)
|
||||
f.write(f"### `{r['id']}` — {r['target']} (`{g}`, {r.get('latency')}s, "
|
||||
f"phase={r.get('phase')}, source={r.get('source')})\n\n")
|
||||
f.write(f"**ASSIGNMENT:** find the top issue in `{r['target']}`\n\n")
|
||||
reply = r.get("reply")
|
||||
f.write(f"**WORKER {r['id']}:** {reply if reply else '_(no reply)_ ' + str(r.get('detail'))}\n\n")
|
||||
if r.get("transitions"):
|
||||
f.write(f"_status: {r['transitions']}_\n")
|
||||
if g != "OK":
|
||||
f.write(f"\n> **GAP — {g}:** {note}\n")
|
||||
f.write("\n")
|
||||
if meta["gaps"]:
|
||||
f.write("## Fleet-level gaps\n\n")
|
||||
for gp in meta["gaps"]:
|
||||
f.write(f"- {gp}\n")
|
||||
return path
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Bridge fan-out issue-hunt test (1 primary, N workers)")
|
||||
ap.add_argument("--base", default="http://127.0.0.1:8765")
|
||||
ap.add_argument("--profile", default=None)
|
||||
ap.add_argument("--repo", default=str(REPO_ROOT))
|
||||
ap.add_argument("--out", default=str(HERE))
|
||||
ap.add_argument("--poll-timeout", type=int, default=300)
|
||||
ap.add_argument("--keep-workers", action="store_true")
|
||||
ap.add_argument("--assignments", default=None)
|
||||
args = ap.parse_args()
|
||||
|
||||
assignments = DEFAULT_ASSIGNMENTS
|
||||
if args.assignments:
|
||||
assignments = json.loads(pathlib.Path(args.assignments).read_text())
|
||||
|
||||
n = len(assignments)
|
||||
barrier = threading.Barrier(n) # releases exactly when all n ready workers reach it
|
||||
print(f"[{now()}] fan-out issue-hunt: 1 primary vs {n} workers "
|
||||
f"(profile={args.profile or 'default'}, repo={args.repo})\n")
|
||||
|
||||
# Spawn + ready + fire + poll all workers concurrently; the barrier makes every send fire
|
||||
# together once all are ready, so delivery pressure hits the daemon simultaneously.
|
||||
records = []
|
||||
with ThreadPoolExecutor(max_workers=n) as ex:
|
||||
futures = [ex.submit(worker_lifecycle, args.base, args.profile, args.repo,
|
||||
args.poll_timeout, a, barrier) for a in assignments]
|
||||
for fut in futures:
|
||||
records.append(fut.result())
|
||||
|
||||
records.sort(key=lambda r: [a["id"] for a in assignments].index(r["id"]))
|
||||
|
||||
print()
|
||||
for r in records:
|
||||
g, note = grade(r)
|
||||
print(f"[{r['id']:11}] {g:8} {str(r.get('latency','?')):6}s "
|
||||
f"phase={r.get('phase')} source={r.get('source')}")
|
||||
print(f" status: {r.get('transitions') or '(none)'}")
|
||||
print(f" reply: {((r.get('reply') or '(none) ' + str(r.get('detail'))).strip()[:200])}")
|
||||
print(f" note: {note}\n")
|
||||
|
||||
gaps = crosstalk_report(records)
|
||||
meta = {"profile": args.profile or "default", "gaps": gaps}
|
||||
out_dir = pathlib.Path(args.out)
|
||||
out_dir.mkdir(parents=True, exist_ok=True)
|
||||
path = write_transcript(out_dir, records, meta)
|
||||
|
||||
grades = [grade(r)[0] for r in records]
|
||||
counts = {g: grades.count(g) for g in ("OK", "DEGRADED", "EMPTY", "FAILED", "WEDGE") if grades.count(g)}
|
||||
print("=" * 72)
|
||||
print(f"FAN-OUT ISSUE-HUNT SUMMARY — 1 primary vs {n} workers")
|
||||
print(" channel: " + " ".join(f"{g}:{v}" for g, v in counts.items()))
|
||||
print(f" transcript: {path}")
|
||||
if gaps:
|
||||
print(" FLEET GAPS:")
|
||||
for gp in gaps:
|
||||
print(f" ⚠ {gp}")
|
||||
else:
|
||||
print(" isolation: clean — distinct tickets, distinct replies, each on its own file")
|
||||
delivered = all(r.get("ticket") for r in records)
|
||||
replied = all(g in ("OK", "DEGRADED") for g in grades)
|
||||
if delivered and replied and not gaps:
|
||||
print(" RESULT: PASS — all workers delivered concurrently, replied, and stayed isolated.")
|
||||
elif delivered and replied:
|
||||
print(" RESULT: PASS (with notes) — all delivered & replied, but see FLEET GAPS.")
|
||||
else:
|
||||
print(" RESULT: FAIL — a worker did not deliver or did not reply (see GAP notes).")
|
||||
print("=" * 72)
|
||||
|
||||
if not args.keep_workers:
|
||||
for r in records:
|
||||
if r.get("spawned") and r.get("pane"):
|
||||
try:
|
||||
http(args.base, "DELETE", f"/workers/{r['pane']}")
|
||||
print(f"[{now()}] [{r['id']}] stopped (pane {r['pane']})")
|
||||
except Exception as e: # noqa: BLE001
|
||||
print(f"[{now()}] [{r['id']}] stop failed (ignore): {e}")
|
||||
|
||||
sys.exit(0 if (delivered and replied and not gaps) else 1)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,16 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
||||
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
||||
fleetd.yaml
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -0,0 +1,142 @@
|
||||
# CB-307 — Active push-to-primary + reminder loop (the reliability layer)
|
||||
|
||||
**Status:** design (2026-07-19). Builds directly on the shipped durable landing zone
|
||||
(`AmqpReplyInbox`, main `2bc5f3a`, dogfooded live). gitea #5.
|
||||
|
||||
## Why this exists
|
||||
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `fleet_send` is
|
||||
open: it lands in `agent.<target>.inbox` on the broker and survives a daemon bounce. But
|
||||
delivery is still **pull** — the primary only sees the reply if it happens to call
|
||||
`fleet_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
the primary works on something else.
|
||||
|
||||
This layer makes delivery **active**: the bridge *pushes* a nudge to the primary the moment
|
||||
a reply lands, and keeps reminding (bounded) until the primary drains it. At-least-once,
|
||||
dedup by `msgId`, and — critically — it never loses the reply even if every push fails,
|
||||
because the durable inbox is the backstop.
|
||||
|
||||
## The hard constraint it works around
|
||||
|
||||
The bridge is an MCP **server**; the primary is an MCP **client**. A server cannot call
|
||||
into a client. So "push to the primary" cannot be an MCP response — it needs a *sideband*
|
||||
channel. The chosen channel: **inject a synthetic user-turn into the primary's own herdr
|
||||
terminal pane** — the same mechanism the bridge already uses to deliver tasks to workers,
|
||||
pointed at the primary's pane instead.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
W["worker"] -->|"fleet_reply (no open send)"| MS["MessageService.reply"]
|
||||
MS -->|"inbox.publish"| INBOX[("agent.<target>.inbox<br/>(durable, LavinMQ)")]
|
||||
MS -->|"notify"| LOOP["ReplyPushLoop"]
|
||||
LOOP -->|"status-gated inject"| PANE["primary's herdr pane"]
|
||||
PANE -->|"primary drains"| DRAIN["fleet_poll(target)<br/>= peek + ack"]
|
||||
DRAIN -->|"inbox now empty"| LOOP
|
||||
LOOP -.->|"still non-empty →<br/>re-inject on backoff"| PANE
|
||||
classDef store fill:#2c5282,stroke:#1a365d,color:#ffffff;
|
||||
class INBOX store
|
||||
```
|
||||
|
||||
*Figure 1 — a reply lands in the durable inbox; the push loop nudges the primary's pane;
|
||||
the primary's drain acks it; a still-full inbox triggers a bounded re-nudge.*
|
||||
|
||||
## Three increments
|
||||
|
||||
### Increment 1 — learn & store the primary's terminal_id
|
||||
|
||||
**Finding (seam map):** `ConnectionIdentity.resolve(remoteAddr, remotePort)` already returns
|
||||
the caller's herdr `terminal_id` for *every* MCP call, via `PaneLocator.terminalForPid`
|
||||
(walks `pane.list`, matches the caller PID to a pane's process tree). It is non-null whenever
|
||||
the caller runs in a herdr pane on this host. Today it's discarded for the primary
|
||||
(`presence.markPresent` is a no-op on it).
|
||||
|
||||
**Plan:** a single-slot `PrimaryRegistry` (thread-safe) holding the primary's `terminal_id`.
|
||||
Populate it from the **orchestration-side** MCP tools — `fleet_send`, `fleet_spawn`,
|
||||
`fleet_poll`, `fleet_list`, `fleet_status`, `fleet_profiles` — capturing
|
||||
`callerTerminal(exchange)` when it is (a) non-null and (b) **not** a registered worker
|
||||
session in `SessionManager`. That caller is, by construction, the primary. Worker-side tools
|
||||
(`fleet_reply`, `fleet_ask`) never set it.
|
||||
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `FleetConfig`
|
||||
(nested record, same shape as `Broker`). Lets an operator pin it, or supply it when
|
||||
derivation can't (see degrade case).
|
||||
- **Degrade:** if the primary is off-host or in a non-herdr terminal, `terminalForPid`
|
||||
returns null and no override is set → **the registry stays empty → the push loop is a
|
||||
no-op and we fall back to pull** (today's behaviour). The reply is never lost; it's just
|
||||
not actively pushed. This is a safe, explicit degradation, not a failure.
|
||||
|
||||
### Increment 2 — the push loop
|
||||
|
||||
A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
(`MessageService.reply` → the `inbox.publish` branch, `MessageService.java:192`).
|
||||
|
||||
- **Inject a nudge, not the payload.** The injected turn tells the primary *to drain*
|
||||
(e.g. "Worker `<target>` returned a reply — run `fleet_poll(target=<target>)` to collect
|
||||
it"), it does **not** carry the reply text. Rationale: replies can be large/multiline and
|
||||
terminal injection would mangle them; the drain response is the clean transport. Keeps the
|
||||
push idempotent — re-nudging is harmless.
|
||||
- **Ack = drain.** The primary draining (`drainReplies` = peek + ack) is the acknowledgement.
|
||||
The loop's **stop condition is `inbox.peek(target).isEmpty()`** — the reply is gone from the
|
||||
inbox because it was acked. No new `fleet_ack` tool needed for v1 (see Increment 3).
|
||||
- **Status-gated injection (mechanism (b), chosen).** A dedicated lightweight scheduled loop,
|
||||
**not** the worker `Injector`. It injects via `AgentControl.send(primaryTerminal, nudge)`
|
||||
(the same herdr `agent.send` = `pane send-text` + submit that delivers to workers) only when
|
||||
`AgentControl.status(primaryTerminal).injectable()` (IDLE/BLOCKED) — never mid-turn. This keeps
|
||||
the primary path fully isolated from `WorkerPresence`/`StatusPoller` (which are worker-scoped),
|
||||
and makes it unit-testable with a fake `AgentControl` + an injected clock (per the CB-306
|
||||
`LongSupplier` clock + `Runnable` sleeper seam). Rejected (a) reuse-the-Injector: it would force
|
||||
the primary terminal into the worker poller set and couple to worker-presence semantics — more
|
||||
integration surface, harder to test, no real gain for a bounded reminder.
|
||||
- **Bounded reminder / backoff.** While `peek(target)` stays non-empty, re-inject on a
|
||||
backoff schedule up to a cap (N reminders or a max duration; config
|
||||
`primary.push_reminders` / `primary.push_backoff_ms`). After the cap, **stop reminding** —
|
||||
the reply remains in the durable inbox and the next natural poll (or a later worker reply's
|
||||
nudge) still surfaces it. Bounded so the bridge never spams the primary.
|
||||
|
||||
### Increment 3 — optional per-`msgId` `fleet_ack` tool (deferred)
|
||||
|
||||
Drain-as-ack is coarse: it clears *all* pending replies for a target at once. If finer
|
||||
control is ever needed (ack one reply, leave others held), add a `fleet_ack(msgId)` tool
|
||||
mapping to `inbox.ack(target, msgId)` — the port already supports per-`msgId` ack. Not built
|
||||
in v1; the stop-on-empty loop is sufficient.
|
||||
|
||||
## The two subtleties (decided here)
|
||||
|
||||
1. **Which caller is "the primary"?** Connection-derived, not self-reported: the caller whose
|
||||
resolved terminal is non-null **and not a registered worker session**, seen on an
|
||||
orchestration-side tool. This never mislabels a worker (workers are in `SessionManager`)
|
||||
and needs no new env var or argument (identity stays connection-derived, per the existing
|
||||
`FleetMcp` invariant).
|
||||
|
||||
2. **Readiness-gate mismatch → dedicated loop.** The existing `Injector` gates delivery on
|
||||
`ready.test(target)` = `WorkerPresence` (the *worker's* MCP connected). The primary is not
|
||||
in `WorkerPresence`, so reusing `Injector` would mean forcing the primary terminal into the
|
||||
worker `StatusPoller` set and swapping the `ready` predicate — extra integration surface with
|
||||
worker-scoped machinery. Decision: **mechanism (b)** — a small dedicated scheduled loop that
|
||||
calls `AgentControl.status(primaryTerminal).injectable()` then `AgentControl.send(...)`, with
|
||||
an injected clock. Isolated from worker presence, trivially unit-testable, sufficient for a
|
||||
bounded reminder. (Verified live: this primary resolves to `term_656c8cc03e1f0b1`, pane
|
||||
`w2:pY` — the primary genuinely runs in a herdr pane on this host, so the path is exercisable.)
|
||||
|
||||
## Boundary note
|
||||
|
||||
This is the first time the bridge **writes into the primary's pane** — a new direction of
|
||||
control. It stays within the communication-bus identity: the injection is a **nudge** (a
|
||||
synthetic "go drain your replies" turn), **status-gated** so it never interrupts a turn,
|
||||
**bounded** so it never spams, carries **no env** and **never crosses the subscription
|
||||
boundary**. The bridge is signalling the primary that it has mail — not driving its work.
|
||||
|
||||
## Test plan
|
||||
|
||||
- **Unit (hermetic):** `PrimaryRegistry` set/clear/override; the "caller is primary iff
|
||||
non-null terminal AND not a registered session" predicate; the loop's stop-on-empty and
|
||||
bounded-reminder logic with an injected clock + a fake injector (no real herdr).
|
||||
- **Live dogfood (primary-side):** with the daemon on the broker jar + a real worker,
|
||||
delegate a task, let the worker reply after the `fleet_send` window closes, and observe the
|
||||
bridge inject a drain nudge into *this* primary pane; confirm draining stops the reminders;
|
||||
confirm an unreachable primary (registry empty) degrades to pull with no loss.
|
||||
|
||||
## Out of scope
|
||||
|
||||
Multi-host (CB-308) — the push loop is local-only; a remote primary is reached by its own
|
||||
local gateway, not cross-host injection. Federation reuses this loop per-gateway.
|
||||
@@ -0,0 +1,711 @@
|
||||
# fleetd configuration (example). Copy to fleetd.yaml and adjust.
|
||||
#
|
||||
# fleetd is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — fleetd REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
|
||||
# API authentication (CB-501). Governs how a caller that is NOT an on-host worker pane proves it
|
||||
# is the primary. Worker identity never depends on this: a loopback peer PID that maps to a herdr
|
||||
# pane is unforgeable and is always honoured, so turning auth on cannot lock the fleet out.
|
||||
#
|
||||
# mode: loopback-trust → DEFAULT, and the historical behaviour: any loopback caller that is not
|
||||
# a worker is the primary, no credential needed. Sound ONLY because the
|
||||
# OS refuses remote connections to a loopback socket.
|
||||
# mode: token → such a caller must send `Authorization: Bearer <token>`; without it it
|
||||
# is anonymous and authorized for nothing. REQUIRED for a non-loopback
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# FLEETD_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
# location / { proxy_pass http://127.0.0.1:8765; proxy_set_header Authorization $http_authorization; }
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: FLEETD_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from fleet_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
# pushReminders: 5 # max nudges before giving up (default 5)
|
||||
# pushBackoffMs: 15000 # delay between nudges (default 15000)
|
||||
|
||||
# CB-530: MORE THAN ONE LEAD. `primary:` above is singular by construction — every other pane
|
||||
# resolves as a worker — which is right for one lead driving a fleet and wrong the moment two leads
|
||||
# (say a Claude lead and an opencode lead) work as peers: the second is silently demoted and refused
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# tab → the ONLY field identity depends on (CB-579); the exact label of the tab hosting the lead.
|
||||
# Label the tab yourself, or let fleetd label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by fleet_whoami
|
||||
#
|
||||
# A lead's tab must already carry its label (or be launched by fleetd, which labels it) — there is
|
||||
# no terminal id to paste in and nothing to re-pin when the session restarts: the tab survives, so
|
||||
# the same label resolves the same lead again on the next scan.
|
||||
# `fleet_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
# destination for its nudges, and is a separate mechanism from lead identity — see `fleet.leaders:`.
|
||||
#
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing fleetd places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open fleet_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
# Opt-in on purpose — it SPENDS the operator's subscription on its own initiative (each nudge starts
|
||||
# a lead turn nobody asked for), so upgrading the daemon must never switch it on for you. Absent
|
||||
# block = feature off, exactly as before.
|
||||
#
|
||||
# Three knobs, each with a default that errs on the side of not burning context:
|
||||
# idleAfterSeconds: 300 # how long the lead must stay idle before the FIRST nudge (default 300 —
|
||||
# # absorbs normal post-turn pauses; re-prompting every pause burns context)
|
||||
# backoffMs: 60000 # re-check cadence / spacing between nudges past the quiet period (default 60000)
|
||||
# quietNudgeCap: 3 # cap on consecutive nudges that find NOTHING pending, then it stops
|
||||
# # until real state appears (default 3 — never nag an empty fleet forever)
|
||||
# leadHeartbeat:
|
||||
# idleAfterSeconds: 300
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
# per tick.
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
# Math.max(15, intervalSeconds), so a lower value is silently raised, not
|
||||
# rejected.
|
||||
# workingSuspectAfterSeconds → age before a BUSY member is suspected of a stall (default 600).
|
||||
# ENFORCED floor of 300: a lower value is silently raised.
|
||||
# paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by anything. Setting it changes
|
||||
# nothing right now. It exists so a later build can start honouring it without
|
||||
# another config-shape change.
|
||||
# notifications.mode → "webhook" flips what fleet_list REPORTS (healthCoverage: "full" instead
|
||||
# of "detection-only") — it does NOT make fleetd send any webhook call; no
|
||||
# delivery mechanism is implemented yet. Any other value, or omitting the
|
||||
# block, reports "detection-only".
|
||||
# health:
|
||||
# enabled: true
|
||||
# intervalSeconds: 30 # floor 15
|
||||
# workingSuspectAfterSeconds: 600 # floor 300 — how long BUSY with no activity means STALL_SUSPECTED
|
||||
# paneProbeIntervalSeconds: 60 # parsed, but nothing reads it yet — changing it changes nothing
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# Optional socket for member panes. Omit this to use herdrSocket for both leads and members.
|
||||
# memberHerdrSocket: /Users/member/.config/herdr/herdr.sock
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
# it is for; that is the member's role, and roles live under `fleet:` below. Which profile an
|
||||
# unqualified spawn lands on comes from that role's pool, not from a global default.
|
||||
#
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → fleetd mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# ideMcpUrl → opt-in (CB-634), default off. When set, fleetd mounts the IDE Index MCP as a
|
||||
# second inline server named `intellij`, and adds an IDE charter that pins every
|
||||
# ide_* call to the member's own worktree. A URL, not a boolean — host and port
|
||||
# are host-specific. Set it only on a host where the IDE actually runs.
|
||||
# ideProjectDir → repo-relative module dir the IDE opens and the overlay pins (CB-634). Only read
|
||||
# when ideMcpUrl is set. This repo's Maven pom lives in `fleetd/`, not at the
|
||||
# worktree root, so opening the root imports no module and ide_* resolves nothing;
|
||||
# set this to `fleetd`. Omit for a repo whose project is the worktree root.
|
||||
# ideOpenCommand → host command that opens ideProjectDir in the IDE at spawn (CB-634 auto-open).
|
||||
# Only read when ideMcpUrl is set. `{dir}` is replaced with the absolute module
|
||||
# dir and the command runs through `/bin/sh -c`, so set env inline if needed —
|
||||
# e.g. `env DISPLAY=:10.0 idea {dir}`. Best-effort: a failure is logged, never
|
||||
# fails the spawn. Omit to open the member's module by hand. There is no close
|
||||
# half yet — an opened module stays open until the operator closes it.
|
||||
# autoCompactWindow → opt-in, default off. A bounded token window that forces a spawned member to
|
||||
# compact its context instead of running on the backend's own default and dying
|
||||
# mid-turn (losing its fleet_reply — the whole point of the turn — with it).
|
||||
# Validated at config load to [100000, 1000000] — the band Claude Code's own
|
||||
# --autocompact flag accepts.
|
||||
# CROSS-BACKEND SEMANTICS DIFFER: on claude-code this is a launch-time
|
||||
# `--autocompact <tokens>` flag — the member compacts AT this window. opencode
|
||||
# has no equivalent flag (it only forces `compaction.auto: true`, unconditionally,
|
||||
# already), so this is instead applied as the model's `limit.context` in the
|
||||
# generated opencode.json — the member compacts WITHIN this window, not exactly
|
||||
# at it — and only when this profile's `model:` is in `provider/model` form; if it
|
||||
# isn't, fleetd logs a WARN naming the profile rather than silently doing nothing.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
# cwd on an MCP spawn, else the daemon's cwd — never $HOME. See
|
||||
# docs/Worker-Startup-and-Trust.md.
|
||||
# configDir → CLAUDE_CONFIG_DIR for the worker, so it inherits that profile's
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.env, .envrc]. (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
# handed a worker the primary's IDE servers, which are bound to the primary's
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# fleetd neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
# Opt-in by design — omit and the worker gets no PR-create grant (push over
|
||||
# SSH is unaffected). The token value itself is never stored in this file.
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# exhaustedPattern → regex matched against a completion-fallback scrape (CB-578 stage A) to
|
||||
# classify a turn that ended with no fleet_reply as the backend having
|
||||
# refused on a subscription usage limit, rather than a real answer. Opt-in —
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into fleetd itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
# BACKEND_EXHAUSTED classification fires. Two profiles that set the SAME
|
||||
# credentialId share one quarantine — the case this exists for is two models
|
||||
# on one account (e.g. sol and terra both billing one OpenAI credential): an
|
||||
# exhaustion on either one must lock out both, or the fleet just walks onto
|
||||
# the same dead account under the sibling's name. Opt-in — omit and this
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. fleetd hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# fleetd now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.fleet.plist and deploy/fleetd.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
# entry cannot repoint a worker past the SubscriptionGuard — which is checked
|
||||
# against `baseUrl` alone.
|
||||
# Put `defaultMode: "auto"` in each ccs profile so the worker runs autonomously.
|
||||
profiles:
|
||||
gx10: # ccs profile name (NOT a hostname)
|
||||
kind: claude-code # which adapter spawns this profile (default; may omit)
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: FLEETD_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
# weight: relative selection weight for automatic placement (weighted, round-robin, and
|
||||
# fixed's fallback walk). Absent defaults to 1.0. An explicit 0 or negative value means
|
||||
# "never auto-select this profile" (CB-554) — it stays reachable via an explicit
|
||||
# `fleet_spawn{profile:"gx10"}`, which bypasses placement entirely; only automatic
|
||||
# selection skips it.
|
||||
weight: 0.5
|
||||
# maxLoad: max live workers on this profile. Omit for unlimited. An explicit 0 (CB-585) caps
|
||||
# the profile at zero live members — it is excluded from automatic placement and an explicit
|
||||
# `fleet_spawn{profile:"gx10"}` against it is refused too; a cap holds even when the profile
|
||||
# is named directly. Negative is refused at config load — there is no sane meaning for it.
|
||||
maxLoad: 2
|
||||
# subscription: true
|
||||
# THE KNOB THAT DECIDES WHO PAYS (CB-539). Default false. When true, this profile's members
|
||||
# run on the OPERATOR'S OWN Claude subscription instead of a metered endpoint — every spawn
|
||||
# bills your plan and eats your usage limit. Off-subscription is the whole point of this
|
||||
# daemon, so treat `true` as a deliberate exception, not a convenience.
|
||||
#
|
||||
# What changes when it is set (ClaudeCodeLauncher):
|
||||
# - no ANTHROPIC_BASE_URL and no ANTHROPIC_AUTH_TOKEN are injected — the member inherits
|
||||
# the operator's own Claude Code auth, which is exactly why it bills the plan;
|
||||
# - SubscriptionGuard never vets it, because there is no baseUrl to vet;
|
||||
# - no token is required, so `tokenEnv` is irrelevant here.
|
||||
#
|
||||
# MUTUALLY EXCLUSIVE with `baseUrl` — setting both is refused at config load (CB-542). On the
|
||||
# subscription path no guard would vet the URL, so allowing both would be a way around the
|
||||
# guard rather than a configuration.
|
||||
#
|
||||
# GOTCHA 1 — it is invisible to the startup secret check. `Fleetd.reportRequiredSecrets`
|
||||
# skips subscription profiles on purpose (they need no token), so a boot log that reports
|
||||
# every secret as fine says nothing about these profiles.
|
||||
#
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".env", ".envrc"] # the default; never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
# autoCompactWindow: 250000 # opt-in: bound member context; claude-code compacts AT this, opencode within it (model limit.context)
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
weight: 0.5
|
||||
maxLoad: 2
|
||||
# Pin an auto-compact window BELOW the served model's context ceiling. The global
|
||||
# ~/.claude/settings.json value is shared by every ccs instance and the primary, so the
|
||||
# per-profile override belongs here. Equal to the ceiling means auto-compact never fires
|
||||
# before the server rejects the prompt, which kills a worker mid-turn (CB-523).
|
||||
env:
|
||||
CLAUDE_CODE_AUTO_COMPACT_WINDOW: "280000"
|
||||
# CB-402: a second coding-agent kind, proving the PeerLauncher SPI is provider-neutral.
|
||||
# opencode is provider-agnostic and uses NONE of Claude's private seams: no ANTHROPIC_BASE_URL /
|
||||
# SubscriptionGuard (so it needs no `guard` host entry), no --mcp-config / --append-system-prompt.
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → fleet_send →
|
||||
# structured fleet_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
#
|
||||
# CB-508: point an opencode profile at your OWN OpenAI-compatible endpoint (local vLLM, llama.cpp,
|
||||
# LM Studio, TGI…) instead of opencode's gateway. Setting `baseUrl` on a `kind: opencode` profile
|
||||
# makes the bridge emit a custom `provider` block into the generated opencode.json — opencode has
|
||||
# no ANTHROPIC_BASE_URL seam, so this is how the endpoint is pinned.
|
||||
# baseUrl → a bare host:port gets `/v1` appended (where these servers mount the API); a URL that
|
||||
# already has a path is used verbatim, so a custom mount point still works.
|
||||
# model → MUST be "<provider>/<model>". The provider half names the generated block; the model
|
||||
# half must match an id the server reports at /v1/models. One field drives both the
|
||||
# declaration and the `-m` flag, so they cannot drift apart. A bare model name with a
|
||||
# baseUrl set is rejected at spawn rather than silently using the default gateway.
|
||||
# tokenEnv → optional; its value becomes the provider apiKey. Most local servers ignore the key,
|
||||
# so a placeholder is used when unset (the AI SDK still requires a non-empty one).
|
||||
# NOTE: no `guard` entry is needed even with a baseUrl set. The SubscriptionGuard exists to stop a
|
||||
# worker borrowing the primary's Anthropic subscription, and an opencode process has no Anthropic
|
||||
# credential path at all.
|
||||
# opencode-local:
|
||||
# kind: opencode
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
# How an unqualified spawn chooses a profile: fixed (default, reproduces pre-CB-518 behaviour),
|
||||
# round-robin, or weighted. Omitting this key is a strict no-op for existing configs.
|
||||
#
|
||||
# `weighted` IS NOT "cheapest first" — read this before you set weights (CB-589).
|
||||
# It is smooth weighted round-robin: it spreads spawns across EVERY profile that has a free slot,
|
||||
# in weight ratio. It has no idea which profile costs money. So with local:10 / paid:2 you do not
|
||||
# get "use local, overflow to paid" — you get roughly one spawn in six going to the paid profile
|
||||
# while the local box still has a free slot.
|
||||
#
|
||||
# There is a sharper second effect. The policy's running score map lives for the daemon's whole
|
||||
# life. While a profile is at maxLoad it is filtered out and its score FREEZES, so the paid
|
||||
# profiles keep accumulating against it. When the local slot frees up it returns with a stale
|
||||
# score and can LOSE the next pick — a paid spawn while the free box sits idle.
|
||||
#
|
||||
# Until a real cost-first policy exists, the workaround is to make the ratio decisive rather than
|
||||
# proportional: give the free profile a weight so large that it wins every pick it is eligible
|
||||
# for, and paid profiles only ever take genuine overflow. On this host that is local weight 100
|
||||
# against paid weights of ~1.
|
||||
#
|
||||
# The gotcha with that workaround: it expresses a PREFERENCE ORDER through a RATIO knob. Add a
|
||||
# future profile at weight 150 and it silently outranks the free box, with nothing to warn you.
|
||||
# Re-check the weights whenever you add a profile.
|
||||
placement: weighted
|
||||
|
||||
# How long a credential sits out after a BACKEND_EXHAUSTED classification (CB-578 stage B), in
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded fleetd keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. fleetd checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
# Not every key can move under a running daemon, and the difference is about what already exists
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Fleetd.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
# with nothing in the deferred list — but has NO effect until you restart. Treat it
|
||||
# as deferred in practice, even though today's reload output does not say so.
|
||||
# DEFERRED → accepted into the new config, but the wiring built at startup keeps the old value
|
||||
# until you restart: `lifecycle:`, `leadHeartbeat:`, `guard:`, `worktreeRoot:`,
|
||||
# `spawnReadyTimeoutMs` / `spawnReadyPollMs`, `quarantineCooldownSeconds` (CB-578
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
# bound, the broker connection is open, and the auth mode decides who may reach the
|
||||
# port that is already listening.
|
||||
#
|
||||
# A changed COLD key refuses the WHOLE reload — not the hot half applied and the cold half warned
|
||||
# about. A half-applied reload would leave the daemon matching no file on disk, which is the worst
|
||||
# thing a reload can do to an operator debugging one. A file that fails to parse or fails a startup
|
||||
# validator is refused the same way, and the running config stays live.
|
||||
# configReload:
|
||||
# enabled: true
|
||||
# intervalSeconds: 10
|
||||
|
||||
# THE FLEET (CB-557) — who the daemon may run, and under which role. This one block replaced four
|
||||
# older keys: `leaders:`, `members:`, `leadScan:` and `defaultProfile:`.
|
||||
#
|
||||
# A member is anything a lead spawns, and every member has two INDEPENDENT attributes:
|
||||
# role — which contract: architect, dev or reviewer. It picks the launch charter, the role
|
||||
# file, the playbook skill and the authz row.
|
||||
# profile — which backend: one of the `profiles:` keys above (model, CLI adapter, cost).
|
||||
# They vary on their own. A reviewer may run on the same profile as the dev whose diff it reads,
|
||||
# which is why the two cannot be one field.
|
||||
#
|
||||
# The ROLE IS THE CONTAINING KEY, not a `role:` field. That is not only tidier: a misspelled role
|
||||
# used to parse into a member with no contract at all, while a misspelled pool name here simply
|
||||
# declares nothing.
|
||||
#
|
||||
# Each pool lists the profiles that role MAY run on — these are pools, not identities. That is also
|
||||
# what replaced `defaultProfile:`: an unqualified spawn names a role, and that role's pool supplies
|
||||
# the candidates, in definition order. A dev and a reviewer staying anonymous is exactly compatible
|
||||
# with being listed here; the entry key just names the entry.
|
||||
fleet:
|
||||
# Optional launch-charter text, keyed only by the singular role wire names: architect, dev,
|
||||
# reviewer. Changes are HOT and reach the next spawn without a daemon restart. Do not put secrets
|
||||
# here: a later launch step writes this text to a world-readable temp file, and ${ENV} interpolation
|
||||
# is deliberately not supported.
|
||||
charters:
|
||||
architect: |-
|
||||
You are an architect in this fleet. You refine work before anyone builds it:
|
||||
scope, acceptance criteria, risks, and a unit split. You read the repo and
|
||||
write analysis. You never commit production code and never open a PR.
|
||||
A design task is worked by two architects. Design alone first, then exchange
|
||||
and say plainly where you disagree. Do not concede just to agree.
|
||||
dev: |-
|
||||
You implement the one unit you were given, and nothing else. You test it,
|
||||
commit it, and open your own pull request. You never merge.
|
||||
reviewer: |-
|
||||
You review the diff you were given. You report bugs, risks and missing tests.
|
||||
You do not change code.
|
||||
|
||||
# Optional. Template for a member tab's label; {role}, {profile}, {model} and {n} are substituted.
|
||||
# {n} counts per role+profile, so `dev: sonnet #2` really is the second sonnet dev. Because {role}
|
||||
# comes from a closed enum, a generated label can never begin with a lead's tabPrefix.
|
||||
# tabLabel: "{role}: {profile} #{n}"
|
||||
|
||||
# Panes that orchestrate rather than are orchestrated. A lead may now be CREATED as well as
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
# inside it restarts — the terminal id changes; the tab, and its label, do not, so no config edit
|
||||
# follows a restart.
|
||||
#
|
||||
# A lead the daemon launches is labelled BY the daemon with this same `tab:` value, so it is found
|
||||
# by the same scan. A lead counts as live only when herdr also reports a running agent in that
|
||||
# tab — a label left behind by a session that died does not block the relaunch, and a tab that is
|
||||
# gone entirely drops out of the next scan rather than being remembered forever.
|
||||
#
|
||||
# An auto-launched lead is NOT a member: it gets no worker reply charter, is never registered with
|
||||
# the session lifecycle (the idle reaper would kill your orchestrator), and stays on the
|
||||
# subscription — ANTHROPIC_BASE_URL/AUTH_TOKEN are stripped from its env whatever the profile says.
|
||||
#
|
||||
# GET THE `tab:` VALUE RIGHT. A pane that does not match any configured `tab:` (a typo, a renamed
|
||||
# tab, a pane no entry names at all) is not recognised as a lead — it resolves as an ordinary
|
||||
# WORKER instead, silently, and every orchestration call it makes (spawn/stop/send/drain) is
|
||||
# refused. There is no error at startup for this: an unmatched pane is simply not a lead. If your
|
||||
# primary suddenly can't spawn or send, check this section first.
|
||||
# leaders:
|
||||
# opus-5.0:
|
||||
# profile: opus # omit to never create this lead, only recognise it
|
||||
# instances: 1 # desired live count; only the shortfall is launched. 0 = off
|
||||
# tab: "lead: opus-5.0" # REQUIRED — the exact tab label this lead lives in
|
||||
# tabPrefix: "lead:" # only used to guard against a worker tabLabel colliding with
|
||||
# # this convention at startup; plays no part in matching a lead
|
||||
# scanIntervalSeconds: 10 # rescan cadence, and the worst case before a new tab is seen
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
||||
# kind: claude # descriptive; reported by fleet_whoami
|
||||
# gpt-sol-5.6:
|
||||
# tab: "lead: gpt-sol-5.6"
|
||||
# kind: opencode
|
||||
# model: openai/gpt-5.6-terra
|
||||
|
||||
# architects:
|
||||
# architect-1:
|
||||
# profile: opus # a strong model, on the operator's subscription
|
||||
# architect-2:
|
||||
# profile: sol # a different vendor on purpose — two architects that share a
|
||||
# # model share its blind spots
|
||||
developers:
|
||||
gx10:
|
||||
profile: gx10
|
||||
# reviewers:
|
||||
# gx10:
|
||||
# profile: gx10 # the same backend may serve two roles; that is the point
|
||||
|
||||
# Subscription boundary. A worker's base_url host MUST be one of these; the primary
|
||||
# must carry none. Every profile above must have its host listed here.
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
- gx01.gw
|
||||
|
||||
# Member credential policy (CB-596, gitea issue #82). A herdr pane runs a LOGIN shell, and that
|
||||
# shell re-sources the operator's own secret store — so a spawned member inherits every credential
|
||||
# the operator's shell holds, not just the ones fleetd means to give it. Measured on this host:
|
||||
# 31 credential names, all set, with only ONE (GITEA_ACCESS_TOKEN) blocked before this — and that
|
||||
# block was a single name hardcoded in HerdrPeerLauncher.java, not driven by this file. This block
|
||||
# replaces that hardcoded shadow with a config-driven list of names.
|
||||
#
|
||||
# ROUND-2 CORRECTION, measured live: the pane-creation env overlay below (applied at tab.create /
|
||||
# pane.split, BEFORE the pane's login shell runs) does NOT survive that login shell for any name
|
||||
# secrets.sh actually exports — the shell re-exports it afterwards and overwrites the sentinel.
|
||||
# Proof: GITEA_ACCESS_TOKEN comes back blocked only because secrets.sh itself carries a guarded
|
||||
# export (`[ -n "${BRIDGED_MEMBER:-}" ] || export GITEA_ACCESS_TOKEN=...`) — that guard, not this
|
||||
# file, is what wins. No other name in `known` below has a matching guard in secrets.sh yet (1
|
||||
# guard measured against 33 export lines there). So today this block's overlay is REAL protection
|
||||
# only for a name secrets.sh does not export, or a peer kind whose pane never runs a login shell —
|
||||
# for everything secrets.sh exports and guards, the guard in secrets.sh (out of scope for this
|
||||
# ticket) is what actually blocks it, not this list. An exec-time fix (winning after the login
|
||||
# shell finishes, before the agent process starts) was attempted and found to have no seam in the
|
||||
# current herdr protocol — AgentControl.start takes a fixed `kind` (herdr resolves the executable)
|
||||
# plus trailing CLI args for that binary, not an arbitrary argv or an env map; only tab.create /
|
||||
# pane.split accept `env`, and that is this same pane-creation overlay. See gitea #82 for the open
|
||||
# design question this leaves.
|
||||
#
|
||||
# DENY-BY-DEFAULT, NOT A DENY-LIST. A deny-list (name the bad ones, let everything else through) is
|
||||
# silently wrong the moment the operator's store gains a new secret — nothing would ever report it.
|
||||
# Deny-by-default inverts that: `known` bounds the blast radius to names actually enumerated below,
|
||||
# and EVERY one of them is blocked UNLESS it is also in `allow`. Omitting this block entirely (the
|
||||
# shipped default) blocks NOTHING — unlike most optional blocks in this file, absence here is a real
|
||||
# gap, not a safe "feature off". A name that is neither `known` nor `allow`-ed is not silently let
|
||||
# through either: the daemon logs a WARN naming any credential-shaped env var it finds on neither
|
||||
# list (never its value), so a secret added to the store later does not go unnoticed forever.
|
||||
#
|
||||
# policy → "deny-by-default" (the default; also accepted spelled "deny-list") overlays each
|
||||
# known-but-not-allowed name BEFORE the pane's login shell runs — real protection only
|
||||
# where that shell does not re-export the name (see ROUND-2 CORRECTION above). An
|
||||
# unrecognized value refuses to start, naming it.
|
||||
# policy → "allow-list" (CB-633) moves the control to a per-spawn ZDOTDIR directory the daemon
|
||||
# generates and passes through tab.create's env map. Each generated startup file sources
|
||||
# its ~/ counterpart FIRST and then runs the scrub, so the scrub happens after the
|
||||
# operator's whole chain and no sourced file can undo it.
|
||||
# The scrub is sourced from BOTH the generated .zshrc and the generated .zlogin, because
|
||||
# herdr does not open the same kind of shell everywhere: macOS panes run a LOGIN zsh (so
|
||||
# .zlogin runs), Linux panes run a plain interactive zsh (so .zlogin never runs at all).
|
||||
# A scrub in .zlogin alone would be a control that silently does nothing on Linux.
|
||||
# The allow-list is DERIVED, never typed:
|
||||
# every profile's tokenEnv/gitTokenEnv/gitHostEnv values and env-map keys, plus an
|
||||
# infrastructure set (PATH HOME SHELL TERM LANG LC_* TMPDIR USER LOGNAME PWD SHLVL EDITOR
|
||||
# PAGER JAVA_HOME XDG_* ZDOTDIR), plus whatever keys this spawn's own env overlay carries.
|
||||
# Adding a profile can therefore only widen the list, never break another spawn's scrub.
|
||||
# Under this policy `known`/`allow` below become REPORTING ONLY — they feed the gap WARN,
|
||||
# they are no longer a control. If the member's login shell is NOT zsh, the daemon logs a
|
||||
# loud WARN saying protection is off and falls back to deny-by-default's overlay.
|
||||
# Each pane writes a scrub-report.txt naming how many variables it kept of how many it
|
||||
# saw; the daemon logs that "allowed N of M" line when the pane stops. If the report is
|
||||
# MISSING the daemon logs a WARN instead — the scrub cannot then be confirmed to have
|
||||
# run, and a silently-dead control is exactly what this policy exists to prevent.
|
||||
# allow → credential names a member legitimately needs. Under deny-by-default, left OUT of the
|
||||
# pane's env overlay entirely, so the value the pane's own (login) shell exports passes
|
||||
# through untouched. Under allow-list: reporting only.
|
||||
# known → every credential name the operator's store is known to export. Under deny-by-default,
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAuthSock → whether SSH_AUTH_SOCK may pass through under allow-list ("allow") or must be
|
||||
# blanked like any other non-derived name ("block", the default). This is a decision you
|
||||
# have to make explicitly: SSH_AUTH_SOCK is a handle to YOUR ssh-agent, and a member
|
||||
# holding it can sign with your keys — it sits in no secret file and looks like no
|
||||
# credential, which is why it slipped past three earlier tickets (gitea #110). Blocking
|
||||
# it breaks git over SSH inside members (push/fetch authenticate as you); use HTTPS
|
||||
# remotes or scoped deploy keys instead of allowing it lightly.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAuthSock: block # allow-list only; see the sshAuthSock note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
# - WORKER_GITEA_TOKEN # the repo-scoped forge token a member needs to open its own PR (CB-302)
|
||||
# - CONTEXT7_TOKEN # already decided as allowed by CB-593
|
||||
# - GITEA_HOST # not a credential — a hostname, paired with the forge token above
|
||||
# known:
|
||||
# - AI_GATEWAY_TOKEN
|
||||
# - BESZEL_ADMIN_EMAIL
|
||||
# - BESZEL_ADMIN_PASSWORD
|
||||
# - BESZEL_HUB_URL
|
||||
# - BESZEL_KEY
|
||||
# - BESZEL_UNIVERSAL_TOKEN
|
||||
# - BRAIN_MCP_TOKEN
|
||||
# - CF_ACCOUNT_ID
|
||||
# - CF_API_TOKEN
|
||||
# - CF_USER_TOKEN
|
||||
# - CONFLUENCE_API_TOKEN
|
||||
# - CONFLUENCE_USERNAME
|
||||
# - CONTEXT7_TOKEN
|
||||
# - GITEA_ACCESS_TOKEN
|
||||
# - GITEA_HOST
|
||||
# - GITLAB_OAUTH_CLIENT_SECRET
|
||||
# - GITLAB_PERSONAL_ACCESS_TOKEN
|
||||
# - GRAFANA_ADMIN_PASSWORD
|
||||
# - GRAFANA_ADMIN_USER
|
||||
# - HASS_TOKEN
|
||||
# - HW_PASSWORD
|
||||
# - HW_USER
|
||||
# - LTMS_API_KEY
|
||||
# - MEMORY_MCP_TOKEN
|
||||
# - METRICS_PUSH_TOKEN
|
||||
# - OPENCODE_AUTOMODE_MODEL
|
||||
# - TELEGRAM_BOT_TOKEN
|
||||
# - TELEGRAM_CHAT_ID
|
||||
# - TS_API_KEY
|
||||
# - TS_AUTHKEY
|
||||
# - WORKER_GITEA_TOKEN
|
||||
|
||||
# Spawn-readiness gate (CB-306). The launcher blocks until the worker's herdr status is
|
||||
# injectable (IDLE/BLOCKED/DONE) or the timeout elapses. 0 disables the gate.
|
||||
# NOTE: keys are camelCase — config is bound by plain Jackson with no naming strategy and
|
||||
# unknown keys are ignored, so a snake_case key would be silently dropped (default kept).
|
||||
# spawnReadyTimeoutMs: 20000
|
||||
# spawnReadyPollMs: 300
|
||||
|
||||
# Worktree provisioning root (CB-301-ext). Where per-worker git worktrees are checked out so
|
||||
# each worker owns an isolated branch instead of sharing the primary's tree. Omit to default
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
# contextCap → force-release a session after this many delegated turns
|
||||
# drainTimeoutSeconds → seconds to wait for BUSY sessions on shutdown before forced teardown
|
||||
# clearAfterTurn → whether a reusable worker discards its conversation context after every
|
||||
# completed delegated turn (default false). Works for claude-code workers
|
||||
# only — any other peer kind (e.g. opencode) logs "context reset is
|
||||
# unsupported for peer kind …" once and the reset is a no-op.
|
||||
# lifecycle:
|
||||
# idleTtlSeconds: 300
|
||||
# contextCap: 10
|
||||
# drainTimeoutSeconds: 5
|
||||
# clearAfterTurn: false
|
||||
|
||||
# Durable reply delivery (CB-307 Stage 2). OMIT this block entirely to keep the default
|
||||
# in-memory, soft-state reply inbox (late worker replies are held only until a daemon bounce).
|
||||
# Set a broker uri to swap in the AMQP-backed inbox: worker replies with no open send are held
|
||||
# on a durable per-target queue (agent.<target>.inbox) and survive a restart — the broker
|
||||
# redelivers anything the primary had not yet drained. Production default is LavinMQ; a stock
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# uriEnv → CB-151: name of a host env var holding the AMQP URI, preferred over `uri` (wins
|
||||
# whenever set). The URI carries `user:pass@` inline, so naming a variable keeps the
|
||||
# password out of fleetd.yaml — same pattern as auth.tokenEnv/Profile.tokenEnv. A
|
||||
# uriEnv that resolves to an unset or blank variable is treated as NOT configured and
|
||||
# the daemon falls back to the in-memory inbox, warning loudly.
|
||||
# prefetch → CB-527: consumer basicQos, capping how many unacked messages the inbox holds
|
||||
# in-heap per owned target (the rest sits on the broker's durable queue instead of
|
||||
# growing the JVM heap). Default 32 when omitted.
|
||||
# broker:
|
||||
# uriEnv: LAVINMQ_URI
|
||||
# prefetch: 32
|
||||
|
||||
# Shared cross-host LEADER coordination broker. OMIT this block to leave lead-to-lead messaging
|
||||
# off entirely (config-only in this ticket — nothing here wires it into a live LeadMailbox yet).
|
||||
# This is a SEPARATE AMQP vhost from `broker:` above: member/worker inboxes always stay on the
|
||||
# per-fleet `broker:` vhost, and this vhost carries only leader-to-leader traffic, so two fleets
|
||||
# whose members must never see each other can still share one coordination vhost for their leads.
|
||||
# uriEnv → name of a host env var holding the coordination AMQP URI, same convention as
|
||||
# broker.uriEnv (keeps the credential out of fleetd.yaml). Wins over `uri` when set.
|
||||
# selfId → this daemon's own lead coord-id — the name its mailbox is owned under
|
||||
# (lead.<selfId>.inbox), e.g. "mac-opus" or "fleet01-lead". Must be globally unique
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
# terminal → pin the primary's herdr terminal id. Omit to learn it from the connection on
|
||||
# the first orchestration-side MCP call (the normal case). An off-host or
|
||||
# non-herdr primary leaves this unresolved → the loop is a no-op and delivery
|
||||
# degrades to pull; the reply is still never lost.
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just fleetd-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off fleet_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
# primary:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
+264
@@ -0,0 +1,264 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<project xmlns="http://maven.apache.org/POM/4.0.0"
|
||||
xmlns:xsi="http://www.w3.org/2001/XMLSchema-instance"
|
||||
xsi:schemaLocation="http://maven.apache.org/POM/4.0.0 http://maven.apache.org/xsd/maven-4.0.0.xsd">
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
<groupId>dev.ltms</groupId>
|
||||
<artifactId>fleetd</artifactId>
|
||||
<version>1.0.0</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<name>fleetd</name>
|
||||
<description>fleet message server: sole gateway between a lead session, its members, and herdr</description>
|
||||
|
||||
<properties>
|
||||
<maven.compiler.release>25</maven.compiler.release>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<mainClass>dev.ltms.fleet.Fleetd</mainClass>
|
||||
|
||||
<jackson.version>2.19.0</jackson.version>
|
||||
<javalin.version>6.7.0</javalin.version>
|
||||
<jetty.version>11.0.25</jetty.version>
|
||||
<mcp.version>2.0.0</mcp.version>
|
||||
<slf4j.version>2.0.16</slf4j.version>
|
||||
<logback.version>1.5.18</logback.version>
|
||||
<junit.version>5.11.4</junit.version>
|
||||
<amqp.version>5.22.0</amqp.version>
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
Dependency security (validate with the JetBrains analyzer's Mend.io check on this pom).
|
||||
Deps are pinned to the latest available versions. Residual advisories with NO upstream fix,
|
||||
accepted for this loopback-bound daemon that processes no untrusted config:
|
||||
- jetty-http 11.0.25 (via Javalin): CVE-2026-2332, CVE-2025-11143 — Jetty 11 is EOL;
|
||||
fixed only in Jetty 12, which needs a Javalin major (6.x rides Jetty 11).
|
||||
- logback-core 1.5.18: CVE-2025-11226, CVE-2026-1225 — both require a MALICIOUS
|
||||
logback.xml (attacker with config write already has code execution); ours is trusted.
|
||||
- jackson-core 2.19.0: WS-2026-0003 — "insufficient information", no fixed version published.
|
||||
- tools.jackson.core (Jackson 3) 3.0.3 via the MCP SDK: CVE-2026-29062 (nesting-depth
|
||||
resource exhaustion). The SDK 2.0.0 is pinned to Jackson 3.0.3 + jackson-annotations
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
skew). Javalin 6.x rides Jetty 11; a move to Jetty 12 needs a Javalin major. -->
|
||||
<dependencyManagement>
|
||||
<dependencies>
|
||||
<dependency>
|
||||
<groupId>org.eclipse.jetty</groupId>
|
||||
<artifactId>jetty-bom</artifactId>
|
||||
<version>${jetty.version}</version>
|
||||
<type>pom</type>
|
||||
<scope>import</scope>
|
||||
</dependency>
|
||||
<!-- The MCP SDK (Jackson 3) needs jackson-annotations with JsonFormat.Shape.POJO
|
||||
(the 3.0 line); it shares the com.fasterxml.jackson.annotation package with our
|
||||
Jackson 2.19 databind, so both must resolve to the same jar. 3.0 is built to work
|
||||
with Jackson 2.19 databind too — pin it to reconcile the two. -->
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.core</groupId>
|
||||
<artifactId>jackson-annotations</artifactId>
|
||||
<version>3.0-rc5</version>
|
||||
</dependency>
|
||||
<!-- Testcontainers 1.20.4 pulls commons-compress 1.24.0 (test scope), which carries
|
||||
CVE-2024-25710 (8.1) + CVE-2024-26308 — both fixed in 1.26.0. Pin the patched line.
|
||||
Test-scope only (never shipped in the jar), but bumped per the CVE policy. -->
|
||||
<dependency>
|
||||
<groupId>org.apache.commons</groupId>
|
||||
<artifactId>commons-compress</artifactId>
|
||||
<version>${commons-compress.version}</version>
|
||||
</dependency>
|
||||
<!-- Testcontainers 1.20.4 also pulls commons-lang3 3.16.0 (test scope): CVE-2025-48924
|
||||
(uncontrolled recursion in ClassUtils), fixed in 3.18.0. Pin the patched line.
|
||||
Test-scope only (never shipped in the jar), bumped per the CVE policy. -->
|
||||
<dependency>
|
||||
<groupId>org.apache.commons</groupId>
|
||||
<artifactId>commons-lang3</artifactId>
|
||||
<version>${commons-lang3.version}</version>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
</dependencyManagement>
|
||||
|
||||
<dependencies>
|
||||
<!-- JSON + YAML (config, herdr wire format, REST bodies) -->
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.core</groupId>
|
||||
<artifactId>jackson-databind</artifactId>
|
||||
<version>${jackson.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>com.fasterxml.jackson.dataformat</groupId>
|
||||
<artifactId>jackson-dataformat-yaml</artifactId>
|
||||
<version>${jackson.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- REST: the testability surface. MCP tools are thin adapters over these endpoints. -->
|
||||
<dependency>
|
||||
<groupId>io.javalin</groupId>
|
||||
<artifactId>javalin</artifactId>
|
||||
<version>${javalin.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- MCP server: the SERVER face. Streamable-HTTP servlet mounted on Javalin's Jetty at
|
||||
/mcp, exposing fleet_send/fleet_reply/fleet_status as thin adapters over REST. -->
|
||||
<dependency>
|
||||
<groupId>io.modelcontextprotocol.sdk</groupId>
|
||||
<artifactId>mcp</artifactId>
|
||||
<version>${mcp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Broker client (CB-307 Stage 2): AMQP 0-9-1. Default deploy targets LavinMQ; this same
|
||||
client speaks to RabbitMQ unchanged (URI-only swap), so integration tests run against a
|
||||
stock RabbitMQ container. Only wired when a broker: block is present in config; absent →
|
||||
the in-memory ReplyInbox. -->
|
||||
<dependency>
|
||||
<groupId>com.rabbitmq</groupId>
|
||||
<artifactId>amqp-client</artifactId>
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
<artifactId>slf4j-api</artifactId>
|
||||
<version>${slf4j.version}</version>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>ch.qos.logback</groupId>
|
||||
<artifactId>logback-classic</artifactId>
|
||||
<version>${logback.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Tests -->
|
||||
<dependency>
|
||||
<groupId>org.junit.jupiter</groupId>
|
||||
<artifactId>junit-jupiter</artifactId>
|
||||
<version>${junit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- Testcontainers RabbitMQ: spins a real broker for the @Tag("contract") AMQP integration
|
||||
test only. Excluded from the default build (contract group), so `mvn clean install`
|
||||
stays hermetic and green without Docker; run under -Pcontract with Docker present. -->
|
||||
<dependency>
|
||||
<groupId>org.testcontainers</groupId>
|
||||
<artifactId>rabbitmq</artifactId>
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
<dependency>
|
||||
<groupId>org.testcontainers</groupId>
|
||||
<artifactId>junit-jupiter</artifactId>
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
||||
launchd plist together. The installed plist names fleetd/target/fleetd.jar and
|
||||
KeepAlive is armed, so this name, the plist, and the wrapper must move as one. -->
|
||||
<finalName>fleetd</finalName>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-compiler-plugin</artifactId>
|
||||
<version>3.14.0</version>
|
||||
</plugin>
|
||||
|
||||
<!--
|
||||
Coverage (CB-509). Build-time tooling only — never a compile/runtime dependency, so
|
||||
it adds nothing to the shipped jar. Report lands at target/site/jacoco/index.html and
|
||||
target/site/jacoco/jacoco.csv. No `check` rule / threshold is wired: a coverage gate
|
||||
rewards writing tests that execute lines, which is the failure mode this project is
|
||||
trying to avoid, not encourage.
|
||||
-->
|
||||
<plugin>
|
||||
<groupId>org.jacoco</groupId>
|
||||
<artifactId>jacoco-maven-plugin</artifactId>
|
||||
<version>0.8.13</version>
|
||||
<executions>
|
||||
<execution>
|
||||
<id>prepare-agent</id>
|
||||
<goals><goal>prepare-agent</goal></goals>
|
||||
</execution>
|
||||
<execution>
|
||||
<id>report</id>
|
||||
<phase>test</phase>
|
||||
<goals><goal>report</goal></goals>
|
||||
</execution>
|
||||
</executions>
|
||||
</plugin>
|
||||
|
||||
<!-- Unit tests run by default; contract tests (live herdr) are tag-excluded. -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-surefire-plugin</artifactId>
|
||||
<version>3.5.2</version>
|
||||
<configuration>
|
||||
<excludedGroups>${excludedGroups}</excludedGroups>
|
||||
</configuration>
|
||||
</plugin>
|
||||
|
||||
<!-- Runnable fat jar: java -jar target/fleetd.jar -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-shade-plugin</artifactId>
|
||||
<version>3.6.0</version>
|
||||
<executions>
|
||||
<execution>
|
||||
<phase>package</phase>
|
||||
<goals><goal>shade</goal></goals>
|
||||
<configuration>
|
||||
<transformers>
|
||||
<transformer implementation="org.apache.maven.plugins.shade.resource.ManifestResourceTransformer">
|
||||
<mainClass>${mainClass}</mainClass>
|
||||
</transformer>
|
||||
<transformer implementation="org.apache.maven.plugins.shade.resource.ServicesResourceTransformer"/>
|
||||
</transformers>
|
||||
</configuration>
|
||||
</execution>
|
||||
</executions>
|
||||
</plugin>
|
||||
</plugins>
|
||||
</build>
|
||||
|
||||
<profiles>
|
||||
<!-- `mvn test` excludes contract tests. `mvn test -Pcontract` runs them against a live herdr. -->
|
||||
<profile>
|
||||
<id>default-excludes</id>
|
||||
<activation><activeByDefault>true</activeByDefault></activation>
|
||||
<properties><excludedGroups>contract</excludedGroups></properties>
|
||||
</profile>
|
||||
<profile>
|
||||
<id>contract</id>
|
||||
<properties><excludedGroups/></properties>
|
||||
<build>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<artifactId>maven-surefire-plugin</artifactId>
|
||||
<configuration>
|
||||
<!-- Docker-engine compat (see "Running the contract tests" in
|
||||
docs/CB-307-Reliable-Delivery.md): Testcontainers 1.20.4's docker-java
|
||||
client defaults to Docker API 1.32 when no version is set, but modern
|
||||
engines (OrbStack on this dev host, min 1.40) reject that as too old —
|
||||
which surfaces as "Could not find a valid Docker environment". Pinning
|
||||
api.version=1.43 works on OrbStack and Docker 24+, and is overridable
|
||||
per-host via -Dapi.version. Only active under -Pcontract, so the
|
||||
default hermetic build never sets it. -->
|
||||
<systemPropertyVariables>
|
||||
<api.version>1.43</api.version>
|
||||
</systemPropertyVariables>
|
||||
</configuration>
|
||||
</plugin>
|
||||
</plugins>
|
||||
</build>
|
||||
</profile>
|
||||
</profiles>
|
||||
</project>
|
||||
@@ -0,0 +1,913 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.ConfigWatcher;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||
import dev.ltms.fleet.lead.LeadLauncher;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.inject.TurnListener;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.fleet.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.fleet.msg.AmqpReplyInbox;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadMailbox;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.rest.FleetApp;
|
||||
import dev.ltms.fleet.session.GitWorktrees;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.session.SessionReaper;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.member.OpenCodeLauncher;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* {@code fleetd} entry point. Wires the real herdr socket client to the REST app and
|
||||
* starts listening. Before anything else it asserts its own environment is clean —
|
||||
* {@code fleetd} is not a Claude process and must never carry a base_url.
|
||||
*/
|
||||
public final class Fleetd {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Fleetd.class);
|
||||
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
/**
|
||||
* CB-637: how often the lead coordination loop looks for peer messages. A few seconds — slow
|
||||
* enough that an idle fleet is not polling a broker in a tight loop, fast enough that a peer
|
||||
* lead's message is not left sitting once the local lead reaches a turn boundary. The mailbox
|
||||
* pushes into the loop's held set on its own consumer thread, so this interval bounds only the
|
||||
* pane delivery, never the receive.
|
||||
*/
|
||||
private static final long LEAD_COORD_INTERVAL_MS = 3_000L;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
/**
|
||||
* CB-632/CB-634: prefer {@code fleetd.yaml} in {@code dir}; fall back to the legacy
|
||||
* {@code bridged.yaml} when the new name is not there. The product renamed to {@code fleetd},
|
||||
* but a deployment whose local config file is still {@code bridged.yaml} keeps working until
|
||||
* that file is renamed.
|
||||
*/
|
||||
static Path chooseDefaultConfigFile(Path dir) {
|
||||
Path fleetd = dir.resolve("fleetd.yaml");
|
||||
if (Files.exists(fleetd)) {
|
||||
return fleetd;
|
||||
}
|
||||
return dir.resolve("bridged.yaml");
|
||||
}
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = args.length > 0 ? Path.of(args[0]) : chooseDefaultConfigFile(Path.of(""));
|
||||
// The config file was renamed bridged.yaml -> fleetd.yaml. Name the file we actually
|
||||
// loaded, whichever of the two names it carries.
|
||||
log.info("Using configuration file {}", configPath);
|
||||
FleetConfig cfg = FleetConfig.load(configPath);
|
||||
// CB-594: report which secret env vars the config actually needs, by name, before anything
|
||||
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
|
||||
// boots fine either way — this is the only thing that says so out loud.
|
||||
reportRequiredSecrets(cfg);
|
||||
// CB-596: an absent (or empty) memberCredentials: block blocks NOTHING — no credential
|
||||
// name is hardcoded any more to fall back on. Say so loudly, the same way a missing
|
||||
// secret is reported above, so upgrading past this commit never silently drops CB-592's
|
||||
// protection.
|
||||
reportMemberCredentialsGap(cfg);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched fleetd must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
cfg.validateCharters();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
UnixSocketHerdrClient memberHerdr = cfg.memberHerdrSocket() != null && !cfg.memberHerdrSocket().isBlank()
|
||||
? UnixSocketHerdrClient.connect(Path.of(cfg.memberHerdrSocket()), new com.fasterxml.jackson.databind.ObjectMapper())
|
||||
: herdr;
|
||||
AtomicReference<Supplier<Map<String, String>>> leadsRef = new AtomicReference<>(Map::of);
|
||||
HerdrRouter router = new HerdrRouter(herdr, memberHerdr,
|
||||
target -> leadsRef.get().get().containsKey(target));
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, FleetConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, FleetConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(router.memberAgents(), router.memberSpaces(), guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(router.memberAgents(), router.memberSpaces(),
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
// The cooldown is deferred (see FleetConfig#quarantineCooldownSeconds): it is read once
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine);
|
||||
// CB-504: under supervision (launchd/systemd) fleetd can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// The lead and the members now share ONE workspace (the operator asked for a single
|
||||
// "session" with many tabs), so no workspace can be excluded — the lead lives in the
|
||||
// members' space by design. A lead is told from a member by its exact tab label alone:
|
||||
// a lead carries its configured `tab`, a member its `worker: {profile} #{n}` template,
|
||||
// and the two never collide. (The scanner still supports an exclusion set for a split
|
||||
// layout; the fleet's policy here is simply not to use one.)
|
||||
//
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
// This must use the lead daemon: scanning member tabs would demote the lead to a worker.
|
||||
leads = new LeadTabScanner(herdr, tabToName, Set.of(),
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, shared fleet space)",
|
||||
tabToName.keySet(), scanIntervalSeconds);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
leadsRef.set(leads);
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(router.leadAgents(), router.leadSpaces(), cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
// Compiled once at startup, keyed by profile name; a profile with no exhaustedPattern is
|
||||
// simply absent here, so its workers keep today's completion-fallback behaviour unchanged.
|
||||
Map<String, Pattern> exhaustedPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasExhaustedPattern()) {
|
||||
exhaustedPatternsByProfile.put(name, Pattern.compile(profile.exhaustedPattern()));
|
||||
}
|
||||
});
|
||||
ExhaustedPatternLookup exhaustedPatterns = target -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.map(profileName -> config.get().profiles().get(profileName))
|
||||
.ifPresent(profile -> {
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}' classified "
|
||||
+ "BACKEND_EXHAUSTED): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
});
|
||||
AgentControl agents = router.memberAgents();
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns, exhaustionSink);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Predicate<String> deliverable = deliverableTo(presence, leads);
|
||||
Injector injector = new Injector(router, turnListener, deliverable,
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(router, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block selects the AMQP-backed durable adapter; absent (or
|
||||
// unusable), fleetd stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox = selectReplyInbox(cfg.broker(), System.getenv(), AmqpReplyInbox::open);
|
||||
// CB-637: this daemon's lead-to-lead mailbox on the SHARED coordination vhost — a separate
|
||||
// broker from the reply inbox by design (see FleetConfig.Coordinator). Absent a coordinator:
|
||||
// block this is null and every lead path below is simply not wired, which is exactly the
|
||||
// behaviour before this ticket. It owns a broker connection, so keep the reference for the
|
||||
// ordered shutdown hook.
|
||||
final LeadMailbox leadMailbox = openLeadMailbox(cfg.coordinator(), System.getenv(), LeadMailbox::open);
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open fleet_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = FleetMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, router.leadAgents(), replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, router.leadAgents(), replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(router, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(),
|
||||
cfg.health().workingSuspectAfterOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(detail -> {
|
||||
// CB-578 stage C, acceptance criterion 10: a failed ticket's detail should tell a lead
|
||||
// where to re-dispatch onto the same tree, not just that the worker vanished.
|
||||
String reason = "the worker session was released before it replied";
|
||||
if (detail.worktreePath() != null) {
|
||||
reason += "; worktree=" + detail.worktreePath() + " branch=" + detail.branch()
|
||||
+ " snapshot=" + (detail.snapshotRef() != null ? detail.snapshotRef() : "none");
|
||||
}
|
||||
// CB-584 (issue #65 criterion 5): also name the agent session, so a lead can resume the
|
||||
// member's conversation instead of only re-dispatching a fresh one onto the same files.
|
||||
if (detail.agentSessionId() != null) {
|
||||
reason += " agentSessionId=" + detail.agentSessionId();
|
||||
}
|
||||
messages.abandon(detail.terminalId(), reason);
|
||||
replyInbox.release(detail.terminalId());
|
||||
primaryRegistry.forgetDelegation(detail.terminalId()); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): fleet_send/fleet_reply/fleet_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
// CB-185: a caller's pane can live on either daemon (a lead's on the lead daemon, a
|
||||
// member's on the member daemon) — search both, lead first. Collapses to one scan when
|
||||
// memberHerdrSocket is unset (herdr == memberHerdr).
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr, memberHerdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting fleetd");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new FleetMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
new FleetMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}),
|
||||
new FleetMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine),
|
||||
leadMailbox);
|
||||
|
||||
// CB-637: the receive half. Only constructed when a lead mailbox actually opened — with no
|
||||
// coordinator (or an unreachable one) there is nothing to deliver, so no scheduler is
|
||||
// created and no thread runs. It reads the SAME live lead supplier the injector's
|
||||
// deliverability gate does, so a lead found by the tab scan after startup is reachable
|
||||
// without a restart.
|
||||
final LeadCoordLoop leadCoordLoop;
|
||||
final ScheduledExecutorService leadCoordSchedulerRef;
|
||||
if (leadMailbox != null) {
|
||||
var leadCoordScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-leadcoord-").unstarted(r));
|
||||
leadCoordLoop = new LeadCoordLoop(leadMailbox, router.leadAgents(), leads, leadCoordScheduler,
|
||||
LEAD_COORD_INTERVAL_MS);
|
||||
leadCoordLoop.start();
|
||||
leadCoordSchedulerRef = leadCoordScheduler;
|
||||
} else {
|
||||
leadCoordLoop = null;
|
||||
leadCoordSchedulerRef = null;
|
||||
}
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (leadCoordLoop != null) leadCoordLoop.close(); // CB-637: stop delivering peer-lead messages
|
||||
if (leadCoordSchedulerRef != null) leadCoordSchedulerRef.shutdownNow();
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
// CB-637: the coordination connection goes with it — after the loop that reads it has
|
||||
// stopped, so no tick can be mid-ack against a closed channel.
|
||||
if (leadMailbox != null) {
|
||||
try {
|
||||
leadMailbox.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
router.close();
|
||||
}));
|
||||
|
||||
// CB-185: give FleetApp both daemons — /healthz must require both to answer and
|
||||
// GET /sessions must merge across both, or a down/unpolled member daemon is invisible.
|
||||
Javalin app = new FleetApp(herdr, memberHerdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics, deliverable).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("fleetd listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a spawned
|
||||
* member whose agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> member's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code FleetMcp} marks presence
|
||||
* for every spawned member (worker and architect), deliberately, since that map doubles as the
|
||||
* member roster's availability signal and a lead counted there would show up as an available
|
||||
* member. So without the second disjunct a lead is permanently un-deliverable: every
|
||||
* lead→lead send sat on the gate for {@code READINESS_GRACE_POLLS} (~60s) and then failed
|
||||
* having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #selectReplyInbox}: production binds {@link AmqpReplyInbox#open}. */
|
||||
@FunctionalInterface
|
||||
interface AmqpOpener {
|
||||
ReplyInbox open(String uri, int prefetch);
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #openLeadMailbox}: production binds {@link LeadMailbox#open}. */
|
||||
@FunctionalInterface
|
||||
interface LeadMailboxOpener {
|
||||
LeadMailbox open(String uri, String selfCoordId, int prefetch);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-637: open this daemon's lead-to-lead mailbox, or return {@code null} to leave the feature
|
||||
* off. Package-private and env-injected for the same reason as {@link #selectReplyInbox}: the
|
||||
* selection is then testable without a broker or a mutable process environment.
|
||||
*
|
||||
* <p>Every "off" path returns {@code null}, and each says why at the level it deserves:
|
||||
*
|
||||
* <ul>
|
||||
* <li>no {@code coordinator:} block — silent. Lead coordination is opt-in; an operator who
|
||||
* never configured it does not need to be told it is off on every boot.</li>
|
||||
* <li>a block whose {@code uriEnv} does not resolve — INFO, the same "you moved to the secret
|
||||
* store and the variable is not there" case {@code selectReplyInbox} warns about.</li>
|
||||
* <li>a configured broker but no {@code selfId} — WARN. This one is a half-finished config: a
|
||||
* mailbox is named after the coord-id that owns it, so with no id there is no queue to own
|
||||
* and no {@code from} to send as. Loud, because the operator plainly intended the feature.</li>
|
||||
* <li>the broker refuses at boot — WARN, and carry on. Mirrors {@code openAmqpOrFallback}: a
|
||||
* coordination broker that is down must never take a whole fleet's daemon with it, and the
|
||||
* fleet still works exactly as it did before this feature existed.</li>
|
||||
* </ul>
|
||||
*/
|
||||
static LeadMailbox openLeadMailbox(FleetConfig.Coordinator coordinator, Map<String, String> env,
|
||||
LeadMailboxOpener opener) {
|
||||
if (coordinator == null) {
|
||||
return null; // opt-in: nothing configured, nothing to say
|
||||
}
|
||||
String uri = coordinator.effectiveUri(env);
|
||||
if (uri == null) {
|
||||
log.info("lead coordination: OFF — coordinator{} has no usable broker uri",
|
||||
coordinator.uriEnv() == null ? "" : ".uriEnv=" + coordinator.uriEnv());
|
||||
return null;
|
||||
}
|
||||
if (coordinator.selfId() == null || coordinator.selfId().isBlank()) {
|
||||
log.warn("coordinator.selfId is unset — lead coordination is OFF. A lead mailbox is the "
|
||||
+ "queue named after the coord-id that owns it, so with no id there is nothing to "
|
||||
+ "own and no sender identity to publish as. Set coordinator.selfId to a name that "
|
||||
+ "is unique across every daemon sharing {} and restart fleetd.",
|
||||
stripCredentials(uri));
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
LeadMailbox mailbox = opener.open(uri, coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
log.info("lead coordination: ON as coord-id {} (prefetch={})",
|
||||
coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
return mailbox;
|
||||
} catch (IllegalStateException e) {
|
||||
log.warn("cannot reach the AMQP coordination broker ({}) — lead-to-lead messaging is OFF "
|
||||
+ "for this process lifetime. fleet_send{{coordId}} will report it as "
|
||||
+ "not configured, and peer messages already queued stay on the broker until "
|
||||
+ "a restart picks them up. Reason: {}",
|
||||
stripCredentials(uri), reasonOf(e));
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-151/152: pick the reply inbox. A usable broker — a literal {@code uri}, or a {@code
|
||||
* uriEnv} whose variable resolves (both read from {@code env}) — selects the durable AMQP inbox.
|
||||
* Everything else falls back to the in-memory inbox: no broker block, a blank {@code uri}, a
|
||||
* {@code uriEnv} whose variable is unset or blank, or a broker unreachable at boot. The two
|
||||
* lossy paths warn <em>loudly</em> — never silently — because what is lost is durable,
|
||||
* cross-restart reply delivery. Package-private and env-injected so the selection is testable
|
||||
* without a real broker or a mutable process environment.
|
||||
*/
|
||||
static ReplyInbox selectReplyInbox(FleetConfig.Broker broker, Map<String, String> env, AmqpOpener amqp) {
|
||||
if (broker == null) {
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
if (broker.hasUriEnv()) {
|
||||
// uriEnv is authoritative whenever set (CB-151): the operator moved off clear text, so
|
||||
// it must not quietly fall back onto a stale literal uri.
|
||||
if (broker.uri() != null && !broker.uri().isBlank()) {
|
||||
log.info("broker.uri is ignored because broker.uriEnv={} is set", broker.uriEnv());
|
||||
}
|
||||
String effectiveUri = broker.effectiveUri(env);
|
||||
if (effectiveUri == null) {
|
||||
log.warn("broker.uriEnv={} is unset or blank — durable AMQP reply inbox DISABLED. "
|
||||
+ "Replies are soft-state and will not survive a restart. Set {} in the "
|
||||
+ "daemon's environment (see scripts/redeploy-fleetd.sh) and restart to "
|
||||
+ "use the durable broker inbox.",
|
||||
broker.uriEnv(), broker.uriEnv());
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
log.info("reply inbox: AMQP broker (durable) via env var {} (prefetch={})",
|
||||
broker.uriEnv(), broker.prefetchOrDefault());
|
||||
return openAmqpOrFallback(effectiveUri, broker.prefetchOrDefault(), "uriEnv " + broker.uriEnv(), amqp);
|
||||
}
|
||||
// No uriEnv: the literal uri path (existing behaviour).
|
||||
if (broker.effectiveUri(env) == null) {
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
log.info("reply inbox: AMQP broker (durable) (prefetch={})", broker.prefetchOrDefault());
|
||||
return openAmqpOrFallback(broker.uri(), broker.prefetchOrDefault(), "uri", amqp);
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the AMQP inbox, falling back to the in-memory inbox for this process lifetime if the
|
||||
* broker cannot be reached at boot (CB-152). Not silent: the warning says durable delivery is
|
||||
* off, replies are soft-state and will not survive a restart, plus the source that failed and
|
||||
* the URI <em>with credentials stripped</em>. Never retries in the background — a broker that
|
||||
* drops <em>after</em> startup already self-heals via the connection factory's automatic
|
||||
* recovery; only the boot path is changed here.
|
||||
*/
|
||||
private static ReplyInbox openAmqpOrFallback(String effectiveUri, int prefetch, String source,
|
||||
AmqpOpener amqp) {
|
||||
try {
|
||||
return amqp.open(effectiveUri, prefetch);
|
||||
} catch (IllegalStateException e) {
|
||||
log.warn("cannot reach AMQP broker ({}, {}) — falling back to the in-memory reply inbox "
|
||||
+ "for this process lifetime. Durable, cross-restart reply delivery is OFF; "
|
||||
+ "replies are soft-state and will not survive a restart. Reason: {}",
|
||||
source, stripCredentials(effectiveUri), reasonOf(e));
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
}
|
||||
|
||||
/** An AMQP URI carries {@code user:pass@} inline — show the host/port, never the credentials. */
|
||||
static String stripCredentials(String uri) {
|
||||
return uri == null ? null : uri.replaceAll("://[^@/]*@", "://");
|
||||
}
|
||||
|
||||
/** The deepest cause's class and message — the outermost {@code IllegalStateException} echoes the URI (with password). */
|
||||
private static String reasonOf(Throwable e) {
|
||||
Throwable t = e;
|
||||
while (t.getCause() != null && t.getCause() != t) {
|
||||
t = t.getCause();
|
||||
}
|
||||
String msg = t.getMessage();
|
||||
return t.getClass().getSimpleName() + (msg == null || msg.isBlank() ? "" : ": " + msg);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: which env vars the loaded config actually needs, and why — every non-{@code
|
||||
* subscription} profile's {@code tokenEnv} (a subscription profile never reads one, see
|
||||
* {@link FleetConfig.Profile#isSubscription()}), plus every profile's {@code gitTokenEnv}
|
||||
* where set (opt-in), plus a configured {@code broker.uriEnv} (CB-151). Derived from the
|
||||
* config, not hard-coded, so a new profile is covered for free. A var required by more than one
|
||||
* profile is one entry naming every profile that needs it. Deliberately excludes {@code
|
||||
* auth.tokenEnv}: that one is already enforced loudly, by a startup throw in {@code main()} —
|
||||
* about 370 lines <em>below</em> this method's call site
|
||||
* ({@link #reportRequiredSecrets(FleetConfig)}), not a few lines above it. That throw only
|
||||
* fires when {@code auth.mode: token} is configured; under the default loopback-trust mode it
|
||||
* never runs, and {@code auth.tokenEnv} is simply not required.
|
||||
*
|
||||
* <p>Package-private and pure (no I/O, no logging) so the derivation is unit-testable without
|
||||
* capturing log output; {@link #reportRequiredSecrets(FleetConfig)} is the logging caller.
|
||||
*/
|
||||
static Map<String, List<String>> requiredSecretEnvVars(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (!profile.isSubscription()) {
|
||||
requiredBy.computeIfAbsent(profile.tokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' tokenEnv");
|
||||
}
|
||||
if (profile.hasGitToken()) {
|
||||
requiredBy.computeIfAbsent(profile.gitTokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' gitTokenEnv");
|
||||
}
|
||||
});
|
||||
FleetConfig.Broker broker = cfg.broker();
|
||||
if (broker != null && broker.hasUriEnv()) {
|
||||
requiredBy.computeIfAbsent(broker.uriEnv(), _ -> new ArrayList<>())
|
||||
.add("broker uriEnv");
|
||||
}
|
||||
return requiredBy;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: log, by name only, which required env vars (see {@link #requiredSecretEnvVars}) are
|
||||
* set in the daemon's own process environment — the environment every profile's {@code
|
||||
* tokenEnv}/{@code gitTokenEnv} is read from at spawn time (see
|
||||
* {@code HerdrPeerLauncher.resolveEnv}). Never logs a value, a prefix, or a length.
|
||||
*
|
||||
* <p>A missing entry only warns — it must never refuse to start. A daemon that boots and says
|
||||
* what is wrong is strictly more useful than one that will not boot at all.
|
||||
*/
|
||||
private static void reportRequiredSecrets(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = requiredSecretEnvVars(cfg);
|
||||
if (requiredBy.isEmpty()) {
|
||||
log.info("startup secrets: no profile references a token env var — nothing to check");
|
||||
return;
|
||||
}
|
||||
Map<String, String> env = System.getenv();
|
||||
requiredBy.forEach((varName, sources) -> {
|
||||
String value = env.get(varName);
|
||||
if (value != null && !value.isBlank()) {
|
||||
log.info("startup secret {}: set ({})", varName, String.join(", ", sources));
|
||||
} else {
|
||||
log.warn("startup secret {}: MISSING ({}) — the daemon will start anyway, and this "
|
||||
+ "failure stays invisible until a worker actually needs it. Fix "
|
||||
+ "${SHARED_ENV}/tools/secrets.sh and restart fleetd from a LOGIN "
|
||||
+ "shell (see scripts/redeploy-fleetd.sh).",
|
||||
varName, String.join(", ", sources));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: {@code known:} empty (block absent entirely, or present but empty) means {@link
|
||||
* FleetConfig.MemberCredentials#blockedSet()} is empty too — every member pane inherits the
|
||||
* operator's whole secret store, unblocked, exactly the defect this ticket fixes. Unlike a
|
||||
* missing token ({@link #reportRequiredSecrets}), there is no name to point at: the point is
|
||||
* that the block itself is missing. Warn once at startup and say what to add; never refuse to
|
||||
* start over it — see {@link #reportRequiredSecrets} for why a daemon that boots and says
|
||||
* what is wrong beats one that will not boot at all.
|
||||
*
|
||||
* <p>Package-private so the test can capture the log directly, the same way {@link
|
||||
* #requiredSecretEnvVars} is exposed for {@link #reportRequiredSecrets}'s own test.
|
||||
*/
|
||||
static void reportMemberCredentialsGap(FleetConfig cfg) {
|
||||
FleetConfig.MemberCredentials creds = cfg.memberCredentials();
|
||||
if (creds != null && !creds.known().isEmpty()) {
|
||||
log.info("memberCredentials: policy={}, {} known name(s), {} allowed — blocking {} on "
|
||||
+ "every spawn{}",
|
||||
creds.policy(), creds.known().size(), creds.allow().size(), creds.blockedSet().size(),
|
||||
creds.isAllowList()
|
||||
? " (allow-list: known/allow are reporting only — the control is the derived ZDOTDIR scrub)"
|
||||
: "");
|
||||
return;
|
||||
}
|
||||
log.warn("memberCredentials: absent or empty — the daemon will start anyway, and every "
|
||||
+ "member pane inherits the operator's WHOLE secret store, unblocked (CB-592's "
|
||||
+ "protection is lost). Add a memberCredentials: block (policy/allow/known) to "
|
||||
+ "fleetd.yaml — see fleetd.example.yaml — and restart.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Fleetd() {
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,72 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.time.Instant;
|
||||
import java.time.ZoneOffset;
|
||||
import java.time.format.DateTimeFormatter;
|
||||
|
||||
/**
|
||||
* Append-only record of privileged actions (CB-505).
|
||||
*
|
||||
* <p>Writes JSON lines to a dedicated {@code audit} logger — its own appender, separate from the
|
||||
* chatty app log — so the trail stays greppable and can later be shipped without dragging debug
|
||||
* noise along.
|
||||
*
|
||||
* <p><strong>Message content is never recorded.</strong> This bridge carries the user's source
|
||||
* code, diffs, and prompts; an audit trail that quietly accumulated them would be a transcript
|
||||
* archive wearing a security control's clothing. Records carry <em>who / what / against what /
|
||||
* outcome</em> and correlation ids only.
|
||||
*/
|
||||
public final class AuditLog {
|
||||
|
||||
private static final Logger AUDIT = LoggerFactory.getLogger("audit");
|
||||
private static final DateTimeFormatter TS =
|
||||
DateTimeFormatter.ofPattern("yyyy-MM-dd'T'HH:mm:ss.SSSXXX").withZone(ZoneOffset.UTC);
|
||||
|
||||
private AuditLog() {
|
||||
}
|
||||
|
||||
/** Record an allowed action. */
|
||||
public static void allowed(Principal caller, Authz.Action action, String target) {
|
||||
write(caller, action, target, "allowed", null);
|
||||
}
|
||||
|
||||
/** Record a refused action and why. */
|
||||
public static void denied(Principal caller, Authz.Action action, String target, String reason) {
|
||||
write(caller, action, target, "denied", reason);
|
||||
}
|
||||
|
||||
/** Record an action that was authorized but then failed downstream (guard, timeout, herdr). */
|
||||
public static void failed(Principal caller, Authz.Action action, String target, String reason) {
|
||||
write(caller, action, target, "failed", reason);
|
||||
}
|
||||
|
||||
private static void write(Principal caller, Authz.Action action, String target,
|
||||
String outcome, String reason) {
|
||||
Principal c = caller != null ? caller : Principal.anonymous();
|
||||
StringBuilder sb = new StringBuilder(200);
|
||||
// The timestamp is built here rather than by the appender pattern: a pattern that wrapped
|
||||
// literal braces around the message collides with logback's own variable substitution.
|
||||
sb.append("{\"ts\":\"").append(TS.format(Instant.now())).append('"')
|
||||
.append(",\"role\":\"").append(c.role()).append('"')
|
||||
.append(",\"actor\":\"").append(esc(c.describe())).append('"')
|
||||
.append(",\"pid\":").append(c.pid())
|
||||
.append(",\"action\":\"").append(action).append('"')
|
||||
.append(",\"target\":").append(target == null ? "null" : '"' + esc(target) + '"')
|
||||
.append(",\"outcome\":\"").append(outcome).append('"');
|
||||
if (reason != null) {
|
||||
sb.append(",\"reason\":\"").append(esc(reason)).append('"');
|
||||
}
|
||||
sb.append('}');
|
||||
// The appender supplies the timestamp, so it cannot disagree with the app log's clock.
|
||||
AUDIT.info(sb.toString());
|
||||
}
|
||||
|
||||
/** Minimal JSON string escaping — these values are ids and short reasons, never free text. */
|
||||
private static String esc(String s) {
|
||||
return s.replace("\\", "\\\\").replace("\"", "\\\"")
|
||||
.replace("\n", "\\n").replace("\r", "\\r").replace("\t", "\\t");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,82 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||
*
|
||||
* <p>Most of these rules are already true de facto — {@code FleetMcp} derives a worker's identity
|
||||
* from the connection rather than reading it from an argument, so a worker has never been able to
|
||||
* reply <em>as</em> another worker over MCP. What was missing is that the REST surface trusted the
|
||||
* session id in the URL path, and neither surface checked role at all. This class makes the
|
||||
* invariant explicit and testable rather than emergent.
|
||||
*/
|
||||
public final class Authz {
|
||||
|
||||
private Authz() {
|
||||
}
|
||||
|
||||
/** A privileged operation, named for the audit trail. */
|
||||
public enum Action {
|
||||
/** Spawn a worker peer. */
|
||||
SPAWN,
|
||||
/** Tear a worker peer down. */
|
||||
STOP,
|
||||
/** Deliver a turn to a session (or answer a worker's question). */
|
||||
SEND,
|
||||
/** A worker's terminal reply for its own turn. */
|
||||
REPLY,
|
||||
/** A worker's mid-turn question to the primary. */
|
||||
ASK,
|
||||
/** Collect held replies from a session's inbox. */
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
READ,
|
||||
/** Scrape the metrics endpoint. */
|
||||
METRICS
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code caller} may perform {@code action} against {@code targetSession}.
|
||||
*
|
||||
* @param targetSession the session id in the request path; only consulted for the worker-scoped
|
||||
* actions ({@code REPLY}, {@code ASK}), ignored otherwise, may be
|
||||
* {@code null}
|
||||
*/
|
||||
public static boolean permits(Principal caller, Action action, String targetSession) {
|
||||
if (caller == null || caller.isAnonymous()) {
|
||||
return false; // authenticated as nothing ⇒ authorized for nothing
|
||||
}
|
||||
return switch (action) {
|
||||
// Fleet lifecycle is the primary's alone — spawn, stop, drain. An architect
|
||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
||||
// escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
// excluded — sending would be it escalating.
|
||||
case SEND -> caller.isPrimary() || caller.isArchitect();
|
||||
|
||||
// The load-bearing rule: a caller acts only as the pane it occupies. CB-532 widened who
|
||||
// that can be — a lead answering another lead is replying for its OWN terminal, which
|
||||
// this already permits — while the rule itself is unchanged, and is what stops anyone
|
||||
// forging a reply for a rendezvous someone else is waiting on. An architect's own pane
|
||||
// passes through the same check, so it can answer a funnel that delegated to it. An
|
||||
// unnamed primary (token/loopback, no pane) owns nothing and is still excluded.
|
||||
case REPLY, ASK -> caller.ownsSession(targetSession);
|
||||
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Why a request was refused, for the error body. Distinguishes "you are nobody" from "you are
|
||||
* somebody, but not the right somebody" — the first is a credential problem (401), the second
|
||||
* an authorization one (403).
|
||||
*/
|
||||
public static boolean isUnauthenticated(Principal caller) {
|
||||
return caller == null || caller.isAnonymous();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,272 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Resolves every caller to a {@link Principal}, for both entry paths into the core (CB-501).
|
||||
*
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code FleetMcp}
|
||||
* calls the service layer directly and is mounted as a raw servlet (so it never passes through a
|
||||
* Javalin filter), while the REST routes historically resolved no identity at all. Both now
|
||||
* delegate here, so the authorization rules are stated once instead of drifting apart.
|
||||
*
|
||||
* <p><strong>Resolution order</strong> — connection identity first, token second, nothing third:
|
||||
* <ol>
|
||||
* <li>A loopback peer PID that maps to a pane named by {@code leaders:}, by the legacy
|
||||
* {@code primary.terminal} pin, or by an operator-labelled lead tab (CB-307, CB-530, CB-531)
|
||||
* ⇒ {@link Role#PRIMARY}, carrying that lead's
|
||||
* name. The pane mapping is as unforgeable as a worker's, and the config explicitly names
|
||||
* that pane as a lead's own — without this rule a lead running <em>inside</em> a herdr pane
|
||||
* is misread as a worker and locked out of orchestration. More than one pane may be named,
|
||||
* so two leads can work as peers rather than one being demoted.</li>
|
||||
* <li>A loopback peer PID that maps to a pane bound to a CB-548 architect slot ⇒
|
||||
* {@link Role#ARCHITECT}, carrying the slot name. Just unforgeable as a worker's, and
|
||||
* resolved from the <em>live</em> terminal→slot binding (never a request argument), before
|
||||
* the generic worker fallback.</li>
|
||||
* <li>A loopback peer PID that maps to any other herdr pane ⇒ {@link Role#WORKER}. This is
|
||||
* unforgeable (the OS reports the PID, herdr owns the PID→pane map) and is honoured
|
||||
* regardless of auth mode, so enabling auth never breaks the fleet.</li>
|
||||
* <li>Otherwise, under {@code token} mode, a valid bearer token ⇒ {@link Role#PRIMARY}.</li>
|
||||
* <li>Otherwise, under {@code loopback-trust}, a loopback caller ⇒ {@link Role#PRIMARY}
|
||||
* (the historical behaviour, now an explicit configured choice).</li>
|
||||
* <li>Otherwise {@link Role#ANONYMOUS}.</li>
|
||||
* </ol>
|
||||
*/
|
||||
public final class CallerResolver {
|
||||
|
||||
private final ConnectionIdentity identity;
|
||||
private final boolean tokenMode;
|
||||
private final byte[] expectedToken; // null unless tokenMode
|
||||
/**
|
||||
* terminal_id → lead name; empty when nothing is pinned. CB-530.
|
||||
*
|
||||
* <p>A supplier rather than a map because the registry is no longer fixed at startup: CB-531
|
||||
* discovers leads by scanning herdr for operator-labelled tabs, so a lead that opens its tab
|
||||
* after the daemon booted must still be recognised. Consulted per resolve; the scanner behind
|
||||
* it is TTL-cached, so this is a map lookup in the common case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> leadTerminals;
|
||||
/**
|
||||
* terminal_id → architect slot name; empty when nothing is configured. CB-548.
|
||||
*
|
||||
* <p>Like {@link #leadTerminals}, a supplier rather than a fixed map, so a binding injected
|
||||
* after startup — when the later spawn lifecycle establishes a live architect session, or an
|
||||
* operator pins one — takes effect without a restart. Consulted per resolve; today's wiring
|
||||
* in {@code Fleetd} reads a constant from config, which is the degenerate live case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> architectTerminals;
|
||||
private final Function<String, MemberRole> memberSlotRoles;
|
||||
private final Function<String, String> memberSlotNames;
|
||||
|
||||
/** Loopback-trust resolver: no token required, historical behaviour. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity) {
|
||||
this(identity, false, null, Map.of());
|
||||
}
|
||||
|
||||
/** As {@link #CallerResolver(ConnectionIdentity, boolean, String, Map)} with no leads pinned. Test-only. */
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token) {
|
||||
this(identity, tokenMode, token, Map.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* Single-pin form for the legacy {@code primary.terminal}-only configuration — one lead, named
|
||||
* {@code primary}.
|
||||
*
|
||||
* <p>A static factory rather than a fourth constructor overload on purpose: {@code String} and
|
||||
* {@code Map} overloads are ambiguous for a literal {@code null} argument, which is a compile
|
||||
* error at the call site and exactly the shape "unpinned" is written in.
|
||||
*
|
||||
* @param pinnedPrimaryTerminal the primary's own herdr {@code terminal_id}
|
||||
* ({@code null}/blank = unpinned)
|
||||
*/
|
||||
static CallerResolver pinnedTo(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token, String pinnedPrimaryTerminal) {
|
||||
return new CallerResolver(identity, tokenMode, token,
|
||||
pinnedPrimaryTerminal == null || pinnedPrimaryTerminal.isBlank()
|
||||
? Map.of() : Map.of(pinnedPrimaryTerminal, "primary"));
|
||||
}
|
||||
|
||||
/**
|
||||
* @param identity connection-based worker identification
|
||||
* @param tokenMode when true, a non-worker caller must present a valid bearer token
|
||||
* @param token the expected bearer token; required (non-blank) when {@code tokenMode}
|
||||
* @param leadTerminals herdr {@code terminal_id} → lead name for every configured lead
|
||||
* (CB-530). A caller resolving to one of these panes is that lead — a
|
||||
* {@link Role#PRIMARY} — rather than a worker. Empty = nothing pinned,
|
||||
* so every pane resolves as a worker.
|
||||
*/
|
||||
CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Map<String, String> leadTerminals) {
|
||||
this(identity, tokenMode, token, fixed(leadTerminals), null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-registry form: {@code leadTerminals} is consulted on every resolve, so leads discovered
|
||||
* after startup (CB-531's tab scan) take effect without a restart.
|
||||
*
|
||||
* <p>A static factory rather than a fourth constructor overload, for the same reason as
|
||||
* {@link #pinnedTo}: {@code Map} and {@code Supplier} overloads are ambiguous for a literal
|
||||
* {@code null}.
|
||||
*/
|
||||
static CallerResolver withLeads(ConnectionIdentity identity, boolean tokenMode,
|
||||
String token,
|
||||
Supplier<Map<String, String>> leadTerminals) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live registry form that can confirm a bound slot is an architect slot.
|
||||
*
|
||||
* <p>This is the only public construction path. It keeps terminal bindings and slot roles in
|
||||
* the same {@link MemberRegistry}, so a configured architect can resolve as an architect.
|
||||
*/
|
||||
public static CallerResolver withLeadsAndMembers(ConnectionIdentity identity,
|
||||
boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
MemberRegistry members) {
|
||||
return new CallerResolver(identity, tokenMode, token, leadTerminals,
|
||||
members == null ? null : members::snapshot,
|
||||
members == null ? null : members::roleForSlot,
|
||||
members == null ? null : members::nameForSlot);
|
||||
}
|
||||
|
||||
private static Supplier<Map<String, String>> fixed(Map<String, String> leadTerminals) {
|
||||
Map<String, String> snapshot = leadTerminals == null ? Map.of() : Map.copyOf(leadTerminals);
|
||||
return () -> snapshot;
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles) {
|
||||
this(identity, tokenMode, token, leadTerminals, architectTerminals, memberSlotRoles, null);
|
||||
}
|
||||
|
||||
private CallerResolver(ConnectionIdentity identity, boolean tokenMode, String token,
|
||||
Supplier<Map<String, String>> leadTerminals,
|
||||
Supplier<Map<String, String>> architectTerminals,
|
||||
Function<String, MemberRole> memberSlotRoles,
|
||||
Function<String, String> memberSlotNames) {
|
||||
if (tokenMode && (token == null || token.isBlank())) {
|
||||
throw new IllegalArgumentException(
|
||||
"auth.mode=token requires a non-empty token; check that the env var named by "
|
||||
+ "auth.tokenEnv is exported to the daemon's environment");
|
||||
}
|
||||
this.identity = identity;
|
||||
this.tokenMode = tokenMode;
|
||||
this.expectedToken = tokenMode ? token.getBytes(StandardCharsets.UTF_8) : null;
|
||||
this.leadTerminals = leadTerminals == null ? Map::of : leadTerminals;
|
||||
this.architectTerminals = architectTerminals == null ? Map::of : architectTerminals;
|
||||
this.memberSlotRoles = memberSlotRoles == null ? _ -> null : memberSlotRoles;
|
||||
this.memberSlotNames = memberSlotNames == null ? Function.identity() : memberSlotNames;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-recognised leads, {@code terminal_id → name} (CB-535).
|
||||
*
|
||||
* <p>Deliberately read from the same supplier {@link #resolve} consults, rather than from a
|
||||
* second copy handed to the roster: a lead that is <em>listed</em> but would not <em>resolve</em>
|
||||
* (or the reverse) is an address a peer cannot actually reach, and the two answers drifting apart
|
||||
* is precisely the confusion this exists to end. Live, so a lead discovered by the tab scan after
|
||||
* startup appears without a restart.
|
||||
*/
|
||||
public Map<String, String> leads() {
|
||||
return leadTerminals.get();
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-recognised architect slots, {@code terminal_id → slot name} (CB-548).
|
||||
*
|
||||
* <p>Read from the same supplier {@link #resolve} consults, so a slot that is <em>listed</em>
|
||||
* here but would not <em>resolve</em> (or the reverse) cannot drift apart. Live for the same
|
||||
* reason as {@link #leads()}.
|
||||
*/
|
||||
public Map<String, String> members() {
|
||||
return architectTerminals.get();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the caller of a request.
|
||||
*
|
||||
* @param remoteAddr the connection's remote address
|
||||
* @param remotePort the connection's remote port (used for the peer-PID lookup)
|
||||
* @param authorizationHeader the raw {@code Authorization} header, or {@code null}
|
||||
*/
|
||||
public Principal resolve(String remoteAddr, int remotePort, String authorizationHeader) {
|
||||
ConnectionIdentity.Caller c = identity.resolve(remoteAddr, remotePort);
|
||||
if (c.terminal() != null) {
|
||||
String lead = leadTerminals.get().get(c.terminal());
|
||||
if (lead != null) {
|
||||
// The config names this pane as a lead's own. The pane mapping is exactly as
|
||||
// unforgeable as a worker's, so it outranks the token path — no credential needed.
|
||||
// Checked before the architect registry so a pane named in BOTH is still the lead
|
||||
// (CB-548 preserves every existing leader behaviour).
|
||||
return Principal.leader(lead, c.terminal(), c.pid());
|
||||
}
|
||||
String slot = architectTerminals.get().get(c.terminal());
|
||||
if (slot != null && memberSlotRoles.apply(slot) == MemberRole.ARCHITECT) {
|
||||
// The config/live binding names this pane as an architect slot's own. Same
|
||||
// unforgeable pane mapping; the live binding, never a request argument, decides.
|
||||
// Check the slot role too: this defence in depth prevents a bad lifecycle bind from
|
||||
// escalating a dev or reviewer into an architect. Checked before the worker fallback.
|
||||
return Principal.architect(memberSlotNames.apply(slot), c.terminal(), c.pid());
|
||||
}
|
||||
return Principal.worker(c.terminal(), c.pid()); // unforgeable; never token-gated
|
||||
}
|
||||
|
||||
if (tokenMode) {
|
||||
return presentedTokenMatches(authorizationHeader)
|
||||
? Principal.primary(c.pid())
|
||||
: Principal.anonymous();
|
||||
}
|
||||
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
String presented = bearerValue(authorizationHeader);
|
||||
if (presented == null) {
|
||||
return false;
|
||||
}
|
||||
// Constant-time: MessageDigest.isEqual does not short-circuit on the first differing byte,
|
||||
// so a token cannot be recovered a byte at a time by timing the response.
|
||||
return MessageDigest.isEqual(presented.getBytes(StandardCharsets.UTF_8), expectedToken);
|
||||
}
|
||||
|
||||
/** Extract the credential from {@code Authorization: Bearer <token>}, or {@code null}. */
|
||||
private static String bearerValue(String header) {
|
||||
if (header == null) {
|
||||
return null;
|
||||
}
|
||||
String h = header.trim();
|
||||
if (h.length() < 7 || !h.regionMatches(true, 0, "Bearer ", 0, 7)) {
|
||||
return null;
|
||||
}
|
||||
String token = h.substring(7).trim();
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
}
|
||||
@@ -0,0 +1,232 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
* profile it points at, plus the <em>live</em> bindings from a live architect's herdr terminal to
|
||||
* its slot.
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
*/
|
||||
public final class MemberRegistry implements MemberLifecycle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MemberRegistry.class);
|
||||
|
||||
/**
|
||||
* One flattened {@code fleet:} entry.
|
||||
*
|
||||
* <p>Flattened because a slot name is unique only <em>within</em> its pool — {@code sonnet} may
|
||||
* legitimately be both a developer and a reviewer — while a terminal binds to exactly one thing.
|
||||
* The qualified {@link #key()} is what that binding uses.
|
||||
*
|
||||
* @param name the slot's key inside its pool
|
||||
* @param role the pool it came from
|
||||
* @param profile the backend it runs on
|
||||
*/
|
||||
public record Entry(String name, MemberRole role, String profile) {
|
||||
/** {@code "architect:opus"} — unique across pools, unlike {@link #name()}. */
|
||||
public String key() {
|
||||
return role.wireName() + ":" + name;
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
fleet.pool(role).forEach((name, slot) -> {
|
||||
if (slot != null) {
|
||||
Entry e = new Entry(name, role, slot.profile());
|
||||
flat.put(e.key(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code fleet_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
synchronized (terminalToSlot) {
|
||||
return Map.copyOf(terminalToSlot);
|
||||
}
|
||||
}
|
||||
|
||||
/** The slot a live terminal is bound to, or {@code null} if it is not an architect slot. */
|
||||
public String slotForTerminal(String terminal) {
|
||||
if (terminal == null) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
return terminalToSlot.get(terminal);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under — what the spawn lifecycle reads.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind {@code terminal} to {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it stands a slot up. The bind is atomic and preserves
|
||||
* the two cardinality invariants: a terminal may occupy at most one slot, and a slot may host at
|
||||
* most one terminal. Binding the same terminal to the same slot again is a harmless no-op.
|
||||
*
|
||||
* @param slot a configured slot name, or the bind is refused
|
||||
* @param terminal the pane that will act as this architect
|
||||
* @return {@code true} if the binding is now {@code terminal → slot}; {@code false} if it was
|
||||
* refused — an unknown slot, a terminal already bound to a different slot, or a slot
|
||||
* already hosting a different terminal
|
||||
*/
|
||||
public boolean bind(String slot, String terminal) {
|
||||
if (slot == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!isSlot(slot)) {
|
||||
return false; // unknown slot — nothing to bind to
|
||||
}
|
||||
String existingSlot = terminalToSlot.get(terminal);
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare-safe unbind of {@code expectedTerminal} from {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it tears a slot down. Only the exact binding
|
||||
* {@code expectedTerminal → slot} is removed; if that terminal was since rebound to a different
|
||||
* slot (or the slot to a different terminal), the call is a no-op returning {@code false} — a
|
||||
* stale unbind must never remove a replacement.
|
||||
*
|
||||
* @param slot the slot the caller believes the terminal is bound to
|
||||
* @param expectedTerminal the terminal it expects to be bound there
|
||||
* @return {@code true} if {@code expectedTerminal → slot} was removed; {@code false} if nothing
|
||||
* was (no such binding, or the binding had already moved)
|
||||
*/
|
||||
public boolean unbind(String slot, String expectedTerminal) {
|
||||
if (slot == null || expectedTerminal == null) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
String current = terminalToSlot.get(expectedTerminal);
|
||||
if (current == null || !slot.equals(current)) {
|
||||
return false; // absent, or a replacement/moved binding — leave it in place
|
||||
}
|
||||
terminalToSlot.remove(expectedTerminal);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind only architect sessions to a free slot with the resolved profile.
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
String slot = slotForTerminal(terminal);
|
||||
if (slot != null) {
|
||||
unbind(slot, terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,127 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* A resolved caller: its {@link Role}, and — for a worker — the herdr {@code terminal_id} that
|
||||
* identifies which worker it is (CB-501).
|
||||
*
|
||||
* @param role what this caller is authorized to act as
|
||||
* @param terminal the herdr {@code terminal_id} of the pane this caller occupies — a worker's, or
|
||||
* (since CB-532) a named lead's; {@code null} for an unnamed primary resolved off a
|
||||
* token or loopback trust, and for {@code ANONYMOUS}
|
||||
* @param pid the connecting process id, or {@code -1} when not resolvable (audit context)
|
||||
* @param name for a lead resolved from the CB-530 {@code leaders:} registry, which lead it is;
|
||||
* for an architect resolved from the CB-548 {@code architects:} registry, which
|
||||
* slot it occupies; {@code null} for every other caller, including an unnamed primary
|
||||
*/
|
||||
public record Principal(Role role, String terminal, long pid, String name) {
|
||||
|
||||
/**
|
||||
* Three-arg form for the callers that have no name to carry (workers, anonymous, and the
|
||||
* token/loopback primary paths). Kept so adding CB-530's {@code name} did not churn every
|
||||
* construction site — and so a reconstructed principal without a stashed name still works.
|
||||
*/
|
||||
public Principal(Role role, String terminal, long pid) {
|
||||
this(role, terminal, pid, null);
|
||||
}
|
||||
|
||||
/** A caller authenticated as nothing — the default when no check establishes anything else. */
|
||||
public static Principal anonymous() {
|
||||
return new Principal(Role.ANONYMOUS, null, -1);
|
||||
}
|
||||
|
||||
/** The orchestrating session, unnamed (token or loopback-trust path). */
|
||||
public static Principal primary(long pid) {
|
||||
return new Principal(Role.PRIMARY, null, pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* A named lead from the {@code leaders:} registry (CB-530).
|
||||
*
|
||||
* <p>Carries {@link Role#PRIMARY}: a lead <em>is</em> a primary as far as authorization goes,
|
||||
* so every existing {@code isPrimary()} gate keeps working unchanged and the role table needed
|
||||
* no new entry. The name is reporting only — it lets {@code fleet_whoami} say <em>which</em>
|
||||
* lead is asking once more than one is configured.
|
||||
*
|
||||
* <p><strong>CB-532: a lead now carries the terminal it was matched by.</strong> Under CB-530 it
|
||||
* deliberately did not, because {@code terminal} meant "which worker pane" everywhere and a
|
||||
* non-null one would have enrolled the lead in the worker presence map. That reading was what
|
||||
* made a lead unaddressable: {@link #ownsSession} could never be true for it, so
|
||||
* {@code fleet_reply} was refused and one lead could send to another but never be answered.
|
||||
* The terminal now means "which pane is this caller", the presence map keys on
|
||||
* {@link #isSpawnedMember()} instead, and a lead is a peer that can both send and receive.
|
||||
*/
|
||||
public static Principal leader(String name, String terminal, long pid) {
|
||||
return new Principal(Role.PRIMARY, terminal, pid, name);
|
||||
}
|
||||
|
||||
/** A worker peer, identified by its herdr pane. */
|
||||
public static Principal worker(String terminal, long pid) {
|
||||
return new Principal(Role.WORKER, terminal, pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* An architect (CB-548), identified by the slot it occupies and the pane bound to it.
|
||||
*
|
||||
* <p>Carries {@link Role#ARCHITECT}. {@code slotName} is reporting only — it lets
|
||||
* {@code fleet_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* (future) spawn lifecycle reads a profile back from. Identity is the {@code terminal}: like a
|
||||
* worker's it comes from the connection and the live terminal→slot binding, so
|
||||
* {@code ownsSession} works exactly as it does for a worker — an architect acts as its own
|
||||
* pane and no other.
|
||||
*/
|
||||
public static Principal architect(String slotName, String terminal, long pid) {
|
||||
return new Principal(Role.ARCHITECT, terminal, pid, slotName);
|
||||
}
|
||||
|
||||
public boolean isPrimary() {
|
||||
return role == Role.PRIMARY;
|
||||
}
|
||||
|
||||
public boolean isArchitect() {
|
||||
return role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isWorker() {
|
||||
return role == Role.WORKER;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller is a spawned member with its own pane.
|
||||
*
|
||||
* <p>Both workers and architects are spawned members. A lead is excluded because recording it
|
||||
* as present would count it as an available member in the roster.
|
||||
*/
|
||||
public boolean isSpawnedMember() {
|
||||
return role == Role.WORKER || role == Role.ARCHITECT;
|
||||
}
|
||||
|
||||
public boolean isAnonymous() {
|
||||
return role == Role.ANONYMOUS;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether this caller may act <em>as</em> {@code sessionId} — the "own session only" rule that
|
||||
* keeps one peer from replying or asking on another's behalf.
|
||||
*
|
||||
* <p>The rule is about <em>identity</em>, not rank: a caller may act as the pane it demonstrably
|
||||
* occupies, and as no other. CB-532 dropped the extra {@code isWorker()} conjunct that used to
|
||||
* be here. It was not what enforced the rule — {@code terminal.equals(sessionId)} is, and that
|
||||
* terminal comes from the connection, so it cannot be forged either way. All the conjunct did
|
||||
* was make a lead permanently unable to answer anyone, since a lead's terminal was null and a
|
||||
* lead is not a worker. A caller with no terminal at all (an off-host or token-authenticated
|
||||
* primary) still owns nothing, which is the case the null check covers.
|
||||
*/
|
||||
public boolean ownsSession(String sessionId) {
|
||||
return terminal != null && terminal.equals(sessionId);
|
||||
}
|
||||
|
||||
/** Short, non-sensitive description for audit lines and error details. */
|
||||
public String describe() {
|
||||
return switch (role) {
|
||||
case WORKER -> "worker:" + terminal;
|
||||
case ARCHITECT -> "architect:" + name;
|
||||
case PRIMARY -> name == null ? "primary" : "leader:" + name;
|
||||
case ANONYMOUS -> "anonymous";
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* What a caller is allowed to be on the bus (CB-501).
|
||||
*
|
||||
* <p>The ordering matters conceptually: {@link #PRIMARY} is the <em>most</em> privileged role
|
||||
* (it spawns, stops, sends to any session, and drains any inbox), not the least. Before CB-501
|
||||
* the daemon reached {@code PRIMARY} by <em>failing</em> every other check — any caller that did
|
||||
* not resolve to a known worker pane was treated as the primary. That is inverted here:
|
||||
* {@link #ANONYMOUS} is the fallback, and {@code PRIMARY} must be established.
|
||||
*/
|
||||
public enum Role {
|
||||
|
||||
/**
|
||||
* The orchestrating session. Established either by being a loopback caller that is not a
|
||||
* worker pane (under {@code loopback-trust}) or by presenting a valid bearer token (under
|
||||
* {@code token} mode).
|
||||
*/
|
||||
PRIMARY,
|
||||
|
||||
/**
|
||||
* A worker peer, identified by its herdr pane. Unforgeable: derived from the connection's
|
||||
* loopback peer PID via herdr's PID→pane map, never from a request argument.
|
||||
*/
|
||||
WORKER,
|
||||
|
||||
/**
|
||||
* A config-declared architect slot (CB-548): a gateway-local named session on a strong-model
|
||||
* profile that coordinates and delegates turns but does not own the fleet. Unforgeable like a
|
||||
* worker's — derived from the connection's pane and the live terminal→slot binding, never from
|
||||
* a request argument. May {@code SEND} a turn, {@code REPLY}/{@code ASK} only as its own pane,
|
||||
* and {@code READ}/{@code METRICS}; may <em>not</em> {@code SPAWN}/{@code STOP}/{@code DRAIN}
|
||||
* (those stay the primary's, to keep lifecycle in one pair of hands).
|
||||
*/
|
||||
ARCHITECT,
|
||||
|
||||
/** Authenticated as nothing. Authorized for nothing but {@code /healthz}. */
|
||||
ANONYMOUS
|
||||
}
|
||||
@@ -0,0 +1,297 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link FleetConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Fleetd.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(FleetConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public FleetConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
/** The file this ref reloads from, or {@code null} for a {@link #fixed} ref. */
|
||||
public Path path() {
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
public String summary() {
|
||||
if (error != null) {
|
||||
return "config reload refused — " + error;
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
}
|
||||
return "config reloaded";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the file, validate it, and swap it in when nothing cold changed.
|
||||
*
|
||||
* <p>Never throws: a reload is a best-effort operation on a daemon that is already serving, and
|
||||
* a bad edit must not take it down. Every failure path leaves the previous config live and is
|
||||
* reported through the returned {@link Outcome}.
|
||||
*/
|
||||
public Outcome reload() {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
FleetConfig old = current.get();
|
||||
FleetConfig fresh;
|
||||
try {
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
return Outcome.failed(msg);
|
||||
}
|
||||
|
||||
List<String> cold = changedColdKeys(old, fresh);
|
||||
if (!cold.isEmpty()) {
|
||||
Outcome out = Outcome.refusedCold(cold);
|
||||
log.warn(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
}
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.memberHerdrSocket(), fresh.memberHerdrSocket())) {
|
||||
changed.add("memberHerdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
if (!Objects.equals(old.auth(), fresh.auth())) {
|
||||
changed.add("auth");
|
||||
}
|
||||
// Kept in step with COLD_KEYS so the doc and the code cannot drift apart silently.
|
||||
assert COLD_KEYS.containsAll(changed) : "a cold key was reported that COLD_KEYS omits";
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
}
|
||||
if (!Objects.equals(old.leadHeartbeat(), fresh.leadHeartbeat())) {
|
||||
changed.add("leadHeartbeat");
|
||||
}
|
||||
if (!Objects.equals(old.guard(), fresh.guard())) {
|
||||
changed.add("guard");
|
||||
}
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, FleetConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, FleetConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
if (!before.keySet().equals(after.keySet())) {
|
||||
Set<String> diff = new LinkedHashSet<>(before.keySet());
|
||||
diff.addAll(after.keySet());
|
||||
diff.removeIf(p -> before.containsKey(p) && after.containsKey(p));
|
||||
changed.add("profiles (added/removed: " + String.join(", ", diff) + ")");
|
||||
}
|
||||
// An EXISTING profile's launch settings are deferred too, and this is easy to get wrong:
|
||||
// `HerdrPeerLauncher` takes `Map.copyOf(profiles)` at construction and `spawn` resolves the
|
||||
// profile out of that snapshot, so a reloaded model/baseUrl/argv/env never reaches a launch.
|
||||
// Only weight and maxLoad are genuinely hot, because placement reads them through the
|
||||
// supplier on the composite rather than from the adapter's copy. Without this check a
|
||||
// changed model would report "config reloaded" and silently do nothing — the worst outcome
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
FleetConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
});
|
||||
if (!relaunch.isEmpty()) {
|
||||
changed.add("profiles." + String.join("/", relaunch) + " launch settings "
|
||||
+ "(model, baseUrl, argv, env, …) — the launcher holds a startup snapshot");
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
&& Objects.equals(a.argv(), b.argv())
|
||||
&& Objects.equals(a.placement(), b.placement())
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
// CB-634: the IDE MCP mount is a launch flag, fixed at spawn like mcpUrl — a
|
||||
// reload changes it only for members spawned after, so a changed value is deferred.
|
||||
&& Objects.equals(a.ideMcpUrl(), b.ideMcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,96 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Polls {@code fleetd.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* (CB-559). Opt-in through {@code configReload.enabled}.
|
||||
*
|
||||
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
||||
* native backend — it falls back to polling internally anyway, at an interval this code does not
|
||||
* control — and editors save config files in ways that produce a different event mix per editor
|
||||
* (write-in-place, write-and-rename, write-temp-and-swap). A modified-time check treats all of them
|
||||
* the same and is a single {@code stat} per tick, which at a ten-second cadence costs nothing worth
|
||||
* measuring.
|
||||
*
|
||||
* <p><strong>A missing or unreadable file is not a reason to act.</strong> Many editors briefly
|
||||
* unlink the file during a save. Reloading on "it vanished" would mean reloading from a file that no
|
||||
* longer exists; reporting an error every tick would bury the log. So an unreadable file is skipped
|
||||
* silently and the next tick tries again — the running config stays live, which is the correct
|
||||
* outcome either way.
|
||||
*/
|
||||
public final class ConfigWatcher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigWatcher.class);
|
||||
|
||||
private final ConfigRef ref;
|
||||
private final long intervalSeconds;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
|
||||
private volatile long lastSeenMillis;
|
||||
|
||||
public ConfigWatcher(ConfigRef ref, long intervalSeconds) {
|
||||
this.ref = ref;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.lastSeenMillis = modifiedMillis(ref.path());
|
||||
this.scheduler = Executors.newSingleThreadScheduledExecutor(r -> {
|
||||
Thread t = new Thread(r, "config-watcher");
|
||||
// A daemon thread: an operator's config watch must never be the reason the JVM refuses
|
||||
// to exit after everything else has shut down.
|
||||
t.setDaemon(true);
|
||||
return t;
|
||||
});
|
||||
}
|
||||
|
||||
/** Begin watching. A ref with no file (a fixed one) is a no-op rather than an error. */
|
||||
public void start() {
|
||||
if (ref.path() == null) {
|
||||
log.debug("config watch not started — this config has no file behind it");
|
||||
return;
|
||||
}
|
||||
scheduler.scheduleWithFixedDelay(this::tick, intervalSeconds, intervalSeconds,
|
||||
TimeUnit.SECONDS);
|
||||
log.info("config watch: {} re-read when it changes (every {}s)", ref.path(), intervalSeconds);
|
||||
}
|
||||
|
||||
/** One poll. Never throws — an exception here would silently cancel the schedule. */
|
||||
void tick() {
|
||||
try {
|
||||
long now = modifiedMillis(ref.path());
|
||||
if (now == 0 || now == lastSeenMillis) {
|
||||
return;
|
||||
}
|
||||
// Stamp BEFORE reloading. A file whose reload is refused (a bad edit, or a cold key)
|
||||
// must not be retried every tick — that would log the same refusal forever. The next
|
||||
// save moves the timestamp again and earns a fresh attempt.
|
||||
lastSeenMillis = now;
|
||||
ref.reload();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("config watch tick failed, still watching: {}", e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
private static long modifiedMillis(Path path) {
|
||||
if (path == null) {
|
||||
return 0;
|
||||
}
|
||||
try {
|
||||
return Files.getLastModifiedTime(path).toMillis();
|
||||
} catch (IOException e) {
|
||||
return 0; // mid-save, or gone: say nothing and try again next tick
|
||||
}
|
||||
}
|
||||
|
||||
/** Stop polling. Called from the daemon's ordered shutdown hook, alongside the other loops. */
|
||||
public void stop() {
|
||||
scheduler.shutdownNow();
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.guard;
|
||||
package dev.ltms.fleet.guard;
|
||||
|
||||
/** Thrown when the subscription boundary would be violated. Never swallow this. */
|
||||
public class GuardException extends RuntimeException {
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.guard;
|
||||
package dev.ltms.fleet.guard;
|
||||
|
||||
import java.net.URI;
|
||||
import java.net.URISyntaxException;
|
||||
@@ -16,7 +16,7 @@ import java.util.Set;
|
||||
* means its traffic would leave the subscription. That is a hard stop.</li>
|
||||
* </ul>
|
||||
*
|
||||
* Both checks throw {@link GuardException} on violation. {@code bridged} calls
|
||||
* Both checks throw {@link GuardException} on violation. {@code fleetd} calls
|
||||
* {@link #assertWorker} before spawning a worker and {@link #assertPrimaryClean}
|
||||
* against its own environment at startup.
|
||||
*/
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
|
||||
/**
|
||||
* Pure classifier. Collection and repair are deliberately outside this package.
|
||||
* {@link HealthState#ERROR_ON_SCREEN} is not decided yet because it needs a bounded pane detection
|
||||
* read and an adapter-specific fatal signature; status facts alone must not guess it.
|
||||
*/
|
||||
public final class FleetHealth {
|
||||
private FleetHealth() { }
|
||||
|
||||
public static HealthDecision decide(HealthSnapshot s, HealthPrior prior, long nowNanos) {
|
||||
if (s.controlLinkDown()) return result(HealthState.CONTROL_LINK_DOWN, false);
|
||||
if (s.targetNotFound()) return result(HealthState.GONE, false);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING && !s.present() && s.readinessGraceElapsed()) {
|
||||
return result(HealthState.NEVER_READY, false);
|
||||
}
|
||||
if (s.orphanedDelegation()) return result(HealthState.DELEGATION_ORPHANED, false);
|
||||
boolean disagreement = s.sessionState() == MemberSession.State.BUSY && s.acceptedDelivery()
|
||||
&& (s.liveStatus() == AgentStatus.IDLE || s.liveStatus() == AgentStatus.DONE);
|
||||
if (disagreement && prior.busyButDone()) return result(HealthState.TURN_BOUNDARY_LOST, true);
|
||||
if (s.stalled()) return result(HealthState.STALL_SUSPECTED, disagreement);
|
||||
if (s.replyStranded()) return result(HealthState.REPLY_STRANDED, disagreement);
|
||||
if (s.queuedDelivery() || s.inboxMessage()) return result(HealthState.WORK_PENDING, disagreement);
|
||||
if (s.sessionState() == MemberSession.State.SPAWNING) return result(HealthState.STARTING, disagreement);
|
||||
if (s.acceptedDelivery() && s.liveStatus() == AgentStatus.BLOCKED) {
|
||||
return result(HealthState.BLOCKED_AMBIGUOUS, disagreement);
|
||||
}
|
||||
// An accepted delivery remains bridge work even when herdr is late, unknown, or has already
|
||||
// reported DONE once. It cannot be IDLE until the delegation has resolved.
|
||||
if (s.acceptedDelivery()) return result(HealthState.WORKING, disagreement);
|
||||
return result(HealthState.IDLE, disagreement);
|
||||
}
|
||||
|
||||
private static HealthDecision result(HealthState state, boolean disagreement) {
|
||||
return new HealthDecision(state, new HealthPrior(disagreement));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,209 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long intervalSeconds;
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
* for one tick during an ordinary race — an async ticket exists before its virtual thread has
|
||||
* reached {@code rendezvous.open()}, so for that instant nothing is accepted or queued behind
|
||||
* it. {@code decide} maps the field straight to {@code DELEGATION_ORPHANED} with no cross-tick
|
||||
* smoothing of its own, so a single racy read would log a fault that clears on the next tick.
|
||||
* Requiring two consecutive observations costs one interval of latency on a real orphan and
|
||||
* removes that false positive entirely.
|
||||
*/
|
||||
private final Map<String, Integer> orphanStreaks = new HashMap<>();
|
||||
|
||||
/** How many consecutive ticks a target must look orphaned before health reports it (CB-643). */
|
||||
static final int ORPHAN_CONFIRM_TICKS = 2;
|
||||
|
||||
// CB-643: every HealthSnapshot field now carries real evidence. The NOT_YET_OBSERVED placeholder
|
||||
// that stood in for 7 of the 12 is gone, and with it the reason 8 of the 9 fault states were
|
||||
// unreachable — GONE and NEVER_READY included, which is what kept CB-580's failTarget from ever
|
||||
// firing. Do not reintroduce a constant here: a field with no publisher is a dead state, and the
|
||||
// tests pass either way, so nothing else will tell you.
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code FleetMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
long workingSuspectAfterSeconds, BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.workingSuspectAfterNanos = TimeUnit.SECONDS.toNanos(workingSuspectAfterSeconds);
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
List<Agent> agentsNow;
|
||||
boolean controlLinkDown = false;
|
||||
try {
|
||||
agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
} catch (HerdrException error) {
|
||||
agentsNow = List.of();
|
||||
controlLinkDown = true;
|
||||
log.warn("fleet health control link unavailable; classifying roster", error);
|
||||
}
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
long nowNanos = clock.getAsLong();
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean present = agent != null;
|
||||
boolean targetNotFound = !controlLinkDown && !present
|
||||
&& session.state() != MemberSession.State.SPAWNING;
|
||||
boolean readinessGraceElapsed = nowNanos - session.spawnedAtNanos() >= READINESS_GRACE_NANOS;
|
||||
boolean stalled = session.state() == MemberSession.State.BUSY
|
||||
&& nowNanos - session.lastActivityAtNanos() >= workingSuspectAfterNanos;
|
||||
// CB-643: the three message-layer facts CB-640 published. Read them here rather than
|
||||
// leaving them false — that constant is what made 8 of the 9 fault states dead.
|
||||
boolean queuedDelivery = messages.hasQueuedDelivery(session.terminalId());
|
||||
boolean replyStranded = messages.hasStrandedReply(session.terminalId());
|
||||
boolean orphanedDelegation = confirmOrphan(session.terminalId(),
|
||||
messages.hasOrphanedDelegation(session.terminalId()));
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, queuedDelivery,
|
||||
messages.hasInboxMessage(session.terminalId()), present, targetNotFound, controlLinkDown,
|
||||
readinessGraceElapsed, orphanedDelegation, replyStranded, stalled);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
nowNanos);
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
orphanStreaks.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// Any unclassified collection failure must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Debounce {@link MessageService#hasOrphanedDelegation} across ticks (CB-643). Returns true only
|
||||
* once {@code observed} has held for {@link #ORPHAN_CONFIRM_TICKS} consecutive ticks; a single
|
||||
* false reading resets the streak, so a transient race never reaches the classifier.
|
||||
*/
|
||||
private boolean confirmOrphan(String target, boolean observed) {
|
||||
if (!observed) {
|
||||
orphanStreaks.remove(target);
|
||||
return false;
|
||||
}
|
||||
int streak = orphanStreaks.merge(target, 1, Integer::sum);
|
||||
return streak >= ORPHAN_CONFIRM_TICKS;
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,4 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
/** Classification plus the private fact that the next pure decision needs. */
|
||||
public record HealthDecision(HealthState state, HealthPrior prior) { }
|
||||
@@ -0,0 +1,6 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
/** Private cross-tick observation. It is deliberately not a reported health value. */
|
||||
public record HealthPrior(boolean busyButDone) {
|
||||
public static final HealthPrior NONE = new HealthPrior(false);
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
|
||||
/** Read-only facts from one fleet collection tick. */
|
||||
public record HealthSnapshot(MemberSession.State sessionState, AgentStatus liveStatus,
|
||||
boolean acceptedDelivery, boolean queuedDelivery, boolean inboxMessage,
|
||||
boolean present, boolean targetNotFound, boolean controlLinkDown,
|
||||
boolean readinessGraceElapsed, boolean orphanedDelegation,
|
||||
boolean replyStranded, boolean stalled) { }
|
||||
@@ -0,0 +1,8 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
/** Health classifications reported for a member. */
|
||||
public enum HealthState {
|
||||
STARTING, IDLE, WORKING, WORK_PENDING, BLOCKED_AMBIGUOUS,
|
||||
NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN
|
||||
}
|
||||
@@ -0,0 +1,25 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Counts turns that ended via the completion fallback instead of {@code fleet_reply}.
|
||||
* MUTE is an observation by target and profile, not a classifier state and never suppresses faults.
|
||||
*/
|
||||
public final class MuteCounter {
|
||||
private final Map<String, Integer> byTarget = new ConcurrentHashMap<>();
|
||||
private final Map<String, Integer> byProfile = new ConcurrentHashMap<>();
|
||||
|
||||
/** Record only fallback completion; a structured reply does not make a member mute. */
|
||||
public void observe(String target, String profile, Rendezvous.Kind kind) {
|
||||
if (kind != Rendezvous.Kind.COMPLETION) return;
|
||||
byTarget.merge(target, 1, Integer::sum);
|
||||
byProfile.merge(profile, 1, Integer::sum);
|
||||
}
|
||||
|
||||
public int forTarget(String target) { return byTarget.getOrDefault(target, 0); }
|
||||
public int forProfile(String profile) { return byProfile.getOrDefault(profile, 0); }
|
||||
}
|
||||
@@ -0,0 +1,26 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/** Fixed pane-probe limits. Pane content is never retained here. */
|
||||
public final class PaneBudget {
|
||||
public static final long COOLDOWN_NANOS = 60_000_000_000L;
|
||||
public static final int MAX_PER_TICK = 2;
|
||||
private final Map<String, Long> lastProbe = new HashMap<>();
|
||||
private int cursor;
|
||||
|
||||
public List<String> choose(List<String> candidates, long nowNanos, long configuredCooldownNanos) {
|
||||
long cooldown = Math.max(COOLDOWN_NANOS, configuredCooldownNanos);
|
||||
List<String> out = new ArrayList<>();
|
||||
for (int n = 0; n < candidates.size() && out.size() < MAX_PER_TICK; n++) {
|
||||
String target = candidates.get((cursor + n) % candidates.size());
|
||||
Long last = lastProbe.get(target);
|
||||
if (last == null || nowNanos - last >= cooldown) { out.add(target); lastProbe.put(target, nowNanos); }
|
||||
}
|
||||
if (!candidates.isEmpty()) cursor = (cursor + 1) % candidates.size();
|
||||
return List.copyOf(out);
|
||||
}
|
||||
}
|
||||
+9
-3
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
@@ -15,6 +15,10 @@ import com.fasterxml.jackson.databind.JsonNode;
|
||||
* @param agentType detected agent kind, e.g. {@code "claude"}, or {@code null} before herdr
|
||||
* has detected it (the start-time shape)
|
||||
* @param status current lifecycle state
|
||||
* @param name the unique label the agent was started with — for a bridge worker this is
|
||||
* {@code claude-<profile>-<nonce>-<seq>} (CB-117 keys orphan reaping on the
|
||||
* nonce); {@code null} for agents the bridge did not start, e.g. a user's own
|
||||
* Claude session
|
||||
*/
|
||||
public record Agent(
|
||||
String terminalId,
|
||||
@@ -23,7 +27,8 @@ public record Agent(
|
||||
String tabId,
|
||||
String sessionId,
|
||||
String agentType,
|
||||
AgentStatus status) {
|
||||
AgentStatus status,
|
||||
String name) {
|
||||
|
||||
/** Project a herdr {@code agent} node. Tolerates the start-time shape (no session yet). */
|
||||
public static Agent from(JsonNode a) {
|
||||
@@ -42,6 +47,7 @@ public record Agent(
|
||||
a.path("tab_id").asText(null),
|
||||
sessionId,
|
||||
type,
|
||||
AgentStatus.fromWire(a.path("agent_status").asText(null)));
|
||||
AgentStatus.fromWire(a.path("agent_status").asText(null)),
|
||||
a.path("name").asText(null));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,159 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
|
||||
/**
|
||||
* Domain layer over herdr's native {@code agent.*} namespace — the worker south side.
|
||||
* Chosen in the CB-102 spike over the pane + {@code send_text} fallback because herdr
|
||||
* tracks each worker's Claude session UUID itself.
|
||||
*
|
||||
* <p>Ported to herdr protocol 19 (herdr 0.8.0, CB-521): {@code agent.start} now starts a
|
||||
* <em>supported</em> agent ({@code kind}) into an <em>existing</em> pane, so the worker's
|
||||
* {@code env}/{@code cwd} move to pane creation ({@code tab.create}/{@code pane.split} — see
|
||||
* {@link WorkspaceControl}), and {@code agent.send} is replaced by {@code agent.prompt}
|
||||
* (which submits in one call) plus {@code agent.send_keys} for the raw Enter nudge.
|
||||
*
|
||||
* <p>Every method is one herdr call through the injected {@link HerdrClient}, so this
|
||||
* layer is unit-testable with a fake and contract-tested against a live daemon.
|
||||
*/
|
||||
public final class AgentControl {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
/**
|
||||
* Protocol 19 dropped {@code terminal_id} as an {@code agent.*} target — herdr now resolves
|
||||
* targets by pane id or agent name only, while the bridge keys every session on the terminal.
|
||||
* This caches the terminal→pane mapping (stable for a worker's lifetime) so callers keep
|
||||
* addressing agents by terminal; entries are invalidated on {@code agent_not_found}.
|
||||
*/
|
||||
private final Map<String, String> paneByTerminal = new ConcurrentHashMap<>();
|
||||
|
||||
public AgentControl(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/** One agent-targeted call, translating a terminal id to its pane id (retrying once fresh). */
|
||||
private JsonNode agentCall(String method, String target, Map<String, Object> extra) {
|
||||
String resolved = resolveTarget(target);
|
||||
try {
|
||||
return herdr.call(method, withTarget(resolved, extra));
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_not_found".equals(e.code()) || resolved.equals(target)) throw e;
|
||||
paneByTerminal.remove(target); // the cached pane went away — re-resolve once
|
||||
String fresh = resolveTarget(target);
|
||||
if (fresh.equals(resolved)) throw e;
|
||||
return herdr.call(method, withTarget(fresh, extra));
|
||||
}
|
||||
}
|
||||
|
||||
private static Map<String, Object> withTarget(String target, Map<String, Object> extra) {
|
||||
Map<String, Object> m = new LinkedHashMap<>();
|
||||
m.put("target", target);
|
||||
m.putAll(extra);
|
||||
return m;
|
||||
}
|
||||
|
||||
/** The pane id behind a terminal-id target, or the target verbatim for pane ids / names. */
|
||||
private String resolveTarget(String target) {
|
||||
if (target == null || !target.startsWith("term_")) {
|
||||
return target;
|
||||
}
|
||||
String cached = paneByTerminal.get(target);
|
||||
if (cached != null) {
|
||||
return cached;
|
||||
}
|
||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||
if (target.equals(a.path("terminal_id").asText(null))) {
|
||||
String pane = a.path("pane_id").asText(null);
|
||||
if (pane != null) {
|
||||
paneByTerminal.put(target, pane);
|
||||
return pane;
|
||||
}
|
||||
}
|
||||
}
|
||||
return target; // unknown terminal — let herdr report it against the original target
|
||||
}
|
||||
|
||||
/**
|
||||
* Start an agent into {@code paneId}, which must be sitting at its interactive shell prompt —
|
||||
* the seed pane of a freshly-created worker tab, or a fresh split. The pane's shell already
|
||||
* carries the worker's env ({@code ANTHROPIC_BASE_URL}, token, …) and cwd from pane creation;
|
||||
* herdr resolves the executable from {@code kind} and waits (its default timeout) until the
|
||||
* agent is detected and ready for input.
|
||||
*
|
||||
* @param name unique label for this agent ({@code <kind>-<profile>-<nonce>-<seq>})
|
||||
* @param kind supported agent kind and canonical executable, e.g. {@code "claude"},
|
||||
* {@code "opencode"}
|
||||
* @param args extra arguments after the executable, e.g. {@code --mcp-config …}
|
||||
* @param paneId the pane to start the agent in
|
||||
*/
|
||||
public Agent start(String name, String kind, List<String> args, String paneId) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("name", name);
|
||||
params.put("kind", kind);
|
||||
params.put("pane_id", paneId);
|
||||
params.put("args", args);
|
||||
JsonNode result = herdr.call("agent.start", params);
|
||||
return Agent.from(result.get("agent"));
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code text} to an agent as its next prompt <em>and submit it</em> — herdr's
|
||||
* {@code agent.prompt} pastes the text (embedded newlines preserved verbatim) and submits it
|
||||
* in the same call, replacing the pre-protocol-19 two-event {@code agent.send} dance.
|
||||
*/
|
||||
public void send(String target, String text) {
|
||||
agentCall("agent.prompt", target, Map.of("text", text));
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-send the submit keystroke (Enter) to {@code target}. The submit that accompanies a
|
||||
* delivery can race the paste — especially right as the worker's TUI becomes interactive —
|
||||
* leaving the text unsubmitted; the injector nudges it with this until the worker actually
|
||||
* picks up (CB-113).
|
||||
*/
|
||||
public void submit(String target) {
|
||||
agentCall("agent.send_keys", target, Map.of("keys", List.of("enter")));
|
||||
}
|
||||
|
||||
/**
|
||||
* Read an agent's terminal.
|
||||
*
|
||||
* @param source one of {@code visible|recent|recent_unwrapped|detection}
|
||||
*/
|
||||
public String read(String target, String source) {
|
||||
JsonNode result = agentCall("agent.read", target, Map.of("source", source));
|
||||
return result.path("read").path("text").asText("");
|
||||
}
|
||||
|
||||
/** Current agent record (status, session UUID, pane). */
|
||||
public Agent get(String target) {
|
||||
return Agent.from(agentCall("agent.get", target, Map.of()).get("agent"));
|
||||
}
|
||||
|
||||
/** Just the lifecycle status — what the status-gated injector checks before send. */
|
||||
public AgentStatus status(String target) {
|
||||
return get(target).status();
|
||||
}
|
||||
|
||||
/** All agents herdr tracks — the discovery surface ("what workers exist"). */
|
||||
public List<Agent> list() {
|
||||
JsonNode result = herdr.call("agent.list");
|
||||
List<Agent> out = new ArrayList<>();
|
||||
for (JsonNode a : result.path("agents")) {
|
||||
out.add(Agent.from(a));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Tear a worker down (there is no agent.stop — close its pane). */
|
||||
public void close(String paneId) {
|
||||
herdr.call("pane.close", Map.of("pane_id", paneId));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,39 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
/**
|
||||
* A herdr agent's lifecycle state, as reported by {@code agent_status}. Drives the
|
||||
* status-gated injector: a worker is safe to inject into only when {@link #IDLE},
|
||||
* {@link #BLOCKED}, or {@link #DONE}, never mid-turn ({@link #WORKING}).
|
||||
*/
|
||||
public enum AgentStatus {
|
||||
IDLE,
|
||||
WORKING,
|
||||
BLOCKED,
|
||||
/**
|
||||
* The worker has finished its turn and is settled at an idle prompt. herdr emits this
|
||||
* (observed live alongside {@code idle}) as a turn-complete marker; earlier code mapped the
|
||||
* unrecognized string to {@link #UNKNOWN}, which both wedged delivery (not {@link #injectable})
|
||||
* and mis-fired the CB-109 stall-failure on a worker that had actually answered. It is a
|
||||
* turn-boundary equivalent to {@link #IDLE}: injectable, and a {@code working → done} edge is a
|
||||
* real completion.
|
||||
*/
|
||||
DONE,
|
||||
UNKNOWN;
|
||||
|
||||
/** Map herdr's wire string ({@code idle|working|blocked|done|unknown}) to the enum. */
|
||||
public static AgentStatus fromWire(String s) {
|
||||
if (s == null) return UNKNOWN;
|
||||
return switch (s.toLowerCase()) {
|
||||
case "idle" -> IDLE;
|
||||
case "working" -> WORKING;
|
||||
case "blocked" -> BLOCKED;
|
||||
case "done" -> DONE;
|
||||
default -> UNKNOWN;
|
||||
};
|
||||
}
|
||||
|
||||
/** Whether {@code fleetd} may inject a message now without stepping on a live turn. */
|
||||
public boolean injectable() {
|
||||
return this == IDLE || this == BLOCKED || this == DONE;
|
||||
}
|
||||
}
|
||||
+3
-3
@@ -1,17 +1,17 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
/**
|
||||
* Client face onto the herdr daemon (protocol 14, herdr 0.7.0).
|
||||
*
|
||||
* <p>This is the ONLY thing in {@code bridged} that speaks to herdr. Every method
|
||||
* <p>This is the ONLY thing in {@code fleetd} that speaks to herdr. Every method
|
||||
* maps to a herdr JSON-RPC call over its Unix domain socket. Requests are
|
||||
* newline-delimited JSON with a <em>string</em> id; responses carry either a
|
||||
* {@code result} object (whose {@code type} field discriminates the payload) or an
|
||||
* {@code error} object.
|
||||
*
|
||||
* <p>Higher layers ({@code bridged}'s policy brain, REST endpoints, MCP adapters)
|
||||
* <p>Higher layers ({@code fleetd}'s policy brain, REST endpoints, MCP adapters)
|
||||
* depend on this interface, not on the socket. Tests substitute a fake; the
|
||||
* {@code contract}-tagged suite exercises the real implementation against a running
|
||||
* herdr to catch protocol drift.
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.core.JsonProcessingException;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
/**
|
||||
* Raised when a herdr call fails: transport error, or an {@code error} envelope
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import java.util.Objects;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
/** Routes lead operations and member operations to their owning herdr daemon. */
|
||||
public final class HerdrRouter implements AutoCloseable {
|
||||
private final HerdrClient lead;
|
||||
private final HerdrClient member;
|
||||
private final AgentControl leadAgents;
|
||||
private final AgentControl memberAgents;
|
||||
private final WorkspaceControl leadSpaces;
|
||||
private final WorkspaceControl memberSpaces;
|
||||
private final Predicate<String> isLead;
|
||||
|
||||
public HerdrRouter(HerdrClient lead, HerdrClient member, Predicate<String> isLead) {
|
||||
this.lead = Objects.requireNonNull(lead, "lead");
|
||||
this.member = member != null ? member : lead;
|
||||
this.isLead = Objects.requireNonNull(isLead, "isLead");
|
||||
leadAgents = new AgentControl(this.lead);
|
||||
memberAgents = this.member == this.lead ? leadAgents : new AgentControl(this.member);
|
||||
leadSpaces = new WorkspaceControl(this.lead);
|
||||
memberSpaces = this.member == this.lead ? leadSpaces : new WorkspaceControl(this.member);
|
||||
}
|
||||
|
||||
public AgentControl leadAgents() { return leadAgents; }
|
||||
public WorkspaceControl leadSpaces() { return leadSpaces; }
|
||||
public AgentControl memberAgents() { return memberAgents; }
|
||||
public WorkspaceControl memberSpaces() { return memberSpaces; }
|
||||
public AgentControl agentsFor(String targetId) { return isLead.test(targetId) ? leadAgents : memberAgents; }
|
||||
|
||||
HerdrClient leadClient() { return lead; }
|
||||
HerdrClient memberClient() { return member; }
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
lead.close();
|
||||
if (member != lead) member.close();
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,188 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Discovers which panes host a lead by scanning herdr for tabs the operator labelled by convention
|
||||
* (CB-531), and hands {@link dev.ltms.fleet.auth.CallerResolver} the resulting
|
||||
* {@code terminal_id → lead name} map.
|
||||
*
|
||||
* <p><strong>Why scan at all.</strong> A lead is never spawned — a human opens a tab and starts an
|
||||
* agent in it — so the daemon cannot learn a lead's {@code terminal_id} at creation time the way it
|
||||
* does a worker's. CB-530 solved that by having the operator paste each id into {@code leaders:},
|
||||
* which works but costs a config edit and a daemon restart per lead, and the id is only obtainable
|
||||
* by first starting the session and asking it. Scanning closes that loop: label the tab, and the
|
||||
* pane is recognised on the next resolve.
|
||||
*
|
||||
* <p><strong>CB-579 — matched by name, not prefix.</strong> This used to strip one shared
|
||||
* {@code tabPrefix} off a label to derive the lead's name, and merged a config-supplied
|
||||
* {@code terminal_id} pin over every scan result so the pin could never expire. Both are gone: each
|
||||
* lead now configures its own exact {@code tab} label ({@code fleet.leaders.<name>.tab}), so this
|
||||
* class is handed a {@code tab → name} map up front and matches labels against it exactly
|
||||
* (case-insensitively). There is no merge step — a scan result is the whole answer. That is the
|
||||
* fix for the bug this replaces: a {@code terminal_id} pin surviving in config after the pane it
|
||||
* named was gone, so the daemon kept treating a dead session as a live lead forever.
|
||||
*
|
||||
* <p><strong>Direction of trust.</strong> The label names the lead; it never <em>grants</em>
|
||||
* anything a pane could take for itself. Three properties keep that honest:
|
||||
* <ol>
|
||||
* <li>Worker spaces are excluded wholesale ({@code excludedWorkspaceLabels}), so a worker cannot
|
||||
* become a lead by being placed — as a split, say — inside a matching tab.</li>
|
||||
* <li>A worker cannot rename a tab: {@code tab.rename} is reachable only through
|
||||
* {@link WorkspaceControl}, which no {@code fleet_*} tool exposes. The label is writable by
|
||||
* the human at the terminal and by nobody the bridge is defending against.</li>
|
||||
* <li>The label is a <em>name</em>, not a capability. What a pane may do is decided by
|
||||
* {@code Authz} against the role {@code CallerResolver} returns; a tab that calls itself a
|
||||
* lead still cannot act as one unless the daemon's own registry agrees.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p><strong>CB-558 — fleetd now writes lead labels too.</strong> This class used to be able to say
|
||||
* that fleetd never renames a lead tab, so the label was always the human's own writing and there
|
||||
* was no round-trip from the daemon's rename back into its next decision.
|
||||
* {@code dev.ltms.fleet.lead.LeadLauncher} ends that: an auto-launched lead is labelled by the
|
||||
* daemon and found again by this scan. The trust direction above is unaffected — fleetd writing a
|
||||
* name for a lead it just started is not a pane promoting itself — but <em>staleness</em> becomes
|
||||
* real: a label left behind by a session that has since died would read as a live lead forever.
|
||||
* This scanner does not solve that (its job is naming, and a stale name costs nothing here); the
|
||||
* launcher does, by requiring a running agent in the tab before it counts the lead as live. If you
|
||||
* ever make a decision that <em>removes</em> something based on this map, add the same check.
|
||||
* The remaining hazard is an <em>operator</em> one — a worker {@code tabLabel} template that
|
||||
* happens to start with the same prefix would promote the whole fleet — and that is refused at
|
||||
* startup by {@code FleetConfig.validateLeadTabPrefixes} rather than documented here.
|
||||
*
|
||||
* <p><strong>Caching.</strong> {@link #get()} is on the request path (every resolve), so the scan
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan keeps
|
||||
* the previous answer instead of emptying it — a herdr hiccup must not silently demote a live lead
|
||||
* mid-session.
|
||||
*/
|
||||
public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadTabScanner.class);
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final Map<String, String> tabToName;
|
||||
private final Set<String> excludedWorkspaceLabels;
|
||||
private final long ttlNanos;
|
||||
private final LongSupplier clock;
|
||||
|
||||
private Map<String, String> cached = Map.of();
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
* @param tabToName every configured lead's exact tab label → its name
|
||||
* ({@code fleet.leaders.<name>.tab}), matched case-insensitively
|
||||
* @param excludedWorkspaceLabels workspaces never scanned — the configured worker spaces
|
||||
* @param ttlNanos how long a scan result is reused before the next one
|
||||
* @param clock nanosecond time source ({@code System::nanoTime} in production)
|
||||
*/
|
||||
public LeadTabScanner(HerdrClient herdr, Map<String, String> tabToName,
|
||||
Set<String> excludedWorkspaceLabels, long ttlNanos, LongSupplier clock) {
|
||||
this.herdr = herdr;
|
||||
this.tabToName = normalize(tabToName);
|
||||
this.excludedWorkspaceLabels = excludedWorkspaceLabels == null
|
||||
? Set.of() : Set.copyOf(excludedWorkspaceLabels);
|
||||
this.ttlNanos = ttlNanos;
|
||||
this.clock = clock;
|
||||
}
|
||||
|
||||
/** Keys stripped and lower-cased once, so every lookup is a plain map hit. */
|
||||
private static Map<String, String> normalize(Map<String, String> tabToName) {
|
||||
if (tabToName == null || tabToName.isEmpty()) {
|
||||
return Map.of();
|
||||
}
|
||||
Map<String, String> out = new LinkedHashMap<>();
|
||||
tabToName.forEach((tab, name) -> {
|
||||
if (tab != null && !tab.isBlank() && name != null && !name.isBlank()) {
|
||||
out.put(tab.strip().toLowerCase(Locale.ROOT), name);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* The current {@code terminal_id → lead name} map, rescanning when the cache has expired.
|
||||
*
|
||||
* <p>Synchronized so a burst of concurrent calls produces one scan rather than one each; a scan
|
||||
* is a handful of RPCs over a Unix socket and is rate-limited to one per TTL.
|
||||
*/
|
||||
@Override
|
||||
public synchronized Map<String, String> get() {
|
||||
long now = clock.getAsLong();
|
||||
if (everScanned && now - scannedAtNanos < ttlNanos) {
|
||||
return cached;
|
||||
}
|
||||
// Stamp before scanning, not after: a herdr that is down must cost one attempt per TTL, not
|
||||
// one per request.
|
||||
scannedAtNanos = now;
|
||||
everScanned = true;
|
||||
try {
|
||||
Map<String, String> fresh = scan();
|
||||
if (!fresh.equals(cached)) {
|
||||
log.info("lead panes: {}", fresh);
|
||||
}
|
||||
cached = fresh;
|
||||
} catch (HerdrException e) {
|
||||
log.warn("lead-tab scan failed, keeping the {} lead(s) already known: {}",
|
||||
cached.size(), e.getMessage());
|
||||
}
|
||||
return cached;
|
||||
}
|
||||
|
||||
/** One full pass: labelled tabs → their panes → those panes' terminals. */
|
||||
private Map<String, String> scan() {
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
Workspace ws = Workspace.from(w);
|
||||
if (ws.workspaceId() == null || excludedWorkspaceLabels.contains(ws.label())) {
|
||||
continue;
|
||||
}
|
||||
for (JsonNode t : herdr.call("tab.list", Map.of("workspace_id", ws.workspaceId())).path("tabs")) {
|
||||
Tab tab = Tab.from(t);
|
||||
String name = leadNameOf(tab.label());
|
||||
if (name != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), name);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
if (!nameByTab.isEmpty()) {
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String name = nameByTab.get(p.path("tab_id").asText(null));
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name != null && terminal != null && !terminal.isBlank()) {
|
||||
byTerminal.put(terminal, name);
|
||||
}
|
||||
}
|
||||
}
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead name a tab label declares, or {@code null} if it names none of the configured leads.
|
||||
*
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
return tabToName.get(label.strip().toLowerCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,88 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
* identity (CB-105). Given the PID that opened an MCP connection, {@link #terminalForPid} finds
|
||||
* the agent pane whose process tree contains it, so {@code fleetd} can tell <em>which worker</em>
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
*
|
||||
* <p>CB-185 split the fleet across two herdr daemons — lead operations on one, members on the
|
||||
* other ({@code memberHerdrSocket}). A caller's pane can live on <em>either</em> daemon (a lead's
|
||||
* MCP connection resolves against the lead daemon; a member's against the member daemon), so this
|
||||
* must be able to search more than one client. {@link #PaneLocator(HerdrClient, HerdrClient)}
|
||||
* searches the lead client first, then the member client, and collapses to a single scan when the
|
||||
* two are the same object (the historical single-daemon deployment).
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final List<HerdrClient> herdrs;
|
||||
|
||||
/** Search only this client — the single-daemon deployment. */
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdrs = List.of(herdr);
|
||||
}
|
||||
|
||||
/**
|
||||
* Search {@code lead} first, then {@code member} — the two-daemon deployment (CB-185). When
|
||||
* the caller passes the same client for both (no {@code memberHerdrSocket} configured), this
|
||||
* collapses to one client and one scan, exactly {@link #PaneLocator(HerdrClient)}'s behaviour.
|
||||
*/
|
||||
public PaneLocator(HerdrClient lead, HerdrClient member) {
|
||||
this.herdrs = lead == member ? List.of(lead) : List.of(lead, member);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane on any searched daemon owns it (e.g. the caller is the
|
||||
* primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
for (HerdrClient herdr : herdrs) {
|
||||
String terminal = terminalForPid(herdr, pid);
|
||||
if (terminal != null) {
|
||||
return terminal;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private static String terminalForPid(HerdrClient herdr, long pid) {
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(herdr, paneId, pid)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private static boolean paneOwnsPid(HerdrClient herdr, String paneId, long pid) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
+4
-4
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
@@ -23,9 +23,9 @@ public record Tab(String tabId, String workspaceId, String label, int paneCount)
|
||||
}
|
||||
|
||||
/**
|
||||
* A freshly-created tab together with the placeholder shell pane herdr seeds it with.
|
||||
* The caller starts the worker into {@link #tab()} then closes {@link #rootPaneId()} so
|
||||
* only the worker pane remains.
|
||||
* A freshly-created tab together with the shell pane herdr seeds it with. Under protocol 19
|
||||
* the caller starts the worker <em>into</em> {@link #rootPaneId()} — the seed pane's shell
|
||||
* carries the worker's cwd and env from {@code tab.create}, and becomes the worker pane.
|
||||
*/
|
||||
public record Created(Tab tab, String rootPaneId) {
|
||||
/** Project a {@code tab_created} result ({@code {tab, root_pane}}). */
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
@@ -7,7 +7,7 @@ import com.fasterxml.jackson.databind.JsonNode;
|
||||
* dedicated worker space so they never split or clutter the user's real work spaces.
|
||||
*
|
||||
* @param workspaceId herdr's stable id (e.g. {@code "w4"})
|
||||
* @param label display label shown in herdr's UI (e.g. {@code "bridged-workers"})
|
||||
* @param label display label shown in herdr's UI (e.g. {@code "fleetd-workers"})
|
||||
* @param activeTabId the workspace's currently-focused tab, or {@code null}
|
||||
*/
|
||||
public record Workspace(String workspaceId, String label, String activeTabId) {
|
||||
+42
-7
@@ -1,10 +1,11 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
@@ -40,6 +41,16 @@ public final class WorkspaceControl {
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Every tab in {@code workspaceId}, in herdr's order. */
|
||||
public List<Tab> listTabs(String workspaceId) {
|
||||
JsonNode result = herdr.call("tab.list", Map.of("workspace_id", workspaceId));
|
||||
List<Tab> out = new ArrayList<>();
|
||||
for (JsonNode t : result.path("tabs")) {
|
||||
out.add(Tab.from(t));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The first workspace with this exact label, if any. */
|
||||
public Optional<Workspace> findByLabel(String label) {
|
||||
return listWorkspaces().stream()
|
||||
@@ -65,13 +76,37 @@ public final class WorkspaceControl {
|
||||
}
|
||||
|
||||
/**
|
||||
* A brand-new tab in {@code workspaceId} plus the placeholder shell pane herdr seeds
|
||||
* it with. Start the worker into the tab, then {@code pane.close} the root pane so the
|
||||
* tab holds only the worker.
|
||||
* A brand-new tab in {@code workspaceId} plus the shell pane herdr seeds it with. Under
|
||||
* protocol 19 that seed pane is where the worker <em>starts</em>: its shell carries
|
||||
* {@code cwd} and {@code env} (the worker's {@code ANTHROPIC_BASE_URL} — this is the
|
||||
* subscription-injection seam now), and {@code agent.start} launches the agent into it.
|
||||
*/
|
||||
public Tab.Created createTab(String workspaceId) {
|
||||
JsonNode result = herdr.call("tab.create", Map.of("workspace_id", workspaceId));
|
||||
return Tab.Created.from(result);
|
||||
public Tab.Created createTab(String workspaceId, String cwd, Map<String, String> env) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("workspace_id", workspaceId);
|
||||
if (cwd != null && !cwd.isBlank()) {
|
||||
params.put("cwd", cwd);
|
||||
}
|
||||
if (env != null && !env.isEmpty()) {
|
||||
params.put("env", env);
|
||||
}
|
||||
return Tab.Created.from(herdr.call("tab.create", params));
|
||||
}
|
||||
|
||||
/**
|
||||
* Split the currently-focused tab and return the new pane's id — the legacy pane placement's
|
||||
* seed pane, carrying {@code cwd} and {@code env} exactly as {@link #createTab}'s does.
|
||||
*/
|
||||
public String splitPane(String cwd, Map<String, String> env) {
|
||||
Map<String, Object> params = new LinkedHashMap<>();
|
||||
params.put("direction", "right");
|
||||
if (cwd != null && !cwd.isBlank()) {
|
||||
params.put("cwd", cwd);
|
||||
}
|
||||
if (env != null && !env.isEmpty()) {
|
||||
params.put("env", env);
|
||||
}
|
||||
return herdr.call("pane.split", params).path("pane").path("pane_id").asText(null);
|
||||
}
|
||||
|
||||
/** Give a worker's tab a human label in the tab bar. */
|
||||
@@ -0,0 +1,456 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.time.Duration;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* The CB-106 completion fallback: bridges the {@link Injector}'s turn-completion signal to the
|
||||
* {@link Rendezvous} so a blocking {@code fleet_send} resolves even when the worker finishes its
|
||||
* task without ever calling {@code fleet_reply} — the common case for a real delegated coding task.
|
||||
*
|
||||
* <p>On a confirmed {@code working → idle} boundary it scrapes the worker's recent transcript and
|
||||
* resolves the awaiting send with that tail (a {@link Rendezvous.Kind#COMPLETION} resolution, so the
|
||||
* caller can tell a scrape from a structured reply). It scrapes only when a send is actually waiting
|
||||
* — a fleet worker's own turns, or a send that already timed out, cost no herdr traffic. An explicit
|
||||
* {@code fleet_reply} that raced in first wins; {@link Rendezvous#resolveCompletion} is then a no-op.
|
||||
*
|
||||
* <p>It also handles the CB-109 stall signal ({@link #onTurnFailed}): a worker that ran a turn then
|
||||
* wedged in an {@code unknown} state resolves the send as a failure (with the error screen as
|
||||
* context) rather than leaving it to time out.
|
||||
*
|
||||
* <p>The scrape is cleaned to the last {@code ⏺} assistant block (stripping TUI chrome) and guarded
|
||||
* against misattribution (CB-115): the pane content is baselined on delivery ({@link #onDelivered}),
|
||||
* and a completion whose scrape is unchanged from that baseline — the previous turn's wind-down
|
||||
* sampled as this turn's boundary on a rapid back-to-back send — is suppressed rather than resolving
|
||||
* the send with a stale answer.
|
||||
*
|
||||
* <p><strong>Waiter-specific resolution (CB-116).</strong> On delivery we also capture the exact
|
||||
* {@link Rendezvous} waiter this turn belongs to, and the completion/failure fallbacks resolve
|
||||
* <em>that</em> waiter — never "whatever send is waiting now". A completion fallback runs on a virtual
|
||||
* thread and can land after the worker's {@code fleet_reply} already resolved the turn and the
|
||||
* <em>next</em> send opened its own waiter on the same session; resolving the current waiter would
|
||||
* then deliver turn N's stale scrape as turn N+1's answer. Targeting the captured waiter makes a late
|
||||
* completion a harmless no-op (its waiter is already done) instead of a cross-turn stale reply.
|
||||
*
|
||||
* <p>Wired as the {@link Injector}'s {@link TurnListener}; the handlers hand off to a virtual thread
|
||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||
*/
|
||||
public final class CompletionResolver implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source for the completion scrape. {@code recent} returns the tail of
|
||||
* the transcript (the worker's last output), which is what a delegator wants when the worker
|
||||
* didn't structure a reply.
|
||||
*/
|
||||
static final String SCRAPE_SOURCE = "recent";
|
||||
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
/**
|
||||
* fleetd#164: the floor below which a {@code BUSY -> DONE} transition cannot be real work. A
|
||||
* backend that rejects a turn outright (e.g. an HTTP 400 from the model, before the worker read
|
||||
* a single file or produced a token) drives the exact same confirmed {@code working -> idle}
|
||||
* transition a genuine completion does — just in about a second instead of the many seconds a
|
||||
* real turn costs. {@link #onTurnComplete} cannot tell those two cases apart from the transition
|
||||
* alone, so a turn that settles inside this floor is treated as a crash signature and resolved
|
||||
* as a failure, never as a (possibly empty) success.
|
||||
*/
|
||||
public static final long MIN_TURN_NANOS = Duration.ofSeconds(2).toNanos();
|
||||
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call fleet_reply.]";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
private final LongSupplier nowNanos;
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
* delivering send opened, plus the assistant block present when it was delivered.
|
||||
*
|
||||
* <p>The {@code waiter} is what makes a late fallback safe (CB-116): we resolve it, not "whoever
|
||||
* is waiting now", so a completion that fires after the next send has opened its own waiter is a
|
||||
* no-op rather than a cross-turn stale reply. The {@code baseline} is the CB-115 staleness
|
||||
* reference: a completion scrape equal to it means the worker produced no new output (the previous
|
||||
* turn's wind-down sampled as this boundary), so it is suppressed. Overwritten on each delivery;
|
||||
* cleared when the turn resolves. Package-private so tests can capture and replay a specific turn.
|
||||
*
|
||||
* <p>{@code deliveredAtNanos} (fleetd#164) is the {@link #nowNanos} reading taken at delivery —
|
||||
* the other half of the {@link #MIN_TURN_NANOS} floor check, compared against a fresh reading at
|
||||
* resolution time.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos) {
|
||||
|
||||
/**
|
||||
* Convenience for tests exercising scrape/suppression logic that don't care about turn
|
||||
* timing: back-dates the delivery far enough that {@link #MIN_TURN_NANOS} can never fire.
|
||||
* Not used by production code — {@link #captureBaseline} always records a real reading.
|
||||
*/
|
||||
InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
this(waiter, baseline, Long.MIN_VALUE / 2);
|
||||
}
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* @param exhaustedPatterns CB-578 stage A: per-target lookup for a profile's configured
|
||||
* usage-limit refusal pattern. Required — there is deliberately no
|
||||
* defaulting overload; a caller that does not want the classification
|
||||
* must pass an explicit inert value ({@link ExhaustedPatternLookup#none()}).
|
||||
* @param exhaustionSink CB-578 stage B: notified when a {@code BACKEND_EXHAUSTED}
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, System::nanoTime);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test constructor with an injectable clock (fleetd#164), matching the {@code LongSupplier}
|
||||
* pattern {@link dev.ltms.fleet.session.SessionManager} and {@link dev.ltms.fleet.msg.MessageService}
|
||||
* already use: lets a test place a turn's delivery and its resolution at an exact, controllable
|
||||
* distance apart around the {@link #MIN_TURN_NANOS} floor, without a real sleep. Public (rather
|
||||
* than package-private like those two) because callers that wire a full {@code MessageService}
|
||||
* fixture — e.g. {@code MessageServiceTest} — construct this resolver directly from another
|
||||
* package.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target, token);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target, TurnToken token) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
}
|
||||
String baseline;
|
||||
try {
|
||||
// Clip to the same cap resolve() applies to the tail (line ~134): the CB-115 misattribution
|
||||
// guard compares baseline.equals(tail), so both sides must be the same capped representation.
|
||||
// An unclipped baseline vs a clipped tail would never match for a >MAX_SCRAPE_CHARS block,
|
||||
// defeating the guard and letting a stale completion resolve the send.
|
||||
baseline = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline, nowNanos.getAsLong()));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
InFlight inFlight(String target) {
|
||||
return inFlight.get(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
// Read the in-flight turn on the poller thread — before any next-turn delivery can overwrite
|
||||
// it — then off-load the scrape (a herdr round-trip we must not block polling on) to a vthread.
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("completion-" + target).start(() -> resolve(target, turn));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the completed turn before adapter housekeeping can erase its rendered output. This is
|
||||
* intentionally synchronous and used only when a post-turn context reset is enabled; the normal
|
||||
* path remains off-loaded so polling is not blocked by a scrape.
|
||||
*/
|
||||
public void resolveBeforePostAction(String target) {
|
||||
resolve(target, inFlight.get(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, null));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, reason));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
void resolve(String target, InFlight turn) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = turn == null ? null : turn.waiter();
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
// Nobody is blocked on THIS turn (it had no send, or its fleet_reply already won). Skip
|
||||
// the scrape; resolving the current waiter here would be the CB-116 cross-turn stale reply.
|
||||
inFlight.remove(target, turn);
|
||||
return;
|
||||
}
|
||||
// fleetd#164: a BUSY -> DONE transition inside the floor cannot be real work — it's a crash
|
||||
// signature (e.g. a backend HTTP 400 before the worker did anything), not a fast answer. Fail
|
||||
// it before spending a scrape on the ordinary path; the reason still carries whatever is on
|
||||
// screen, since that is usually the backend's own error.
|
||||
long elapsedNanos = nowNanos.getAsLong() - turn.deliveredAtNanos();
|
||||
if (elapsedNanos < MIN_TURN_NANOS) {
|
||||
fail(target, turn, tooFastReason(target, elapsedNanos));
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("completion scrape for {} failed: {}", target, e.getMessage());
|
||||
tail = "";
|
||||
scrapeFailed = true;
|
||||
}
|
||||
// fleetd#164: a scrape nobody could read, and a scrape that read cleanly but produced nothing,
|
||||
// both used to resolve the send as a SUCCESS carrying "" — indistinguishable from a worker that
|
||||
// genuinely finished with nothing to say. That is the defect: fail loudly instead, naming the
|
||||
// member, so a caller (including a lead deciding whether to delegate again) can tell a lost
|
||||
// turn from a real empty answer.
|
||||
if (scrapeFailed || tail.isEmpty()) {
|
||||
fail(target, turn, emptyScrapeReason(target, scrapeFailed));
|
||||
return;
|
||||
}
|
||||
// Misattribution guard (CB-115): if the scrape is byte-identical to the pane content at
|
||||
// delivery, this turn produced no new output — the boundary belongs to the previous turn's
|
||||
// wind-down (common on rapid back-to-back sends). Suppress rather than resolve the send with
|
||||
// a stale answer; the real fleet_reply (or a later genuine completion) resolves it instead.
|
||||
String baseline = turn.baseline();
|
||||
if (baseline != null && baseline.equals(tail)) {
|
||||
log.debug("suppressing misattributed completion for {} (no output change since delivery)",
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
}
|
||||
// CB-578 stage A: a turn that ended with no fleet_reply AND whose scrape matches the
|
||||
// backend's configured usage-limit pattern is a refusal, not an answer. Classify it as
|
||||
// BACKEND_EXHAUSTED rather than handing the caller a scrape that reads like a real reply.
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(assistantBlock, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED (no fleet_reply; scrape "
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
log.warn("completion scrape for {} clipped from {} chars to the {} char cap; "
|
||||
+ "member did not call fleet_reply, so the pane tail is partial",
|
||||
target, originalLength, MAX_SCRAPE_CHARS);
|
||||
}
|
||||
log.debug("resolved send to {} via turn-completion fallback ({} chars scraped)",
|
||||
target, tail.length());
|
||||
}
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
}
|
||||
|
||||
/** Synchronous fail with an optional reason supplied by a dropped worker queue. */
|
||||
void fail(String target, InFlight turn, String explicitReason) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
CompletableFuture<Rendezvous.Resolution> waiter =
|
||||
turn != null ? turn.waiter() : rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason = explicitReason;
|
||||
if (reason == null || reason.isBlank()) {
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164: the failure reason for a turn that settled inside {@link #MIN_TURN_NANOS} — names
|
||||
* the member and both timings, and appends whatever the pane shows (usually the backend's own
|
||||
* error) so the caller sees the cause, not just "it failed".
|
||||
*/
|
||||
private String tooFastReason(String target, long elapsedNanos) {
|
||||
String scrape;
|
||||
try {
|
||||
scrape = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
scrape = "";
|
||||
}
|
||||
String reason = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms) — too fast to be real work, most "
|
||||
+ "likely a backend error before any work started",
|
||||
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
|
||||
return scrape.isBlank() ? reason : reason + ": " + scrape;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164: the failure reason for a scrape that produced zero characters — names the member
|
||||
* and says plainly that the turn produced nothing, so a caller (a lead deciding whether to
|
||||
* delegate again included) never mistakes a lost turn for a genuinely empty reply.
|
||||
*/
|
||||
private static String emptyScrapeReason(String target, boolean scrapeFailed) {
|
||||
return "member " + target + " turn completed with an empty scrape (0 chars) — "
|
||||
+ (scrapeFailed ? "its pane could not be read; " : "")
|
||||
+ "treating as a lost turn, not a real answer";
|
||||
}
|
||||
|
||||
/**
|
||||
* The first line of {@code text} matching {@code pattern}, stripped — the CB-578 stage A
|
||||
* evidence carried in a {@code BACKEND_EXHAUSTED} reason so the operator sees the real refusal
|
||||
* text, never a generic label. {@code null} if no line matches.
|
||||
*/
|
||||
static String firstMatchingLine(String text, Pattern pattern) {
|
||||
if (text == null || text.isEmpty()) return null;
|
||||
for (String line : text.split("\n", -1)) {
|
||||
if (pattern.matcher(line).find()) {
|
||||
return line.strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* the way {@link dev.ltms.fleet.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
return unconfigured.isEmpty()
|
||||
? "full (all profiles configured: " + sorted(allProfiles) + ")"
|
||||
: "partial (configured: " + sorted(configuredProfiles) + "; not configured: " + sorted(unconfigured) + ")";
|
||||
}
|
||||
|
||||
private static List<String> sorted(Set<String> names) {
|
||||
return names.stream().sorted().toList();
|
||||
}
|
||||
|
||||
private static String clip(String s) {
|
||||
if (s == null) return "";
|
||||
String trimmed = s.strip();
|
||||
return trimmed.length() <= MAX_SCRAPE_CHARS
|
||||
? trimmed
|
||||
: trimmed.substring(trimmed.length() - MAX_SCRAPE_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the last assistant message from a raw Claude Code pane scrape (CB-115). Claude Code
|
||||
* prefixes each assistant turn with {@code ⏺}; the delegator wants that answer, not the TUI
|
||||
* chrome around it. Take everything from the final {@code ⏺} onward and stop at the <em>first</em>
|
||||
* hard interface boundary below it — the spinner/status line, input box, {@code ❯} prompt (which
|
||||
* may echo the <em>next</em> turn's text), footer, or tips/warnings. Stopping at the first
|
||||
* boundary (rather than trimming only trailing chrome) is what keeps a following turn's echoed
|
||||
* prompt out of this reply. Blank lines are not boundaries, so a multi-paragraph answer survives;
|
||||
* trailing blanks are trimmed at the end. With no {@code ⏺} marker (an unusual render) the whole
|
||||
* text is scanned the same way, so we never lose the reply.
|
||||
*
|
||||
* <p>Package-private and pure so it is unit-testable without herdr.
|
||||
*/
|
||||
static String lastAssistantBlock(String raw) {
|
||||
if (raw == null || raw.isBlank()) return "";
|
||||
int marker = raw.lastIndexOf('⏺');
|
||||
String block = marker >= 0 ? raw.substring(marker + 1) : raw;
|
||||
StringBuilder out = new StringBuilder();
|
||||
int kept = 0;
|
||||
for (String line : block.split("\n", -1)) {
|
||||
if (isBoundary(line)) break; // first TUI boundary ends the assistant message
|
||||
if (kept++ > 0) out.append('\n');
|
||||
out.append(line);
|
||||
}
|
||||
return out.toString().strip();
|
||||
}
|
||||
|
||||
/**
|
||||
* A hard TUI boundary line that marks the end of an assistant message and the start of interface
|
||||
* chrome (input box, prompt, spinner, footer, tips/warnings). Blank lines are <em>not</em>
|
||||
* boundaries — an answer may contain them — so they are kept and trimmed only if trailing.
|
||||
*/
|
||||
private static boolean isBoundary(String line) {
|
||||
String t = line.strip();
|
||||
if (t.isEmpty()) return false;
|
||||
// A horizontal rule / all box-drawing separators (e.g. "──────").
|
||||
if (t.chars().allMatch(c -> c == '─' || c == '—' || c == '━' || c == '═' || c == '-')) {
|
||||
return true;
|
||||
}
|
||||
String lower = t.toLowerCase();
|
||||
return t.startsWith("╭") || t.startsWith("│") || t.startsWith("╰") || t.startsWith("┌")
|
||||
|| t.startsWith("└") || t.startsWith("❯") || t.startsWith("⏵")
|
||||
|| t.startsWith("⎿") || t.startsWith("⚠")
|
||||
// Status/spinner lines Claude Code renders below a settled or in-flight turn,
|
||||
// e.g. "✻ Baked for 21s", "✶ Forming…".
|
||||
|| t.startsWith("✻") || t.startsWith("✳") || t.startsWith("✽") || t.startsWith("·")
|
||||
|| t.startsWith("●") || t.startsWith("◐") || t.startsWith("✢") || t.startsWith("✶")
|
||||
|| lower.contains("auto mode") || lower.contains("for shortcuts")
|
||||
|| lower.contains("esc to interrupt") || lower.contains("bypass permissions");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,27 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured usage-limit refusal pattern (CB-578 stage A): how
|
||||
* {@link CompletionResolver} tells a backend that refused on a subscription usage limit — the
|
||||
* worker's pane stays healthy, but the account is exhausted — apart from a genuine completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its refusal differently, so a hardcoded sentence would only ever match one of them.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustedPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Inert lookup — no profile has a pattern configured, so the classification never fires and
|
||||
* the completion fallback behaves exactly as before CB-578 stage A. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload.
|
||||
*/
|
||||
static ExhaustedPatternLookup none() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a {@code BACKEND_EXHAUSTED}
|
||||
* classification to a waiting send (CB-578 stage B) — never on a race that lost (see
|
||||
* {@link CompletionResolver#resolve}, which only calls this after
|
||||
* {@code Rendezvous.resolveExhausted} returns {@code true}).
|
||||
*
|
||||
* <p>{@link CompletionResolver} knows only {@code target} (a herdr terminal id); it has no notion of
|
||||
* profiles or credentials, so mapping {@code target} to whatever should be quarantined is entirely
|
||||
* the sink's job — see {@code Fleetd.main}'s wiring, which resolves target → session → profile →
|
||||
* {@code effectiveCredentialId()} and calls {@code BackendQuarantine.quarantine} on it.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustionSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
*/
|
||||
void onExhausted(String target, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on exhaustion. The explicit stand-in a caller (or a test not
|
||||
* exercising this feature) passes instead of a defaulting overload, exactly like
|
||||
* {@link ExhaustedPatternLookup#none()}.
|
||||
*/
|
||||
static ExhaustionSink none() {
|
||||
return (target, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,434 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayDeque;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Deque;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* The status-gated injector (CB-103): the single writer that delivers a message into a
|
||||
* worker only when it is safe — {@code idle} or {@code blocked}, never mid-turn.
|
||||
*
|
||||
* <p>Per target it holds a FIFO queue and delivers <strong>at most one message per turn</strong>:
|
||||
* after a send it waits for the worker to pick the message up (go {@code working}) before
|
||||
* delivering the next, so two rapid deliveries never interleave into one turn. Each
|
||||
* status-check-then-send for a target is serialized on the target's monitor, closing the
|
||||
* TOCTOU window between "is it idle?" and "send" — one writer per worker.
|
||||
*
|
||||
* <p><strong>Polling caveat.</strong> This is driven by {@link #onStatus} sampling (a
|
||||
* {@code StatusPoller}), not by reliable status <em>edges</em>. A turn can begin and end
|
||||
* entirely between two polls, so the {@code working} pickup may never be sampled. To avoid
|
||||
* wedging a queue forever, an awaited pickup is released after {@link #PICKUP_GRACE_POLLS}
|
||||
* consecutive injectable samples (the worker has plainly moved on). Only {@code working} — not
|
||||
* a transient {@code unknown} — counts as a real pickup, so a detection glitch can't prematurely
|
||||
* release the latch. Perfectly reliable turn boundaries require a herdr {@code events.subscribe}
|
||||
* stream; that is the intended upgrade and would replace only the sampling, not this queue.
|
||||
*
|
||||
* <p><strong>Turn completion (CB-106).</strong> Beyond delivery, the injector reports when a
|
||||
* delegated turn <em>finishes</em>: after a delivery is picked up (a real {@code working} sample),
|
||||
* the next injectable sample is a confirmed {@code working → idle} boundary and fires
|
||||
* {@link TurnListener#onTurnComplete}. Completion is only ever synthesized from a <em>confirmed</em>
|
||||
* turn — the pickup-grace path (a turn too fast to sample) unwedges the queue but does not fire
|
||||
* completion, since without a sampled {@code working} there is no trustworthy "the worker just
|
||||
* finished the task" signal to act on.
|
||||
*/
|
||||
public final class Injector {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Injector.class);
|
||||
|
||||
/**
|
||||
* How many consecutive injectable samples (with no {@code working} in between) after a send
|
||||
* before we assume the turn completed unobserved and release the pickup latch. At the
|
||||
* default 250ms poll interval this is a ~2s grace — far longer than a worker takes to start
|
||||
* a turn, so it only fires on a genuinely missed pickup edge.
|
||||
*/
|
||||
private static final int PICKUP_GRACE_POLLS = 8;
|
||||
|
||||
/**
|
||||
* How many consecutive {@code unknown} samples while a delegation is outstanding before we
|
||||
* declare it stalled and fire {@link TurnListener#onTurnFailed} (CB-109). A worker wedged in a
|
||||
* state herdr can't classify (e.g. an API-error screen) stays {@code unknown} indefinitely and
|
||||
* would otherwise never resolve; any {@code working}/{@code idle} sample resets the streak, so a
|
||||
* transient detection glitch cannot trip it. At the 250ms poll interval this is ~30s — far longer
|
||||
* than any real detection blip, and still vastly better than the async send's timeout.
|
||||
*/
|
||||
private static final int TURN_STALL_GRACE_POLLS = 120;
|
||||
|
||||
/**
|
||||
* How many consecutive injectable samples a queued-but-undelivered message may wait on the
|
||||
* {@link #ready} gate before we give up and fail it (CB-114). The gate holds a message out of a
|
||||
* worker's boot window (herdr reports {@code idle} while its Claude is still starting), but a
|
||||
* worker whose Claude crashes during boot — or never connects the bridge MCP — stays "idle and
|
||||
* not ready" forever: {@link #ready} never accepts it, the message is never delivered, and the
|
||||
* target would be polled indefinitely with its caller's future never completing. After this
|
||||
* grace the queued messages are failed and the target released. At the 250ms poll interval this
|
||||
* is ~60s — deliberately longer than {@link #TURN_STALL_GRACE_POLLS}, since a first boot (spawn
|
||||
* + model load + MCP connect) legitimately takes longer than an in-turn detection blip.
|
||||
*/
|
||||
private static final int READINESS_GRACE_POLLS = 240;
|
||||
|
||||
/**
|
||||
* The single source for the injector poll cadence — how often the {@link StatusPoller} drives
|
||||
* {@link #onStatus} at. {@code Fleetd} passes this to every {@link StatusPoller} it constructs,
|
||||
* and this class reads it to state the readiness grace in seconds on the CB-562 expiry log
|
||||
* instead of hardcoding "60s". One constant, so a cadence change cannot silently desync a log
|
||||
* that claims a grace duration.
|
||||
*/
|
||||
public static final long POLL_INTERVAL_MILLIS = 250;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final HerdrRouter router;
|
||||
private final TurnListener turnListener;
|
||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
||||
private final ConcurrentHashMap<String, Target> targets = new ConcurrentHashMap<>();
|
||||
|
||||
/** Delivery only; completion signalling is a no-op and every target is treated as available. */
|
||||
public Injector(AgentControl agents) {
|
||||
this(agents, TurnListener.NOOP);
|
||||
}
|
||||
|
||||
/** Delivery plus turn-completion signalling (CB-106); every target is treated as available. */
|
||||
public Injector(AgentControl agents, TurnListener turnListener) {
|
||||
this(agents, turnListener, _ -> true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Delivery, completion signalling (CB-106), and a readiness gate (CB-113): a message is delivered
|
||||
* only when {@code ready} accepts the target — i.e. the worker's Claude has connected the bridge
|
||||
* MCP. This holds the first delivery out of the worker's boot window, where herdr already reports
|
||||
* {@code idle} but the TUI would drop an injected paste.
|
||||
*/
|
||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready) {
|
||||
this(agents, turnListener, ready, _ -> {
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Delivery, completion signalling (CB-106), a readiness gate (CB-113), and readiness cleanup
|
||||
* (CB-114): {@code forget} is invoked with a target when its worker is gone — dropped
|
||||
* (pane crash) or timed out on the readiness gate — so its stale presence/readiness is cleared
|
||||
* and does not linger past the worker's life.
|
||||
*/
|
||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this.agents = agents;
|
||||
this.router = null;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
}
|
||||
|
||||
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this.agents = null;
|
||||
this.router = router;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
}
|
||||
|
||||
private AgentControl agentsFor(String target) {
|
||||
return router != null ? router.agentsFor(target) : agents;
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
private static final class Target {
|
||||
final Deque<Pending> queue = new ArrayDeque<>();
|
||||
boolean awaitingPickup; // sent a message, waiting for the worker to pick it up
|
||||
int injectableSincePickup; // consecutive injectable samples while awaitingPickup
|
||||
boolean awaitingCompletion; // a delivered message's turn is not yet known-complete
|
||||
boolean turnObserved; // saw a real `working` sample since that delivery (turn ran)
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
boolean postTurnObserved;
|
||||
int injectableSincePostTurnPickup;
|
||||
|
||||
synchronized void add(Pending p) {
|
||||
queue.add(p);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Queue {@code text} for delivery to {@code target} (a herdr {@code terminal_id}). Returns
|
||||
* immediately with a future that completes when the message is actually sent — the worker
|
||||
* may be mid-turn, in which case delivery waits for the next injectable window.
|
||||
*
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, token, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
return t;
|
||||
});
|
||||
return delivered;
|
||||
}
|
||||
|
||||
/**
|
||||
* Feed a fresh status observation for {@code target}. Delivers the head of the queue iff the
|
||||
* worker is injectable and no earlier message is still awaiting pickup. Serialized per target
|
||||
* so the check and the send cannot race another delivery to the same worker; the delivered
|
||||
* future is completed <em>after</em> the monitor is released so a caller's continuation never
|
||||
* runs on the poller thread while it holds the lock.
|
||||
*/
|
||||
public void onStatus(String target, AgentStatus status) {
|
||||
Target t = targets.get(target);
|
||||
if (t == null) return;
|
||||
|
||||
Pending sent = null;
|
||||
RuntimeException sendError = null;
|
||||
boolean turnCompleted = false;
|
||||
boolean turnFailed = false;
|
||||
boolean resubmit = false;
|
||||
boolean startPostTurn = false;
|
||||
List<Pending> notReady = null; // queued messages failed because the worker never became ready
|
||||
synchronized (t) {
|
||||
if (status == AgentStatus.WORKING) {
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
t.postTurnObserved = true;
|
||||
}
|
||||
// Definitive pickup: the worker is busy on our last message, and (if a delivery is
|
||||
// outstanding) a real turn is now confirmed to be running.
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.unknownSinceTurn = 0;
|
||||
t.notReadySincePoll = 0;
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
t.unknownSinceTurn = 0;
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
if (++t.injectableSincePostTurnPickup >= PICKUP_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
} else {
|
||||
resubmit = true;
|
||||
}
|
||||
} else if (t.postTurnObserved) {
|
||||
t.postTurnObserved = false;
|
||||
}
|
||||
if (t.awaitingPickup) {
|
||||
if (++t.injectableSincePickup >= PICKUP_GRACE_POLLS) {
|
||||
// Pickup edge was never sampled (turn faster than the poll, or status lag).
|
||||
// Release the latch rather than wedge — and give up on synthesizing a
|
||||
// completion for this message, since without a confirmed `working` we cannot
|
||||
// trust that a task-processing turn actually ran.
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
} else {
|
||||
// Delivered but still idle → the worker hasn't picked it up; the submit
|
||||
// keystroke likely raced the paste (esp. right as the TUI became ready).
|
||||
// Re-nudge Enter (CB-113) until the worker starts (WORKING) or the grace ends.
|
||||
resubmit = true;
|
||||
}
|
||||
}
|
||||
if (!t.awaitingPickup) {
|
||||
// A confirmed turn (a `working` sample was seen) that has now returned to idle is
|
||||
// a trustworthy `working → idle` completion boundary.
|
||||
if (t.awaitingCompletion && t.turnObserved) {
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
turnCompleted = true;
|
||||
if (turnListener.hasPostTurnAction(target)) {
|
||||
t.postTurnPending = true;
|
||||
startPostTurn = true;
|
||||
}
|
||||
}
|
||||
// Deliver the next queued message only once the prior turn is fully settled, so a
|
||||
// completion is never confused with the pickup of the following message — and only
|
||||
// once the worker is available (CB-113), so we never paste into its boot window.
|
||||
if (!t.awaitingCompletion && !t.postTurnPending
|
||||
&& !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
Pending p = t.queue.peek();
|
||||
if (p != null && ready.test(target)) {
|
||||
t.notReadySincePoll = 0;
|
||||
try {
|
||||
agentsFor(target).send(target, p.text());
|
||||
t.queue.poll();
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
t.injectableSincePickup = 0;
|
||||
sent = p;
|
||||
} catch (RuntimeException e) {
|
||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
||||
// rather than blocking the queue behind it.
|
||||
t.queue.poll();
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
} else if (p != null && ++t.notReadySincePoll >= READINESS_GRACE_POLLS) {
|
||||
// The worker has been idle-but-not-ready for the whole grace: its Claude
|
||||
// never connected the bridge MCP (crashed during boot, or wedged on a
|
||||
// startup prompt). The readiness gate would hold this message forever, so
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
||||
+ "became deliverable, so failing {} queued message(s) that never "
|
||||
+ "reached its pane",
|
||||
target, READINESS_GRACE_POLLS,
|
||||
READINESS_GRACE_POLLS * POLL_INTERVAL_MILLIS / 1000, notReady.size());
|
||||
t.queue.clear();
|
||||
t.notReadySincePoll = 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// UNKNOWN (or any other non-injectable, non-working): not a safe window nor a
|
||||
// reliable pickup signal, so we never deliver or release the pickup latch here. But
|
||||
// an outstanding delegation whose worker has gone unresponsive — stuck in a state
|
||||
// herdr can't classify (CB-109) — will never yield a working→idle boundary. After a
|
||||
// sustained streak, declare it failed so the awaiting send resolves rather than
|
||||
// riding out the async timeout. (This also frees a delivery that wedged before it
|
||||
// was ever picked up, which the injectable-only pickup grace could never release.)
|
||||
if (t.awaitingCompletion && ++t.unknownSinceTurn >= TURN_STALL_GRACE_POLLS) {
|
||||
t.awaitingPickup = false;
|
||||
t.awaitingCompletion = false;
|
||||
t.turnObserved = false;
|
||||
t.unknownSinceTurn = 0;
|
||||
turnFailed = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
// completion awaited), so the map cannot grow without bound across short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
|
||||
// Fire listeners / herdr calls after releasing the monitor so nothing runs on the poller
|
||||
// thread while it holds the target lock.
|
||||
if (resubmit) {
|
||||
try {
|
||||
agentsFor(target).submit(target); // nudge a raced Enter so the pending paste submits
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
||||
}
|
||||
}
|
||||
if (notReady != null) {
|
||||
// Worker never became available: forget its (never-set) readiness, unblock every queued
|
||||
// caller, and route the awaiting send through the same failure path as a stalled turn so
|
||||
// a blocking or async waiter resolves WORKER_FAILED rather than riding out the timeout.
|
||||
forget.accept(target);
|
||||
RuntimeException cause = new IllegalStateException(
|
||||
target + " never became available (no bridge MCP connection within the boot window)");
|
||||
for (Pending p : notReady) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
if (turnCompleted) {
|
||||
if (startPostTurn) {
|
||||
boolean started = turnListener.onTurnCompleteWithPostAction(target);
|
||||
synchronized (t) {
|
||||
t.postTurnPending = false;
|
||||
if (started) {
|
||||
t.awaitingPostTurnPickup = true;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
}
|
||||
if (t.queue.isEmpty() && !t.awaitingPostTurnPickup) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
turnListener.onTurnComplete(target);
|
||||
}
|
||||
}
|
||||
if (turnFailed) {
|
||||
turnListener.onTurnFailed(target);
|
||||
}
|
||||
if (sent != null) {
|
||||
if (sendError != null) {
|
||||
log.warn("inject to {} failed, dropped message: {}", target, sendError.getMessage());
|
||||
sent.delivered().completeExceptionally(sendError);
|
||||
} else {
|
||||
// Baseline the pane's pre-turn content so a misattributed completion (no new output)
|
||||
// can't resolve this send with the previous turn's stale answer (CB-115).
|
||||
turnListener.onDelivered(target, sent.token());
|
||||
sent.delivered().complete(null);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Targets the poller must keep sampling: those with a queued message, an awaited pickup, or an
|
||||
* awaited turn completion (so the {@code working → idle} boundary is observed).
|
||||
*/
|
||||
public Set<String> activeTargets() {
|
||||
return targets.entrySet().stream()
|
||||
.filter(e -> {
|
||||
synchronized (e.getValue()) {
|
||||
Target t = e.getValue();
|
||||
return !t.queue.isEmpty() || t.awaitingPickup || t.awaitingCompletion
|
||||
|| t.postTurnPending || t.awaitingPostTurnPickup || t.postTurnObserved;
|
||||
}
|
||||
})
|
||||
.map(java.util.Map.Entry::getKey)
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Forget a target whose worker is gone, failing every still-queued message so awaiting callers
|
||||
* unblock instead of hanging forever. If a message had already been <em>delivered</em> but its
|
||||
* turn was not yet resolved (CB-110 — the worker vanished mid-turn, e.g. its pane crashed), fire
|
||||
* {@link TurnListener#onTurnFailed} for it: a delivered message is no longer in the queue, so
|
||||
* failing queued waiters alone would leave that send's rendezvous hanging until the async
|
||||
* timeout. Futures and listeners are completed after the monitor is released.
|
||||
*/
|
||||
public void drop(String target, Throwable cause) {
|
||||
Target t = targets.remove(target);
|
||||
if (t == null) return;
|
||||
List<Pending> pending;
|
||||
boolean hadDeliveredTurn;
|
||||
synchronized (t) {
|
||||
pending = new ArrayList<>(t.queue);
|
||||
t.queue.clear();
|
||||
hadDeliveredTurn = t.awaitingCompletion;
|
||||
t.awaitingCompletion = false;
|
||||
t.awaitingPickup = false;
|
||||
t.postTurnPending = false;
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.postTurnObserved = false;
|
||||
}
|
||||
log.warn("{} is gone, dropping its queue: {} message(s) failed{}; cause: {}", target,
|
||||
pending.size(),
|
||||
hadDeliveredTurn ? " (including one turn already in flight whose completion was never confirmed)" : "",
|
||||
cause.getMessage());
|
||||
forget.accept(target); // the worker is gone — clear its readiness/presence too (CB-114)
|
||||
for (Pending p : pending) {
|
||||
p.delivered().completeExceptionally(cause);
|
||||
}
|
||||
// A queued send has no in-flight record, while a delivered turn does. CompletionResolver
|
||||
// handles both forms and resolves its waiter at most once.
|
||||
turnListener.onTurnFailed(target, cause.getMessage());
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user