Compare commits
407 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| d678783af7 | |||
| 799014e99d | |||
| 69e09b10fa | |||
| b9d09e044e | |||
| fd8650cda4 | |||
| 11050e24ed | |||
| 769f282408 | |||
| 6f71f40047 | |||
| fb36c5238f | |||
| cc9cdc938b | |||
| 5fede82468 | |||
| 1515025804 | |||
| 7754f53662 | |||
| a507f7b31b | |||
| 71c322f104 | |||
| fde2c15627 | |||
| 2830735644 | |||
| 4e3ac91a22 | |||
| 29d3f0b41f | |||
| cbc732444f | |||
| 6f828b8c38 | |||
| 145a8c8862 | |||
| 7057291739 | |||
| 2302b3bc11 | |||
| 3a004dc1b3 | |||
| a9a3c12232 | |||
| 94ec77a1bc | |||
| 49f285cfda | |||
| 5a12ae7930 | |||
| 127e6832a9 | |||
| 6442a583ae | |||
| 24b96d29ae | |||
| 4721771052 | |||
| 2f71a30bd7 | |||
| 22cdebbdbe | |||
| 154971c2b8 | |||
| fef287c346 | |||
| a37acd5ee3 | |||
| d6ef0c8013 | |||
| dfb70871b4 | |||
| 4ee7b16929 | |||
| 8557289dc0 | |||
| dd2efd8541 | |||
| 5af786d135 | |||
| 380eb63277 | |||
| 6c61355f8f | |||
| 96c406b968 | |||
| a5d81c3f70 | |||
| 92c0f164f1 | |||
| 9e813ec179 | |||
| 395b3b5c46 | |||
| 6e9e464d62 | |||
| 105c065615 | |||
| 01492059d4 | |||
| f84824ee29 | |||
| c4d40fbc2b | |||
| 7c684e40d3 | |||
| 6938f52155 | |||
| 0df34f3220 | |||
| 457458437f | |||
| 3759c41f99 | |||
| 09159f2857 | |||
| 29cd1194c2 | |||
| 815e8f8b23 | |||
| 1e60ac0745 | |||
| 650a4c146b | |||
| 23f299e105 | |||
| dbf6fef0e9 | |||
| d292522d00 | |||
| 0241e0d3a8 | |||
| e4973eb8a4 | |||
| b6b88c5f1c | |||
| 86dddfe240 | |||
| 0d5944af63 | |||
| 4a5030a5c6 | |||
| c1c8794c48 | |||
| 65a78932c1 | |||
| f429ca1a50 | |||
| 73aab3f83e | |||
| 887aca0183 | |||
| 3fd23ecafa | |||
| f379847942 | |||
| ea12107497 | |||
| 591df91de1 | |||
| 6a814176f0 | |||
| d11d1d157c | |||
| d057d56156 | |||
| d703ce1313 | |||
| b32a30fd47 | |||
| 464dbc0930 | |||
| a8cadd9150 | |||
| 57b8c0b56d | |||
| 147f50c19e | |||
| eee4d576a2 | |||
| b4f9d7f53a | |||
| 3aca53b967 | |||
| 4aa1fae296 | |||
| 0c865032f9 | |||
| ea41bbf6b9 | |||
| 7b918c51ff | |||
| 554395b104 | |||
| 823976c1b5 | |||
| c8388a7f92 | |||
| 02e6aef98c | |||
| 6d493bc7bb | |||
| e5cb51a90e | |||
| e545c08082 | |||
| efa0deb9b2 | |||
| ca47e90c01 | |||
| fa1f49675b | |||
| 8426c3528f | |||
| 65f98ba910 | |||
| d05205d1eb | |||
| 667254df47 | |||
| 2926cd1784 | |||
| c801851c66 | |||
| de70aa38f1 | |||
| 53a533afb4 | |||
| 77ad88631b | |||
| d88017807b | |||
| 2159a5a94a | |||
| b9c2cf69f4 | |||
| 8beae50fe7 | |||
| b2a58cb966 | |||
| b8b25cf74c | |||
| f159ca7d27 | |||
| 83f2aea60f | |||
| 002329adb5 | |||
| a49671ceb9 | |||
| 76672ff016 | |||
| 9379f92c23 | |||
| 21ff63b11d | |||
| 21844b54d7 | |||
| 4769481515 | |||
| bbbb4c1eb3 | |||
| 38c248e617 | |||
| 9debc0de27 | |||
| f34361b263 | |||
| be5ba22c75 | |||
| 85c90d440a | |||
| e97502d550 | |||
| aef14ff46e | |||
| cba516bda4 | |||
| ba51e0c6cc | |||
| 086c59848e | |||
| 0c10079755 | |||
| ece2091b53 | |||
| b3f917e6f5 | |||
| fa39a5f55e | |||
| d5128a1d35 | |||
| 61097e5cf0 | |||
| ef507bcd12 | |||
| 94f50e507a | |||
| 75b15086b0 | |||
| dab9645906 | |||
| e93b5f6512 | |||
| bfabe13e8f | |||
| 7e49c6eca2 | |||
| 11cbfa79b4 | |||
| c2c2746922 | |||
| 6ed70700a0 | |||
| 0087645da4 | |||
| 19cacf5b62 | |||
| 4dd12083ab | |||
| 30e21adec7 | |||
| 719b79f892 | |||
| 66e5247b6d | |||
| 1e41bd63b4 | |||
| 4887d03d88 | |||
| 7d5434455d | |||
| f71ee4926e | |||
| 282a2fc2b8 | |||
| f04e934b94 | |||
| 3fae35c357 | |||
| 18aecbfe67 | |||
| 5d75f72473 | |||
| e028a0ae54 | |||
| 2fa673d4c0 | |||
| 9020d01b40 | |||
| 1006805027 | |||
| 27aefbf9a0 | |||
| d42c2bc204 | |||
| 3916adc372 | |||
| fa97f598dd | |||
| ea9aa4fd77 | |||
| a2b8caf6b5 | |||
| b5ddbe5757 | |||
| d223a93039 | |||
| 96d8191149 | |||
| f0e7ac73d6 | |||
| e2fe861b4d | |||
| 279d6f5fbd | |||
| 34480cebef | |||
| 9d0bf14c46 | |||
| 5a3ab5764c | |||
| 3bad9f5785 | |||
| 8308c0b68f | |||
| 3b3063eb2b | |||
| df9086263d | |||
| eb568ff451 | |||
| ac790e4cce | |||
| 6b5f3f472f | |||
| 38dec72152 | |||
| 0e8bfb74fc | |||
| 51f7b0a3ca | |||
| 21c539f22e | |||
| bd2774b5f1 | |||
| c50f5b2d61 | |||
| 01a840cc14 | |||
| c4deef08be | |||
| c796eac09c | |||
| e897e5257b | |||
| 2afa3652bb | |||
| 80092ff359 | |||
| 9d37f3aa29 | |||
| eaf89abaf6 | |||
| d895f02bc1 | |||
| 43206cac2f | |||
| 8bba3a8184 | |||
| 5cf3ca9a89 | |||
| ac474981e4 | |||
| ba04b2359b | |||
| 2e5b63f6f6 | |||
| 3437d6313d | |||
| 743377d6cd | |||
| 5952d559c7 | |||
| 205ad823b0 | |||
| d654ccb818 | |||
| a89dcc9b7e | |||
| 0d7b4fb026 | |||
| 321d8dcbb5 | |||
| ef8c97871e | |||
| 26bafe824b | |||
| 838a701109 | |||
| e5eb3534c7 | |||
| 959c83534f | |||
| 31b028e860 | |||
| 7840e9adf6 | |||
| c3672f5472 | |||
| cf54aed451 | |||
| bbf68f3e3c | |||
| 776743cbe2 | |||
| 826e0aeb2a | |||
| c935b181dd | |||
| ee932fd85b | |||
| fe2e5ede34 | |||
| c325054242 | |||
| 7662e2d0c8 | |||
| 4877992a70 | |||
| 1c051c4e47 | |||
| adb7a67880 | |||
| 723fe494e9 | |||
| 0f08b93659 | |||
| 432c1d92d1 | |||
| 60b7e67b42 | |||
| 3fbd43fe3f | |||
| 32ebf065ac | |||
| 1178b3f684 | |||
| 2a434ced2f | |||
| cc919aa2b6 | |||
| 6417b0edd9 | |||
| 39c7ce76f3 | |||
| b40f477210 | |||
| 5c56cb347f | |||
| e694deace3 | |||
| 748367b7d6 | |||
| dcf5fb3be3 | |||
| f9d2ee2a2b | |||
| 866c7f2e9a | |||
| d9168de43e | |||
| c3fa1136d4 | |||
| cabcd87b66 | |||
| bf0e09b1a2 | |||
| e1eb50ce65 | |||
| 049e7d9d54 | |||
| ff3b49cd1e | |||
| 0373b6c41b | |||
| 445a45f6e1 | |||
| 966c58a3b8 | |||
| de026b8f8a | |||
| 97f6c33a45 | |||
| 735c837604 | |||
| 457dc0330d | |||
| a1a9015217 | |||
| 5a811a3695 | |||
| 1fdaa74eb3 | |||
| 7d4a4339c2 | |||
| 847e8bd3fa | |||
| f9fb387427 | |||
| a55079afbd | |||
| e18ad4723b | |||
| 6de8ac8972 | |||
| a052975420 | |||
| 6d82ca95a4 | |||
| 8067ee4ec4 | |||
| 3743789e8d | |||
| 388aba7632 | |||
| ad587eafa3 | |||
| 08ce9aef11 | |||
| ee5f8b932b | |||
| 4ac688b6d9 | |||
| a814d1ef00 | |||
| ea98856130 | |||
| a49e96835a | |||
| 63c19dcba7 | |||
| 2823349c8e | |||
| 6fc301d62c | |||
| bc99d64786 | |||
| 615af4ed0a | |||
| 045d229728 | |||
| d1fd5700f5 | |||
| 2d55b0b9a5 | |||
| d89ae94a2e | |||
| e6193c4098 | |||
| b66f0677ed | |||
| 23ada1981e | |||
| a22480c117 | |||
| 24f404f989 | |||
| a237fbff9d | |||
| 31d5516991 | |||
| 17af61e8dd | |||
| 6af87b6ad6 | |||
| 5ba05d0bdb | |||
| fc655e78c2 | |||
| 25726a5ae7 | |||
| 11c3ff67b6 | |||
| d867c87100 | |||
| b0c4cedfab | |||
| 85417d5215 | |||
| 4accc746bd | |||
| 21c4c8cbef | |||
| 430f5b0dae | |||
| 42731833d0 | |||
| 65acf066ad | |||
| ee8f570fd7 | |||
| fa3f910d44 | |||
| 46ac6e4e38 | |||
| 3bfa82839b | |||
| 65ccf2e4ad | |||
| 7a3b27f76f | |||
| 82e7be564c | |||
| c5e24197bf | |||
| bcb402b688 | |||
| 7c4170ff6d | |||
| a6095743f0 | |||
| 66e776d178 | |||
| 26f64cba45 | |||
| 51047848f1 | |||
| d8c0b657e8 | |||
| 312c0584ce | |||
| da2625acfb | |||
| 97d9cebc59 | |||
| 3ce76a5d69 | |||
| 2e138a199b | |||
| 450a5ed9c5 | |||
| edabccd885 | |||
| 1dbe3a03fc | |||
| 6058b8472b | |||
| f29968c334 | |||
| 7b98cca967 | |||
| 2757bc7185 | |||
| 7655f1b51a | |||
| a5efb7c676 | |||
| 5a8cf4cb4d | |||
| a3fc7e4df8 | |||
| d811b30df3 | |||
| 3f4ac2b24e | |||
| 4644359128 | |||
| 95a8dbcea9 | |||
| 83b50753fe | |||
| 6f968d59a4 | |||
| d432df8e5c | |||
| 4b822731e6 | |||
| e38eac1a33 | |||
| 8101290933 | |||
| 6e7fc12f89 | |||
| 50df14a50f | |||
| ccd882f2fc | |||
| ecc590f344 | |||
| 3b7cdf9365 | |||
| 9b50dd69d8 | |||
| 08f1a79800 | |||
| 51bdec22e7 | |||
| d43f670285 | |||
| c135583402 | |||
| 8d2893b67e | |||
| 555715ced9 | |||
| 446a9d395c | |||
| 8311f2db7f | |||
| f4570ff274 | |||
| 3aea4e1ec6 | |||
| 516366b796 | |||
| d82d0157cb | |||
| 1deb0c90d4 | |||
| 41a3114d03 | |||
| 76e4b577a0 | |||
| 03473a286b | |||
| 5cabd09705 | |||
| cc47672b7c | |||
| 37edd9134b | |||
| 80f167b1f7 | |||
| bd6547fca3 | |||
| 7e97f5bff5 | |||
| 7d41ccccee | |||
| bf616e192a | |||
| f8bd5d0c51 | |||
| ac40de1d30 |
@@ -1,15 +1,15 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the bridged MCP gateway.",
|
||||
"name": "fleetd",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleet",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the bridged MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"description": "Mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.2.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
name: architect
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
name: dev
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,274 @@
|
||||
---
|
||||
name: fleets-status
|
||||
description: Report the status of every fleet that shares one LavinMQ instance. Use for local daemon health, broker-wide fleet presence, and cross-host lead coordination checks.
|
||||
---
|
||||
|
||||
# Status of every fleet on the shared LavinMQ instance
|
||||
|
||||
**The headline: always report what is missing.** This skill starts with the local fleet, then adds
|
||||
broker-wide facts when its read-only credential exists. A missing fleet must appear as `unknown` or
|
||||
`not reachable`, with the reason and the fix. Never leave it out.
|
||||
|
||||
The known topology has one LavinMQ instance on `10.10.20.13` (`fleet01`). AMQP uses port `5672`,
|
||||
and the management API uses port `15672`. The Mac fleet owns vhost `/mac`. The fleet01 fleet owns
|
||||
vhost `/fleet01`.
|
||||
|
||||
## 1. Protect credentials before any probe
|
||||
|
||||
**Hard rule — never print `LAVINMQ_URI`.** It is an AMQP URI with its password inline. It only
|
||||
resolves in a login shell because `${SHARED_ENV}/tools/secrets.sh` supplies it. A non-login shell
|
||||
can make every broker probe look empty.
|
||||
|
||||
- Never run `echo "$LAVINMQ_URI"`.
|
||||
- Never put `${LAVINMQ_URI:-something}` in output. That form expands to the secret value when set.
|
||||
- Parse the user, host, and password into shell or Python variables. Use them without printing them.
|
||||
- Prefer `resolves` or `does not resolve` over any part of the value.
|
||||
- Every command that can read `LAVINMQ_URI` must send all output through this redaction before it
|
||||
reaches the report:
|
||||
|
||||
```bash
|
||||
sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
```
|
||||
|
||||
**The `g` flag is not optional.** Without it `sed` replaces only the first match on each line, so a
|
||||
line carrying two URIs leaks the second one. `scripts/redeploy-fleetd.sh --check` prints lines like
|
||||
that. Checked on 2026-08-27: without `g`, `amqp://u1:p1@h1/mac and http://u2:p2@h2:15672/api`
|
||||
redacts the first pair and prints `u2:p2` in the clear.
|
||||
|
||||
Keep `pipefail` on when applying that filter. Otherwise the filter can hide a failed probe. Apply
|
||||
the same no-print rule to the management password below, even though it is not in an AMQP URI.
|
||||
|
||||
## 2. Tier 1 — this fleet (always run)
|
||||
|
||||
Start here even when the broker tier is blocked. Work from the local fleetd checkout.
|
||||
|
||||
First run the read-only deployment check. It already checks the daemon process, deployed jar versus
|
||||
the checkout `HEAD`, launchd state, and whether each configured token resolves in a login shell.
|
||||
Do not copy those checks into new shell code. The script reads `LAVINMQ_URI`, so redact all output:
|
||||
|
||||
```bash
|
||||
set -o pipefail
|
||||
scripts/redeploy-fleetd.sh --check 2>&1 \
|
||||
| sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
git rev-parse HEAD
|
||||
```
|
||||
|
||||
Treat jar drift as a top-level warning. A merge is not a deployment. State the running jar result
|
||||
as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into a match.
|
||||
|
||||
Report the process identifier (PID) and uptime too:
|
||||
|
||||
```bash
|
||||
PIDS="$(pgrep -f 'target/fleetd.jar' || true)"
|
||||
if [ -z "$PIDS" ]; then
|
||||
printf '%s\n' 'fleetd: not running'
|
||||
else
|
||||
for PID in $PIDS; do
|
||||
ps -p "$PID" -o pid=,etime=,lstart=,command=
|
||||
done
|
||||
fi
|
||||
```
|
||||
|
||||
Read the full health response. Keep the HTTP status because `503` means fleetd is running but herdr
|
||||
is not reachable. Report both `herdr.version` and `herdr.protocol` when present:
|
||||
|
||||
```bash
|
||||
curl -sS --max-time 5 -w '\nHTTP %{http_code}\n' http://127.0.0.1:8765/healthz
|
||||
```
|
||||
|
||||
Call `fleet_whoami`, then call `fleet_list`. Preserve its sections in the report:
|
||||
|
||||
- `leads`, including which row is this lead;
|
||||
- `members`, including state, role, profile, branch, and worktree when present;
|
||||
- every per-profile `capacity` row, including `maxLoad`, `live`, `free`, and quarantine facts;
|
||||
- the exact `healthCoverage` value.
|
||||
|
||||
Do not describe an empty `members` list as an empty fleet. It says only that no members are spawned.
|
||||
Also do not hide a profile with `free: 0`; say whether load or credential quarantine caused it.
|
||||
|
||||
Show only WARN, ERROR, and SEVERE lines after the last `fleetd listening` line. This anchor stops an
|
||||
old incident from looking current:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
path = Path("fleetd/fleetd.out")
|
||||
if not path.exists():
|
||||
print("cannot check current WARN/ERROR: fleetd/fleetd.out does not exist")
|
||||
else:
|
||||
lines = path.read_text(errors="replace").splitlines()
|
||||
starts = [i for i, line in enumerate(lines) if "fleetd listening" in line]
|
||||
if not starts:
|
||||
print("cannot anchor WARN/ERROR: no 'fleetd listening' line exists")
|
||||
else:
|
||||
current = lines[starts[-1]:]
|
||||
alerts = [line for line in current if re.search(r"\b(?:WARN|ERROR|SEVERE)\b", line)]
|
||||
print(f"current WARN/ERROR/SEVERE count: {len(alerts)}")
|
||||
for line in alerts[-50:]:
|
||||
print(line)
|
||||
PY
|
||||
```
|
||||
|
||||
**What this tier cannot see:** it proves facts only about the Mac daemon at `127.0.0.1:8765`.
|
||||
It cannot show the fleet01 daemon, broker queue depth, or broker consumers. The fleet01 REST service
|
||||
at `10.10.20.13:8765` is not reachable from the Mac. Say this in the report rather than omitting
|
||||
fleet01.
|
||||
|
||||
**But fleet01 IS reachable over SSH — checked 2026-08-28.** An older version of this line said SSH
|
||||
was denied. That is true only for the user `dai.ha`. The host alias `fleet01` maps to user `ltms`,
|
||||
and `ssh fleet01` works with key auth:
|
||||
|
||||
```bash
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=6 fleet01 'echo $(id -un)@$(hostname)'
|
||||
```
|
||||
|
||||
So fleet01's daemon PID, uptime, jar and `/healthz` **can** be reported — over SSH, not over REST.
|
||||
Do that rather than writing `not reachable`. `ltms` also has passwordless sudo there.
|
||||
|
||||
## 3. Tier 2 — the shared broker (run when management access exists)
|
||||
|
||||
**This tier is blocked today.** The AMQP user in `LAVINMQ_URI` can connect on port `5672`, but gets
|
||||
HTTP `401` from the management API on port `15672`. An AMQP connection does not grant monitoring
|
||||
access.
|
||||
|
||||
The operator must create a separate, read-only LavinMQ management user with the `monitoring` tag.
|
||||
It needs access to inspect both `/mac` and `/fleet01`. Store its values as
|
||||
`LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD` in
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Do not reuse or print the AMQP URI. Full multi-fleet status stays
|
||||
blocked until this user exists.
|
||||
|
||||
When both variables resolve, run this from a login shell. It calls `GET /api/overview`,
|
||||
`GET /api/vhosts`, `GET /api/queues`, and `GET /api/connections`. It prints selected status fields,
|
||||
but never the user, password, Authorization header, or AMQP URI:
|
||||
|
||||
```bash
|
||||
zsh -lc 'python3 - "$@"' -- <<'PY'
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
base = "http://10.10.20.13:15672"
|
||||
user = os.environ.get("LAVINMQ_MANAGEMENT_USER", "")
|
||||
password = os.environ.get("LAVINMQ_MANAGEMENT_PASSWORD", "")
|
||||
if not user or not password:
|
||||
print("broker tier: BLOCKED — management credential does not resolve in a login shell")
|
||||
sys.exit(0)
|
||||
|
||||
token = base64.b64encode(f"{user}:{password}".encode()).decode()
|
||||
|
||||
def get(path):
|
||||
request = urllib.request.Request(
|
||||
base + path,
|
||||
headers={"Authorization": "Basic " + token, "Accept": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=5) as response:
|
||||
return json.load(response)
|
||||
|
||||
try:
|
||||
overview = get("/api/overview")
|
||||
vhosts = get("/api/vhosts")
|
||||
queues = get("/api/queues")
|
||||
connections = get("/api/connections")
|
||||
except urllib.error.HTTPError as error:
|
||||
print(f"broker tier: BLOCKED — management API returned HTTP {error.code}")
|
||||
sys.exit(0)
|
||||
except Exception as error:
|
||||
print(f"broker tier: BLOCKED — management API is not reachable: {type(error).__name__}")
|
||||
sys.exit(0)
|
||||
|
||||
fleet_names = {"/mac": "Mac fleet", "/fleet01": "fleet01 fleet"}
|
||||
print(json.dumps({
|
||||
"overview": {
|
||||
"lavinmq_version": overview.get("lavinmq_version"),
|
||||
"rabbitmq_version": overview.get("rabbitmq_version"),
|
||||
"queue_totals": overview.get("queue_totals", {}),
|
||||
"object_totals": overview.get("object_totals", {}),
|
||||
},
|
||||
"fleets": [
|
||||
{
|
||||
"fleet": fleet_names.get(vhost.get("name"), "UNKNOWN FLEET"),
|
||||
"vhost": vhost.get("name"),
|
||||
"queues": [
|
||||
{
|
||||
"name": queue.get("name"),
|
||||
"messages": queue.get("messages", 0),
|
||||
"messages_ready": queue.get("messages_ready", 0),
|
||||
"messages_unacknowledged": queue.get("messages_unacknowledged", 0),
|
||||
"consumers": queue.get("consumers", 0),
|
||||
}
|
||||
for queue in queues if queue.get("vhost") == vhost.get("name")
|
||||
],
|
||||
"connections": [
|
||||
{
|
||||
"name": connection.get("name"),
|
||||
"peer_host": connection.get("peer_host"),
|
||||
"state": connection.get("state"),
|
||||
}
|
||||
for connection in connections if connection.get("vhost") == vhost.get("name")
|
||||
],
|
||||
}
|
||||
for vhost in vhosts
|
||||
],
|
||||
}, indent=2, sort_keys=True))
|
||||
PY
|
||||
```
|
||||
|
||||
Map `/mac` to the Mac fleet and `/fleet01` to the fleet01 fleet. Keep any other vhost in the
|
||||
report as `UNKNOWN FLEET`; do not drop it. For each vhost, total the ready, unacknowledged, and all
|
||||
messages. Report every queue's consumer count and each live connection.
|
||||
|
||||
**A vhost with queues but zero consumers means that fleet's daemon is down while its durable state
|
||||
survives. Call this out as a top-level warning.** This is the main reason to use the management API
|
||||
instead of calling each remote daemon.
|
||||
|
||||
**What this tier cannot see:** without the new `monitoring` credential it cannot enumerate any
|
||||
vhost, queue, depth, consumer, or connection. With the credential it still cannot report fleet01's
|
||||
daemon PID, uptime, jar revision, `/healthz`, herdr version, or member capacity. Those need reachable
|
||||
fleet01 REST or SSH access, which the Mac does not have today.
|
||||
|
||||
## 4. Tier 3 — cross-fleet lead coordination
|
||||
|
||||
Use the queue data from Tier 2. Select queues whose names match `lead.<coordId>.inbox`. Report each
|
||||
queue's vhost, depth, consumer count, and the `coordId` between the prefix and suffix.
|
||||
|
||||
- A lead inbox with a consumer shows that a lead mailbox is live on that vhost.
|
||||
- A durable lead inbox with zero consumers shows saved coordination state, but no live receiver.
|
||||
- No lead inbox is not proof that coordination is disabled. The daemon may be down before declaring
|
||||
its queue, or this account may not be allowed to see the vhost.
|
||||
|
||||
This Mac fleet currently sets both `broker.uriEnv` and `coordinator.uriEnv` to the same variable,
|
||||
`LAVINMQ_URI`. Therefore its coordinator connects to `/mac`. Cross-host `fleet_send{coordId}` routes
|
||||
only when both leads share the same coordinator vhost. If the fleet01 lead uses `/fleet01` for its
|
||||
coordinator, the leads cannot see each other and the send will not route.
|
||||
|
||||
**Open question:** the fleet01 coordinator vhost has not been checked. Surface this question in
|
||||
every report until a live `lead.<coordId>.inbox` consumer or fleet01's config proves the answer. Do
|
||||
not claim that fleet01 uses `/fleet01` just because its member queues do.
|
||||
|
||||
Also compare these broker facts with the `leads` rows from local `fleet_list`. A missing remote lead
|
||||
is `not visible from this coordinator`, not `down`, unless the broker consumer facts prove it.
|
||||
|
||||
**What this tier cannot see:** without Tier 2 management access it cannot list lead inboxes or their
|
||||
consumers. Even with that access, a stopped fleet01 daemon leaves only durable queue history. That
|
||||
history cannot prove which coordinator URI its current config would use after restart.
|
||||
|
||||
## 5. Report all fleets
|
||||
|
||||
Use one row per known or discovered fleet. Include blocked rows.
|
||||
|
||||
| Fleet | Daemon | Deployment | Herdr | Members/capacity | Queues/consumers | Lead coordination | Cannot check |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| Mac (`/mac`) | PID + uptime | jar vs `HEAD` | health + version + protocol | `fleet_list` + `healthCoverage` | facts or blocked reason | inbox facts or open question | exact missing facts |
|
||||
| fleet01 (`/fleet01`) | reachable/down/unknown | value or `not reachable` | value or `not reachable` | value or `not reachable` | facts or blocked reason | inbox facts plus coordinator-vhost question | exact missing facts and fix |
|
||||
|
||||
Add rows for unknown vhosts. End with three short sections: `Current warnings`, `Checks that were
|
||||
blocked`, and `Operator action`. Until the management user exists, `Operator action` must say:
|
||||
|
||||
> Create a read-only LavinMQ management user with the `monitoring` tag and access to `/mac` and
|
||||
> `/fleet01`. Put its user and password in `${SHARED_ENV}/tools/secrets.sh` as
|
||||
> `LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD`.
|
||||
@@ -0,0 +1,102 @@
|
||||
---
|
||||
name: hunter
|
||||
description: Defect-hunt procedure for a fleetd worker — sweep an assigned package for real bugs and report several ranked findings without fixing anything. Load this when the lead asks you to hunt or audit a scope rather than review one diff. Do NOT load `reviewer` for this; the two want different output.
|
||||
---
|
||||
|
||||
# Hunter worker — procedure
|
||||
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies.
|
||||
|
||||
**This skill is not `reviewer`.** `reviewer` judges one diff and reports the *single* most
|
||||
important issue in about 90 words. A hunt sweeps a whole package and reports *several* findings
|
||||
in a long structured form. Loading both gives you two contradictory output contracts, and the
|
||||
usual result is a worker that writes a good report into its terminal and ends the turn without
|
||||
sending it. Load exactly one.
|
||||
|
||||
## 0. Read this before you read code: how the report gets home
|
||||
|
||||
Your terminal reaches nobody. The lead sees **only** the text inside your `fleet_reply` call.
|
||||
|
||||
A long report is exactly the case where this goes wrong, so plan for it:
|
||||
|
||||
- **Write the report into the `fleet_reply` argument itself.** Do not compose it in your terminal
|
||||
and then summarise it into the call.
|
||||
- If the report is long, **send it anyway** — one `fleet_reply` with everything.
|
||||
- If you end the turn without replying, the bridge scrapes your pane instead. That scrape carries
|
||||
at most the last 4000 characters, and on a hunt it usually captures the tail of the lead's own
|
||||
brief rather than your findings. The lead then has nothing and has to ask you again.
|
||||
|
||||
## 1. Change nothing
|
||||
|
||||
A hunt is read-only. Do not edit a production file, do not "quickly fix" what you find, and do
|
||||
not run a formatter. You may run the build and tests to *check* a claim, and you should say so
|
||||
when you did.
|
||||
|
||||
## 2. Read the whole scope first
|
||||
|
||||
Read every file in the assigned package before you judge any of it. A defect that a caller
|
||||
elsewhere in the same package makes unreachable is not a defect, and you cannot know that from
|
||||
one file.
|
||||
|
||||
Stay inside the scope. If a defect there depends on a class outside it, read that class to
|
||||
confirm — but the defect itself must live in the scope you were given.
|
||||
|
||||
## 3. The bar — this matters more than the count
|
||||
|
||||
**Name the path into the bad state.** Say which caller, in which state, reaches it. A defect on
|
||||
paper is not a reachable defect. If you cannot name that path, keep the finding but mark it
|
||||
`unproven` and say exactly what you could not check. Do not drop it, and do not dress it up.
|
||||
|
||||
**Say which direction the harm goes.** Data loss, privilege escalation and silent wrong answers
|
||||
are worth reporting even when the window is narrow. A finding whose worst outcome is a worse log
|
||||
line is not worth a block.
|
||||
|
||||
Two workers once ran the same scope: the one that applied the direction-of-harm filter found ten
|
||||
real defects, the one that did not found none. Fewer findings the lead can act on beat many the
|
||||
lead has to triage.
|
||||
|
||||
## 4. Shapes that have produced real merged fixes here
|
||||
|
||||
Read for these first:
|
||||
|
||||
1. **A one-way gate.** A guard added after an incident closes only the direction that incident
|
||||
came from. Do not only ask what closes the gate — ask **which states still open it**.
|
||||
2. **A value read once, then used later to authorise something destructive**, after something
|
||||
else has had a chance to change it.
|
||||
3. **A failure downgraded to a value that looks like a legitimate result** — `-1`, `null`, an
|
||||
empty list, `false` — which a caller then trusts.
|
||||
4. **A lock held for one half of a read-modify-write and not the other**, or two collections
|
||||
updated under different locks.
|
||||
5. **A comment or javadoc stating an invariant the code no longer keeps.** Comments are
|
||||
load-bearing in this repo; a stale one has already caused a bug.
|
||||
|
||||
## 5. What you cannot check, and must not claim you did
|
||||
|
||||
- `fleetd/fleetd.yaml` is gitignored and **absent from your worktree**. You cannot read it. If a
|
||||
finding depends on live configuration, name the key and say you could not check it.
|
||||
- `.mcp.json`, `opencode.json` and `.autoenv` in your worktree are neutralised stubs, not the
|
||||
repo's real files.
|
||||
- The `wiki/` submodule pointer is months old. Do not cite it.
|
||||
|
||||
Reporting a fact you took from the lead's brief as something you measured yourself is a false
|
||||
report, even when the fact is correct. Say where each fact came from.
|
||||
|
||||
## 6. The report — what goes in `fleet_reply`
|
||||
|
||||
One block per finding, most severe first:
|
||||
|
||||
```
|
||||
FINDING N — <one line>
|
||||
file:line
|
||||
Path in: <which caller, in which state, reaches this>
|
||||
Direction: <data loss | escalation | silent wrong answer | outage | ...>
|
||||
Window/trigger: <when it actually happens>
|
||||
Confidence: <confirmed by reading | unproven — say what you could not check>
|
||||
Why nothing else catches it: <the guard or test you checked, and why it misses>
|
||||
```
|
||||
|
||||
End with one line naming every file you read, so the lead knows the denominator.
|
||||
|
||||
**Nothing clears the bar?** Reply `NO FINDINGS`, name the files you read, and say what you ruled
|
||||
out. A clean sweep is a valid result; an invented defect is worse than none.
|
||||
@@ -1,11 +1,11 @@
|
||||
---
|
||||
name: implementer
|
||||
description: Implementer-role procedure for a bridged worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over bridged.
|
||||
description: Implementer-role procedure for a fleetd worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over fleetd.
|
||||
---
|
||||
|
||||
# Implementer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge, never commit `.mcp.json` or `wiki/`) is in **`CLAUDE.md` → Bridge communication →
|
||||
Worker** and already applies. This skill is only the *implement-and-hand-off procedure*.
|
||||
|
||||
@@ -39,6 +39,19 @@ a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
**Never run `git stash` (or `git stash pop`/`apply`/`drop`).** Your worktree is isolated, but the
|
||||
stash is **not**: `refs/stash` is one stack shared by the primary's checkout and every other
|
||||
worker's worktree of this repo. Measured on 2026-09-04 — `git stash list` from a worker's worktree
|
||||
and from the primary's tree returned byte-identical output. So a `git stash` you run can be popped
|
||||
into someone else's tree, and a `git stash pop` you run can drop **another worker's** uncommitted
|
||||
edits on top of yours. This has already happened here: two workers were running in parallel and one
|
||||
of them had its in-progress edit silently overwritten by the other's stash.
|
||||
|
||||
The branch is your isolation, so use it instead. To set work aside, commit it on your own branch
|
||||
(`git commit -m "wip: ..."`) and carry on; to try something and back out, use
|
||||
`git diff > /tmp/<your-branch>.patch` then `git checkout -- <file>`. Both stay inside your worktree.
|
||||
If you find a stash entry you did not create, leave it alone and say so in your report.
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
@@ -49,7 +62,7 @@ test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-t
|
||||
your worktree*:
|
||||
|
||||
```bash
|
||||
cd "$(git rev-parse --show-toplevel)/bridged" && mvn clean install
|
||||
cd "$(git rev-parse --show-toplevel)/fleetd" && mvn clean install
|
||||
echo "exit=$?"
|
||||
```
|
||||
|
||||
@@ -85,7 +98,7 @@ host (`GITEA_HOST`) into your env for exactly this — the token can create a PR
|
||||
merge**.
|
||||
|
||||
```bash
|
||||
API="${GITEA_HOST%/}/api/v1/repos/lms/claude-bridge/pulls"
|
||||
API="${GITEA_HOST%/}/api/v1/repos/fleet/fleetd/pulls"
|
||||
BRANCH="$(git branch --show-current)"
|
||||
curl -sS -X POST "$API" \
|
||||
-H "Authorization: token ${GITEA_TOKEN}" \
|
||||
@@ -103,7 +116,7 @@ fix it if the cause is yours (e.g. branch not pushed yet), and report the failur
|
||||
inventing a URL. If `GITEA_TOKEN` is unset your profile was not granted PR-create: push the branch
|
||||
and report its name so the lead opens the PR.
|
||||
|
||||
## 6. Hand off — what goes in `bridge_reply`
|
||||
## 6. Hand off — what goes in `fleet_reply`
|
||||
|
||||
The reply is the entire handoff; the lead cannot see your terminal.
|
||||
|
||||
@@ -130,7 +143,7 @@ sequenceDiagram
|
||||
I->>G: git push -u origin HEAD
|
||||
I->>G: POST /pulls (GITEA_TOKEN) — open PR to main
|
||||
G-->>I: html_url
|
||||
I->>L: bridge_reply(PR url, branch, files, tests)
|
||||
I->>L: fleet_reply(PR url, branch, files, tests)
|
||||
Note over L,G: the lead reviews the PR and merges on green — you never merge
|
||||
```
|
||||
|
||||
|
||||
@@ -53,7 +53,7 @@ Map each entry from `.mcp.json`:
|
||||
"$schema": "https://opencode.ai/config.json",
|
||||
"instructions": ["CLAUDE.md"],
|
||||
"mcp": {
|
||||
"bridged": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"fleetd": { "type": "remote", "url": "http://127.0.0.1:8765/mcp", "enabled": true },
|
||||
"context7": { "type": "remote", "url": "https://example.dev/mcp", "enabled": true,
|
||||
"headers": { "Authorization": "Bearer {env:CONTEXT7_TOKEN}" } },
|
||||
"gitea": { "type": "local", "command": ["gitea-mcp", "-t", "stdio"], "enabled": true,
|
||||
|
||||
@@ -1,15 +1,20 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Reviewer-role procedure for a bridged worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over bridged.
|
||||
description: Reviewer-role procedure for a fleetd worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over fleetd.
|
||||
---
|
||||
|
||||
# Reviewer worker — procedure
|
||||
|
||||
The turn contract (one `bridge_reply`, `bridge_ask` for the lead's decisions, honest reporting,
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies. This
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
**Wrong skill for a sweep.** This one reviews *one* diff or scope and reports the *single* most
|
||||
important issue. If the lead asked you to hunt or audit a whole package for several defects, load
|
||||
`hunter` instead and ignore this file — the two want different output, and following both is how a
|
||||
worker ends its turn with a good report that never gets sent.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
@@ -24,13 +29,13 @@ wrong.
|
||||
covers what it was given.
|
||||
- Do **not** edit files or run the build. You review; the owner acts.
|
||||
|
||||
## 3. Reach for `bridge_ask` only for a genuine fork
|
||||
## 3. Reach for `fleet_ask` only for a genuine fork
|
||||
|
||||
Ambiguous requirement, a missing acceptance criterion, "intended or a bug?", or two defensible
|
||||
fixes with different consequences — those are the lead's call, and guessing produces a
|
||||
confident-but-wrong finding. Anything you could settle by reading more code is yours to settle.
|
||||
|
||||
## 4. The finding — what goes in `bridge_reply`
|
||||
## 4. The finding — what goes in `fleet_reply`
|
||||
|
||||
Report the **single most important** real issue in the scope, in these four lines, under
|
||||
~90 words:
|
||||
|
||||
@@ -14,7 +14,7 @@ jobs:
|
||||
# does not depend on the wiki repo being reachable.
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The runner image ships an older default-jdk; bridged sets maven.compiler.release=25, so
|
||||
# The runner image ships an older default-jdk; fleetd sets maven.compiler.release=25, so
|
||||
# provision the JDK explicitly rather than apt-installing whatever "default" means today.
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
@@ -32,7 +32,7 @@ jobs:
|
||||
mvn -version
|
||||
|
||||
- name: Build and test
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
# This IS the mock-socket surface CB-503 asks for: the pom's `default-excludes` profile
|
||||
# already sets excludedGroups=contract, so the @Tag("contract") tests — which need a live
|
||||
# herdr socket and a RabbitMQ container — are excluded without any flag here. Everything
|
||||
@@ -46,7 +46,7 @@ jobs:
|
||||
# the log instead, where they are actually readable.
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||
# straight to it — no Docker, no skipped tests. This separation (build job hermetic and
|
||||
# Docker-free; contract job broker-provided) is deliberate — see the default-excludes/contract
|
||||
# profiles in bridged/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# profiles in fleetd/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# exactly as in the build job above.
|
||||
contract:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -69,8 +69,6 @@ jobs:
|
||||
env:
|
||||
RABBITMQ_DEFAULT_USER: guest
|
||||
RABBITMQ_DEFAULT_PASS: guest
|
||||
ports:
|
||||
- 5672:5672
|
||||
env:
|
||||
# Service containers are reachable from the job by their network alias on their internal port.
|
||||
AMQP_URI: amqp://guest:guest@rabbitmq:5672
|
||||
@@ -93,12 +91,12 @@ jobs:
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
- name: Contract tests
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
|
||||
+4
-4
@@ -14,8 +14,8 @@
|
||||
.env
|
||||
.envrc
|
||||
|
||||
# Daemon runtime artefacts. bridged appends its log wherever it is launched from, so both the
|
||||
# repo root and bridged/ collect one; neither belongs in git.
|
||||
bridged.out
|
||||
bridged/bridged.out
|
||||
# Daemon runtime artefacts. fleetd appends its log wherever it is launched from, so both the
|
||||
# repo root and fleetd/ collect one; neither belongs in git.
|
||||
fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
+1
-1
@@ -1,3 +1,3 @@
|
||||
[submodule "wiki"]
|
||||
path = wiki
|
||||
url = ssh://git@git.ltms.dev:2224/lms/claude-bridge.wiki.git
|
||||
url = ssh://git@git.ltms.dev:2224/fleet/fleetd.wiki.git
|
||||
|
||||
@@ -0,0 +1,24 @@
|
||||
---
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
You never commit production code and never open a pull request.
|
||||
|
||||
A design task is worked by two architects. Design alone first, then exchange and
|
||||
say plainly where you disagree. Do not concede just to agree.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Report only work you actually did and the real output of checks you ran. Do not
|
||||
claim a result from a tool you could not use. The primary's IDE tools are not yours.
|
||||
A mounted forge tool may use a blocked credential and fail by design.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,29 @@
|
||||
---
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
the worktree root and branch before you edit. Use only paths under that root.
|
||||
|
||||
Do only the assigned scope. Note anything outside that scope in one line and do not
|
||||
investigate it further. Use `fleet_ask{question}` only when a decision belongs to
|
||||
the lead, such as an unclear requirement or two defensible fixes. Do not ask about
|
||||
something you can decide by reading more code.
|
||||
|
||||
Implement the change and run the full required build in your worktree. Read the
|
||||
complete output and report its real result. Do not hide failures with a pipe. State
|
||||
only checks you actually ran. The primary's IDE tools are not yours. A mounted forge
|
||||
tool may use a blocked credential and fail by design.
|
||||
|
||||
Stage only files you changed. Never use `git add -A` or `git add .`. Never commit
|
||||
`.mcp.json` or `wiki/`. Commit with a clear message, push your branch, and open your
|
||||
own pull request against `main`. Never merge.
|
||||
|
||||
Your handoff must name the pull request or why it was not created, the branch, the
|
||||
files changed, the build result, and any caveat for review.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -0,0 +1,34 @@
|
||||
---
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
Read the whole assigned scope before judging it. Review only that scope. If you see
|
||||
something outside it, note it in one line and do not investigate it further. Do not
|
||||
run the build. The owner makes changes and runs checks.
|
||||
|
||||
Use `fleet_ask{question}` only when a decision belongs to the lead, such as an
|
||||
unclear requirement or two defensible fixes. Do not ask about something you can
|
||||
decide by reading more code.
|
||||
|
||||
Report the single most important real issue in this form:
|
||||
|
||||
```
|
||||
1. <path>:<line>
|
||||
2. issue: <one sentence: what is wrong and why it matters>
|
||||
3. fix: <one line: the concrete change>
|
||||
4. severity: high | medium | low
|
||||
```
|
||||
|
||||
If there is no real issue, report `NO ISSUE` and one line saying why. A clean review
|
||||
is valid. Do not invent an issue. Use high for a wrong result, data loss, security,
|
||||
or a hang or crash on a real path. Use medium for an edge-path bug or a correctness
|
||||
risk under load or concurrency. Use low for clarity, a latent foot-gun, or a smell
|
||||
with no current failure.
|
||||
|
||||
The launcher provides the required bridge reply instructions for every member.
|
||||
@@ -4,35 +4,43 @@
|
||||
|
||||
> **Canonical block.** Everything down to §Layering is the portable bridge charter, copied verbatim
|
||||
> into every project that mounts the bridge MCP. Keep it byte-identical with the template in the
|
||||
> wiki ([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
>
|
||||
> **Anything you measure in an addendum is perishable.** Date it, give the command that
|
||||
> re-measures it and what each outcome means, and tell the reader to delete the section once
|
||||
> it stops reproducing. The four parts work together: deciding what would falsify a claim is
|
||||
> the expensive step, and a reader in the middle of another task will not pay it, so a bare
|
||||
> "verify before relying on this" costs the same space and does nothing. The case this is for
|
||||
> is a note that goes stale as a live restriction — it will tell a future session it cannot do
|
||||
> the thing at the moment doing it becomes the job.
|
||||
|
||||
If no `bridge_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
`bridged` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
`fleetd` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **member**) mount the *same* MCP server and talk only
|
||||
through its `bridge_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
through its `fleet_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
|
||||
### Which role am I? — settle this before acting
|
||||
|
||||
**Every role reads this file.** A member runs in a git worktree of this same repo, so it inherits
|
||||
this `CLAUDE.md` verbatim, and every rule below is role-conditional.
|
||||
|
||||
**Call `bridge_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
**Call `fleet_whoami`.** It returns `primary`, `worker`, or `architect`, resolved by the daemon from
|
||||
your connection — unforgeable, and the same resolution its authorization gate uses. A worker also
|
||||
carries its `sessionId`, `profile`, `worktree` and `branch`; an architect carries the slot name it
|
||||
was bound to. Don't infer what you can ask.
|
||||
|
||||
Only if that call is unavailable, fall back to these — each is one-way, so keep reading until one
|
||||
fires: the reply charter in your system prompt (*"You are an off-subscription worker in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; bridge tools prefixed `mcp__bridge__*` ⇒ **spawned
|
||||
fires: the reply charter in your system prompt (*"You are a spawned member in the
|
||||
claude-bridge fleet"*) ⇒ **spawned member**; fleet tools prefixed `mcp__fleet__*` ⇒ **spawned
|
||||
member** (the launcher fixes that mount name; a primary's mount is named by whoever wrote its
|
||||
`.mcp.json`, so it varies); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
`.mcp.json`, so it varies — and a member spawned before CB-632 still says `mcp__bridge__*`); `ANTHROPIC_BASE_URL` set ⇒ **spawned member** (Claude-model members run
|
||||
on a clean env, so its *absence* proves nothing). None of these separate a worker from an architect —
|
||||
only `bridge_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
only `fleet_whoami` does. **Still unsure ⇒ act as a worker**, the most restricted member role. The
|
||||
two mistakes are not symmetric: a primary acting as a worker is refused by the authorization gate —
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `bridge_reply`,
|
||||
loud and self-correcting — while a member acting as the primary ends its turn with no `fleet_reply`,
|
||||
and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
### Invariants — both roles, no exceptions
|
||||
@@ -41,7 +49,7 @@ and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
stays on subscription; only the bridge puts a member off it, at spawn. Mounting the bridge must
|
||||
never move a session across that boundary.
|
||||
2. **The bridge is the only channel.** Text you print in your terminal reaches nobody — the other
|
||||
side cannot see your screen. An answer that isn't in a `bridge_*` call is silently discarded.
|
||||
side cannot see your screen. An answer that isn't in a `fleet_*` call is silently discarded.
|
||||
3. **Identity comes from the connection, never an argument.** Workers never pass a target; you
|
||||
cannot act as another session. Spawn/stop/drain are lead-only; **send is lead or architect**;
|
||||
reply/ask are only-as-itself — any peer may answer for its own pane, and for no other. A call
|
||||
@@ -57,10 +65,10 @@ and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
|
||||
**Delegate by default — that is the job.** With the bridge mounted you are an orchestrator on a
|
||||
metered subscription, and workers are cheap, parallel, and disposable. The default answer to "who
|
||||
does this?" is **a worker**, not you. Reach for `bridge_send` before you reach for `Edit`. The steps
|
||||
does this?" is **a worker**, not you. Reach for `fleet_send` before you reach for `Edit`. The steps
|
||||
below are the procedure — run them in order, every task, not only the big ones.
|
||||
|
||||
0. **Know your role** — `bridge_whoami`, once per session, before anything else.
|
||||
0. **Know your role** — `fleet_whoami`, once per session, before anything else.
|
||||
1. **Split.** Write the unit list. Every unit carries: scope · the files or PR in question ·
|
||||
acceptance criteria · exactly what to report back. A unit with no acceptance criteria is not
|
||||
ready to delegate — refine it or keep it.
|
||||
@@ -70,17 +78,20 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
delegate. The keep-list is closed: the conversation with the user, decomposition and planning,
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `bridge_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
4. **Then send them all** — `bridge_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model, cost and
|
||||
LIVENESS, not in tier, so the default is rarely what you want. The default is whatever the
|
||||
daemon reports, and on a host where it sits on an exhausted or withdrawn credential every
|
||||
unqualified spawn fails — sometimes loudly, sometimes as a member that spawns fine and then
|
||||
produces nothing. `fleet_profiles` reports the default; check it once per session.
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
brief instead. The brief is self-contained — the worker sees your message and the repo, nothing
|
||||
of your context, your plan, or your screen.
|
||||
5. **Collect** — `bridge_poll{ticket}` → `bridge_ack{ticket, msgId}`. Answer a worker's `bridge_ask`
|
||||
with `bridge_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`bridge_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||
5. **Collect** — `fleet_poll{ticket}` → `fleet_ack{target, msgId}`. Answer a worker's `fleet_ask`
|
||||
with `fleet_send{turnId, content}` — **not** `sessionId`. A worker gone quiet is diagnosed with
|
||||
`fleet_status`, never by reading its terminal; it also reports an open question and the `turnId`
|
||||
that answers it. **A worker's ask waits ~55 seconds, and no nudge makes that longer** — so never
|
||||
brief a worker to "ask me". Decide before you delegate, or give it an explicit default.
|
||||
6. **Verify yourself.** Re-run the build and the checks. A worker cannot run your IDE tooling, any
|
||||
@@ -93,11 +104,23 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
read it yourself.
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `bridge_stop{paneId}`.
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
**If the forge refuses you the merge** — a protected branch, a token without the grant — the
|
||||
adjudication is still yours. Read the diff, decide, and hand the operator a merge-ready queue
|
||||
with the refusal quoted. Never report a PR as merged, and never call one "ready to merge"
|
||||
without having read the diff yourself. A refusal is exactly when that shortcut is tempting,
|
||||
because no action is left that forces you to look, and taking it turns this step into
|
||||
forwarding a reviewer's verdict — which is delegating the merge by proxy, two lines above.
|
||||
**Test a refusal; do not read it off a permissions field.** A protected branch holds its merge
|
||||
rights separately from the repository permissions, so that field can say yes while the merge is
|
||||
refused, and still say no after a grant makes it work. Probe instead, with a request that cannot
|
||||
succeed on its merits, so a rejection can only mean the refusal. Treat a transport failure as a
|
||||
third answer that proves nothing: a timeout, a DNS error or a bad URL is not a refusal, and
|
||||
counting it as one makes you sure of something you never measured.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
prefer `wait:false` + `bridge_poll` for anything non-trivial: a blocking `bridge_send` is capped by
|
||||
prefer `wait:false` + `fleet_poll` for anything non-trivial: a blocking `fleet_send` is capped by
|
||||
*your own* MCP client call timeout (~60s), well below the task's real runtime.
|
||||
|
||||
**Delegating does not delegate responsibility.** Workers open PRs; you are the gate. Never delegate
|
||||
@@ -105,21 +128,22 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
|
||||
| Intent | Tool |
|
||||
|---|---|
|
||||
| Confirm your own role | `bridge_whoami` |
|
||||
| See backends available | `bridge_profiles` |
|
||||
| Start a member | `bridge_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `bridge_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `bridge_status{sessionId}` |
|
||||
| Delegate (blocking) | `bridge_send{sessionId, content}` |
|
||||
| Delegate (long task) | `bridge_send{sessionId, content, wait:false}` → ticket → `bridge_poll{ticket}` |
|
||||
| Answer a member's `bridge_ask` | `bridge_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** | `bridge_send{sessionId: <their terminal>, content}` — `bridge_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `bridge_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `bridge_poll{target}` · then `bridge_ack{target, msgId}` |
|
||||
| Tear down a member | `bridge_stop{paneId}` |
|
||||
| Confirm your own role | `fleet_whoami` |
|
||||
| See backends available | `fleet_profiles` |
|
||||
| Start a member | `fleet_spawn{role?, profile?, cwd?, worktree?, ticket?, sessionName?, resumeSessionId?}` → `sessionId` + `paneId` |
|
||||
| See the fleet | `fleet_list` → `leads` (your peers) + `members` (each carries `agentSessionId` when its backend knows one) · one peer's state: `fleet_status{sessionId}` |
|
||||
| Delegate (blocking) | `fleet_send{sessionId, content}` |
|
||||
| Delegate (long task) | `fleet_send{sessionId, content, wait:false}` → ticket → `fleet_poll{ticket}` |
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_reply{content}` — the one case a lead replies |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
`bridge_list` returns `leads` alongside `members`; your own row carries `self: true`. Every other row
|
||||
`fleet_list` returns `leads` alongside `members`; your own row carries `self: true`. Every other row
|
||||
is a peer — an orchestrator with its own context, its own members, and its own judgment. An empty
|
||||
`members` array means no members are spawned; it says nothing about peers.
|
||||
|
||||
@@ -139,8 +163,13 @@ The traffic between leads is coordination and nothing else:
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
4. **Ask a peer to read your project addendum.** Your addendum is instruction surface: every future
|
||||
session on your host obeys it, and a wrong one is obeyed just as faithfully as a right one. The
|
||||
author is the worst reader of their own qualifier placement — measured here, one addendum carried
|
||||
two defects and a non-author found both. If you have no peer, at least re-read it asking "which
|
||||
sentence goes false first, and would a reader reach the caveat before acting?"
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `bridge_reply`, and push back on
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
|
||||
@@ -148,11 +177,11 @@ you.
|
||||
|
||||
1. **Load the playbook skill the lead named** before doing anything else.
|
||||
2. **Do the assigned scope only.** Note anything you spot outside it in one line; don't go hunt it.
|
||||
3. **`bridge_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
3. **`fleet_ask{question}`** when a decision is genuinely the lead's (ambiguous requirement, two
|
||||
defensible fixes, "bug or intended?"). It blocks and you resume the *same* turn with the answer.
|
||||
Don't ask what you could decide yourself.
|
||||
4. **End the turn with exactly one `bridge_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `bridge_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
4. **End the turn with exactly one `fleet_reply{content}`**, carrying your complete answer. This is
|
||||
the whole handoff. No `fleet_reply` ⇒ the sender gets nothing and the exchange stalls.
|
||||
Do **not** lean on the completion fallback to carry your answer for you: when you end a turn
|
||||
without replying, the bridge scrapes your pane, and it can return only the last 4000 characters.
|
||||
A clipped scrape is marked as partial, but the missing text is gone — your report reaches the
|
||||
@@ -171,33 +200,77 @@ you.
|
||||
|
||||
| Layer | Scope | Reaches |
|
||||
|---|---|---|
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `bridge_reply`* | every spawned member, at launch, every peer kind — never a lead |
|
||||
| the launcher's reply charter | the one rule that must survive with no repo: *end every turn with `fleet_reply`* | every spawned member, at launch, every peer kind — never a lead |
|
||||
| **this section** | protocol + orchestration policy | primary **and** every member that reads the repo — tracked in git, so worktrees inherit it |
|
||||
| role agent definition files | role contract and per-job procedure | a member whose launcher binds its role to the matching file in its worktree |
|
||||
| role playbook skills | per-job procedure (commit/PR recipe, finding format) | a member told to load one |
|
||||
| the bridge's own docs | design detail, flows, error model | on demand |
|
||||
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. Peers that don't read
|
||||
`CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they* must obey belongs in the
|
||||
charter, not here.
|
||||
A rule belongs in **exactly one** layer — the outermost one that must obey it. A member without a
|
||||
repo checkout still gets the launcher's reply charter, which is why that one rule stays there.
|
||||
Peers that don't read `CLAUDE.md` (non-Claude adapters) get the charter only, so any rule *they*
|
||||
must obey belongs in the charter, not here.
|
||||
|
||||
## Project addendum — claude-bridge (not part of the canonical block)
|
||||
|
||||
- **This repo is the bridge.** The daemon is `bridged`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/BridgeMcp` (tools), `auth/Authz` (the role table),
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **`fleet_profiles`/`fleet_list` report two separate outage states, and they are not the same
|
||||
thing.** *Quarantined* (CB-578) means the backend told us it is out of capacity — a long,
|
||||
1800s-default cooldown. *Cooling off* (fleetd #201/#227) means a profile's credential threw two
|
||||
distinct backend errors (a non-exhaustion failure such as an HTTP 5xx) within 60 seconds — a
|
||||
short, fixed 60s cooldown, not configurable per profile. Each check runs independently, so a
|
||||
profile can show both at once. In the JSON: a cooling profile carries `credentialId` and
|
||||
`coolingOffForSeconds`; a quarantined profile carries `quarantinedForSeconds`; a profile hit by
|
||||
both carries all three fields, and either state alone already sets that profile's `free` to `0`.
|
||||
A `fleet_spawn` naming a cooling-off profile is refused before it ever reaches the backend
|
||||
adapter, with a message naming the credential and the remaining seconds ("cooling off after
|
||||
repeated backend errors") — distinct wording from a quarantine refusal, so don't conflate the
|
||||
two when reading a spawn failure.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR),
|
||||
`reviewer` (one diff → one structured finding) and `hunter` (sweep a package → several ranked
|
||||
findings, change nothing). Name exactly one in every delegation. **`reviewer` and `hunter` are
|
||||
not interchangeable** — `reviewer` caps the answer at one finding in about 90 words, so naming
|
||||
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
||||
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
||||
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace).
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace) and
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
- **This repo is also a Claude Code marketplace, and ships a plugin.** `.claude-plugin/marketplace.json`
|
||||
points at `plugin/`, which carries the MCP mount and the `setup` skill
|
||||
(`/claude-bridge:setup` — make any project bridge-ready). It was added in CB-527 and then went
|
||||
unmentioned by every instruction file, so it drifted and a later session planned it from scratch
|
||||
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
||||
structural, not bugs: a plugin cannot carry the role agent files, because
|
||||
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
||||
worktree; and a plugin cannot deliver anything to members at all, because
|
||||
`ClaudeCodeLauncher.java:285` exports `CLAUDE_CONFIG_DIR` and every Claude profile here sets it,
|
||||
so a member never reads the operator's plugin store. **The plugin is the lead-side surface;
|
||||
member-facing assets travel in the worktree.**
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `bridge_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` §6, kept out of this file because it loads into every
|
||||
session's context.
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
committed copies would otherwise mount the primary's IDE and forge servers (fleetd #134). The
|
||||
worktree's copy of each is a stub, **not** the repo's real file, so a worker that reads one and
|
||||
reports what it found is reporting on the stub. The daemon logs a per-spawn summary, but the
|
||||
worker cannot see that log. From inside its own worktree a worker — or a lead debugging one —
|
||||
reads the list with `git config --worktree --get-all fleet.neutralizedConfig`, and the
|
||||
consequence with `git config --worktree --get fleet.neutralizedConfigNote`. Never brief a worker
|
||||
to edit one of these files: the edit cannot be committed, and it will not tell you so.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, the turn-done
|
||||
fallback and status gating — are diagrammed in `docs/MCP-Contract.md`. That page is now flows
|
||||
only: its pre-build tool catalogue, parameter tables and REST paths were deleted rather than
|
||||
corrected, because a hand-maintained second copy of the tool surface is what drifted for a month
|
||||
while this line pointed every session at it (CB-609 / #114). **The live MCP schema is the tool
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `bridged` holds the jar it was started with, so a
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
@@ -208,9 +281,9 @@ and stopping it kills the worker's own channel mid-turn.
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-bridged.sh --check # report state, change nothing
|
||||
scripts/redeploy-bridged.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-bridged.sh --yes # skip the drain prompt (fleet already checked)
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
@@ -227,17 +300,17 @@ if the script is unavailable or a step fails, this is what it was protecting you
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `bridge_list`, then `bridge_stop` each member, and collect anything
|
||||
you still want with `bridge_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `bridge_whoami` and confirm it still answers `primary`. The
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `bridged listening` line at the end of
|
||||
`bridged/bridged.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
@@ -246,7 +319,7 @@ it is one auditable command, so the operator allow-lists it once instead of appr
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-bridged.sh:*)"] } }
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
@@ -264,15 +337,15 @@ Before you call any work done, check the row that matches what you touched:
|
||||
|
||||
| You changed… | Re-read and update… |
|
||||
|---|---|
|
||||
| a `bridge_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| a `fleet_*` tool — added, removed, renamed, or its params/semantics | the primary's intent→tool table; any rule that names that tool |
|
||||
| `Authz` / the role table | invariant 3, and the primary-only vs worker-only claims |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `bridge_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__bridge__*`), and the layering table's top row |
|
||||
| `ConnectionIdentity` / how a caller is resolved | the `fleet_whoami` paragraph and the fallback ladder |
|
||||
| `REPLY_CHARTER`, or a launcher's mount/flags | the fallback ladder (`mcp__fleet__*`), and the layering table's top row |
|
||||
| the injector / status gating | invariant 4 |
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `bridged.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
|
||||
That last row is not bookkeeping. Chapters 1–10 answer *how is this built* and *why this way*;
|
||||
none of them has a home for *what can it do and how do I turn it on*, so for twenty tickets a
|
||||
@@ -283,7 +356,7 @@ is a Roadmap line. A change that touches none of the three earns no entry, and t
|
||||
outcome rather than an omission.
|
||||
|
||||
Then **propagate**: the block in this file and the template in the wiki
|
||||
([Use Cases](https://git.ltms.dev/lms/claude-bridge/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable `CLAUDE.md`
|
||||
block*) must stay byte-identical, and other projects carrying the block need the same edit. Verify
|
||||
rather than trust:
|
||||
|
||||
@@ -306,10 +379,10 @@ PY
|
||||
|
||||
Two IDE MCP servers are connected: **intellij-index** (semantic code intelligence) and
|
||||
**jetbrains** (file problems, reformat, debugger). IntelliJ has multiple projects open; our
|
||||
module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
module is **`fleetd`**. Always pass these to IDE MCP tools:
|
||||
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/bridged`
|
||||
- IDE paths are relative to `bridged/` (e.g. `src/main/java/dev/ltms/bridged/...`)
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/fleetd`
|
||||
- IDE paths are relative to `fleetd/` (e.g. `src/main/java/dev/ltms/fleet/...`)
|
||||
|
||||
### After editing any file — mandatory
|
||||
|
||||
@@ -323,7 +396,7 @@ module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
whole-project gate before declaring work done or committing.
|
||||
|
||||
**Whenever dependencies change (or a `pom.xml` edit), validate CVEs with
|
||||
`jetbrains get_file_problems{filePath: "bridged/pom.xml"}`** — its Mend.io check reflects the
|
||||
`jetbrains get_file_problems{filePath: "fleetd/pom.xml"}`** — its Mend.io check reflects the
|
||||
dependencies on disk. (Note: `ide_diagnostics` / intellij-index does NOT re-resolve dependencies
|
||||
after a pom edit without a full Maven reimport, so it reports stale CVE results — don't trust it
|
||||
for this.) Treat a CVE warning like any other: bump to a patched version and confirm
|
||||
|
||||
@@ -9,32 +9,33 @@ Sibling of [`crush-bridge`](https://git.ltms.dev/systems/vms) (which drives a he
|
||||
process*, so it inherits `CLAUDE.md`, hooks, skills, and MCP — just pointed at a
|
||||
cheaper/local model.
|
||||
|
||||
## Leading approach — herdr-centric message server (`bridged`)
|
||||
## Leading approach — herdr-centric message server (`fleetd`)
|
||||
|
||||
A small always-on message server, **`bridged`**, controls
|
||||
A small always-on message server, **`fleetd`**, controls
|
||||
[herdr](https://herdr.dev) (an agent multiplexer) over its Unix-socket API and exposes a
|
||||
clean 2-way messaging API as an **MCP server that both the primary and the workers mount** —
|
||||
one unified Claude setup and the **sole communication gateway** (REST/SSE stays for non-Claude
|
||||
clients; any broker is `bridged`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `bridged` owns
|
||||
clients; any broker is `fleetd`-internal, below the gateway).
|
||||
herdr owns the PTYs, multiplexing, persistence, and **agent-status events**; `fleetd` owns
|
||||
policy (subscription boundary, session lifecycle, status-gated delivery) and the client
|
||||
contract. The worker `claude` launches with `ANTHROPIC_BASE_URL=https://ollama.ltms.dev` + a
|
||||
bearer token; the primary Opus stays env-clean and calls `bridged`'s MCP tools.
|
||||
contract. A Claude member launches with `ANTHROPIC_BASE_URL` pointed at the gateway,
|
||||
`https://llm.ltms.dev/anthropic`, plus a bearer token; the lead stays env-clean and calls
|
||||
`fleetd`'s MCP tools. See the wiki's **[13 User Guide](wiki/13-User-Guide.md)** to run it.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["bridged — standalone daemon (not a claude process)"]
|
||||
subgraph BD["fleetd — standalone daemon (not a claude process)"]
|
||||
SRV["SERVER face<br/>MCP · REST/SSE · policy"]
|
||||
CLI["CLIENT face<br/>status-gated injector · herdr socket"]
|
||||
SRV --> CLI
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
M["ollama.ltms.dev<br/>(worker model)"]
|
||||
M["llm.ltms.dev<br/>(the one gateway)"]
|
||||
|
||||
OPUS -->|"MCP bridge_send (blocks)"| SRV
|
||||
W -.->|"MCP bridge_reply"| SRV
|
||||
OPUS -->|"MCP fleet_send (blocks)"| SRV
|
||||
W -.->|"MCP fleet_reply"| SRV
|
||||
CLI -->|"Unix socket<br/>send_text · events.subscribe"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
W -->|"inference"| M
|
||||
@@ -46,24 +47,26 @@ flowchart LR
|
||||
```
|
||||
|
||||
- **Subscription boundary:** the *primary* never sets `ANTHROPIC_BASE_URL` (stays on
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `bridged` itself is a
|
||||
Pro/Max). Only the *secondary* process is off-subscription — and `fleetd` itself is a
|
||||
plain daemon (no Anthropic quota), so it may poll/subscribe freely.
|
||||
- **One gateway (unified MCP setup):** `bridged` is the **sole communication path** for every
|
||||
- **One gateway (unified MCP setup):** `fleetd` is the **sole communication path** for every
|
||||
Claude session. Primary and workers each mount it as an MCP server (one `claude mcp add`
|
||||
line, same on both) and talk over MCP tools — `bridge_send` / `bridge_reply` /
|
||||
`bridge_status` (with `bridge_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `bridged`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`bridge_send`);
|
||||
`bridged` holds it open until the worker calls `bridge_reply` or its turn hits
|
||||
line, same on both) and talk over MCP tools — `fleet_send` / `fleet_reply` /
|
||||
`fleet_status` (with `fleet_ask` planned for the blocked-worker path). **No Claude session
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `fleetd`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
**Tool naming:** the tools are `fleet_*` (renamed from `bridge_*` in CB-622). The old
|
||||
`bridge_*` names were removed in CB-634 — only `fleet_*` answers now.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`fleet_send`);
|
||||
`fleetd` holds it open until the worker calls `fleet_reply` or its turn hits
|
||||
`agent_status=done`, then returns the reply as the tool result. No cross-turn busy-poll, so
|
||||
no quota burn. SSE is an optional side-channel for humans/dashboards watching status.
|
||||
- **Worker → primary** rides `bridged`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `bridged` **injects the primary's idle pane** when it's
|
||||
- **Worker → primary** rides `fleetd`'s **MCP rendezvous** — the reply resolves the primary's
|
||||
blocking call (or, for detached work, `fleetd` **injects the primary's idle pane** when it's
|
||||
ready), so *no keystroke-into-primary and no broker are involved, even single-host*. The one
|
||||
exception: a split-host primary that isn't a herdr pane wakes via its own `Stop`-hook, which
|
||||
polls **`bridged`** (never a broker). See the wiki for the two topologies.
|
||||
polls **`fleetd`** (never a broker). See the wiki for the two topologies.
|
||||
- **Different model per process** sidesteps Claude Code's lack of per-subagent provider
|
||||
routing — the worker isn't a subagent, it's its own configured process.
|
||||
- **AgentAPI** ([`coder/agentapi`](https://github.com/coder/agentapi)) is retained only as a
|
||||
@@ -72,11 +75,11 @@ flowchart LR
|
||||
|
||||
## Docs
|
||||
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/lms/claude-bridge/wiki)**,
|
||||
Full design, setup, and operations live in the **[wiki](https://git.ltms.dev/fleet/fleetd/wiki)**,
|
||||
vendored here as a submodule under [`wiki/`](./wiki):
|
||||
|
||||
```bash
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/lms/claude-bridge.git
|
||||
git clone --recurse-submodules ssh://git@git.ltms.dev:2224/fleet/fleetd.git
|
||||
# or, after a plain clone:
|
||||
git submodule update --init
|
||||
```
|
||||
@@ -86,7 +89,7 @@ Gitea wiki.
|
||||
|
||||
## Status
|
||||
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`bridged`** message server is built and in
|
||||
🟢 **Implemented & dogfooded** — the herdr-centric **`fleetd`** message server is built and in
|
||||
real use: an Opus primary delegates tasks to off-subscription workers that reply through the
|
||||
bridge (code reviews delegated this way have produced committed bug fixes). Selected as the
|
||||
primary approach 2026-07-11, superseding the AgentAPI plan (2026-07-08); AgentAPI retained as a
|
||||
@@ -97,9 +100,9 @@ tests run separately via `mvn test -Pcontract`):
|
||||
|
||||
- **Core gateway** — herdr socket client (contract-tested vs live 0.7.0); guard-checked worker
|
||||
spawn with `ANTHROPIC_BASE_URL` injected only into the worker's env; status-gated injector;
|
||||
blocking `bridge_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `bridge_send` / `bridge_reply` / `bridge_status` (messaging) and `bridge_spawn`
|
||||
/ `bridge_list` / `bridge_stop` / `bridge_profiles` / `bridge_poll` (fleet). Caller identity is
|
||||
blocking `fleet_send` with reply rendezvous; MCP server as a thin adapter over the REST core.
|
||||
- **MCP tools** — `fleet_send` / `fleet_reply` / `fleet_status` (messaging) and `fleet_spawn`
|
||||
/ `fleet_list` / `fleet_stop` / `fleet_profiles` / `fleet_poll` (fleet). Caller identity is
|
||||
connection-based (loopback peer PID → herdr pane), so the same mount serves primary and workers.
|
||||
- **Delivery reliability** — completion fallback (a confirmed `working→idle` turn resolves a
|
||||
send); async fire-and-poll (beats the caller's MCP call timeout for long tasks); and failure
|
||||
@@ -107,7 +110,7 @@ tests run separately via `mvn test -Pcontract`):
|
||||
- **Fleet** — multiple worker profiles, each with an independent base_url guard check; workers
|
||||
inherit the primary's working directory (never `$HOME`); a readiness gate holds delivery until
|
||||
a worker's Claude has connected the bridge MCP (no paste lost into its boot window).
|
||||
- **Blocked-worker path** — `bridge_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
- **Blocked-worker path** — `fleet_ask` reverse rendezvous: a worker pauses its delegated turn to
|
||||
ask the primary and resumes the *same* turn with the answer (CB-205).
|
||||
- **Session lifecycle** — session manager with spawn/reuse/recycle, `idle_ttl` reaper, `context_cap`,
|
||||
and graceful drain on shutdown (CB-301/CB-303); per-worker git worktrees on their own branch with
|
||||
|
||||
@@ -1,14 +0,0 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from bridged.example.yaml)
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -1,683 +0,0 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.config.ConfigRef;
|
||||
import dev.ltms.bridged.config.ConfigWatcher;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrClient;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.LeadTabScanner;
|
||||
import dev.ltms.bridged.lead.LeadLauncher;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.CompletionResolver;
|
||||
import dev.ltms.bridged.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.bridged.inject.ExhaustionSink;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.inject.StatusPoller;
|
||||
import dev.ltms.bridged.inject.TurnListener;
|
||||
import dev.ltms.bridged.inject.MemberPresence;
|
||||
import dev.ltms.bridged.auth.MemberRegistry;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.mcp.BridgeMcp;
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.health.FleetHealthMonitor;
|
||||
import dev.ltms.bridged.mcp.PrimaryRegistry;
|
||||
import dev.ltms.bridged.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.bridged.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.bridged.msg.AmqpReplyInbox;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.ReplyInbox;
|
||||
import dev.ltms.bridged.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.bridged.msg.ReplyPushLoop;
|
||||
import dev.ltms.bridged.rest.BridgedApp;
|
||||
import dev.ltms.bridged.session.GitWorktrees;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.session.SessionReaper;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.member.CompositePeerLauncher;
|
||||
import dev.ltms.bridged.member.HerdrPeerLauncher;
|
||||
import dev.ltms.bridged.member.OpenCodeLauncher;
|
||||
import dev.ltms.bridged.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* {@code bridged} entry point. Wires the real herdr socket client to the REST app and
|
||||
* starts listening. Before anything else it asserts its own environment is clean —
|
||||
* {@code bridged} is not a Claude process and must never carry a base_url.
|
||||
*/
|
||||
public final class Bridged {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Bridged.class);
|
||||
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = Path.of(args.length > 0 ? args[0] : "bridged.yaml");
|
||||
BridgedConfig cfg = BridgedConfig.load(configPath);
|
||||
// CB-594: report which secret env vars the config actually needs, by name, before anything
|
||||
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
|
||||
// boots fine either way — this is the only thing that says so out loud.
|
||||
reportRequiredSecrets(cfg);
|
||||
// CB-596: an absent (or empty) memberCredentials: block blocks NOTHING — no credential
|
||||
// name is hardcoded any more to fall back on. Say so loudly, the same way a missing
|
||||
// secret is reported above, so upgrading past this commit never silently drops CB-592's
|
||||
// protection.
|
||||
reportMemberCredentialsGap(cfg);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched bridged must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
cfg.validateCharters();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, BridgedConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, BridgedConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(agents, spaces,
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
// The cooldown is deferred (see BridgedConfig#quarantineCooldownSeconds): it is read once
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine);
|
||||
// CB-504: under supervision (launchd/systemd) bridged can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(BridgedConfig.Profile::workspace)
|
||||
.filter(Objects::nonNull)
|
||||
.collect(Collectors.toSet());
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
leads = new LeadTabScanner(herdr, tabToName, memberSpaces,
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, member spaces {} excluded)",
|
||||
tabToName.keySet(), scanIntervalSeconds, memberSpaces);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(agents, spaces, cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
// Compiled once at startup, keyed by profile name; a profile with no exhaustedPattern is
|
||||
// simply absent here, so its workers keep today's completion-fallback behaviour unchanged.
|
||||
Map<String, Pattern> exhaustedPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasExhaustedPattern()) {
|
||||
exhaustedPatternsByProfile.put(name, Pattern.compile(profile.exhaustedPattern()));
|
||||
}
|
||||
});
|
||||
ExhaustedPatternLookup exhaustedPatterns = target -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.map(profileName -> config.get().profiles().get(profileName))
|
||||
.ifPresent(profile -> {
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}' classified "
|
||||
+ "BACKEND_EXHAUSTED): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
});
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns, exhaustionSink);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.bridged.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Injector injector = new Injector(agents, turnListener, deliverableTo(presence, leads),
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block (with a uri) selects the AMQP-backed durable adapter;
|
||||
// absent, bridged stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox;
|
||||
if (cfg.broker() != null && cfg.broker().isConfigured()) {
|
||||
replyInbox = AmqpReplyInbox.open(cfg.broker().uri(), cfg.broker().prefetchOrDefault());
|
||||
log.info("reply inbox: AMQP broker (durable) at {} (prefetch={})",
|
||||
cfg.broker().uri(), cfg.broker().prefetchOrDefault());
|
||||
} else {
|
||||
replyInbox = new InMemoryReplyInbox();
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
}
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open bridge_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = BridgedMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, agents, replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, agents, replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(detail -> {
|
||||
// CB-578 stage C, acceptance criterion 10: a failed ticket's detail should tell a lead
|
||||
// where to re-dispatch onto the same tree, not just that the worker vanished.
|
||||
String reason = "the worker session was released before it replied";
|
||||
if (detail.worktreePath() != null) {
|
||||
reason += "; worktree=" + detail.worktreePath() + " branch=" + detail.branch()
|
||||
+ " snapshot=" + (detail.snapshotRef() != null ? detail.snapshotRef() : "none");
|
||||
}
|
||||
// CB-584 (issue #65 criterion 5): also name the agent session, so a lead can resume the
|
||||
// member's conversation instead of only re-dispatching a fresh one onto the same files.
|
||||
if (detail.agentSessionId() != null) {
|
||||
reason += " agentSessionId=" + detail.agentSessionId();
|
||||
}
|
||||
messages.abandon(detail.terminalId(), reason);
|
||||
replyInbox.release(detail.terminalId());
|
||||
primaryRegistry.forgetDelegation(detail.terminalId()); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): bridge_send/bridge_reply/bridge_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting bridged");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
BridgeMcp mcp = new BridgeMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new BridgeMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
new BridgeMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}),
|
||||
new BridgeMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine));
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
herdr.close();
|
||||
}));
|
||||
|
||||
Javalin app = new BridgedApp(herdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("bridged listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a spawned
|
||||
* member whose agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> member's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code BridgeMcp} marks presence
|
||||
* for every spawned member (worker and architect), deliberately, since that map doubles as the
|
||||
* member roster's availability signal and a lead counted there would show up as an available
|
||||
* member. So without the second disjunct a lead is permanently un-deliverable: every
|
||||
* lead→lead send sat on the gate for {@code READINESS_GRACE_POLLS} (~60s) and then failed
|
||||
* having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: which env vars the loaded config actually needs, and why — every non-{@code
|
||||
* subscription} profile's {@code tokenEnv} (a subscription profile never reads one, see
|
||||
* {@link BridgedConfig.Profile#isSubscription()}), plus every profile's {@code gitTokenEnv}
|
||||
* where set (opt-in). Derived from the config, not hard-coded, so a new profile is covered for
|
||||
* free. A var required by more than one profile is one entry naming every profile that needs
|
||||
* it. Deliberately excludes {@code auth.tokenEnv}: that one is already enforced loudly, by a
|
||||
* startup throw in {@code main()} — about 370 lines <em>below</em> this method's call site
|
||||
* ({@link #reportRequiredSecrets(BridgedConfig)}), not a few lines above it. That throw only
|
||||
* fires when {@code auth.mode: token} is configured; under the default loopback-trust mode it
|
||||
* never runs, and {@code auth.tokenEnv} is simply not required.
|
||||
*
|
||||
* <p>Package-private and pure (no I/O, no logging) so the derivation is unit-testable without
|
||||
* capturing log output; {@link #reportRequiredSecrets(BridgedConfig)} is the logging caller.
|
||||
*/
|
||||
static Map<String, List<String>> requiredSecretEnvVars(BridgedConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (!profile.isSubscription()) {
|
||||
requiredBy.computeIfAbsent(profile.tokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' tokenEnv");
|
||||
}
|
||||
if (profile.hasGitToken()) {
|
||||
requiredBy.computeIfAbsent(profile.gitTokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' gitTokenEnv");
|
||||
}
|
||||
});
|
||||
return requiredBy;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: log, by name only, which required env vars (see {@link #requiredSecretEnvVars}) are
|
||||
* set in the daemon's own process environment — the environment every profile's {@code
|
||||
* tokenEnv}/{@code gitTokenEnv} is read from at spawn time (see
|
||||
* {@code HerdrPeerLauncher.resolveEnv}). Never logs a value, a prefix, or a length.
|
||||
*
|
||||
* <p>A missing entry only warns — it must never refuse to start. A daemon that boots and says
|
||||
* what is wrong is strictly more useful than one that will not boot at all.
|
||||
*/
|
||||
private static void reportRequiredSecrets(BridgedConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = requiredSecretEnvVars(cfg);
|
||||
if (requiredBy.isEmpty()) {
|
||||
log.info("startup secrets: no profile references a token env var — nothing to check");
|
||||
return;
|
||||
}
|
||||
Map<String, String> env = System.getenv();
|
||||
requiredBy.forEach((varName, sources) -> {
|
||||
String value = env.get(varName);
|
||||
if (value != null && !value.isBlank()) {
|
||||
log.info("startup secret {}: set ({})", varName, String.join(", ", sources));
|
||||
} else {
|
||||
log.warn("startup secret {}: MISSING ({}) — the daemon will start anyway, and this "
|
||||
+ "failure stays invisible until a worker actually needs it. Fix "
|
||||
+ "${SHARED_ENV}/tools/secrets.sh and restart bridged from a LOGIN "
|
||||
+ "shell (see scripts/redeploy-bridged.sh).",
|
||||
varName, String.join(", ", sources));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: {@code known:} empty (block absent entirely, or present but empty) means {@link
|
||||
* BridgedConfig.MemberCredentials#blockedSet()} is empty too — every member pane inherits the
|
||||
* operator's whole secret store, unblocked, exactly the defect this ticket fixes. Unlike a
|
||||
* missing token ({@link #reportRequiredSecrets}), there is no name to point at: the point is
|
||||
* that the block itself is missing. Warn once at startup and say what to add; never refuse to
|
||||
* start over it — see {@link #reportRequiredSecrets} for why a daemon that boots and says
|
||||
* what is wrong beats one that will not boot at all.
|
||||
*
|
||||
* <p>Package-private so the test can capture the log directly, the same way {@link
|
||||
* #requiredSecretEnvVars} is exposed for {@link #reportRequiredSecrets}'s own test.
|
||||
*/
|
||||
static void reportMemberCredentialsGap(BridgedConfig cfg) {
|
||||
BridgedConfig.MemberCredentials creds = cfg.memberCredentials();
|
||||
if (creds != null && !creds.known().isEmpty()) {
|
||||
log.info("memberCredentials: {} known name(s), {} allowed — blocking {} on every spawn",
|
||||
creds.known().size(), creds.allow().size(), creds.blockedSet().size());
|
||||
return;
|
||||
}
|
||||
log.warn("memberCredentials: absent or empty — the daemon will start anyway, and every "
|
||||
+ "member pane inherits the operator's WHOLE secret store, unblocked (CB-592's "
|
||||
+ "protection is lost). Add a memberCredentials: block (policy/allow/known) to "
|
||||
+ "bridged.yaml — see bridged.example.yaml — and restart.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Bridged() {
|
||||
}
|
||||
}
|
||||
@@ -1,21 +0,0 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
}
|
||||
@@ -1,291 +0,0 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link BridgedConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Bridged.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Bridged.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<BridgedConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<BridgedConfig> current;
|
||||
|
||||
public ConfigRef(Path path, BridgedConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(BridgedConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public BridgedConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
/** The file this ref reloads from, or {@code null} for a {@link #fixed} ref. */
|
||||
public Path path() {
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
public String summary() {
|
||||
if (error != null) {
|
||||
return "config reload refused — " + error;
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart bridged to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
}
|
||||
return "config reloaded";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the file, validate it, and swap it in when nothing cold changed.
|
||||
*
|
||||
* <p>Never throws: a reload is a best-effort operation on a daemon that is already serving, and
|
||||
* a bad edit must not take it down. Every failure path leaves the previous config live and is
|
||||
* reported through the returned {@link Outcome}.
|
||||
*/
|
||||
public Outcome reload() {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
BridgedConfig old = current.get();
|
||||
BridgedConfig fresh;
|
||||
try {
|
||||
fresh = BridgedConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
return Outcome.failed(msg);
|
||||
}
|
||||
|
||||
List<String> cold = changedColdKeys(old, fresh);
|
||||
if (!cold.isEmpty()) {
|
||||
Outcome out = Outcome.refusedCold(cold);
|
||||
log.warn(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
}
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
if (!Objects.equals(old.auth(), fresh.auth())) {
|
||||
changed.add("auth");
|
||||
}
|
||||
// Kept in step with COLD_KEYS so the doc and the code cannot drift apart silently.
|
||||
assert COLD_KEYS.containsAll(changed) : "a cold key was reported that COLD_KEYS omits";
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(BridgedConfig old, BridgedConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
}
|
||||
if (!Objects.equals(old.leadHeartbeat(), fresh.leadHeartbeat())) {
|
||||
changed.add("leadHeartbeat");
|
||||
}
|
||||
if (!Objects.equals(old.guard(), fresh.guard())) {
|
||||
changed.add("guard");
|
||||
}
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, BridgedConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, BridgedConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
if (!before.keySet().equals(after.keySet())) {
|
||||
Set<String> diff = new LinkedHashSet<>(before.keySet());
|
||||
diff.addAll(after.keySet());
|
||||
diff.removeIf(p -> before.containsKey(p) && after.containsKey(p));
|
||||
changed.add("profiles (added/removed: " + String.join(", ", diff) + ")");
|
||||
}
|
||||
// An EXISTING profile's launch settings are deferred too, and this is easy to get wrong:
|
||||
// `HerdrPeerLauncher` takes `Map.copyOf(profiles)` at construction and `spawn` resolves the
|
||||
// profile out of that snapshot, so a reloaded model/baseUrl/argv/env never reaches a launch.
|
||||
// Only weight and maxLoad are genuinely hot, because placement reads them through the
|
||||
// supplier on the composite rather than from the adapter's copy. Without this check a
|
||||
// changed model would report "config reloaded" and silently do nothing — the worst outcome
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
BridgedConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
});
|
||||
if (!relaunch.isEmpty()) {
|
||||
changed.add("profiles." + String.join("/", relaunch) + " launch settings "
|
||||
+ "(model, baseUrl, argv, env, …) — the launcher holds a startup snapshot");
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(BridgedConfig.Profile a, BridgedConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
&& Objects.equals(a.argv(), b.argv())
|
||||
&& Objects.equals(a.placement(), b.placement())
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Bridged.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
}
|
||||
}
|
||||
@@ -1,151 +0,0 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long intervalSeconds;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
|
||||
// These facts need the evidence publishers introduced by later M4 units. They are not negatives.
|
||||
private static final boolean NOT_YET_OBSERVED = false;
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code BridgeMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<Agent> agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, NOT_YET_OBSERVED,
|
||||
messages.hasInboxMessage(session.terminalId()), agent != null, NOT_YET_OBSERVED,
|
||||
NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
clock.getAsLong());
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// A list failure is health evidence, and must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
@@ -1,59 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
* identity (CB-105). Given the PID that opened an MCP connection, {@link #terminalForPid} finds
|
||||
* the agent pane whose process tree contains it, so {@code bridged} can tell <em>which worker</em>
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane owns it (e.g. the caller is the primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(paneId, pid)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private boolean paneOwnsPid(String paneId, long pid) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -1,372 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* The CB-106 completion fallback: bridges the {@link Injector}'s turn-completion signal to the
|
||||
* {@link Rendezvous} so a blocking {@code bridge_send} resolves even when the worker finishes its
|
||||
* task without ever calling {@code bridge_reply} — the common case for a real delegated coding task.
|
||||
*
|
||||
* <p>On a confirmed {@code working → idle} boundary it scrapes the worker's recent transcript and
|
||||
* resolves the awaiting send with that tail (a {@link Rendezvous.Kind#COMPLETION} resolution, so the
|
||||
* caller can tell a scrape from a structured reply). It scrapes only when a send is actually waiting
|
||||
* — a fleet worker's own turns, or a send that already timed out, cost no herdr traffic. An explicit
|
||||
* {@code bridge_reply} that raced in first wins; {@link Rendezvous#resolveCompletion} is then a no-op.
|
||||
*
|
||||
* <p>It also handles the CB-109 stall signal ({@link #onTurnFailed}): a worker that ran a turn then
|
||||
* wedged in an {@code unknown} state resolves the send as a failure (with the error screen as
|
||||
* context) rather than leaving it to time out.
|
||||
*
|
||||
* <p>The scrape is cleaned to the last {@code ⏺} assistant block (stripping TUI chrome) and guarded
|
||||
* against misattribution (CB-115): the pane content is baselined on delivery ({@link #onDelivered}),
|
||||
* and a completion whose scrape is unchanged from that baseline — the previous turn's wind-down
|
||||
* sampled as this turn's boundary on a rapid back-to-back send — is suppressed rather than resolving
|
||||
* the send with a stale answer.
|
||||
*
|
||||
* <p><strong>Waiter-specific resolution (CB-116).</strong> On delivery we also capture the exact
|
||||
* {@link Rendezvous} waiter this turn belongs to, and the completion/failure fallbacks resolve
|
||||
* <em>that</em> waiter — never "whatever send is waiting now". A completion fallback runs on a virtual
|
||||
* thread and can land after the worker's {@code bridge_reply} already resolved the turn and the
|
||||
* <em>next</em> send opened its own waiter on the same session; resolving the current waiter would
|
||||
* then deliver turn N's stale scrape as turn N+1's answer. Targeting the captured waiter makes a late
|
||||
* completion a harmless no-op (its waiter is already done) instead of a cross-turn stale reply.
|
||||
*
|
||||
* <p>Wired as the {@link Injector}'s {@link TurnListener}; the handlers hand off to a virtual thread
|
||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||
*/
|
||||
public final class CompletionResolver implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source for the completion scrape. {@code recent} returns the tail of
|
||||
* the transcript (the worker's last output), which is what a delegator wants when the worker
|
||||
* didn't structure a reply.
|
||||
*/
|
||||
static final String SCRAPE_SOURCE = "recent";
|
||||
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call bridge_reply.]";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
* delivering send opened, plus the assistant block present when it was delivered.
|
||||
*
|
||||
* <p>The {@code waiter} is what makes a late fallback safe (CB-116): we resolve it, not "whoever
|
||||
* is waiting now", so a completion that fires after the next send has opened its own waiter is a
|
||||
* no-op rather than a cross-turn stale reply. The {@code baseline} is the CB-115 staleness
|
||||
* reference: a completion scrape equal to it means the worker produced no new output (the previous
|
||||
* turn's wind-down sampled as this boundary), so it is suppressed. Overwritten on each delivery;
|
||||
* cleared when the turn resolves. Package-private so tests can capture and replay a specific turn.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* @param exhaustedPatterns CB-578 stage A: per-target lookup for a profile's configured
|
||||
* usage-limit refusal pattern. Required — there is deliberately no
|
||||
* defaulting overload; a caller that does not want the classification
|
||||
* must pass an explicit inert value ({@link ExhaustedPatternLookup#none()}).
|
||||
* @param exhaustionSink CB-578 stage B: notified when a {@code BACKEND_EXHAUSTED}
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target, token);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target, TurnToken token) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
}
|
||||
String baseline;
|
||||
try {
|
||||
// Clip to the same cap resolve() applies to the tail (line ~134): the CB-115 misattribution
|
||||
// guard compares baseline.equals(tail), so both sides must be the same capped representation.
|
||||
// An unclipped baseline vs a clipped tail would never match for a >MAX_SCRAPE_CHARS block,
|
||||
// defeating the guard and letting a stale completion resolve the send.
|
||||
baseline = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
InFlight inFlight(String target) {
|
||||
return inFlight.get(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
// Read the in-flight turn on the poller thread — before any next-turn delivery can overwrite
|
||||
// it — then off-load the scrape (a herdr round-trip we must not block polling on) to a vthread.
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("completion-" + target).start(() -> resolve(target, turn));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the completed turn before adapter housekeeping can erase its rendered output. This is
|
||||
* intentionally synchronous and used only when a post-turn context reset is enabled; the normal
|
||||
* path remains off-loaded so polling is not blocked by a scrape.
|
||||
*/
|
||||
public void resolveBeforePostAction(String target) {
|
||||
resolve(target, inFlight.get(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, null));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, reason));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
void resolve(String target, InFlight turn) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = turn == null ? null : turn.waiter();
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
// Nobody is blocked on THIS turn (it had no send, or its bridge_reply already won). Skip
|
||||
// the scrape; resolving the current waiter here would be the CB-116 cross-turn stale reply.
|
||||
inFlight.remove(target, turn);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
} catch (RuntimeException e) {
|
||||
// The worker finished but we couldn't read its screen — still resolve the send so the
|
||||
// caller unblocks; an empty tail beats hanging until the caller's timeout.
|
||||
log.warn("completion scrape for {} failed; resolving with an empty tail: {}",
|
||||
target, e.getMessage());
|
||||
tail = "";
|
||||
scrapeFailed = true;
|
||||
}
|
||||
// Misattribution guard (CB-115): if the scrape is byte-identical to the pane content at
|
||||
// delivery, this turn produced no new output — the boundary belongs to the previous turn's
|
||||
// wind-down (common on rapid back-to-back sends). Suppress rather than resolve the send with
|
||||
// a stale answer; the real bridge_reply (or a later genuine completion) resolves it instead.
|
||||
// A scrape that failed to read is exempt — an empty tail there is "couldn't see", not "no change".
|
||||
String baseline = turn.baseline();
|
||||
if (!scrapeFailed && baseline != null && baseline.equals(tail)) {
|
||||
log.debug("suppressing misattributed completion for {} (no output change since delivery)",
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
}
|
||||
// CB-578 stage A: a turn that ended with no bridge_reply AND whose scrape matches the
|
||||
// backend's configured usage-limit pattern is a refusal, not an answer. Classify it as
|
||||
// BACKEND_EXHAUSTED rather than handing the caller a scrape that reads like a real reply.
|
||||
if (!scrapeFailed) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(assistantBlock, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED (no bridge_reply; scrape "
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
log.warn("completion scrape for {} clipped from {} chars to the {} char cap; "
|
||||
+ "member did not call bridge_reply, so the pane tail is partial",
|
||||
target, originalLength, MAX_SCRAPE_CHARS);
|
||||
}
|
||||
log.debug("resolved send to {} via turn-completion fallback ({} chars scraped)",
|
||||
target, tail.length());
|
||||
}
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
}
|
||||
|
||||
/** Synchronous fail with an optional reason supplied by a dropped worker queue. */
|
||||
void fail(String target, InFlight turn, String explicitReason) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
CompletableFuture<Rendezvous.Resolution> waiter =
|
||||
turn != null ? turn.waiter() : rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason = explicitReason;
|
||||
if (reason == null || reason.isBlank()) {
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The first line of {@code text} matching {@code pattern}, stripped — the CB-578 stage A
|
||||
* evidence carried in a {@code BACKEND_EXHAUSTED} reason so the operator sees the real refusal
|
||||
* text, never a generic label. {@code null} if no line matches.
|
||||
*/
|
||||
static String firstMatchingLine(String text, Pattern pattern) {
|
||||
if (text == null || text.isEmpty()) return null;
|
||||
for (String line : text.split("\n", -1)) {
|
||||
if (pattern.matcher(line).find()) {
|
||||
return line.strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* the way {@link dev.ltms.bridged.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
return unconfigured.isEmpty()
|
||||
? "full (all profiles configured: " + sorted(allProfiles) + ")"
|
||||
: "partial (configured: " + sorted(configuredProfiles) + "; not configured: " + sorted(unconfigured) + ")";
|
||||
}
|
||||
|
||||
private static List<String> sorted(Set<String> names) {
|
||||
return names.stream().sorted().toList();
|
||||
}
|
||||
|
||||
private static String clip(String s) {
|
||||
if (s == null) return "";
|
||||
String trimmed = s.strip();
|
||||
return trimmed.length() <= MAX_SCRAPE_CHARS
|
||||
? trimmed
|
||||
: trimmed.substring(trimmed.length() - MAX_SCRAPE_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the last assistant message from a raw Claude Code pane scrape (CB-115). Claude Code
|
||||
* prefixes each assistant turn with {@code ⏺}; the delegator wants that answer, not the TUI
|
||||
* chrome around it. Take everything from the final {@code ⏺} onward and stop at the <em>first</em>
|
||||
* hard interface boundary below it — the spinner/status line, input box, {@code ❯} prompt (which
|
||||
* may echo the <em>next</em> turn's text), footer, or tips/warnings. Stopping at the first
|
||||
* boundary (rather than trimming only trailing chrome) is what keeps a following turn's echoed
|
||||
* prompt out of this reply. Blank lines are not boundaries, so a multi-paragraph answer survives;
|
||||
* trailing blanks are trimmed at the end. With no {@code ⏺} marker (an unusual render) the whole
|
||||
* text is scanned the same way, so we never lose the reply.
|
||||
*
|
||||
* <p>Package-private and pure so it is unit-testable without herdr.
|
||||
*/
|
||||
static String lastAssistantBlock(String raw) {
|
||||
if (raw == null || raw.isBlank()) return "";
|
||||
int marker = raw.lastIndexOf('⏺');
|
||||
String block = marker >= 0 ? raw.substring(marker + 1) : raw;
|
||||
StringBuilder out = new StringBuilder();
|
||||
int kept = 0;
|
||||
for (String line : block.split("\n", -1)) {
|
||||
if (isBoundary(line)) break; // first TUI boundary ends the assistant message
|
||||
if (kept++ > 0) out.append('\n');
|
||||
out.append(line);
|
||||
}
|
||||
return out.toString().strip();
|
||||
}
|
||||
|
||||
/**
|
||||
* A hard TUI boundary line that marks the end of an assistant message and the start of interface
|
||||
* chrome (input box, prompt, spinner, footer, tips/warnings). Blank lines are <em>not</em>
|
||||
* boundaries — an answer may contain them — so they are kept and trimmed only if trailing.
|
||||
*/
|
||||
private static boolean isBoundary(String line) {
|
||||
String t = line.strip();
|
||||
if (t.isEmpty()) return false;
|
||||
// A horizontal rule / all box-drawing separators (e.g. "──────").
|
||||
if (t.chars().allMatch(c -> c == '─' || c == '—' || c == '━' || c == '═' || c == '-')) {
|
||||
return true;
|
||||
}
|
||||
String lower = t.toLowerCase();
|
||||
return t.startsWith("╭") || t.startsWith("│") || t.startsWith("╰") || t.startsWith("┌")
|
||||
|| t.startsWith("└") || t.startsWith("❯") || t.startsWith("⏵")
|
||||
|| t.startsWith("⎿") || t.startsWith("⚠")
|
||||
// Status/spinner lines Claude Code renders below a settled or in-flight turn,
|
||||
// e.g. "✻ Baked for 21s", "✶ Forming…".
|
||||
|| t.startsWith("✻") || t.startsWith("✳") || t.startsWith("✽") || t.startsWith("·")
|
||||
|| t.startsWith("●") || t.startsWith("◐") || t.startsWith("✢") || t.startsWith("✶")
|
||||
|| lower.contains("auto mode") || lower.contains("for shortcuts")
|
||||
|| lower.contains("esc to interrupt") || lower.contains("bypass permissions");
|
||||
}
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a {@code BACKEND_EXHAUSTED}
|
||||
* classification to a waiting send (CB-578 stage B) — never on a race that lost (see
|
||||
* {@link CompletionResolver#resolve}, which only calls this after
|
||||
* {@code Rendezvous.resolveExhausted} returns {@code true}).
|
||||
*
|
||||
* <p>{@link CompletionResolver} knows only {@code target} (a herdr terminal id); it has no notion of
|
||||
* profiles or credentials, so mapping {@code target} to whatever should be quarantined is entirely
|
||||
* the sink's job — see {@code Bridged.main}'s wiring, which resolves target → session → profile →
|
||||
* {@code effectiveCredentialId()} and calls {@code BackendQuarantine.quarantine} on it.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustionSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
*/
|
||||
void onExhausted(String target, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on exhaustion. The explicit stand-in a caller (or a test not
|
||||
* exercising this feature) passes instead of a defaulting overload, exactly like
|
||||
* {@link ExhaustedPatternLookup#none()}.
|
||||
*/
|
||||
static ExhaustionSink none() {
|
||||
return (target, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -1,66 +0,0 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
|
||||
/**
|
||||
* Resolves <em>who is calling</em> an MCP tool from the connection alone — the anti-spoofing
|
||||
* identity model of the MCP contract. It ties the connection's loopback peer PID (from the OS)
|
||||
* to a herdr agent pane (from herdr), yielding the caller's worker {@code terminal_id}. A caller
|
||||
* that maps to no worker pane — the primary, or an off-host client — resolves to {@code null}.
|
||||
*
|
||||
* <p>Both sources are authoritative and unforgeable: the OS reports the real connecting PID, and
|
||||
* herdr owns the PID→pane mapping. A worker cannot claim to be another worker, nor the primary.
|
||||
* Single-host only (the herd shares the {@code bridged} host); the token path is the split-host
|
||||
* fallback.
|
||||
*/
|
||||
public final class ConnectionIdentity {
|
||||
|
||||
private final PaneLocator panes;
|
||||
private final PeerPidLookup pids;
|
||||
private final ProcessCwdLookup cwds;
|
||||
|
||||
/** Identity only (no cwd resolution — {@link #cwdForPid} returns {@code null}). */
|
||||
public ConnectionIdentity(PaneLocator panes, PeerPidLookup pids) {
|
||||
this(panes, pids, _ -> null);
|
||||
}
|
||||
|
||||
/** Identity plus cwd resolution (CB-112 — inherit the primary's directory on spawn). */
|
||||
public ConnectionIdentity(PaneLocator panes, PeerPidLookup pids, ProcessCwdLookup cwds) {
|
||||
this.panes = panes;
|
||||
this.pids = pids;
|
||||
this.cwds = cwds;
|
||||
}
|
||||
|
||||
/**
|
||||
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
||||
*/
|
||||
public record Caller(String terminal, long pid) {
|
||||
}
|
||||
|
||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||
public Caller resolve(String remoteAddr, int remotePort) {
|
||||
if (!isLoopback(remoteAddr)) {
|
||||
return new Caller(null, -1); // only same-host callers can be workers
|
||||
}
|
||||
long pid = pids.pidForLocalPort(remotePort);
|
||||
return new Caller(panes.terminalForPid(pid), pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* The calling worker's {@code terminal_id}, or {@code null} if the caller is not a known
|
||||
* on-host worker (treat as the primary).
|
||||
*/
|
||||
public String callerTerminal(String remoteAddr, int remotePort) {
|
||||
return resolve(remoteAddr, remotePort).terminal();
|
||||
}
|
||||
|
||||
/** The working directory of {@code pid} (the primary's cwd on an MCP spawn), or {@code null}. */
|
||||
public String cwdForPid(long pid) {
|
||||
return pid > 0 ? cwds.cwdForPid(pid) : null;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String addr) {
|
||||
return "127.0.0.1".equals(addr) || "::1".equals(addr) || "0:0:0:0:0:0:0:1".equals(addr);
|
||||
}
|
||||
}
|
||||
@@ -1,373 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>Claude Code</strong> — the safe path from a
|
||||
* delegation request to a running off-subscription Claude.
|
||||
*
|
||||
* <p>Everything transport-related (tab/pane placement, the CB-306 spawn-readiness gate, unique
|
||||
* naming, CB-117 orphan reap, teardown, listing, cwd resolution) lives in the base. This class
|
||||
* supplies only the two Claude-specific seams:
|
||||
* <ul>
|
||||
* <li>the {@code claude} name prefix (so reap matches {@code claude-*} panes, never another
|
||||
* adapter's), and</li>
|
||||
* <li>{@link #buildLaunch}, which encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em> touching
|
||||
* herdr, and mount the bridge MCP + reply charter as inline launch flags. A worker's base_url
|
||||
* lives in the env map handed to herdr and nowhere else; {@code bridged}'s own environment is
|
||||
* never mutated, and nothing is written to the worker's profile.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "claude";
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so
|
||||
* existing deployments and tests keep the legacy non-blocking spawn semantics.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with spawn-ready gate enabled. The gate polls {@code agents.status()}
|
||||
* until the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet, memberCredentials);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply
|
||||
* fakes for the clock ({@code nowMillis}) and poll-loop wait ({@code sleeper}). The
|
||||
* {@code sleeper} is never called when the gate is disabled ({@code spawnReadyTimeoutMs == 0}).
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param guard subscription-boundary guard (checked before spawning)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (e.g. {@code () -> Thread.sleep(pollMs)}); it
|
||||
* already encodes the poll interval, so the 8th positional argument
|
||||
* (poll ms) is accepted for API symmetry but otherwise unused here
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus an injectable host-env-names source for the CB-596
|
||||
* criterion-4 gap detector. Test seam only — every production call site leaves this at the
|
||||
* default (the real {@code System.getenv()} key set) via the constructor above.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: assert the profile's base_url is on
|
||||
* the allowlist <em>before</em> any herdr call, then build the worker env with
|
||||
* {@code ANTHROPIC_*}, the parity-neutral git-forge grant, and the bridge MCP + reply charter
|
||||
* mounted as inline launch flags. When the request carries session identity (CB-547a) it is
|
||||
* applied here — see {@link #applySessionIdentity}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
// CB-539: a profile may deliberately opt into the subscription (subscription: true) when no
|
||||
// off-subscription endpoint exists for it — e.g. `sonnet` on `ccs`. That profile gets no
|
||||
// ANTHROPIC_BASE_URL/AUTH_TOKEN (there is nothing to point them at) and the guard's base_url
|
||||
// requirement is skipped FOR IT ONLY. Every other profile keeps the hard boundary below.
|
||||
boolean onSubscription = cfg.isSubscription();
|
||||
String baseUrl = cfg.baseUrl();
|
||||
|
||||
if (onSubscription) {
|
||||
// NO SILENT CONTRADICTION: subscription:true + a baseUrl state opposite intents; refuse
|
||||
// loudly rather than pick a winner.
|
||||
if (baseUrl != null && !baseUrl.isBlank()) {
|
||||
throw new IllegalStateException("profile '" + cfg.profile()
|
||||
+ "' sets both subscription: true and a baseUrl ('" + baseUrl + "') — the two "
|
||||
+ "are contradictory: a subscription profile must not point at an endpoint. "
|
||||
+ "Drop baseUrl, or drop subscription: true.");
|
||||
}
|
||||
// Visible without anyone going looking for it: this worker bills the subscription.
|
||||
log.warn("spawning profile '{}' on the Claude subscription (subscription: true) — this "
|
||||
+ "worker WILL bill the operator's subscription", cfg.profile());
|
||||
} else {
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
}
|
||||
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
if (onSubscription) {
|
||||
// CB-542 belt-and-braces: on the subscription path no guard vets these two keys, and the
|
||||
// profile's env: is layered in by baseEnv — so strip any that rode in there. Config load
|
||||
// already rejects this (loudly, naming the profile); this makes the boundary hold even
|
||||
// for a profile built in code that never passed through that validation.
|
||||
workerEnv.remove("ANTHROPIC_BASE_URL");
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so bridged chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
// handle is known BEFORE the agent has written anything; a resume spawn adopts its prior
|
||||
// id via -r and passes no --session-id (the two conflict). Both are injected before the
|
||||
// model flag so --model keeps outranking the operator's own argv.
|
||||
// mutableArgv: argvWithBridge may hand back the profile's own (immutable) List.of when it
|
||||
// has neither MCP nor a charter — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithBridge(cfg, spec.charter()));
|
||||
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
|
||||
return new Launch(workerEnv, argvWithModel(argv, cfg), agentSessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
}
|
||||
if (resuming) {
|
||||
argv.add("-r");
|
||||
argv.add(resumeSessionId);
|
||||
return resumeSessionId;
|
||||
}
|
||||
String minted = UUID.randomUUID().toString();
|
||||
argv.add("--session-id");
|
||||
argv.add(minted);
|
||||
return minted;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus an inline {@code --mcp-config} when {@code worker.mcpUrl} is set and
|
||||
* {@code --append-system-prompt} when the base composed a charter. Neither touches the profile's
|
||||
* config; both are pure command-line flags. This inline-flag mount is Claude Code specific —
|
||||
* other adapters mount MCP and instructions their own way.
|
||||
*/
|
||||
private List<String> argvWithBridge(BridgedConfig.Profile cfg, String charter) {
|
||||
if (!cfg.hasMcp() && charter == null) {
|
||||
return cfg.argv();
|
||||
}
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
if (cfg.hasMcp()) {
|
||||
String mcpJson = "{\"mcpServers\":{\"bridge\":{\"type\":\"http\",\"url\":\""
|
||||
+ cfg.mcpUrl() + "\"}}}";
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpJson);
|
||||
}
|
||||
if (charter != null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(charter);
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin the model on the command line as well as in {@code ANTHROPIC_MODEL} (CB-533).
|
||||
*
|
||||
* <p>The env var alone is not a reliable pin for this adapter, because the argv is usually a
|
||||
* launcher rather than {@code claude} itself — {@code ["ccs", "<profile>"]} — and {@code ccs}
|
||||
* exports its profile's own model family ({@code ANTHROPIC_MODEL}, {@code DEFAULT_OPUS/SONNET/
|
||||
* HAIKU}, {@code CLAUDE_CODE_SUBAGENT_MODEL}) over whatever it inherited. A worker profile that
|
||||
* set {@code model:} therefore got silently overruled by its own launcher. Claude Code's
|
||||
* {@code --model} flag outranks the environment, and {@code ccs <profile> [claude-args...]}
|
||||
* passes trailing arguments through, so the flag survives the wrapper.
|
||||
*
|
||||
* <p>Appended last so it also outranks anything in the operator's own {@code argv}. Profiles
|
||||
* that deliberately leave {@code model:} unset (letting {@code ccs} own model selection, as
|
||||
* {@code gx10} does) are untouched — this adds nothing when there is nothing to add. This is
|
||||
* the {@code kind: claude} counterpart of the opencode adapter's {@code -m provider/model}.
|
||||
*/
|
||||
private static List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() == null || cfg.model().isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withModel = mutableArgv(argv);
|
||||
withModel.add("--model");
|
||||
withModel.add(cfg.model());
|
||||
return withModel;
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.CONTEXT_RESET, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_NAME, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
String target = agentTarget(id);
|
||||
if (target == null) {
|
||||
return false;
|
||||
}
|
||||
// This deliberately bypasses Injector: /clear is housekeeping, not a delegated turn.
|
||||
agents().send(target, "/clear");
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (Claude prefix), kept for direct unit testing -------------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a Claude Code bridge worker started by a <em>different</em> process
|
||||
* than {@code currentNonce}. A thin {@code claude}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -1,995 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.HerdrException;
|
||||
import dev.ltms.bridged.herdr.Tab;
|
||||
import dev.ltms.bridged.herdr.Workspace;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.CharterReceipt;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ConcurrentMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Abstract base for {@link PeerLauncher} adapters that materialize a peer as a <em>herdr</em>
|
||||
* agent (a CLI coding agent running in a herdr tab/pane). It owns everything that is the same
|
||||
* regardless of <em>which</em> coding agent runs: tab/pane placement, the CB-306 spawn-readiness
|
||||
* gate, unique naming, CB-117 orphan reap, teardown, {@link #list() listing}, and cwd resolution.
|
||||
*
|
||||
* <p>Two seams are peer-specific and supplied by the concrete adapter:
|
||||
* <ul>
|
||||
* <li>{@code namePrefix} (constructor arg) — the label prefix ({@code claude}, {@code opencode})
|
||||
* that drives both unique naming and the orphan-reap pattern, so each adapter reaps only its
|
||||
* own kind of pane and never another's.</li>
|
||||
* <li>{@link #buildLaunch(BridgedConfig.Profile, LaunchSpec)} — the peer-specific env map + argv, including any
|
||||
* subscription/guard check, MCP mount, and instruction injection. The base never sees how the
|
||||
* peer is configured; it only places and starts the returned {@link Launch}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Placement: in the default {@code tab} policy a peer lands in its own tab inside a dedicated
|
||||
* worker space (found-or-created once, then shared), so peers never split or clutter the user's
|
||||
* real work spaces. Teardown removes the peer's pane <em>and</em> its now-empty tab, tolerating an
|
||||
* already-gone peer so a repeated DELETE is harmless.
|
||||
*/
|
||||
public abstract class HerdrPeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
|
||||
/** herdr rejects a duplicate agent {@code name}; we retry a bumped name this many times. */
|
||||
private static final int NAME_RETRIES = 8;
|
||||
|
||||
/**
|
||||
* Retries for {@code agent.start} against a seed pane whose shell has not reached its prompt
|
||||
* yet — {@code tab.create}/{@code pane.split} return as soon as the pane exists, and herdr
|
||||
* refuses to start an agent in a pane that is not "an available shell" ({@code agent_pane_busy}).
|
||||
*/
|
||||
private static final int SHELL_READY_RETRIES = 20;
|
||||
|
||||
private final String namePrefix; // label prefix: naming + reap scheme
|
||||
private final AgentControl agents;
|
||||
private final WorkspaceControl spaces;
|
||||
private final Map<String, BridgedConfig.Profile> profiles; // profile name → spawn settings
|
||||
private final String defaultProfile; // profile a no-arg spawn uses (nullable)
|
||||
|
||||
/** Host env lookup (injectable for tests); adapters read it in {@link #buildLaunch}. */
|
||||
protected final Function<String, String> env;
|
||||
|
||||
private final AtomicLong nameSeq = new AtomicLong(); // per-peer counter (herdr agent names only)
|
||||
|
||||
/**
|
||||
* Live fleet config, read once per spawn. A null supplier or value leaves tab labels at their
|
||||
* default and supplies no role charter. A profile's own {@code tabLabel} still overrides it.
|
||||
*
|
||||
* <p>CB-559: a supplier rather than a snapshot, so a config reload affects the next launch
|
||||
* without a restart. Existing tabs keep the label they were given.
|
||||
*/
|
||||
private final Supplier<BridgedConfig.Fleet> fleet;
|
||||
|
||||
/**
|
||||
* CB-596: the live {@code memberCredentials:} policy, read once per spawn (same hot-reload shape
|
||||
* as {@link #fleet}). {@code null} — either the supplier itself, or what it returns — means no
|
||||
* policy is configured and {@link #applyMemberCredentialPolicy} shadows nothing.
|
||||
*/
|
||||
private final Supplier<BridgedConfig.MemberCredentials> memberCredentials;
|
||||
|
||||
/**
|
||||
* Enumerates the daemon's own process environment variable NAMES ONLY, never values — the CB-596
|
||||
* criterion-4 gap detector's data source (see {@link #logCredentialGap}). Injectable for tests;
|
||||
* production always resolves to the real {@code System.getenv()} key set.
|
||||
*
|
||||
* <p>Deliberately the daemon's own environment, not the spawned pane's: nothing in the herdr
|
||||
* client surface lets the daemon read back an arbitrary command's output from a pane before the
|
||||
* peer starts in it, so there is no channel to inspect the pane's environment directly. The
|
||||
* daemon's own process is started the same way (a login shell sourcing the same secret store —
|
||||
* see CB-592's investigation of {@code secrets.sh}), so on a single-host deployment its env
|
||||
* mirrors what the pane's login shell is about to export.
|
||||
*/
|
||||
private final Supplier<Set<String>> hostEnvNames;
|
||||
|
||||
/** The final instruction always requires a bridge reply when the bridge MCP is mounted. */
|
||||
protected static final String REPLY_CHARTER =
|
||||
"You are a spawned member in the claude-bridge fleet. Every message you receive arrives "
|
||||
+ "through the bridge, and the ONLY channel back to the sender is the bridge_reply MCP tool. "
|
||||
+ "Text you write in your terminal is NOT sent anywhere — the sender cannot see your screen, "
|
||||
+ "so an in-terminal answer is silently discarded. Therefore you MUST end EVERY turn by calling "
|
||||
+ "bridge_reply with `content` set to your complete response. This holds for every message without "
|
||||
+ "exception — tasks, questions, clarifications, acknowledgements, and ordinary back-and-forth "
|
||||
+ "conversation. Call bridge_reply exactly once, as the final action of your turn, with your full "
|
||||
+ "answer in `content`; never wait for confirmation first. If you end a turn without calling "
|
||||
+ "bridge_reply, the sender receives nothing and the exchange stalls.";
|
||||
|
||||
/**
|
||||
* Tab numbers, counted per {@code role/profile} pair (CB-557).
|
||||
*
|
||||
* <p>Deliberately not {@link #nameSeq}. That counter is shared by every profile this launcher
|
||||
* serves, because its job is to make herdr <em>agent names</em> unique. Reusing it for the tab
|
||||
* label made the numbers global, so sibling tabs read {@code #4}, {@code #9}, {@code #17} — gaps
|
||||
* that look like a member died. Counting per role+profile makes {@code dev: sonnet #2} mean the
|
||||
* second sonnet dev, which is what a reader assumes it means.
|
||||
*
|
||||
* <p>Resets when the daemon restarts, and that is fine: the label is a human-facing hint, not an
|
||||
* identity. Identity is {@link PeerHandle#id()}.
|
||||
*/
|
||||
private final ConcurrentMap<String, AtomicLong> labelSeq = new ConcurrentHashMap<>();
|
||||
|
||||
private final long spawnReadyTimeoutMs; // 0 = disable gate (legacy non-blocking spawn)
|
||||
private final LongSupplier nowMillis; // monotonic clock (injectable for tests)
|
||||
private final Runnable sleeper; // sleep/wait hook (injectable for tests; never real-sleep in unit tests)
|
||||
|
||||
// Per-process token mixed into each peer name so a fresh process (nameSeq back at 0) cannot
|
||||
// collide with same-profile peers that outlived a restart. See startUniquelyNamed.
|
||||
private final String nameNonce = String.format("%06x", new SecureRandom().nextInt(1 << 24));
|
||||
|
||||
// CB-519: PeerHandle.id() is a host-unique opaque UUID, decoupled from the herdr pane id. The
|
||||
// routing/registry key is the UUID; the herdr pane id is a launcher-private placement/teardown
|
||||
// coordinate. This map bridges the two so stop(id) can resolve a host-unique key back to the
|
||||
// exact pane it must tear down. The pane id is launcher-private (never the routing key) — see
|
||||
// PeerHandle.id().
|
||||
private final ConcurrentMap<String, String> paneByAgentId = new ConcurrentHashMap<>();
|
||||
private final AtomicBoolean resetUnsupportedLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* @param namePrefix label prefix for this peer kind (drives naming and reap)
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured peer profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (never called when the gate is disabled); the poll
|
||||
* interval is baked into this hook, so the base needs no poll field
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(namePrefix, agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus the live {@code fleet} config (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once per spawn; {@code null} ⇒ default tab label and no
|
||||
* role charter. A separate constructor rather than a new parameter on the one
|
||||
* above, so every existing call site keeps the default without an edit.
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(namePrefix, agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, fleet, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus the live {@code memberCredentials} policy (CB-596).
|
||||
*
|
||||
* @param memberCredentials live member-credential policy, read once per spawn; {@code null} ⇒
|
||||
* no policy configured, so a spawn shadows nothing. A separate
|
||||
* constructor rather than a new parameter on the one above, so every
|
||||
* existing call site keeps the pre-CB-596 default without an edit.
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
this(namePrefix, agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, fleet, memberCredentials, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus an injectable {@link #hostEnvNames} source for the CB-596 gap detector. Test
|
||||
* seam only — every production call site leaves this {@code null} and gets the real host env.
|
||||
*/
|
||||
protected HerdrPeerLauncher(String namePrefix, AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
this.fleet = fleet;
|
||||
this.namePrefix = namePrefix;
|
||||
this.agents = agents;
|
||||
this.spaces = spaces;
|
||||
this.profiles = Map.copyOf(profiles);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.env = env;
|
||||
this.spawnReadyTimeoutMs = spawnReadyTimeoutMs;
|
||||
this.nowMillis = nowMillis;
|
||||
this.sleeper = sleeper;
|
||||
this.memberCredentials = memberCredentials;
|
||||
this.hostEnvNames = hostEnvNames != null ? hostEnvNames : () -> System.getenv().keySet();
|
||||
}
|
||||
|
||||
// --- adapter seams -------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Build the peer-specific launch for {@code cfg}: the environment map and argv handed to herdr.
|
||||
* Any subscription/guard check, MCP mount, and instruction injection happen here. The env map
|
||||
* and argv are adapter-private; the base only places and starts what is returned.
|
||||
*/
|
||||
protected abstract Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec);
|
||||
|
||||
/** Direct transport access for peer-specific, non-turn control operations. */
|
||||
protected final AgentControl agents() {
|
||||
return agents;
|
||||
}
|
||||
|
||||
/** Resolve the public peer id to the launcher's private herdr target. */
|
||||
protected final String agentTarget(String id) {
|
||||
return paneByAgentId.get(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
if (resetUnsupportedLogged.compareAndSet(false, true)) {
|
||||
log.warn("context reset is unsupported for peer kind {}; clearAfterTurn is a no-op",
|
||||
namePrefix);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* A peer-specific launch: the herdr {@code env} map and {@code argv}, plus — for an adapter
|
||||
* that carries durable session identity (CB-547a) — the peer's OWN session id
|
||||
* ({@link PeerHandle#agentSessionId()}), known before the peer has written anything. Null for
|
||||
* a launch that carries no identity.
|
||||
*/
|
||||
protected record Launch(Map<String, String> env, List<String> argv, String agentSessionId) {
|
||||
|
||||
/** A launch without a discoverable agent session id (an adapter that carries none). */
|
||||
Launch(Map<String, String> env, List<String> argv) {
|
||||
this(env, argv, null);
|
||||
}
|
||||
}
|
||||
|
||||
/** All per-spawn values adapters may need, including the base-composed effective charter. */
|
||||
protected record LaunchSpec(String sessionName, String resumeSessionId, MemberRole role, String charter) {
|
||||
}
|
||||
|
||||
// --- profile surface -----------------------------------------------------------------------
|
||||
|
||||
/** The configured peer profile names (what {@code spawn(profile)} accepts). */
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return profiles.keySet();
|
||||
}
|
||||
|
||||
/** The parity-overlay file list for {@code profileName} (default list when unset). */
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
return cfg == null ? List.of() : cfg.parityOverlay();
|
||||
}
|
||||
|
||||
/** The profile a no-argument spawn uses, or {@code null} if none is configured. */
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>One {@link HerdrPeerLauncher} instance always serves exactly one adapter kind, so every
|
||||
* profile it owns shares that adapter's {@link #capabilities()} — {@code profileName} only
|
||||
* needs validating (throwing on an unknown profile, same as {@link #spawn}), not routing.
|
||||
*/
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
requireProfile(profileName);
|
||||
return capabilities();
|
||||
}
|
||||
|
||||
/** The configured profiles, for adapter capability decisions (e.g. any git-token grant). */
|
||||
protected Collection<BridgedConfig.Profile> profileConfigs() {
|
||||
return profiles.values();
|
||||
}
|
||||
|
||||
/** Resolve {@code profileName} (null/blank → default) to its config, or throw with the options. */
|
||||
protected BridgedConfig.Profile requireProfile(String profileName) {
|
||||
String name = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (name == null || name.isBlank()) {
|
||||
throw new IllegalArgumentException("no default worker profile is configured — "
|
||||
+ "pass a profile; configured: " + profiles.keySet());
|
||||
}
|
||||
BridgedConfig.Profile cfg = profiles.get(name);
|
||||
if (cfg == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile '" + name
|
||||
+ "' — configured: " + profiles.keySet());
|
||||
}
|
||||
return cfg;
|
||||
}
|
||||
|
||||
// --- spawn ---------------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* A started peer plus the launch's agent-session id (the resume handle, or null) and the
|
||||
* charter receipt (CB-571) the base composed for it.
|
||||
*/
|
||||
private record Spawned(Agent agent, String agentSessionId, CharterReceipt receipt) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer. {@code profileName} null/blank → the default profile. The working directory
|
||||
* (CB-112) is resolved by {@link #resolveCwd}: an explicit {@code requestedCwd}, else the
|
||||
* profile's configured {@code cwd}, else {@code callerCwd} (the primary's cwd, when the spawn
|
||||
* came from the primary over MCP), else the daemon's cwd — never assumed to be {@code $HOME}.
|
||||
* The adapter's {@link #buildLaunch} runs before any herdr call.
|
||||
*/
|
||||
protected Agent spawnInternal(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, null, null).agent();
|
||||
}
|
||||
|
||||
/** Pre-CB-557 shape: no explicit role, so the tab is labelled as a {@code dev}. */
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId,
|
||||
MemberRole.DEV);
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a peer with session identity (CB-547a). The session values, role, and charter are
|
||||
* threaded from the {@link SpawnRequest} into {@link #buildLaunch(BridgedConfig.Profile,
|
||||
* LaunchSpec)}, and the launch's resolved agent-session id is returned alongside the agent so
|
||||
* the caller can put it on the {@link PeerHandle}.
|
||||
*/
|
||||
protected Spawned spawnInternal(String profileName, String requestedCwd, String callerCwd,
|
||||
String sessionName, String resumeSessionId, MemberRole role) {
|
||||
BridgedConfig.Profile cfg = requireProfile(profileName);
|
||||
BridgedConfig.Fleet liveFleet = fleet == null ? null : fleet.get();
|
||||
String roleCharter = liveFleet == null ? null : liveFleet.charterFor(role);
|
||||
String replyCharter = cfg.hasMcp() ? REPLY_CHARTER : null;
|
||||
String charter = roleCharter == null ? replyCharter
|
||||
: replyCharter == null ? roleCharter : roleCharter + "\n\n" + replyCharter;
|
||||
// CB-571: fingerprint the exact composed charter bytes once, here in the base, before the
|
||||
// string leaves for an adapter — so Claude and OpenCode derive the same digest. A failed
|
||||
// start has no bridge_spawn result and no roster row, so the failure log below is the only
|
||||
// surface the byte count can appear on. The charter text itself is never logged.
|
||||
CharterReceipt receipt = CharterReceipt.compose(role, cfg.profile(), roleCharter, charter);
|
||||
try {
|
||||
Launch launch = buildLaunch(cfg, new LaunchSpec(sessionName, resumeSessionId, role, charter));
|
||||
String cwd = resolveCwd(requestedCwd, cfg, callerCwd);
|
||||
Agent agent = cfg.tabPlacement()
|
||||
? spawnInTab(cfg, launch.env(), launch.argv(), cwd, role, liveFleet)
|
||||
: spawnAsPane(cfg, launch.env(), launch.argv(), cwd, charter);
|
||||
logCharterReceipt(receipt, true);
|
||||
return new Spawned(agent, launch.agentSessionId(), receipt);
|
||||
} catch (RuntimeException e) {
|
||||
logCharterReceipt(receipt, false);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The one place the charter's size and digest appear in the logs. {@code success} true after a
|
||||
* start, false from the failure path of {@link #spawnInternal} where no handle or roster row
|
||||
* exists to carry the receipt. Always metadata only — never the charter text.
|
||||
*/
|
||||
private static void logCharterReceipt(CharterReceipt receipt, boolean success) {
|
||||
String role = receipt.role() == null ? "" : receipt.role().wireName();
|
||||
if (success) {
|
||||
log.info("spawned role={} profile={} charterSource={} charterSha256={} charterBytes={}",
|
||||
role, receipt.profile(), receipt.charterSource(),
|
||||
receipt.charterSha256(), receipt.charterBytes());
|
||||
} else {
|
||||
log.warn("spawn failed; charter role={} profile={} charterSource={} charterSha256={} charterBytes={}",
|
||||
role, receipt.profile(), receipt.charterSource(),
|
||||
receipt.charterSha256(), receipt.charterBytes());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The next tab number for {@code role} on {@code profile}, starting at 1.
|
||||
*
|
||||
* <p>Starts at 1 rather than 0 because the number is read by a person: {@code "dev: sonnet #1"}
|
||||
* is the first one, and {@code #0} invites the question of where {@code #1} went.
|
||||
*/
|
||||
private long nextLabelSeq(MemberRole role, String profile) {
|
||||
String key = (role == null ? "" : role.wireName()) + "/" + profile;
|
||||
return labelSeq.computeIfAbsent(key, _ -> new AtomicLong()).incrementAndGet();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to {@link #spawnInternal} and wraps the resulting herdr {@link Agent} in a
|
||||
* {@link WorkerHandle} whose {@link PeerHandle#id()} is a fresh <em>host-unique</em> opaque
|
||||
* UUID (CB-519), deliberately decoupled from the herdr pane id: the id is the registry/routing
|
||||
* key and must never collide across daemon processes on the same host, while the herdr pane id
|
||||
* stays a launcher-private placement/teardown coordinate, remembered here so {@link #stop}
|
||||
* can resolve the host-unique key back to its pane. When {@code spawnReadyTimeoutMs > 0},
|
||||
* blocks until the peer's herdr status is injectable or the timeout elapses; on timeout the
|
||||
* pane is closed (no orphan) and a {@link PeerUnreachableException} is thrown.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
Spawned spawned = spawnInternal(req.profileName(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
Agent agent = spawned.agent();
|
||||
String paneId = agent.paneId();
|
||||
if (spawnReadyTimeoutMs > 0) {
|
||||
waitUntilInjectableOrThrow(paneId);
|
||||
}
|
||||
// CB-519: the handle id is a host-unique UUID; the herdr pane it maps to stays internal.
|
||||
String id = UUID.randomUUID().toString();
|
||||
paneByAgentId.put(id, paneId);
|
||||
return new WorkerHandle(id, agent.terminalId(), requireProfile(req.profileName()).profile(),
|
||||
req.sessionName(), spawned.agentSessionId(), spawned.receipt());
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return effectiveCwd(req.profileName(), req.requestedCwd(), req.callerCwd());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-301: the effective working directory a spawn for {@code profileName} would use, without
|
||||
* actually spawning.
|
||||
*/
|
||||
private String effectiveCwd(String profileName, String requestedCwd, String callerCwd) {
|
||||
return resolveCwd(requestedCwd, requireProfile(profileName), callerCwd);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-112 cwd resolution: spawn arg → profile config → the primary's cwd → the daemon's cwd.
|
||||
* Never returns {@code null}/blank: {@code "."} (the daemon's own working directory) is the
|
||||
* guaranteed last resort so a pathological environment with an unset {@code user.dir} still
|
||||
* honours the "never assume {@code $HOME}" contract rather than letting herdr default the pane.
|
||||
*/
|
||||
private static String resolveCwd(String requestedCwd, BridgedConfig.Profile cfg, String callerCwd) {
|
||||
return firstNonBlank(requestedCwd, cfg.cwd(), callerCwd, System.getProperty("user.dir"), ".");
|
||||
}
|
||||
|
||||
private static String firstNonBlank(String... values) {
|
||||
for (String v : values) {
|
||||
if (v != null && !v.isBlank()) return v;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Dedicated worker space → own tab (carrying cwd+env) → start the peer into the seed pane. */
|
||||
private Agent spawnInTab(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd, MemberRole role, BridgedConfig.Fleet liveFleet) {
|
||||
Workspace space = spaces.ensureWorkspace(cfg.workspace());
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId(), cwd, workerEnv);
|
||||
log.info("spawning {} profile={} space={} tab={} cwd={}",
|
||||
namePrefix, cfg.profile(), space.workspaceId(), tab.tab().tabId(), cwd);
|
||||
|
||||
Started started;
|
||||
try {
|
||||
if (tab.rootPaneId() == null) {
|
||||
// Protocol 19 starts the agent INTO the seed pane — without one there is nowhere
|
||||
// to start, and a partial tab would be left behind.
|
||||
throw new IllegalStateException("tab " + tab.tab().tabId()
|
||||
+ " had no seed pane in the create response — cannot start a peer in it");
|
||||
}
|
||||
started = startUniquelyNamed(cfg, argv, tab.rootPaneId());
|
||||
} catch (RuntimeException e) {
|
||||
// The peer never started — don't leave the tab we just created orphaned.
|
||||
// Best-effort cleanup; never let it mask the real spawn failure.
|
||||
try {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to close orphaned tab {} after spawn error: {}",
|
||||
tab.tab().tabId(), cleanup.getMessage());
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
|
||||
// The peer is LIVE now, in the seed pane itself (no shell pane to drop — protocol 19).
|
||||
// Labelling is cosmetic: it must not fail the spawn or orphan the running peer — on error
|
||||
// we log and still return it so the caller gets its paneId and can tear it down.
|
||||
tidy("label tab " + tab.tab().tabId(),
|
||||
() -> spaces.renameTab(tab.tab().tabId(),
|
||||
cfg.renderTabLabel(
|
||||
liveFleet == null ? null : liveFleet.tabLabel(),
|
||||
role, nextLabelSeq(role, cfg.profile()))));
|
||||
log.info("{} started pane={} tab={} terminal={}",
|
||||
namePrefix, started.agent().paneId(), started.agent().tabId(), started.agent().terminalId());
|
||||
return started.agent();
|
||||
}
|
||||
|
||||
/** Run a best-effort post-start cleanup step, logging (not throwing) on failure. */
|
||||
private void tidy(String what, Runnable step) {
|
||||
try {
|
||||
step.run();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("post-start step failed ({}) — peer is running regardless: {}", what, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Legacy placement: split the currently-focused tab; the peer still starts in {@code cwd}.
|
||||
*
|
||||
* <p>CB-571: this is the one legacy log that printed the full argv, and the charter travels
|
||||
* inside argv — so the charter text went to the daemon log on every pane-placement spawn. The
|
||||
* {@code spawnInTab} path never logs argv, so only this site is fixed. {@code charter} is the
|
||||
* composed charter, if any; its argv element is replaced by its digest so the log still shows
|
||||
* which args were passed without exposing the charter prose.
|
||||
*/
|
||||
private Agent spawnAsPane(BridgedConfig.Profile cfg, Map<String, String> workerEnv,
|
||||
List<String> argv, String cwd, String charter) {
|
||||
log.info("spawning {} (pane placement) profile={} cwd={} argv={}",
|
||||
namePrefix, cfg.profile(), cwd, redactCharter(argv, charter));
|
||||
String paneId = spaces.splitPane(cwd, workerEnv);
|
||||
if (paneId == null) {
|
||||
throw new IllegalStateException("pane.split returned no pane — cannot start a peer");
|
||||
}
|
||||
Agent peer = startUniquelyNamed(cfg, argv, paneId).agent();
|
||||
log.info("{} started pane={} terminal={}", namePrefix, peer.paneId(), peer.terminalId());
|
||||
return peer;
|
||||
}
|
||||
|
||||
/**
|
||||
* A copy of {@code argv} with an element equal to {@code charter} replaced by its digest, so
|
||||
* the pane log never prints the charter prose. The charter is handed to an adapter as one argv
|
||||
* element, so exact-equality is the right match; every other argument passes through unchanged.
|
||||
*/
|
||||
private static List<String> redactCharter(List<String> argv, String charter) {
|
||||
if (charter == null || charter.isBlank() || argv == null || argv.isEmpty()) {
|
||||
return argv;
|
||||
}
|
||||
String digest = CharterReceipt.digestOf(charter);
|
||||
return argv.stream()
|
||||
.map(a -> a.equals(charter) ? "<charter sha256=" + digest + ">" : a)
|
||||
.toList();
|
||||
}
|
||||
|
||||
/** A started peer together with the sequence its unique name/label used. */
|
||||
private record Started(Agent agent, long seq) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Start the peer under a unique herdr agent name. herdr requires each running agent's
|
||||
* {@code name} to be distinct (a 2nd identical {@code name} fails {@code agent_name_taken}) —
|
||||
* the exact case that makes multiple peers useful. The name is
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>}: {@code seq} distinguishes peers within this process,
|
||||
* and the per-process {@code nonce} keeps a fresh process (whose {@code seq} restarts at 0) from
|
||||
* colliding with same-profile peers that outlived a restart. The retry is a belt-and-braces
|
||||
* backstop for the astronomically unlikely nonce+seq clash; the name is a label only — herdr
|
||||
* detects kind and status from terminal output, not from it.
|
||||
*/
|
||||
private Started startUniquelyNamed(BridgedConfig.Profile cfg, List<String> argv, String paneId) {
|
||||
// Protocol 19 resolves the executable from the agent kind (== namePrefix here), so
|
||||
// argv[0] — the configured executable — is dropped and only the extra args are passed.
|
||||
List<String> args = argv.isEmpty() ? argv : argv.subList(1, argv.size());
|
||||
HerdrException last = null;
|
||||
for (int attempt = 0; attempt < NAME_RETRIES; attempt++) {
|
||||
long seq = nameSeq.incrementAndGet();
|
||||
String name = namePrefix + "-" + cfg.profile() + "-" + nameNonce + "-" + seq;
|
||||
try {
|
||||
return new Started(startAwaitingShellPrompt(name, args, paneId), seq);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_name_taken".equals(e.code())) throw e;
|
||||
log.debug("peer name '{}' taken, retrying", name);
|
||||
last = e;
|
||||
}
|
||||
}
|
||||
throw last;
|
||||
}
|
||||
|
||||
/** Start the agent into {@code paneId}, waiting out the seed shell's boot with the sleeper. */
|
||||
private Agent startAwaitingShellPrompt(String name, List<String> args, String paneId) {
|
||||
HerdrException busy = null;
|
||||
for (int attempt = 0; attempt < SHELL_READY_RETRIES; attempt++) {
|
||||
try {
|
||||
return agents.start(name, namePrefix, args, paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!"agent_pane_busy".equals(e.code())) throw e;
|
||||
log.debug("pane {} not at its shell prompt yet, retrying agent.start", paneId);
|
||||
busy = e;
|
||||
sleeper.run();
|
||||
}
|
||||
}
|
||||
throw busy;
|
||||
}
|
||||
|
||||
// --- discovery + reap ----------------------------------------------------------------------
|
||||
|
||||
/** All herdr-tracked agents — discovery for "what peers exist". */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
return agents.list();
|
||||
}
|
||||
|
||||
/**
|
||||
* Reap peer panes left behind by an earlier daemon process (CB-117). herdr keeps a peer's pane
|
||||
* alive across a daemon restart <em>by design</em>, and that pane's id is held only by its
|
||||
* spawner — so a peer whose owning process exited before issuing the matching teardown leaks
|
||||
* with nothing tracking it. On boot we scan herdr for agents whose name matches our
|
||||
* {@code <prefix>-<profile>-<nonce>-<seq>} scheme with a nonce <em>other</em> than this
|
||||
* process's {@link #nameNonce}, and tear each one down (its pane and, via {@link #stop}, its
|
||||
* now-empty dedicated tab). A current-nonce peer is ours and live, so it is left running; a
|
||||
* user's own session carries no such name and is never touched. A peer from a <em>different</em>
|
||||
* adapter (different prefix) is likewise never touched. Best-effort: a failed listing, or a
|
||||
* failure to stop any one peer, is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
List<Agent> all;
|
||||
try {
|
||||
all = agents.list();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("orphan-peer reap skipped — agent.list failed: {}", e.getMessage());
|
||||
return 0;
|
||||
}
|
||||
int reaped = 0;
|
||||
for (Agent a : all) {
|
||||
if (!isForeignWorker(namePrefix, a.name(), nameNonce)) continue;
|
||||
try {
|
||||
stop(a.paneId());
|
||||
reaped++;
|
||||
log.info("reaped orphan {} {} (pane={} tab={}) left by a prior daemon",
|
||||
namePrefix, a.name(), a.paneId(), a.tabId());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("could not reap orphan {} {} (pane={}): {}",
|
||||
namePrefix, a.name(), a.paneId(), e.getMessage());
|
||||
}
|
||||
}
|
||||
if (reaped > 0) {
|
||||
log.info("orphan-peer reap complete — {} stale {} peer(s) removed at startup", reaped, namePrefix);
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The {@code <prefix>-<profile>-<nonce>-<seq>} name pattern; group 1 captures the 6-hex nonce. */
|
||||
static Pattern workerNamePattern(String prefix) {
|
||||
return Pattern.compile(prefix + "-.*-([0-9a-f]{6})-\\d+");
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a peer of kind {@code prefix} started by a <em>different</em> process
|
||||
* than {@code currentNonce} — the reap predicate (CB-117). True only for the prefix's naming
|
||||
* scheme with a foreign nonce: a non-peer name, a different adapter's name, or our own live
|
||||
* nonce is excluded. Pure and package-private so the decision is unit-testable without herdr.
|
||||
*/
|
||||
static boolean isForeignWorker(String prefix, String name, String currentNonce) {
|
||||
String nonce = workerNonce(prefix, name);
|
||||
return nonce != null && !nonce.equals(currentNonce);
|
||||
}
|
||||
|
||||
/** The 6-hex nonce embedded in a {@code prefix} peer name, or {@code null} if not one. */
|
||||
static String workerNonce(String prefix, String name) {
|
||||
if (name == null) return null;
|
||||
Matcher m = workerNamePattern(prefix).matcher(name);
|
||||
return m.matches() ? m.group(1) : null;
|
||||
}
|
||||
|
||||
/** This process's peer-name nonce (a label component only; exposed for reaper tests). */
|
||||
String nameNonce() {
|
||||
return nameNonce;
|
||||
}
|
||||
|
||||
// --- teardown ------------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Tear a peer down: close the pane, and close its tab <em>only</em> when the peer is that tab's
|
||||
* sole occupant. The single-pane check is what makes this safe regardless of how the peer was
|
||||
* placed (or a placement-config change across a restart): a pane-placement peer sitting in one
|
||||
* of the user's shared tabs has siblings, so its tab is never closed — we only ever remove a
|
||||
* tab we created to hold one peer.
|
||||
*
|
||||
* <p>{@code idOrPane} is the {@link PeerHandle#id()} of a peer this launcher spawned (CB-519's
|
||||
* host-unique opaque UUID), resolved through {@link #paneByAgentId} to the pane it must tear
|
||||
* down. An argument that is not one of our ids is treated as a raw herdr pane id — the
|
||||
* {@link #reapOrphanWorkers() orphan-reap} and spawn-gate-timeout paths, plus any caller that
|
||||
* passes a pane directly, keep working without an owning id.
|
||||
*
|
||||
* <p>Resolves the tab from the pane <em>before</em> closing it. An already-gone pane/tab
|
||||
* (repeated DELETE, crashed peer) is treated as success; any other failure propagates so a
|
||||
* genuinely failed teardown is not reported as done.
|
||||
*/
|
||||
@Override
|
||||
public void stop(String idOrPane) {
|
||||
// Teardown knows only the pane, not which profile spawned it. Attempt tab cleanup when any
|
||||
// profile uses tab placement (so the bridge may have created a dedicated peer tab); the
|
||||
// single-occupant check below is what actually protects the user's shared tabs.
|
||||
String paneId = paneByAgentId.remove(idOrPane);
|
||||
if (paneId == null) {
|
||||
paneId = idOrPane; // raw-pane fallback (reap, gate timeout, pane-addressed callers)
|
||||
}
|
||||
WorkspaceControl.PaneLocation loc = usesTabPlacement() ? spaces.locatePane(paneId) : null;
|
||||
try {
|
||||
agents.close(paneId);
|
||||
} catch (HerdrException e) {
|
||||
if (!isAlreadyGone(e)) throw e;
|
||||
log.debug("pane.close({}) ignored — already gone: {}", paneId, e.getMessage());
|
||||
}
|
||||
if (loc != null && loc.tabPaneCount() == 1) {
|
||||
spaces.closeTab(loc.tabId());
|
||||
} else if (loc != null) {
|
||||
log.debug("not closing tab {} — it holds {} panes (not a dedicated peer tab)",
|
||||
loc.tabId(), loc.tabPaneCount());
|
||||
}
|
||||
}
|
||||
|
||||
/** Whether any configured profile places peers in their own tab (so tabs may need cleanup). */
|
||||
private boolean usesTabPlacement() {
|
||||
return profiles.values().stream().anyMatch(BridgedConfig.Profile::tabPlacement);
|
||||
}
|
||||
|
||||
/** True when a herdr error means the target is already gone (safe to treat as done). */
|
||||
private static boolean isAlreadyGone(HerdrException e) {
|
||||
return e.code() != null && e.code().endsWith("_not_found");
|
||||
}
|
||||
|
||||
// --- spawn-readiness gate (CB-306) ---------------------------------------------------------
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until the pane reports an injectable state or the configured
|
||||
* timeout elapses. On timeout, close the pane (self-reap) and throw.
|
||||
*/
|
||||
private void waitUntilInjectableOrThrow(String paneId) {
|
||||
long deadline = nowMillis.getAsLong() + spawnReadyTimeoutMs;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
if (agents.status(paneId).injectable()) {
|
||||
log.debug("peer pane={} reached injectable state", paneId);
|
||||
return;
|
||||
}
|
||||
sleeper.run();
|
||||
}
|
||||
log.warn("peer pane={} did not become injectable within {}ms — closing", paneId, spawnReadyTimeoutMs);
|
||||
stop(paneId);
|
||||
throw new PeerUnreachableException(
|
||||
"worker pane " + paneId + " did not reach injectable state within "
|
||||
+ spawnReadyTimeoutMs + "ms");
|
||||
}
|
||||
|
||||
/**
|
||||
* A concrete {@link PeerHandle} wrapping herdr agent coordinates, the profile that spawned it,
|
||||
* the session identity the launch resolved (CB-547a): the bridge's logical name and the peer's
|
||||
* own session id, both null when the spawn carried no identity — and the charter receipt
|
||||
* (CB-571) the base computed for this launch.
|
||||
*/
|
||||
private record WorkerHandle(String id, String terminalId, String profile,
|
||||
String sessionName, String agentSessionId,
|
||||
CharterReceipt receipt) implements PeerHandle {
|
||||
|
||||
@Override
|
||||
public CharterReceipt charterReceipt() {
|
||||
return receipt;
|
||||
}
|
||||
}
|
||||
|
||||
// --- shared helpers ------------------------------------------------------------------------
|
||||
|
||||
/** Put {@code k → v} only when {@code v} is present (non-null, non-blank). */
|
||||
protected static void putIfPresent(Map<String, String> m, String k, String v) {
|
||||
if (v != null && !v.isBlank()) {
|
||||
m.put(k, v);
|
||||
}
|
||||
}
|
||||
|
||||
/** Host env lookup that tolerates an unconfigured (null/blank) var name — returns null then. */
|
||||
protected String resolveEnv(String name) {
|
||||
return (name == null || name.isBlank()) ? null : env.apply(name);
|
||||
}
|
||||
|
||||
/**
|
||||
* The parity-neutral git-forge token grant (CB-302): when {@code cfg} opts in via
|
||||
* {@code gitTokenEnv} and the token resolves, inject {@code GITEA_TOKEN} plus its paired
|
||||
* {@code GITEA_HOST}. Push over SSH is unaffected; the only incremental grant is PR-create.
|
||||
* Peer-neutral, so every herdr adapter reuses it unchanged.
|
||||
*/
|
||||
protected void applyGitToken(Map<String, String> workerEnv, BridgedConfig.Profile cfg) {
|
||||
if (!cfg.hasGitToken()) {
|
||||
return;
|
||||
}
|
||||
String gitToken = resolveEnv(cfg.gitTokenEnv());
|
||||
if (gitToken != null) {
|
||||
workerEnv.put("GITEA_TOKEN", gitToken);
|
||||
putIfPresent(workerEnv, "GITEA_HOST", resolveEnv(cfg.gitHostEnv()));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: overlay value that shadows any host credential a herdr pane otherwise inherits from
|
||||
* herdr's own login-shell process environment (gitea issue #82, superseding CB-592's single
|
||||
* hardcoded {@code GITEA_ACCESS_TOKEN} name — see {@link #applyMemberCredentialPolicy}). herdr
|
||||
* spawns a pane from its <em>own</em> process environment and layers our map on top —
|
||||
* {@link dev.ltms.bridged.herdr.WorkspaceControl#createTab} and {@code #splitPane} send only
|
||||
* the keys we put in that map, so any key we never mention passes straight through from
|
||||
* herdr's own shell, admin credentials included.
|
||||
*
|
||||
* <p>Deliberately a non-blank sentinel, not {@code ""}. Whether an empty-string overlay value
|
||||
* overrides an inherited variable or is skipped as blank could not be settled by reading this
|
||||
* codebase — herdr's server-side merge is an external process, not something in this repo.
|
||||
* A non-blank replacement sidesteps that ambiguity entirely: {@link #baseEnv}'s own {@code
|
||||
* PATH} seeding already depends on the overlay reliably replacing an inherited value (see its
|
||||
* javadoc), and that is only demonstrated for a non-blank value, so this reuses the same,
|
||||
* proven-reliable shape rather than the unverified one.
|
||||
*
|
||||
* <p><b>MEASURED ON A LIVE PANE, 2026-08-15 (CB-592): this sentinel alone does NOT hold.</b> The
|
||||
* overlay itself works — {@code GITEA_TOKEN} is injected here, is exported by no shell file, and
|
||||
* does reach the pane. The sentinel loses one step later. A herdr pane runs a <em>login</em>
|
||||
* shell, {@code ~/.zprofile} sources {@code ${SHARED_ENV}/tools/secrets.sh}, and that file does a
|
||||
* plain unconditional {@code export GITEA_ACCESS_TOKEN=...}. A login shell overwrites a value
|
||||
* already in the environment, so the real admin token is put back over this sentinel before the
|
||||
* member process ever starts. That defeat applies to <em>every</em> name {@code secrets.sh}
|
||||
* exports, and no launcher-side overlay can win against it.
|
||||
*
|
||||
* <p>So this constant is not the control on its own — {@link #MEMBER_MARKER} is the other half.
|
||||
* Keeping the sentinel is still worth it: it is correct for any peer kind whose pane does not
|
||||
* start a login shell, and it makes the intent explicit at the one place every adapter passes.
|
||||
*/
|
||||
private static final String BLOCKED_CREDENTIAL_SENTINEL =
|
||||
"blocked-by-bridged-cb596-see-gitea-issue-82";
|
||||
|
||||
/**
|
||||
* CB-592: marks a pane as a bridged member so a shell startup file can decline to export
|
||||
* operator-only credentials into it (gitea issue #77).
|
||||
*
|
||||
* <p>This name is deliberately one that {@code secrets.sh} never exports, which is exactly why
|
||||
* it survives the login shell that wipes {@link #BLOCKED_GITEA_ACCESS_TOKEN}. The mechanism is
|
||||
* measured, not assumed: {@code GITEA_TOKEN} is injected the same way, is absent from a login
|
||||
* shell of its own, and was observed set inside a live member pane.
|
||||
*
|
||||
* <p>It is a no-op until the operator guards the export, which is a one-line change in a file
|
||||
* this repo does not own and must not edit unasked:
|
||||
*
|
||||
* <pre>{@code
|
||||
* [ -n "${BRIDGED_MEMBER:-}" ] || export GITEA_ACCESS_TOKEN=...
|
||||
* }</pre>
|
||||
*
|
||||
* <p>Setting the marker now costs nothing and means that edit is the whole remaining fix.
|
||||
*/
|
||||
static final String MEMBER_MARKER = "BRIDGED_MEMBER";
|
||||
|
||||
/**
|
||||
* Seed a worker's environment (CB-511): the daemon's own {@code PATH}, then the profile's
|
||||
* {@code env:} entries, then the CB-592 admin-token shadow.
|
||||
*
|
||||
* <p>Why this exists: bridged passes herdr an explicit env map, and herdr merges it into
|
||||
* <em>its own</em> process environment. So before this, a worker inherited whatever PATH the
|
||||
* herdr server happened to be started with — on this host, one from weeks earlier with no JDK
|
||||
* and no Maven, which left workers unable to run the build they were being asked to run. The
|
||||
* worker's toolchain must follow from configuration, not from how a long-lived daemon was
|
||||
* launched.
|
||||
*
|
||||
* <p>Adapter-specific variables are layered on top of this by {@code buildLaunch} and therefore
|
||||
* win. That ordering is deliberate and load-bearing: it stops a profile's {@code env:} from
|
||||
* overriding {@code ANTHROPIC_BASE_URL} and slipping past {@link
|
||||
* dev.ltms.bridged.guard.SubscriptionGuard}, which is checked against the profile's
|
||||
* {@code baseUrl} and nothing else.
|
||||
*
|
||||
* <p>The CB-596 credential shadow and the CB-592 marker are put in <em>last</em>, after the
|
||||
* profile's own {@code env:}, so no profile — present or future — can restore a blocked
|
||||
* credential, or hide that the pane is a member, by naming either in config. This is the one
|
||||
* place both are applied: every {@code buildLaunch} in every adapter calls this first, so a new
|
||||
* profile, and a peer kind not yet written, gets them for free.
|
||||
*/
|
||||
protected Map<String, String> baseEnv(BridgedConfig.Profile cfg) {
|
||||
Map<String, String> workerEnv = new LinkedHashMap<>();
|
||||
String path = env.apply("PATH");
|
||||
if (path != null && !path.isBlank()) {
|
||||
workerEnv.put("PATH", path);
|
||||
}
|
||||
if (cfg != null && cfg.env() != null) {
|
||||
workerEnv.putAll(cfg.env());
|
||||
}
|
||||
applyMemberCredentialPolicy(workerEnv);
|
||||
workerEnv.put(MEMBER_MARKER, "1");
|
||||
return workerEnv;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: shadow every configured {@code memberCredentials.known} name that is not also
|
||||
* {@code allow}-ed, replacing CB-592's single hardcoded {@code GITEA_ACCESS_TOKEN} name (gitea
|
||||
* issue #82). An allow-listed name is deliberately left unmentioned here — see {@link
|
||||
* #BLOCKED_CREDENTIAL_SENTINEL}'s javadoc for why an overlay entry is the only way to shadow an
|
||||
* inherited value, which is exactly why an allowed name must get NO entry: any entry at all,
|
||||
* blank or not, risks overriding the real value the pane needs.
|
||||
*
|
||||
* <p>No {@code memberCredentials} configured — the supplier is {@code null}, or it resolves to
|
||||
* one whose {@code known} list is empty — shadows nothing. This is a real, config-driven gap
|
||||
* (see {@link BridgedConfig.MemberCredentials}'s javadoc), not a safe default: deny-by-default
|
||||
* only defends names the operator has actually enumerated in {@code known}.
|
||||
*/
|
||||
private void applyMemberCredentialPolicy(Map<String, String> workerEnv) {
|
||||
BridgedConfig.MemberCredentials creds = memberCredentials == null ? null : memberCredentials.get();
|
||||
if (creds == null) {
|
||||
return;
|
||||
}
|
||||
for (String name : creds.blockedSet()) {
|
||||
workerEnv.put(name, BLOCKED_CREDENTIAL_SENTINEL);
|
||||
}
|
||||
logCredentialGap(creds);
|
||||
}
|
||||
|
||||
/** Credential-shaped env var name heuristic for {@link #logCredentialGap} — case-insensitive. */
|
||||
private static final Pattern CREDENTIAL_SHAPED_NAME =
|
||||
Pattern.compile("(?i).*(TOKEN|SECRET|_KEY|APIKEY|PASSWORD|CREDENTIAL|AUTH).*");
|
||||
|
||||
/** Guards {@link #logCredentialGap} to one WARN per launcher instance, not one per spawn. */
|
||||
private final AtomicBoolean credentialGapLogged = new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* CB-596 criterion 4: a credential-shaped host env var name on neither {@code known} nor
|
||||
* {@code allow} is not silently allowed — it is reported. {@link #hostEnvNames} enumerates the
|
||||
* daemon's own environment (see that field's javadoc for why the daemon's env is read rather
|
||||
* than the spawned pane's, which the daemon has no channel to inspect at spawn time); this logs
|
||||
* every such NAME, at WARN, at most once per launcher instance — never a value, a prefix of a
|
||||
* value, or a hash of a value, so the log itself cannot leak anything.
|
||||
*/
|
||||
private void logCredentialGap(BridgedConfig.MemberCredentials creds) {
|
||||
Set<String> covered = new HashSet<>(creds.known());
|
||||
covered.addAll(creds.allow());
|
||||
List<String> gap = hostEnvNames.get().stream()
|
||||
.filter(name -> CREDENTIAL_SHAPED_NAME.matcher(name).matches())
|
||||
.filter(name -> !covered.contains(name))
|
||||
.sorted()
|
||||
.toList();
|
||||
if (gap.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
if (credentialGapLogged.compareAndSet(false, true)) {
|
||||
log.warn("memberCredentials gap: {} credential-shaped env var name(s) are on neither "
|
||||
+ "known: nor allow: — every member pane inherits them UNBLOCKED — {}. "
|
||||
+ "Add each to memberCredentials.known (blocked by default) or .allow "
|
||||
+ "(if a member legitimately needs it).",
|
||||
gap.size(), gap);
|
||||
}
|
||||
}
|
||||
|
||||
/** Defensive copy of {@code argv} plus room to append launch flags. */
|
||||
protected static List<String> mutableArgv(List<String> argv) {
|
||||
return new ArrayList<>(argv);
|
||||
}
|
||||
|
||||
/**
|
||||
* Uninterruptible sleep — the production {@link #sleeper}. Tests supply their own no-op /
|
||||
* fast-faking sleeper so they never real-sleep.
|
||||
*/
|
||||
protected static void sleepUninterruptibly(long ms) {
|
||||
try {
|
||||
Thread.sleep(ms);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
// preserve the interrupt flag but continue — poll loops should not be aborted by an
|
||||
// interrupt that was not meant for them.
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,496 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.CharterReceipt;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>opencode</strong> — an open-source,
|
||||
* provider-agnostic terminal coding agent. Its whole reason for existing is to prove the
|
||||
* {@code PeerLauncher} SPI is genuinely provider-neutral: opencode shares none of Claude Code's
|
||||
* private launch seams, yet reuses every line of shared transport in the base (tab/pane placement,
|
||||
* the CB-306 readiness gate, unique naming + CB-117 reap, teardown, listing, cwd).
|
||||
*
|
||||
* <p>The divergences from {@link ClaudeCodeLauncher}, all confined to {@link #buildLaunch}:
|
||||
* <ul>
|
||||
* <li><strong>No subscription boundary.</strong> opencode carries no {@code ANTHROPIC_BASE_URL}
|
||||
* and there is no {@link dev.ltms.bridged.guard.SubscriptionGuard} — the guard is a
|
||||
* Claude-private concern, not part of the SPI. opencode reads the operator's own provider
|
||||
* credentials from its global {@code auth.json}; the bridge injects none.</li>
|
||||
* <li><strong>File-based MCP mount + instructions.</strong> opencode has no inline
|
||||
* {@code --mcp-config}/{@code --append-system-prompt}. Instead the bridge writes an ephemeral
|
||||
* {@code opencode.json} that declares the bridge as a {@code remote} MCP server and lists a
|
||||
* member-charter file under {@code instructions}, then points the worker at it with
|
||||
* {@code OPENCODE_CONFIG}. This is the one place the launcher touches disk — Claude never did.</li>
|
||||
* <li><strong>Model as a flag.</strong> the {@code provider/model} selector is passed as
|
||||
* {@code -m}, not an env var.</li>
|
||||
* <li><strong>{@code opencode} name prefix</strong> so reap matches {@code opencode-*} panes and
|
||||
* never another adapter's.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "opencode";
|
||||
|
||||
/** Writer for the generated {@code opencode.json}. */
|
||||
private static final ObjectMapper JSON = new ObjectMapper();
|
||||
|
||||
/** Root under which per-spawn opencode config dirs are created (injectable for tests). */
|
||||
private final Path configRoot;
|
||||
|
||||
/**
|
||||
* Session discovery against opencode's on-disk storage ({@link OpenCodeSessionDiscovery}) —
|
||||
* the one seam that knows opencode's private session-file layout. Its root is injectable for
|
||||
* tests so they never touch the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with the spawn-ready gate enabled. Polls {@code agents.status()} until
|
||||
* the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply a
|
||||
* fake clock ({@code nowMillis}), poll-loop wait ({@code sleeper}), and a temp {@code configRoot}
|
||||
* they can inspect the generated {@code opencode.json}/charter under.
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (encodes the poll interval; never called when the
|
||||
* gate is disabled)
|
||||
* @param configRoot existing directory under which per-spawn config dirs are created
|
||||
* @param discoveryRoot opencode's on-disk storage root to scan for session records
|
||||
* (injectable for tests; opencode's layout is matched at
|
||||
* {@link OpenCodeSessionDiscovery})
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, configRoot, discoveryRoot, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<BridgedConfig.Fleet> fleet,
|
||||
Supplier<BridgedConfig.MemberCredentials> memberCredentials) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Builds the opencode launch: no {@code ANTHROPIC_*} and no guard (opencode reads its own
|
||||
* provider credentials); when the profile mounts the bridge MCP or has a member charter,
|
||||
* generate an ephemeral {@code opencode.json} (remote MCP server + member-charter instructions)
|
||||
* and point the worker at it via {@code OPENCODE_CONFIG}; carry the parity-neutral git-forge
|
||||
* grant; and select the model with {@code -m}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
// A config file is needed for the bridge MCP mount, a member charter, or a pinned endpoint (CB-508).
|
||||
if (cfg.hasMcp() || spec.charter() != null || hasCustomProvider(cfg)) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter()).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
return new Launch(workerEnv,
|
||||
argvWithResume(argvWithModel(argvWithAuto(cfg), cfg), spec.resumeSessionId()));
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
*
|
||||
* <p>Note this reuses {@code baseUrl}, the same field the Claude adapter injects as
|
||||
* {@code ANTHROPIC_BASE_URL} — but it does <em>not</em> go through {@code SubscriptionGuard}.
|
||||
* That asymmetry is deliberate and safe: the guard exists to stop a worker borrowing the
|
||||
* primary's Anthropic subscription, and an opencode process has no Anthropic credential path
|
||||
* at all. Pointing it at a local vLLM cannot leak the subscription.
|
||||
*/
|
||||
private static boolean hasCustomProvider(BridgedConfig.Profile cfg) {
|
||||
return cfg.baseUrl() != null && !cfg.baseUrl().isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus the unconditional {@code --auto} flag, which auto-approves the
|
||||
* permissions opencode does not explicitly deny. It is unconditional, not a preference: a
|
||||
* spawned peer has no human at its pane — the bridge spawned it — so one that stops at an
|
||||
* approval prompt is a wedged agent, indistinguishable from a legitimate mid-turn wait and
|
||||
* unable to end its turn with {@code bridge_reply}. opencode's own help calls this
|
||||
* "dangerous!", but the blast radius here is already bounded by design: a worker runs in its
|
||||
* own git worktree on its own branch, is off-subscription, and cannot merge — the lead is the
|
||||
* gate.
|
||||
*/
|
||||
private List<String> argvWithAuto(BridgedConfig.Profile cfg) {
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--auto");
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus, on a resumed spawn, opencode's {@code -s <id>} flag to continue a prior
|
||||
* conversation by its session id. {@code -s, --session <id>} resumes an existing session; on a
|
||||
* fresh spawn (no resume target) no flag is added, letting opencode start a brand-new session.
|
||||
* The id comes from the base launch spec.
|
||||
*/
|
||||
private List<String> argvWithResume(List<String> argv, String id) {
|
||||
if (id == null || id.isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withResume = mutableArgv(argv);
|
||||
withResume.add("-s");
|
||||
withResume.add(id);
|
||||
return withResume;
|
||||
}
|
||||
|
||||
/** The launch argv plus, when a model is configured, the opencode {@code -m provider/model} flag. */
|
||||
private List<String> argvWithModel(List<String> argv, BridgedConfig.Profile cfg) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()) {
|
||||
argv.add("-m");
|
||||
argv.add(cfg.model());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write an ephemeral {@code opencode.json} (and the member-charter file it references) into a
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*/
|
||||
private Path writeConfig(BridgedConfig.Profile cfg, String charterText) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "bridged-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
root.put("$schema", "https://opencode.ai/config.json");
|
||||
// CB-523, opencode side: a worker that runs out of context dies mid-turn, and its reply
|
||||
// — the entire point of the turn — is lost with it. Auto-compaction is therefore not an
|
||||
// operator preference for a bridged worker, it is a condition of the turn contract.
|
||||
//
|
||||
// Stated deliberately even though it is redundant today: OPENCODE_CONFIG is MERGED over
|
||||
// ~/.config/opencode/config.json rather than replacing it, so a worker already inherits
|
||||
// an `auto: true` set at home. We do not want that inheritance to be what the guarantee
|
||||
// rests on — the home file is outside this repo, differs per machine, and is not ours.
|
||||
//
|
||||
// Know the cost before removing it: this key WINS over the home config (verified — an
|
||||
// OPENCODE_CONFIG value overrides the home value, it does not defer to it), so an
|
||||
// operator who sets `compaction.auto: false` at home cannot turn it off for bridged
|
||||
// workers. That is the intended trade for peers we spawn and whose turns we must land;
|
||||
// if per-profile control is ever wanted, add a profile knob rather than dropping this.
|
||||
root.putObject("compaction").put("auto", true);
|
||||
|
||||
if (charterText != null) {
|
||||
Path charter = dir.resolve("member-charter.md");
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp()) {
|
||||
ObjectNode bridge = root.putObject("mcp").putObject("bridge");
|
||||
bridge.put("type", "remote");
|
||||
bridge.put("url", cfg.mcpUrl());
|
||||
bridge.put("enabled", true);
|
||||
}
|
||||
if (hasCustomProvider(cfg)) {
|
||||
addCustomProvider(root, cfg);
|
||||
}
|
||||
|
||||
Path cfgFile = dir.resolve("opencode.json");
|
||||
// Built with Jackson rather than string concatenation: the provider block is nested and
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
"cannot write opencode config for profile " + cfg.profile(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
*
|
||||
* <p>The provider id comes from the {@code provider/model} selector in {@code model:}, so one
|
||||
* field drives both the declaration and the {@code -m} flag and they cannot drift apart.
|
||||
*/
|
||||
private void addCustomProvider(ObjectNode root, BridgedConfig.Profile cfg) {
|
||||
String[] parts = splitModelSelector(cfg);
|
||||
String providerId = parts[0];
|
||||
String modelId = parts[1];
|
||||
|
||||
ObjectNode provider = root.putObject("provider").putObject(providerId);
|
||||
provider.put("npm", "@ai-sdk/openai-compatible");
|
||||
provider.put("name", providerId + " (bridged)");
|
||||
|
||||
ObjectNode options = provider.putObject("options");
|
||||
options.put("baseURL", openAiBaseUrl(cfg.baseUrl()));
|
||||
// vLLM and friends usually ignore the key, but the AI SDK still requires a non-empty one.
|
||||
String token = resolveEnv(cfg.tokenEnv());
|
||||
options.put("apiKey", (token == null || token.isBlank()) ? "bridged-local-noauth" : token);
|
||||
|
||||
provider.putObject("models").putObject(modelId).put("name", modelId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Split {@code model:} into its {@code provider} and {@code model} halves. A pinned endpoint
|
||||
* needs both, so a bare model name is rejected loudly rather than silently falling back to the
|
||||
* default gateway — a worker quietly talking to the wrong endpoint is the failure this avoids.
|
||||
*/
|
||||
private static String[] splitModelSelector(BridgedConfig.Profile cfg) {
|
||||
String model = cfg.model();
|
||||
int slash = model == null ? -1 : model.indexOf('/');
|
||||
if (model == null || model.isBlank() || slash <= 0 || slash == model.length() - 1) {
|
||||
throw new IllegalArgumentException(
|
||||
"profile " + cfg.profile() + " sets baseUrl (a pinned opencode endpoint) so"
|
||||
+ " model: must be \"<provider>/<model>\", e.g."
|
||||
+ " \"local-vllm/deepseek-v4-flash\"; got "
|
||||
+ (model == null ? "null" : '"' + model + '"'));
|
||||
}
|
||||
return new String[]{model.substring(0, slash), model.substring(slash + 1)};
|
||||
}
|
||||
|
||||
/**
|
||||
* The OpenAI-compatible base URL for {@code baseUrl}. A bare {@code host:port} gets {@code /v1}
|
||||
* appended (where these servers put the API); a URL that already carries a path is taken as-is,
|
||||
* so an endpoint mounted somewhere unusual is still reachable.
|
||||
*/
|
||||
private static String openAiBaseUrl(String baseUrl) {
|
||||
String trimmed = baseUrl.trim();
|
||||
while (trimmed.endsWith("/")) {
|
||||
trimmed = trimmed.substring(0, trimmed.length() - 1);
|
||||
}
|
||||
int schemeEnd = trimmed.indexOf("://");
|
||||
String afterScheme = schemeEnd < 0 ? trimmed : trimmed.substring(schemeEnd + 3);
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
}
|
||||
|
||||
/**
|
||||
* A {@link PeerHandle} that delegates everything to the base's worker handle but resolves
|
||||
* {@link #agentSessionId()} lazily through opencode session discovery. Delegate-only, so the
|
||||
* base's id/terminalId/profile semantics (CB-519's host-unique routing key, herdr coordinates)
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String id() {
|
||||
return delegate.id();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String terminalId() {
|
||||
return delegate.terminalId();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String profile() {
|
||||
return delegate.profile();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String sessionName() {
|
||||
return delegate.sessionName();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
}
|
||||
|
||||
@Override
|
||||
public CharterReceipt charterReceipt() {
|
||||
return delegate.charterReceipt();
|
||||
}
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.ORPHAN_REAP, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
// Deliberately NOT SESSION_NAME: opencode has no display-name flag, so the bridge's logical
|
||||
// name can't surface in the peer's own UI — declaring the capability would hide that
|
||||
// asymmetry rather than make it honest. For opencode the name lives only in the bridge's
|
||||
// roster (see PeerHandle.sessionName() returning null).
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(BridgedConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (opencode prefix), kept for direct unit testing -----------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is an opencode bridge worker started by a <em>different</em> process than
|
||||
* {@code currentNonce}. A thin {@code opencode}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -1,122 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a bridged worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every bridged worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A bridged worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
*/
|
||||
String sessionIdForDirectory(String directory) {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
} catch (IOException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,831 +0,0 @@
|
||||
package dev.ltms.bridged.msg;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionException;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.locks.ReentrantLock;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* The blocking delegation feature (CB-104): deliver {@code content} into a worker and block until
|
||||
* the worker returns a <em>structured reply</em> via {@code bridge_reply} (the {@link Rendezvous}),
|
||||
* then hand that reply back. Delivery is the {@link Injector}'s job (the background poller sends it
|
||||
* when the worker is injectable); this service never drives the injector or scrapes the terminal —
|
||||
* completion is the worker's explicit reply, not a guess about {@code agent_status}.
|
||||
*
|
||||
* <p>Sends are serialized per session so exactly one reply can be outstanding per worker, which is
|
||||
* what lets a reply map unambiguously to its send (no cross-talk between concurrent callers).
|
||||
*
|
||||
* <p>If the worker never replies within the timeout, the caller gets a typed "still working" /
|
||||
* "queued" outcome — the message may still be mid-flight. A finished-but-unreplied turn is caught
|
||||
* by the CB-106 completion fallback (see {@link Rendezvous#resolveCompletion}).
|
||||
*
|
||||
* <p><strong>Async fire-and-poll (CB-107).</strong> A caller's MCP client caps a blocking call at
|
||||
* ~60s, but a real delegated task runs for minutes. {@link #sendAsync} therefore runs the same
|
||||
* blocking {@link #send} on a background virtual thread and hands back a <em>ticket</em> the caller
|
||||
* polls with {@link #poll}. The blocking and async paths share one code path (and the same per-target
|
||||
* serialization), so async inherits the reply + completion resolution behaviour for free.
|
||||
*/
|
||||
public final class MessageService {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MessageService.class);
|
||||
|
||||
/**
|
||||
* The window a fire-and-poll send waits for resolution — generous, since no caller is blocked on
|
||||
* it; a real delegated task resolves (reply or completion) well within this, and only a genuinely
|
||||
* hung worker rides it out.
|
||||
*/
|
||||
private static final long ASYNC_TIMEOUT_MS = 30 * 60 * 1_000L;
|
||||
|
||||
/**
|
||||
* How long a finished (terminal) ticket is retained for polling before it is pruned. Package-
|
||||
* private (not {@code private}) so a test can advance an injected clock past it deterministically
|
||||
* instead of duplicating the magic number or sleeping for real.
|
||||
*/
|
||||
static final long TICKET_TTL_NANOS = 10 * 60 * 1_000_000_000L;
|
||||
|
||||
/** Outcome of a blocking send. */
|
||||
public enum Outcome {
|
||||
/** The worker called {@code bridge_reply}; {@code text} holds the structured answer. */
|
||||
REPLIED,
|
||||
/**
|
||||
* The worker's delegated turn finished without a {@code bridge_reply} (CB-106 fallback);
|
||||
* {@code text} is the scraped transcript tail rather than a structured answer.
|
||||
*/
|
||||
COMPLETED_UNREPLIED,
|
||||
/**
|
||||
* The worker ran the turn then wedged in an unrecoverable state (CB-109); {@code text} is the
|
||||
* failure context (e.g. the error screen). Terminal, but not a successful completion.
|
||||
*/
|
||||
WORKER_FAILED,
|
||||
/**
|
||||
* The turn finished without a {@code bridge_reply} and the scrape matched the backend's
|
||||
* configured usage-limit refusal pattern (CB-578 stage A); {@code text} is the reason,
|
||||
* carrying the matched line. The worker's pane is healthy — only its account is refusing —
|
||||
* so this is never reported as a completed reply, and is kept distinct from
|
||||
* {@link #WORKER_FAILED} (a wedged worker) and a session simply going {@code GONE}.
|
||||
*/
|
||||
BACKEND_EXHAUSTED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||
* {@link #answer(String, String, long)} and the turn resumes.
|
||||
*/
|
||||
QUESTION,
|
||||
/** Timed out after the message was delivered — the worker is still working. */
|
||||
TIMED_OUT_WORKING,
|
||||
/** Timed out before delivery — the message is still queued for the worker. */
|
||||
TIMED_OUT_QUEUED,
|
||||
/** Another send to this session was in flight for the whole window. */
|
||||
BUSY,
|
||||
/**
|
||||
* An answer ({@link #answer(String, String, long)}) referenced a {@code turnId} that is no
|
||||
* longer open — the worker's {@code bridge_ask} already timed out or was answered.
|
||||
*/
|
||||
STALE_TURN
|
||||
}
|
||||
|
||||
/**
|
||||
* @param outcome how the send ended (or paused)
|
||||
* @param text the worker's answer when {@link #completed()} (a structured {@code bridge_reply}
|
||||
* for {@link Outcome#REPLIED}, a scraped transcript tail for
|
||||
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
||||
* else {@code null}
|
||||
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
||||
* {@link #answer(String, String, long)}), else {@code null}
|
||||
*/
|
||||
public record Reply(Outcome outcome, String text, String turnId) {
|
||||
/** A reply with no correlation id (the common terminal outcomes). */
|
||||
public Reply(Outcome outcome, String text) {
|
||||
this(outcome, text, null);
|
||||
}
|
||||
|
||||
/** Whether the worker's turn actually finished with an answer (replied or scraped). */
|
||||
public boolean completed() {
|
||||
return outcome == Outcome.REPLIED || outcome == Outcome.COMPLETED_UNREPLIED;
|
||||
}
|
||||
}
|
||||
|
||||
/** How a worker's {@code bridge_ask} (CB-205) resolved. */
|
||||
public enum AskOutcome {
|
||||
/** The primary answered; {@link AskResult#answer} carries it. */
|
||||
ANSWERED,
|
||||
/** No delegation was open to surface the question to — the worker has no one to ask. */
|
||||
NO_WAITER,
|
||||
/** The primary did not answer within the window. */
|
||||
TIMED_OUT
|
||||
}
|
||||
|
||||
/** The outcome of a worker's {@code bridge_ask}: how it resolved and (if answered) the answer. */
|
||||
public record AskResult(AskOutcome outcome, String answer) {
|
||||
}
|
||||
|
||||
/** Lifecycle phase of an async delegation ticket. */
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
PENDING,
|
||||
/** The worker is paused in {@code bridge_ask}; {@link TaskView#reply} and {@link TaskView#turnId} identify it. */
|
||||
ASKING,
|
||||
/** The worker's turn finished; {@link TaskView#reply} holds the answer. */
|
||||
DONE,
|
||||
/** The delegation could not complete (timed out, worker gone, or busy). */
|
||||
FAILED
|
||||
}
|
||||
|
||||
/**
|
||||
* A poll snapshot of an async delegation.
|
||||
*
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, or the question when
|
||||
* {@link #phase} is {@link Phase#ASKING}; otherwise {@code null}
|
||||
* @param replySource {@code "reply"} (structured {@code bridge_reply}) or {@code "transcript"}
|
||||
* (completion scrape) when {@link Phase#DONE}, else {@code null}
|
||||
* @param detail a human note (live worker status while pending, ask state, or failure reason)
|
||||
* @param turnId correlation id for an {@link Phase#ASKING} ticket, else {@code null}
|
||||
*/
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail,
|
||||
String turnId) {
|
||||
}
|
||||
|
||||
/** An in-flight or finished async delegation, keyed by its ticket. */
|
||||
private static final class Task {
|
||||
private final String ticket;
|
||||
private final String target;
|
||||
private final CompletableFuture<Reply> future = new CompletableFuture<>();
|
||||
private final long createdNanos;
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
|
||||
private Task(String ticket, String target, long createdNanos) {
|
||||
this.ticket = ticket;
|
||||
this.target = target;
|
||||
this.createdNanos = createdNanos;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker session's currently-open {@code bridge_ask} question, surfaced so {@code bridge_status}
|
||||
* can show it without the caller needing the ticket first (CB-582). Only covers async
|
||||
* (fire-and-poll) delegations, which track the question on their {@link Task}; a blocking
|
||||
* ({@code wait:true}) send already hands the question straight back to its own caller, so there is
|
||||
* nothing hidden left for {@code bridge_status} to surface in that case.
|
||||
*/
|
||||
public record PendingAsk(String ticket, String question, String turnId) {
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Injector injector;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ReplyInbox inbox;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
// CB-588: injectable so pruneTerminalTickets' 10-minute TICKET_TTL_NANOS can be exercised in a
|
||||
// test without a real wait — same seam SessionManager already uses for its idle reaper (nowNanos).
|
||||
private final LongSupplier nowNanos;
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
/** Async task that owns each exact forward rendezvous waiter. */
|
||||
private final ConcurrentHashMap<CompletableFuture<Rendezvous.Resolution>, Task> asyncTasksByWaiter =
|
||||
new ConcurrentHashMap<>();
|
||||
/** Async tickets paused on a specific {@code bridge_ask} turn. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTurn = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
|
||||
/**
|
||||
* Create with an explicit {@link ReplyInbox} and optional {@link ReplyPushLoop}.
|
||||
*
|
||||
* @param pushLoop nullable — when non-null, the push loop is notified on the no-waiter reply
|
||||
* branch ({@link #reply}) so it can nudge the primary to drain the inbox,
|
||||
* (CB-588) whenever an async ticket started by {@link #sendAsync} reaches a
|
||||
* terminal phase, whenever {@link #poll} hands a terminal ticket to its caller,
|
||||
* and (CB-582) whenever an async ticket's worker pauses mid-turn in
|
||||
* {@code bridge_ask} or that pause ends (answered or lapsed)
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with a metric registry (CB-502). Instrumenting here rather than at the REST and MCP
|
||||
* edges means both surfaces are counted by one piece of code and cannot drift.
|
||||
*
|
||||
* @param metrics nullable — when null, nothing is recorded
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop, Metrics metrics) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, metrics, System::nanoTime);
|
||||
}
|
||||
|
||||
/** Test constructor with an injectable clock (CB-588: exercise the ticket-prune TTL without a real wait). */
|
||||
MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox,
|
||||
ReplyPushLoop pushLoop, Metrics metrics, LongSupplier nowNanos) {
|
||||
this.agents = agents;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
this.pushLoop = pushLoop;
|
||||
this.metrics = metrics;
|
||||
this.nowNanos = nowNanos;
|
||||
}
|
||||
|
||||
/** Create with an explicit {@link ReplyInbox} and no push loop. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox) {
|
||||
this(agents, injector, rendezvous, inbox, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor that uses a default {@link InMemoryReplyInbox}. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous) {
|
||||
this(agents, injector, rendezvous, new InMemoryReplyInbox());
|
||||
}
|
||||
|
||||
/** Current lifecycle status of a worker (the {@code GET /sessions/{id}/status} surface). */
|
||||
public AgentStatus status(String target) {
|
||||
return agents.status(target);
|
||||
}
|
||||
|
||||
/** Read-only delegation fact for fleet views. */
|
||||
public boolean hasAcceptedDelivery(String target) {
|
||||
return rendezvous.isWaiting(target);
|
||||
}
|
||||
|
||||
/** Read-only inbox fact for fleet views. */
|
||||
public boolean hasInboxMessage(String target) {
|
||||
return !inbox.peek(target).isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code bridge_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
* result is <em>not</em> a failure — the reply is held for later drain.
|
||||
*
|
||||
* <p><strong>Do NOT use this for mid-turn questions.</strong> {@code bridge_ask} /
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* @return always {@code true} — the reply either resolved a live send or was queued
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(BridgedMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content);
|
||||
// A rising inbox share is the signal CB-307 exists to make visible: the worker finished but
|
||||
// nobody was waiting, so delivery now depends on the push loop and a drain.
|
||||
count(BridgedMetrics.REPLIES, "path", "inbox");
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
}
|
||||
|
||||
/** Record a counter sample when a registry is wired; a no-op in unit tests. */
|
||||
private void count(String name, String... labels) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(name, labels);
|
||||
}
|
||||
}
|
||||
|
||||
/** Count a send's terminal outcome and pass the reply through unchanged. */
|
||||
private Reply recorded(Reply r) {
|
||||
String label = sendOutcomeLabel(r.outcome());
|
||||
if (label != null) {
|
||||
count(BridgedMetrics.SENDS, "outcome", label);
|
||||
}
|
||||
return r;
|
||||
}
|
||||
|
||||
/** Map a terminal send outcome to its metric label, or {@code null} for non-terminal ones. */
|
||||
private static String sendOutcomeLabel(Outcome o) {
|
||||
return switch (o) {
|
||||
case REPLIED -> "replied";
|
||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Abandon any send still waiting on {@code target} because its session has gone away (CB-516).
|
||||
*
|
||||
* <p>Without this, tearing a worker down left its rendezvous waiter open: a blocking
|
||||
* {@code bridge_send} kept blocking, and an async one kept reporting {@code PENDING} until
|
||||
* {@link #ASYNC_TIMEOUT_MS} — thirty minutes — even though the worker provably no longer
|
||||
* existed and the delegation could never complete. Worse, {@code poll} already had the evidence
|
||||
* (it calls {@code liveStatus} to build its detail string and gets back {@code "unknown"}) and
|
||||
* reported {@code PENDING} anyway.
|
||||
*
|
||||
* <p>Resolving the waiter as a failure — rather than letting it time out — also means the
|
||||
* outcome is counted, so a torn-down delegation stops being invisible to {@code /metrics}.
|
||||
*
|
||||
* @return true if a live waiter was failed
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
boolean failed = waiter != null && !waiter.isDone() && rendezvous.resolveFailure(waiter, reason);
|
||||
boolean asyncFailed = false;
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null
|
||||
&& task.future.complete(new Reply(Outcome.WORKER_FAILED, reason))) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
}
|
||||
if (failed) {
|
||||
log.warn("abandoning the blocked send to {}: {}", target, reason);
|
||||
}
|
||||
return failed || asyncFailed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain (peek + ack) all pending inbox replies for {@code target}.
|
||||
*
|
||||
* <p><strong>The ack happens here, before the caller has the messages</strong> — before the MCP
|
||||
* or REST response carrying them has been written, and long before the client has processed
|
||||
* them. That ordering is what the two adapters disagree about, so do not read this method as
|
||||
* "at-least-once" without qualifying which inbox is behind it (CB-529):
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code InMemoryReplyInbox} — the ack only drops an entry from a local map. The messages
|
||||
* are already in the returned list, so nothing can be lost after this point.
|
||||
* <li>{@code AmqpReplyInbox} — the ack is a broker-side {@code basicAck}. Once it lands the
|
||||
* broker has forgotten the message. If the daemon dies while writing the response, the
|
||||
* reply is gone from the broker <em>and</em> the client never received it. Re-polling
|
||||
* cannot recover it, because there is nothing left to re-deliver.
|
||||
* </ul>
|
||||
*
|
||||
* <p>So the loss window is the response write, and it is a genuine loss rather than a
|
||||
* redelivery. This is accepted, not overlooked: the alternative — ack on the next poll — turns
|
||||
* every normal drain into a double delivery, which costs more than the window it closes. A
|
||||
* caller that needs certainty re-polls; that is idempotent for every case except this one.
|
||||
*
|
||||
* <p>Any change here must be checked against <em>both</em> adapters. The previous version of
|
||||
* this javadoc claimed "the ack is local", which was true when only the in-memory inbox existed
|
||||
* and silently became false when the AMQP adapter landed.
|
||||
*
|
||||
* @return the drained messages, newest last (FIFO); empty list if none
|
||||
*/
|
||||
public List<ReplyInbox.InboxMessage> drainReplies(String target) {
|
||||
var messages = inbox.peek(target);
|
||||
for (var msg : messages) {
|
||||
inbox.ack(target, msg.msgId());
|
||||
}
|
||||
return messages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis) {
|
||||
return send(target, content, timeoutMillis, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
||||
*
|
||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||
* queued, so a throwing hook fails the send cleanly (the waiter it already opened is closed and
|
||||
* nothing is left queued). It is <em>not</em> invoked when the send is {@link Outcome#BUSY}
|
||||
* (lock never taken). A caller uses this to record that <em>it</em> now owns the delegation's
|
||||
* reply routing (CB-548: {@code PrimaryRegistry} delegator ownership) — recording only on
|
||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
return send(target, content, timeoutMillis, onAccepted, null);
|
||||
}
|
||||
|
||||
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
if (task != null && task.future.isDone()) {
|
||||
return task.future.getNow(null);
|
||||
}
|
||||
if (hasAsyncQuestion(target)) {
|
||||
return new Reply(Outcome.BUSY, null); // the worker's current turn is paused for its lead
|
||||
}
|
||||
// Open the waiter BEFORE queueing delivery (CB-548). A fast reply — the worker already
|
||||
// injectable the instant we enqueue — otherwise arrives before the waiter is registered
|
||||
// and orphans into the inbox while this send blocks to the timeout (the enqueue-before-
|
||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
try {
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
TurnToken token = new TurnToken(target, reply);
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content, token);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
return recorded(new Reply(
|
||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
}
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(target, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker's mid-turn question (CB-205 reverse rendezvous): surface {@code question} to the
|
||||
* primary by resolving its open blocking {@code bridge_send}, then block this (worker) call until
|
||||
* the primary answers via {@link #answer} or {@code timeoutMillis} elapses. Identity is the
|
||||
* worker's own session — it does not address the primary.
|
||||
*
|
||||
* <p>Returns {@link AskOutcome#NO_WAITER} when no delegation is open to surface the question to
|
||||
* (nothing to answer it), {@link AskOutcome#ANSWERED} with the primary's answer, or
|
||||
* {@link AskOutcome#TIMED_OUT} if the primary stayed silent. The worker resumes its turn either
|
||||
* way — an answered ask hands back the answer; an unanswered one leaves it to proceed alone.
|
||||
*/
|
||||
public AskResult ask(String workerSession, String question, long timeoutMillis) {
|
||||
Rendezvous.AskTicket ticket = rendezvous.openAsk(workerSession);
|
||||
// Only the freshly-opening caller surfaces the question; a coalesced duplicate simply blocks on
|
||||
// the shared answer future that the fresh owner is already responsible for.
|
||||
if (ticket.fresh()) {
|
||||
// Register the reverse waiter first, then surface the question — so the answer, which can
|
||||
// arrive the instant the primary reacts, always finds an open waiter to resolve.
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(workerSession);
|
||||
Task task = markAsyncQuestion(waiter, question, ticket.turnId());
|
||||
if (!rendezvous.resolveQuestion(workerSession, question, ticket.turnId())) {
|
||||
if (task != null) {
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
}
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
// CB-582: the question just became visible via bridge_poll (Phase.ASKING) for an async
|
||||
// (wait:false) delegation — nudge the lead's own pane the same way a terminal ticket does
|
||||
// (CB-588), since the lead's normal poll cadence is minutes away and the reverse-rendezvous
|
||||
// window (~55s, see BridgeMcp/BridgedApp) is far shorter. A blocking (wait:true) send has
|
||||
// no Task and gets the question directly in its own reply, so task == null there — nothing
|
||||
// to nudge.
|
||||
if (task != null && pushLoop != null) {
|
||||
pushLoop.onQuestionOpened(task.ticket, workerSession, ticket.turnId(), question);
|
||||
}
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("bridge_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the primary's answer for " + workerSession, e);
|
||||
} finally {
|
||||
// Only the fresh owner tears down the shared turn; a duplicate must leave it open.
|
||||
if (ticket.fresh()) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
// CB-582: tear the push loop's copy down at the same point, not only on the three
|
||||
// paths that call clearAsyncQuestion. The answer future can complete exceptionally
|
||||
// (ExecutionException) or the thread be interrupted, and both leave this method by
|
||||
// throwing — the question would stay pending forever, keep being named in nudges
|
||||
// until its own cap, and never be removed from the map. Already-closed is a no-op.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.questionClosed(ticket.turnId());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The primary's answer to a worker's {@code bridge_ask} (CB-205): resolve the worker's blocked
|
||||
* question identified by {@code turnId}, then — like a fresh {@link #send} — block for the worker's
|
||||
* eventual {@code bridge_reply} as it finishes the resumed turn. The worker session is derived from
|
||||
* {@code turnId}, never a caller argument.
|
||||
*
|
||||
* <p>Unlike {@link #send} this does not re-inject through the {@link Injector}: the worker is
|
||||
* mid-turn (already picked up), so the answer flows back through its own open {@code bridge_ask}
|
||||
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
||||
* is unblocked so a reply that lands the instant it resumes is not lost.
|
||||
*/
|
||||
public Reply answer(String turnId, String content, long timeoutMillis) {
|
||||
String workerSession = rendezvous.askSession(turnId);
|
||||
if (workerSession == null) {
|
||||
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
||||
}
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null);
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
clearAsyncQuestion(turnId, false);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
// turn (it never re-entered the injector), so a silent worker rides out the window.
|
||||
return new Reply(Outcome.TIMED_OUT_WORKING, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fire-and-poll variant of {@link #send}: deliver {@code content} to {@code target} on a
|
||||
* background virtual thread and return immediately with a ticket to {@link #poll}. This is how a
|
||||
* long task is delegated without tripping the caller's MCP client call timeout.
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content) {
|
||||
return sendAsync(target, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(String, String)}, with the accepted-delivery hook of
|
||||
* {@link #send(String, String, long, Runnable)} — the running {@code send} invokes {@code onAccepted}
|
||||
* the moment it becomes the accepted target turn, so async flooding records delegator ownership
|
||||
* exactly as the blocking path does (CB-548).
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
Task task = new Task(ticket, target, nowNanos.getAsLong());
|
||||
tasks.put(ticket, task);
|
||||
if (pushLoop != null) {
|
||||
// CB-588: task.future only ever completes on a terminal phase (DONE or a failure) — a
|
||||
// worker paused in bridge_ask leaves it running, per finishAsyncTask's own contract — so
|
||||
// this fires exactly once, from whichever path completes it: finishAsyncTask(task, result)
|
||||
// below on any non-QUESTION outcome of send() — a worker's bridge_reply, the CB-106
|
||||
// completion fallback, a CB-109 wedge, TIMED_OUT, BUSY, or BACKEND_EXHAUSTED — the same
|
||||
// finishAsyncTask reached via answer()'s finishAsyncTask(turnId, result) once a QUESTION
|
||||
// is resolved, completeExceptionally(t) just below when send() itself throws, or a CB-516
|
||||
// abandon() on teardown. Without this, MessageService.reply's rendezvous fast path (the
|
||||
// one an async ticket always takes) never told the push loop anything happened — see the
|
||||
// class javadoc on sendAsync/CB-107.
|
||||
task.future.whenComplete((reply, ex) -> {
|
||||
boolean failed = ex != null || reply == null || !reply.completed();
|
||||
pushLoop.onTicketTerminal(ticket, target, failed);
|
||||
});
|
||||
}
|
||||
asyncExecutor.submit(() -> {
|
||||
try {
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task);
|
||||
if (result.outcome() == Outcome.QUESTION) {
|
||||
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
||||
// just after resolveQuestion wakes this thread.
|
||||
} else {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
log.debug("async send {} -> {}", ticket, target);
|
||||
return ticket;
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket;
|
||||
* otherwise a {@link Phase#PENDING} view (with the live worker status as detail), a
|
||||
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
||||
*/
|
||||
public TaskView poll(String ticket) {
|
||||
Task task = tasks.get(ticket);
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future;
|
||||
if (!f.isDone()) {
|
||||
Reply question = task.question;
|
||||
if (question != null) {
|
||||
return new TaskView(ticket, Phase.ASKING, question.text(), null,
|
||||
"worker is waiting for your answer", question.turnId());
|
||||
}
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target), null);
|
||||
}
|
||||
// CB-588: the ticket is terminal and being handed to the caller right here — tell the push
|
||||
// loop it is collected so a later tick's nudge never names a ticket the lead already has.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.ticketCollected(ticket);
|
||||
}
|
||||
Reply r;
|
||||
try {
|
||||
r = f.getNow(null);
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
Throwable cause = (e instanceof CompletionException ce && ce.getCause() != null) ? ce.getCause() : e;
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage(), null);
|
||||
}
|
||||
if (r.completed()) {
|
||||
String source = r.outcome() == Outcome.REPLIED ? "reply" : "transcript";
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null, null);
|
||||
}
|
||||
// A wedged worker (CB-109) or a backend-exhausted classification (CB-578 stage A) carries
|
||||
// the real cause as its reason; the timeout/busy outcomes carry none, so fall back to the
|
||||
// outcome name.
|
||||
boolean carriesReason = r.outcome() == Outcome.WORKER_FAILED || r.outcome() == Outcome.BACKEND_EXHAUSTED;
|
||||
String detail = carriesReason && r.text() != null
|
||||
? r.text()
|
||||
: "no reply — " + r.outcome().name().toLowerCase();
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
return agents.status(target).name().toLowerCase();
|
||||
} catch (RuntimeException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop finished tickets older than the TTL so {@link #tasks} cannot grow without bound.
|
||||
*
|
||||
* <p>{@code tasks} is the sole authority on whether a ticket still exists — {@link #poll} returns
|
||||
* {@code null} the instant a ticket is gone from here, before it ever reaches the terminal branch
|
||||
* that calls {@link ReplyPushLoop#ticketCollected}. Without telling the push loop about a prune
|
||||
* too, its own {@code pendingTickets} entry would outlive the ticket it names: an unpolled ticket
|
||||
* (or one the reminder cap already gave up on) is pruned here but never collected there, so it
|
||||
* lingers in {@code pendingTickets} forever and rides along on every later nudge to the same lead
|
||||
* — naming a ticket {@code bridge_poll} can no longer find (CB-588 follow-up).
|
||||
*/
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = nowNanos.getAsLong() - TICKET_TTL_NANOS;
|
||||
tasks.entrySet().removeIf(e -> {
|
||||
Task t = e.getValue();
|
||||
boolean expired = t.future.isDone() && t.createdNanos < cutoff;
|
||||
if (expired && pushLoop != null) {
|
||||
pushLoop.ticketCollected(e.getKey());
|
||||
}
|
||||
return expired;
|
||||
});
|
||||
}
|
||||
|
||||
/** Record the active question for an async ticket; blocking sends have no entry and stay unchanged. */
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
// CB-582: tell the push loop first — like ticketCollected, a removal for a turnId it never
|
||||
// nudged about (or already dropped) is a harmless no-op, so this is safe to call unconditionally
|
||||
// rather than threading the guard below through it.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.questionClosed(turnId);
|
||||
}
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.question = null;
|
||||
if (forgetTurn) {
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
task.turnId = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete the async ticket correlated to a specific answered turn. */
|
||||
private void finishAsyncTask(String turnId, Reply result) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
private boolean hasAsyncQuestion(String target) {
|
||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||
}
|
||||
|
||||
/**
|
||||
* The question {@code workerSession} is currently paused on via {@code bridge_ask}, if any
|
||||
* (CB-582) — {@code bridge_status} uses this to show a pending question without the caller
|
||||
* needing the ticket. {@code null} when the session has no open async question (including a
|
||||
* session mid a <em>blocking</em> {@code bridge_ask}, which has no {@link Task} to look up — see
|
||||
* {@link PendingAsk}).
|
||||
*/
|
||||
public PendingAsk pendingAsk(String workerSession) {
|
||||
for (Task task : tasks.values()) {
|
||||
Reply q = task.question;
|
||||
if (q != null && workerSession.equals(task.target)) {
|
||||
return new PendingAsk(task.ticket, q.text(), q.turnId());
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
public void close() {
|
||||
asyncExecutor.shutdown();
|
||||
}
|
||||
|
||||
/** Map a rendezvous {@link Rendezvous.Kind} onto its send {@link Outcome} (shared by send/answer). */
|
||||
private static Outcome outcomeOf(Rendezvous.Kind kind) {
|
||||
return switch (kind) {
|
||||
case REPLY -> Outcome.REPLIED;
|
||||
case COMPLETION -> Outcome.COMPLETED_UNREPLIED;
|
||||
case FAILED -> Outcome.WORKER_FAILED;
|
||||
case BACKEND_EXHAUSTED -> Outcome.BACKEND_EXHAUSTED;
|
||||
case QUESTION -> Outcome.QUESTION;
|
||||
};
|
||||
}
|
||||
|
||||
private static boolean tryLock(ReentrantLock lock, long millis) {
|
||||
try {
|
||||
return lock.tryLock(Math.max(0, millis), TimeUnit.MILLISECONDS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the session send lock", e);
|
||||
}
|
||||
}
|
||||
|
||||
private static long remainingMillis(long deadlineNanos) {
|
||||
return (deadlineNanos - System.nanoTime()) / 1_000_000L;
|
||||
}
|
||||
}
|
||||
@@ -1,107 +0,0 @@
|
||||
package dev.ltms.bridged.peer;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* SPI for materializing a connected peer — the only way the bridge core creates or tears down
|
||||
* a peer process. Every launcher is a first-party, in-tree adapter selected by (future) profile
|
||||
* config; today's single adapter is the {@code ClaudeCodeLauncher} / Claude Code over herdr.
|
||||
*
|
||||
* <p>The core delegates spawn and teardown to this interface without knowing how the peer is set
|
||||
* up. Environment variables, CLI flags, subscription guards, transport (herdr tab/pane) layout,
|
||||
* and naming conventions are all adapter-private — the core sees only the returned
|
||||
* {@link PeerHandle} whose {@code id()} is the registry/routing key.
|
||||
*
|
||||
* <p>The interface is a superset of what {@code SessionManager} and {@code Bridged.main} call
|
||||
* on the concrete launcher today.
|
||||
*/
|
||||
public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The set of {@link Capability capabilities} this launcher declares. A peer whose profile
|
||||
* opts into a git-forge token should include {@link Capability#SELF_PR}; the base set for
|
||||
* the Claude Code herdr adapter is always {@code MID_TURN_ASK, WORKTREE, ORPHAN_REAP}.
|
||||
*/
|
||||
Set<Capability> capabilities();
|
||||
|
||||
/**
|
||||
* The capabilities of the adapter that {@code profileName} resolves to (null/blank → the
|
||||
* default profile, the same resolution {@link #spawn} uses). Distinct from {@link
|
||||
* #capabilities()}, which unions every configured adapter: a caller that must know whether
|
||||
* <em>this</em> profile's backend supports a capability — e.g. {@link Capability#SESSION_RESUME}
|
||||
* before honoring {@link SpawnRequest#resumeSessionId()} — needs the per-profile answer, not
|
||||
* the fleet-wide union, or a mixed fleet could OK a resume that lands on a non-supporting
|
||||
* adapter (CB-584).
|
||||
*
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
Set<Capability> capabilitiesFor(String profileName);
|
||||
|
||||
/**
|
||||
* {@code profileName}/requestedCwd null/blank → default resolution. Returns after the peer
|
||||
* process is live (env + argv + placement complete). Never returns {@code null}.
|
||||
*
|
||||
* @param req the spawn parameters (profile, requested cwd, caller cwd)
|
||||
* @return a handle whose {@link PeerHandle#id()} is the registry/routing key
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
PeerHandle spawn(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The configured worker profile names — the set of names {@code spawn(profileName)} accepts.
|
||||
*/
|
||||
Set<String> profiles();
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
*
|
||||
* @return the resolved absolute path, never null/blank
|
||||
*/
|
||||
String effectiveCwd(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The parity-overlay file list for {@code profileName} (default list when unset). Used by
|
||||
* worktree provisioning to copy config files into the isolated checkout before spawning.
|
||||
*/
|
||||
List<String> parityOverlay(String profileName);
|
||||
|
||||
/**
|
||||
* The set of all agents this launcher currently tracks, transport-specific. Each element
|
||||
* exposes at minimum a pane-like {@code id()} matching this launcher's {@link PeerHandle}
|
||||
* scheme, plus transport-level status. Callers merge this set with the session registry to
|
||||
* build a live roster view.
|
||||
*/
|
||||
List<?> list();
|
||||
|
||||
/**
|
||||
* Reap orphaned peers left behind by a prior daemon process. Only peers whose naming scheme
|
||||
* matches this launcher's and whose nonce differs from the current process are eligible.
|
||||
* Best-effort: a failure to list or to stop any one peer is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
int reapOrphanWorkers();
|
||||
|
||||
/**
|
||||
* Tear a peer down by its registry/routing key ({@link PeerHandle#id()}). Tolerates an
|
||||
* already-gone peer. Also cleans up launcher-private resources (e.g. empty dedicated tabs)
|
||||
* when safe to do so.
|
||||
*/
|
||||
void stop(String id);
|
||||
|
||||
/**
|
||||
* Discard the context of the peer identified by {@code id}. Implementations must bypass normal
|
||||
* bridge delivery/turn accounting. Unsupported peer kinds return {@code false} without sending
|
||||
* a guessed command.
|
||||
*
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
}
|
||||
@@ -1,66 +0,0 @@
|
||||
package dev.ltms.bridged.placement;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
*
|
||||
* <p>Two exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
* <li>Weight 0 (CB-554): {@code fixed} is still automatic selection, so a profile the operator
|
||||
* marked "never auto-select me" ({@code weight <= 0}) must be skipped here exactly as
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code bridge_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined or weight-0 never exercises either path, so today's
|
||||
* behaviour is unchanged.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !weightExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !c.excluded()) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
if (d != null && !d.isBlank()) {
|
||||
boolean dQuarantined = ctx.quarantined().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined && dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and has weight 0 (excluded from automatic selection), and no "
|
||||
+ "available candidate remains");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' has weight 0 (excluded "
|
||||
+ "from automatic selection) and no available candidate remains");
|
||||
}
|
||||
if (dQuarantined) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and no un-quarantined candidate is available");
|
||||
}
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException(
|
||||
"all worker profiles are excluded from automatic selection (quarantined or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
/** Whether {@code profile} carries {@code weight <= 0} (CB-554) among {@code ctx}'s candidates. */
|
||||
private static boolean weightExcluded(PlacementContext ctx, String profile) {
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.profile().equals(profile)) {
|
||||
return c.excluded();
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -1,510 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStreamReader;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* Production {@link Worktrees} implementation that shells {@code git} via {@link ProcessBuilder}.
|
||||
* Non-zero exits become {@link WorktreeException}. Worktree directories live under a configurable
|
||||
* root (default: a sibling {@code .bridged-worktrees} of the repo root) so they are never nested
|
||||
* inside the primary working tree.
|
||||
*/
|
||||
public final class GitWorktrees implements Worktrees {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(GitWorktrees.class);
|
||||
|
||||
/** Project-level MCP config. Present in the repo, so every worktree would otherwise inherit the
|
||||
* primary's IDE server mounts (a CB-523 worker edited the primary checkout; see the isolation
|
||||
* javadoc). Neutralized unconditionally. */
|
||||
private static final String MCP_CONFIG = ".mcp.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .mcp.json}: a valid, explicitly empty server map. */
|
||||
private static final String NEUTRAL_MCP_CONFIG = "{\n \"mcpServers\": {}\n}\n";
|
||||
|
||||
/** OpenCode's repo-level config. Tracked here, so it lands in every worktree, and it mounts the
|
||||
* primary's gitea and context7 servers with the primary's credentials. Neutralized so the worker
|
||||
* gets only the config its launcher writes via {@code OPENCODE_CONFIG}.
|
||||
*
|
||||
* <p>The reason has changed shape and is now stronger. It used to be a crash: the file carried
|
||||
* {@code {file:.secrets/...}} references to gitignored files that never reached a worktree, and
|
||||
* opencode refuses to start on a dangling reference (CB-543). Those credentials now live in one
|
||||
* shell-level store and the file reads them as {@code {env:...}}, so in a worktree the reference
|
||||
* resolves instead of failing. That is worse, not better: a member would silently inherit the
|
||||
* primary's admin-scoped {@code GITEA_ACCESS_TOKEN}. A loud crash became a quiet privilege leak,
|
||||
* so this entry protects a boundary now rather than papering over a startup error. */
|
||||
private static final String OPENCODE_CONFIG = "opencode.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code opencode.json}: a valid, empty JSON object. */
|
||||
private static final String NEUTRAL_OPENCODE_CONFIG = "{}\n";
|
||||
|
||||
/** Autoenv's repo-level config. Not tracked today, but re-landing it must stay safe: autoenv
|
||||
* authorizes by path, so a fresh worktree path is always unauthorized and its interactive prompt
|
||||
* would block every spawn — neutralize it so it can never be committed. */
|
||||
private static final String AUTOENV_CONFIG = ".autoenv";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .autoenv}: a valid, empty env file. */
|
||||
private static final String NEUTRAL_AUTOENV_CONFIG = "";
|
||||
|
||||
/**
|
||||
* A tracked project config that is hostile in a provisioned worktree, and what to replace it
|
||||
* with. {@link #file} is the repo-relative path; {@link #stub} is a neutral but VALID payload for
|
||||
* that file's format — a malformed stub would only trade one crash for another;
|
||||
* {@link #createIfAbsent} keeps {@code .mcp.json}'s long-standing behaviour of writing its stub
|
||||
* even when the repo carries no such file, whereas the others are only touched when present.
|
||||
*/
|
||||
private record WorktreeHostileConfig(String file, String stub, boolean createIfAbsent) {}
|
||||
|
||||
/** The worktree-hostile configs neutralized in every provisioned worktree, in order. */
|
||||
private static final List<WorktreeHostileConfig> WORKTREE_HOSTILE_CONFIGS = List.of(
|
||||
new WorktreeHostileConfig(MCP_CONFIG, NEUTRAL_MCP_CONFIG, true),
|
||||
new WorktreeHostileConfig(OPENCODE_CONFIG, NEUTRAL_OPENCODE_CONFIG, false),
|
||||
new WorktreeHostileConfig(AUTOENV_CONFIG, NEUTRAL_AUTOENV_CONFIG, false)
|
||||
);
|
||||
|
||||
private final String configuredRoot;
|
||||
private final SecureRandom random = new SecureRandom();
|
||||
private final AtomicLong seq = new AtomicLong();
|
||||
|
||||
/** Default constructor: worktree root is derived per-repo as {@code <repoRoot>/../.bridged-worktrees}. */
|
||||
public GitWorktrees() {
|
||||
this(null);
|
||||
}
|
||||
|
||||
/** @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of the repo root. */
|
||||
public GitWorktrees(String configuredRoot) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
String base = (baseRef == null || baseRef.isBlank()) ? "HEAD" : baseRef;
|
||||
String nonce = nonce();
|
||||
Path root = resolveRoot(repoRoot);
|
||||
Path path = root.resolve(nonce);
|
||||
try {
|
||||
Files.createDirectories(root);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
isolateToolSurface(wt);
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Neutralize the worktree's worktree-hostile project configs so a worker inherits only the tools
|
||||
* and environment its launcher mounts (the bridge via {@code --mcp-config}, the opencode config
|
||||
* via {@code OPENCODE_CONFIG}) — never the primary's.
|
||||
*
|
||||
* <p>This is unconditional, and it is not the same job as the parity overlay. The repo's own
|
||||
* committed {@code .mcp.json} declares the primary's IDE servers, so a fresh checkout mounts them
|
||||
* whether or not the overlay copies anything; a worker that inherits them navigates and edits
|
||||
* through tools bound to the <em>primary's</em> IntelliJ project, which silently hands it absolute
|
||||
* paths outside its own worktree. That is not hypothetical: a CB-523 worker made all 59 of its
|
||||
* edits in the primary checkout while compiling its worktree, so every build it ran was of code
|
||||
* that did not contain its changes. {@code opencode.json} is the same trap one tool over — tracked,
|
||||
* so it lands in every worktree, and it mounts gitea and context7 with the primary's own
|
||||
* credentials, which a member must never hold. {@code .autoenv} extends the principle to a
|
||||
* config that is not tracked today: autoenv authorizes by path, so a fresh worktree path is always
|
||||
* unauthorized and its interactive prompt would block every spawn, so re-landing one must be safe.
|
||||
*
|
||||
* <p>Where a config exists it is replaced by a valid neutral stub (an explicitly empty
|
||||
* map/object, or an empty env file — never a deletion, which would still let a later
|
||||
* {@code git checkout} restore the hostile copy). The {@code --skip-worktree} bit keeps the
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
}
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot neutralize " + cfg.file() + " in the worktree: "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing to remove", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.info("removing worktree {}", worktreePath);
|
||||
exec("git", "-C", repoRoot, "worktree", "remove", "--force", worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// A worktree that is already gone holds no work to lose, and it must not break teardown:
|
||||
// git -C <missing-dir> status exits non-zero and would throw where release() is mid-way
|
||||
// through stopping a pane. Mirror remove()'s already-gone tolerance by treating it as clean.
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing can be uncommitted", worktreePath);
|
||||
return false;
|
||||
}
|
||||
// No --untracked-files=no: the exact shape of the work lost in CB-576 was a new file
|
||||
// that was never added, so an untracked-only worktree is still dirty.
|
||||
String out = exec("git", "-C", worktreePath, "status", "--porcelain");
|
||||
return !out.isBlank();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlay == null || overlay.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Path srcRoot = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
Path dstRoot = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (String rel : overlay) {
|
||||
Path src = srcRoot.resolve(rel).normalize();
|
||||
if (!Files.exists(src)) {
|
||||
log.debug("parity overlay source missing — skipping {}", rel);
|
||||
continue;
|
||||
}
|
||||
Path dst = dstRoot.resolve(rel).normalize();
|
||||
try {
|
||||
Files.createDirectories(dst.getParent());
|
||||
Files.copy(src, dst, StandardCopyOption.REPLACE_EXISTING, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
log.debug("copied parity overlay {}", rel);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy overlay " + rel + ": " + e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(dstRoot, rel)) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", rel);
|
||||
log.debug("marked overlay --skip-worktree {}", rel);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
String out = exec("git", "-C", cwd, "rev-parse", "--show-toplevel");
|
||||
return Path.of(out.trim()).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, fixed by CB-587. Stages into a <em>temporary</em> index (never the worktree's
|
||||
* real one, which the worker may still be writing to) — but that temp index is first <em>seeded</em>
|
||||
* from the worktree's real one, rather than starting empty:
|
||||
*
|
||||
* <pre>
|
||||
* cp $(git -C worktree rev-parse --git-path index) <temp>
|
||||
* GIT_INDEX_FILE=<temp> git -C worktree add -A
|
||||
* tree=$(GIT_INDEX_FILE=<temp> git -C worktree write-tree)
|
||||
* commit=$(git -C worktree commit-tree $tree -p HEAD -m message)
|
||||
* git -C worktree update-ref refs/wip/branch $commit
|
||||
* </pre>
|
||||
*
|
||||
* A fresh empty index carries none of the real index's {@code --skip-worktree} /
|
||||
* {@code --assume-unchanged} bits, so {@code add -A} into it stages a skip-worktree file's local
|
||||
* on-disk content even though {@code git status --porcelain} correctly hides that file (CB-587).
|
||||
* Seeding from the real index preserves those bits, so {@code add -A} then skips exactly what
|
||||
* {@code git status} skips. {@code add -A} (never {@code -f}) also still respects
|
||||
* {@code .gitignore} exactly as it would in the real index — a gitignored file staying ignored is
|
||||
* what keeps secrets and local config out of the snapshot's tree. The temporary index file is
|
||||
* removed afterwards regardless of outcome; the worker's real index is never opened for writing.
|
||||
*/
|
||||
@Override
|
||||
public Optional<String> snapshot(String worktreePath, String branch, String message) {
|
||||
if (!Files.exists(Path.of(worktreePath))) {
|
||||
log.debug("worktree {} already gone — nothing to snapshot", worktreePath);
|
||||
return Optional.empty();
|
||||
}
|
||||
Path tempIndex;
|
||||
try {
|
||||
tempIndex = Files.createTempFile("bridged-wip-index-", ".tmp");
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create a temporary index for snapshot: " + e.getMessage(), e);
|
||||
}
|
||||
Map<String, String> indexEnv = Map.of("GIT_INDEX_FILE", tempIndex.toAbsolutePath().toString());
|
||||
try {
|
||||
Path realIndex = resolveRealIndex(worktreePath);
|
||||
try {
|
||||
Files.copy(realIndex, tempIndex, StandardCopyOption.REPLACE_EXISTING);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy the worktree's real index (" + realIndex
|
||||
+ ") into the temporary snapshot index: " + e.getMessage(), e);
|
||||
}
|
||||
exec(indexEnv, "git", "-C", worktreePath, "add", "-A");
|
||||
String tree = exec(indexEnv, "git", "-C", worktreePath, "write-tree").trim();
|
||||
String commit = exec("git", "-C", worktreePath, "commit-tree", tree, "-p", "HEAD", "-m", message).trim();
|
||||
exec("git", "-C", worktreePath, "update-ref", "refs/wip/" + branch, commit);
|
||||
log.info("snapshotted worktree {} to refs/wip/{} commit={}", worktreePath, branch, commit);
|
||||
return Optional.of(commit);
|
||||
} finally {
|
||||
try {
|
||||
Files.deleteIfExists(tempIndex);
|
||||
} catch (IOException e) {
|
||||
log.debug("could not delete temporary snapshot index {}: {}", tempIndex, e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the path of {@code worktreePath}'s real index. Never assume {@code <worktree>/.git/index}:
|
||||
* in a linked worktree {@code .git} is a <em>file</em> pointing at the main repo's
|
||||
* {@code worktrees/<name>/} directory, and that is where the real per-worktree index lives.
|
||||
* {@code git rev-parse --git-path index} resolves this correctly for both a linked worktree and
|
||||
* the main checkout. Throws {@link WorktreeException} — same as every other failure in this
|
||||
* class — if the command fails or the resolved path does not exist, rather than silently
|
||||
* snapshotting from an empty index.
|
||||
*/
|
||||
private Path resolveRealIndex(String worktreePath) {
|
||||
String out = exec("git", "-C", worktreePath, "rev-parse", "--git-path", "index").trim();
|
||||
Path index = Path.of(out);
|
||||
if (!index.isAbsolute()) {
|
||||
index = Path.of(worktreePath).resolve(index).normalize();
|
||||
}
|
||||
if (!Files.exists(index)) {
|
||||
throw new WorktreeException("worktree's real index not found at resolved path " + index
|
||||
+ " (git rev-parse --git-path index reported '" + out + "')");
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
/**
|
||||
* One {@code refs/wip/<branch>} snapshot ref as read by {@link #listWipRefs}: its full ref name,
|
||||
* the snapshot commit's sha, and that commit's committer time in unix millis (the age of the
|
||||
* snapshot — a snapshot is written once and never rewritten, so the commit date is the ref's).
|
||||
*/
|
||||
private record WipRef(String refName, String sha, long committerMillis) {
|
||||
String branch() {
|
||||
return refName.substring("refs/wip/".length());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public WipRefStats wipRefs(String repoRoot) {
|
||||
List<WipRef> refs = listWipRefs(repoRoot);
|
||||
long costBytes = 0;
|
||||
for (WipRef ref : refs) {
|
||||
costBytes += treeSize(repoRoot, ref.sha());
|
||||
}
|
||||
return new WipRefStats(refs.size(), costBytes);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
|
||||
// The rule is documented on Worktrees#pruneWipRefs: delete only a snapshot whose tree
|
||||
// content is already reachable from main AND that is older than minAgeMillis. Reachability
|
||||
// is the floor that keeps a worker's last copy; the age floor keeps a just-written snapshot
|
||||
// from being swept while a lead may still be looking at it.
|
||||
List<WipRef> refs = listWipRefs(repoRoot);
|
||||
if (refs.isEmpty()) {
|
||||
return 0;
|
||||
}
|
||||
long nowMillis = System.currentTimeMillis();
|
||||
// Resolve what main carries once per sweep, not once per ref.
|
||||
Set<String> mainObjects = reachableObjectsFromMain(repoRoot);
|
||||
int deleted = 0;
|
||||
for (WipRef ref : refs) {
|
||||
long ageMillis = nowMillis - ref.committerMillis();
|
||||
if (ageMillis <= minAgeMillis) {
|
||||
continue; // too recent — never swept, even if it looks recoverable (CB-586)
|
||||
}
|
||||
String tree = exec("git", "-C", repoRoot, "rev-parse", ref.sha() + "^{tree}").trim();
|
||||
if (!mainObjects.contains(tree)) {
|
||||
// Last copy of the snapshot's content — the worker's work exists nowhere else.
|
||||
// Never delete automatically (CB-586 criterion 2).
|
||||
continue;
|
||||
}
|
||||
exec("git", "-C", repoRoot, "update-ref", "-d", ref.refName());
|
||||
deleted++;
|
||||
log.info("pruned snapshot ref refs/wip/{} commit={} (age {}h): its tree is already "
|
||||
+ "reachable from main, so the work is preserved; recover from reflog via "
|
||||
+ "git update-ref refs/wip/{} {}",
|
||||
ref.branch(), ref.sha(), TimeUnit.MILLISECONDS.toHours(ageMillis),
|
||||
ref.branch(), ref.sha());
|
||||
}
|
||||
return deleted;
|
||||
}
|
||||
|
||||
/**
|
||||
* Every {@code refs/wip/*} ref (see {@link WipRef}). The committer date is read as a unix
|
||||
* count of seconds and converted to millis. {@code %00} (NUL) separates the fields because a
|
||||
* branch name may contain spaces.
|
||||
*/
|
||||
private List<WipRef> listWipRefs(String repoRoot) {
|
||||
String out = exec("git", "-C", repoRoot, "for-each-ref",
|
||||
"--format=%(refname)%00%(objectname)%00%(committerdate:unix)", "refs/wip/");
|
||||
List<WipRef> refs = new ArrayList<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
String[] parts = line.split("\u0000", -1);
|
||||
if (parts.length == 3 && !parts[1].isBlank()) {
|
||||
refs.add(new WipRef(parts[0], parts[1], Long.parseLong(parts[2]) * 1000L));
|
||||
}
|
||||
}
|
||||
return refs;
|
||||
}
|
||||
|
||||
/**
|
||||
* The set of object shas reachable from {@code main}, or an empty set when {@code main} cannot
|
||||
* be resolved. An empty set is the safe direction: the retention sweep then concludes nothing
|
||||
* is recoverable, so it deletes nothing — a repo with no {@code main} must never cause a
|
||||
* worker's last copy of a snapshot to be dropped on a reachability misreading.
|
||||
*/
|
||||
private Set<String> reachableObjectsFromMain(String repoRoot) {
|
||||
if (exitCode("git", "-C", repoRoot, "rev-parse", "--verify", "main") != 0) {
|
||||
log.debug("refs/wip retention: no 'main' ref in {} — treating nothing as reachable", repoRoot);
|
||||
return Set.of();
|
||||
}
|
||||
String out = exec("git", "-C", repoRoot, "rev-list", "--objects", "main");
|
||||
Set<String> objects = new HashSet<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
int sp = line.indexOf(' ');
|
||||
objects.add(sp < 0 ? line : line.substring(0, sp));
|
||||
}
|
||||
return objects;
|
||||
}
|
||||
|
||||
/** Approximate cost of a snapshot: the sum of every blob's size in its committed tree. */
|
||||
private long treeSize(String repoRoot, String sha) {
|
||||
String out = exec("git", "-C", repoRoot, "ls-tree", "-r", "-l", sha);
|
||||
long total = 0;
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
// ls-tree -l row: "<mode> <type> <object> <size>\t<path>"; the size is only numeric for
|
||||
// blobs (trees read "-"), so gate on the type token and take the 4th whitespace field.
|
||||
String[] parts = line.split("\\s+");
|
||||
if (parts.length >= 4 && "blob".equals(parts[1])) {
|
||||
try {
|
||||
total += Long.parseLong(parts[3]);
|
||||
} catch (NumberFormatException ignored) {
|
||||
// a '-' size (or any anomaly) contributes nothing to the rough figure
|
||||
}
|
||||
}
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/** Resolve the directory that will hold per-session worktree checkouts. */
|
||||
private Path resolveRoot(String repoRoot) {
|
||||
if (configuredRoot != null && !configuredRoot.isBlank()) {
|
||||
return Path.of(configuredRoot).toAbsolutePath().normalize();
|
||||
}
|
||||
Path repo = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
return repo.resolveSibling(".bridged-worktrees");
|
||||
}
|
||||
|
||||
private String nonce() {
|
||||
return String.format("%06x", random.nextInt(1 << 24)) + "-" + seq.incrementAndGet();
|
||||
}
|
||||
|
||||
private boolean isTracked(Path worktreeRoot, String rel) {
|
||||
return exitCode("git", "-C", worktreeRoot.toString(), "ls-files", "--error-unmatch", rel) == 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a command and return its stdout. Non-zero exit → {@link WorktreeException} with both
|
||||
* stdout and stderr (merged by redirectErrorStream).
|
||||
*/
|
||||
private String exec(String... command) {
|
||||
return exec(Map.of(), command);
|
||||
}
|
||||
|
||||
/** Same as {@link #exec(String...)}, with extra environment variables set on the child process. */
|
||||
private String exec(Map<String, String> extraEnv, String... command) {
|
||||
String out;
|
||||
int code;
|
||||
Process p;
|
||||
try {
|
||||
ProcessBuilder pb = new ProcessBuilder(command).redirectErrorStream(true);
|
||||
if (extraEnv != null && !extraEnv.isEmpty()) {
|
||||
pb.environment().putAll(extraEnv);
|
||||
}
|
||||
p = pb.start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try (BufferedReader r = new BufferedReader(new InputStreamReader(p.getInputStream(), StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(Collectors.joining("\n"));
|
||||
} catch (IOException e) {
|
||||
p.destroyForcibly();
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command) + "\n" + out);
|
||||
}
|
||||
code = p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
if (code != 0) {
|
||||
throw new WorktreeException("exit " + code + " for: " + String.join(" ", command)
|
||||
+ (out.isBlank() ? "" : "\n" + out));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
private int exitCode(String... command) {
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command));
|
||||
}
|
||||
return p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,114 +0,0 @@
|
||||
package dev.ltms.bridged;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-594: {@link Bridged#requiredSecretEnvVars(BridgedConfig)} is what decides what the startup
|
||||
* secret report checks — it must derive that set from the config, not a hand-written list, or a
|
||||
* new profile's token silently stops being reported.
|
||||
*/
|
||||
class RequiredSecretEnvVarsTest {
|
||||
|
||||
private static BridgedConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return BridgedConfig.load(f);
|
||||
}
|
||||
|
||||
@Test
|
||||
void collectsATokenEnvPerNonSubscriptionProfile(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
""");
|
||||
|
||||
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
|
||||
|
||||
assertTrue(required.containsKey("AI_GATEWAY_TOKEN"));
|
||||
assertEquals(List.of("profile 'local' tokenEnv"), required.get("AI_GATEWAY_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aSubscriptionProfileNeedsNoTokenEnv(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
opus:
|
||||
subscription: true
|
||||
model: claude-opus-5
|
||||
""");
|
||||
|
||||
assertTrue(Bridged.requiredSecretEnvVars(cfg).isEmpty(),
|
||||
"subscription: true never reads ANTHROPIC_AUTH_TOKEN — see Profile#isSubscription");
|
||||
}
|
||||
|
||||
@Test
|
||||
void gitTokenEnvIsOptInAndCollectedWhenSet(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
""");
|
||||
|
||||
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
|
||||
|
||||
assertTrue(required.containsKey("WORKER_GITEA_TOKEN"));
|
||||
assertEquals(List.of("profile 'local' gitTokenEnv"), required.get("WORKER_GITEA_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void noGitTokenEnvMeansNothingIsRequiredForIt(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
""");
|
||||
|
||||
assertFalse(Bridged.requiredSecretEnvVars(cfg).containsKey("WORKER_GITEA_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aVarSharedByTwoProfilesIsReportedOnceNamingBoth(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
gx:
|
||||
kind: opencode
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
tokenEnv: AI_GATEWAY_TOKEN
|
||||
gitTokenEnv: WORKER_GITEA_TOKEN
|
||||
""");
|
||||
|
||||
Map<String, List<String>> required = Bridged.requiredSecretEnvVars(cfg);
|
||||
|
||||
assertEquals(List.of("profile 'local' tokenEnv", "profile 'gx' tokenEnv"),
|
||||
required.get("AI_GATEWAY_TOKEN"));
|
||||
assertEquals(List.of("profile 'local' gitTokenEnv", "profile 'gx' gitTokenEnv"),
|
||||
required.get("WORKER_GITEA_TOKEN"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void noProfilesMeansNothingIsRequired(@TempDir Path dir) throws Exception {
|
||||
BridgedConfig cfg = load(dir, "bind:\n host: 127.0.0.1\n port: 8765\n");
|
||||
|
||||
assertTrue(Bridged.requiredSecretEnvVars(cfg).isEmpty());
|
||||
}
|
||||
}
|
||||
@@ -1,401 +0,0 @@
|
||||
package dev.ltms.bridged.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-559: re-reading {@code bridged.yaml} under a running daemon.
|
||||
*
|
||||
* <p>The tests that matter here are the refusals. A reload that applies a good file is the easy
|
||||
* half; the half that protects an operator is the one that keeps the running config when the new
|
||||
* file is bad, and the one that refuses a change the running daemon cannot honour.
|
||||
*/
|
||||
class ConfigRefTest {
|
||||
|
||||
/** A minimal file that loads and passes every startup validator. */
|
||||
private static String yaml(String extra) {
|
||||
return """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""" + extra;
|
||||
}
|
||||
|
||||
private static ConfigRef refFor(Path f) {
|
||||
return new ConfigRef(f, BridgedConfig.load(f));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aHotChangeIsAppliedAndReadThroughGet(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
tabLabel: "{role}: {profile} #{n}"
|
||||
developers:
|
||||
a:
|
||||
profile: sonnet
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
assertEquals("{role}: {profile} #{n}", ref.get().fleet().tabLabel());
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
tabLabel: "[{profile}] {role}"
|
||||
developers:
|
||||
a:
|
||||
profile: sonnet
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty());
|
||||
assertEquals("config reloaded", out.summary());
|
||||
assertEquals("[{profile}] {role}", ref.get().fleet().tabLabel());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCharterChangeIsHotAndReachesTheLiveConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: old charter
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
assertEquals("old charter", ref.get().fleet().charterFor(
|
||||
dev.ltms.bridged.peer.MemberRole.ARCHITECT));
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: new charter
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty());
|
||||
assertEquals("new charter", ref.get().fleet().charterFor(
|
||||
dev.ltms.bridged.peer.MemberRole.ARCHITECT));
|
||||
}
|
||||
|
||||
@Test
|
||||
void invalidChartersRefuseReloadAndKeepTheRunningConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: valid charter
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
BridgedConfig before = ref.get();
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: " "
|
||||
"""));
|
||||
ConfigRef.Outcome blank = ref.reload();
|
||||
assertFalse(blank.applied());
|
||||
assertTrue(blank.error().contains("fleet.charters.architect is blank"));
|
||||
assertSame(before, ref.get());
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architetc: valid charter
|
||||
"""));
|
||||
ConfigRef.Outcome unknown = ref.reload();
|
||||
assertFalse(unknown.applied());
|
||||
assertTrue(unknown.error().contains("architetc"));
|
||||
assertTrue(unknown.error().contains("architect"));
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
/**
|
||||
* The point of the whole class: a consumer holding the ref sees the new value without being
|
||||
* rebuilt. A component that captured {@code get()} into a field would still show the old one.
|
||||
*/
|
||||
@Test
|
||||
void aConsumerHoldingTheRefSeesTheNewValue(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
java.util.function.Supplier<String> reader = () -> ref.get().placement();
|
||||
assertEquals("weighted", reader.get());
|
||||
|
||||
Files.writeString(f, yaml("placement: fixed\n"));
|
||||
assertTrue(ref.reload().applied());
|
||||
|
||||
assertEquals("fixed", reader.get());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aChangedColdKeyRefusesTheWholeReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
// Two changes in one file: a cold one (the port) and a hot one (placement).
|
||||
Files.writeString(f, yaml("placement: fixed\n").replace("port: 8765", "port: 9999"));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertEquals(java.util.List.of("bind"), out.coldKeys());
|
||||
assertTrue(out.summary().contains("Restart bridged"), out.summary());
|
||||
// The hot half must NOT have leaked in. A half-applied reload leaves the daemon matching no
|
||||
// file on disk, which is worse for an operator than no reload at all.
|
||||
assertEquals("weighted", ref.get().placement());
|
||||
assertEquals(8765, ref.get().bind().port());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFileThatNoLongerParsesKeepsTheRunningConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
BridgedConfig before = ref.get();
|
||||
|
||||
Files.writeString(f, "profiles:\n sonnet:\n baseUrl: \"unclosed\n");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertTrue(out.summary().startsWith("config reload refused"), out.summary());
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
/** A file that would have refused to boot must not be able to slip in through a reload. */
|
||||
@Test
|
||||
void aFileThatFailsAValidatorKeepsTheRunningConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
BridgedConfig before = ref.get();
|
||||
|
||||
// A member slot naming a profile that does not exist — validateMembers refuses this at
|
||||
// startup, so it must refuse it here too.
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
developers:
|
||||
a:
|
||||
profile: no-such-profile
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aDeletedFileIsRefusedRatherThanCrashing(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
BridgedConfig before = ref.get();
|
||||
|
||||
Files.delete(f);
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
/** A deferred change applies to the snapshot but the operator is told it needs a restart. */
|
||||
@Test
|
||||
void aDeferredChangeIsAppliedAndReported(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
lifecycle:
|
||||
drainTimeoutSeconds: 30
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
lifecycle:
|
||||
drainTimeoutSeconds: 60
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(java.util.List.of("lifecycle"), out.deferred());
|
||||
assertTrue(out.summary().contains("needs a restart") || out.summary().contains("need a restart"),
|
||||
out.summary());
|
||||
assertEquals(60, ref.get().lifecycle().drainTimeoutSeconds());
|
||||
}
|
||||
|
||||
/**
|
||||
* Adding a profile is deferred, not hot: a new backend needs its own launcher, and launchers are
|
||||
* built once at startup. The snapshot carries it so a restart picks it up.
|
||||
*/
|
||||
@Test
|
||||
void addingAProfileIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
haiku:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: haiku
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size());
|
||||
assertTrue(out.deferred().getFirst().contains("haiku"), out.deferred().toString());
|
||||
}
|
||||
|
||||
/**
|
||||
* Changing an existing profile's weight or maxLoad IS hot: placement reads those live through
|
||||
* the supplier on the composite, so the next spawn already sees them.
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesWeightOrMaxLoadIsHot(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
maxLoad: 7
|
||||
weight: 3.0
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty(), out.deferred().toString());
|
||||
assertEquals(7, ref.get().profiles().get("sonnet").maxLoad());
|
||||
}
|
||||
|
||||
/**
|
||||
* Changing an existing profile's MODEL is deferred, not hot — and saying so is the whole point.
|
||||
* HerdrPeerLauncher takes Map.copyOf(profiles) at construction and resolves every spawn out of
|
||||
* that copy, so a reloaded model never reaches a launch. Reporting it as applied would be the
|
||||
* worst outcome a reload can produce: the operator has no reason to doubt a clean "reloaded".
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesLaunchSettingsIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size(), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("sonnet"), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("launch settings"), out.deferred().toString());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("deepseek-v4-flash", ref.get().profiles().get("sonnet").model());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage B: exhaustedPattern is compiled once into Bridged.main's pattern map at startup
|
||||
* (see ExhaustedPatternLookup), so a reload never re-reads it — a changed pattern must be
|
||||
* reported deferred exactly like model/baseUrl, not silently claimed as applied.
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesExhaustedPatternIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
exhaustedPattern: "usage limit has been reached"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
exhaustedPattern: "rate limit exceeded"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size(), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("sonnet"), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("launch settings"), out.deferred().toString());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("rate limit exceeded", ref.get().profiles().get("sonnet").exhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFixedRefHasNoFileAndRefusesToReload() {
|
||||
BridgedConfig cfg = new BridgedConfig(null, null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null).withDefaults();
|
||||
ConfigRef ref = ConfigRef.fixed(cfg);
|
||||
|
||||
assertNull(ref.path());
|
||||
assertSame(cfg, ref.get());
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
}
|
||||
}
|
||||
@@ -1,170 +0,0 @@
|
||||
package dev.ltms.bridged.health;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.function.BiConsumer;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
class FleetHealthMonitorTest {
|
||||
@Test void oneTickUsesOneFleetListForAnyRosterSize() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
BridgedConfig.Profile profile = new BridgedConfig.Profile("test", "http://test:1", null,
|
||||
null, null, null, null, null, null, null, null, null);
|
||||
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(agents, new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("test")), Map.of("test", profile), "test", _ -> "token");
|
||||
SessionManager sessions = new SessionManager(launcher);
|
||||
sessions.acquire("test", null, null, null);
|
||||
sessions.acquire("test", null, null, null);
|
||||
herdr.calls.clear();
|
||||
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox());
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, sessions::roster, messages, scheduler, () -> 1, 60,
|
||||
(_, _) -> { });
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
assertEquals(1, herdr.calls.stream().filter(call -> call.method().equals("agent.list")).count());
|
||||
}
|
||||
|
||||
@Test void failedTickDoesNotStopTheNextTick() {
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false);
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60, (_, _) -> { });
|
||||
monitor.tick();
|
||||
herdr.healthy(true);
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
assertEquals(2, herdr.calls.stream().filter(call -> call.method().equals("agent.list")).count());
|
||||
}
|
||||
|
||||
@Test void faultTransitionLogsOnlyOnceUntilItChanges() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60, (_, _) -> { });
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.stop();
|
||||
assertEquals(1, appender.list.stream().filter(event -> event.getFormattedMessage()
|
||||
.contains("member=term_a state=TURN_BOUNDARY_LOST")).count());
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- CB-580: a member that reaches GONE/NEVER_READY must fail its waiting tickets
|
||||
|
||||
private static FleetHealthMonitor monitorWith(BiConsumer<String, String> failTarget) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
return new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60, failTarget);
|
||||
}
|
||||
|
||||
@Test void terminalTransitionFailsTheTargetOnce() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
assertEquals("term_a", failTarget.calls.get(0).target());
|
||||
assertTrue(failTarget.calls.get(0).reason().contains("GONE"));
|
||||
}
|
||||
|
||||
@Test void neverReadyNamesItselfAsTheReason() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.NEVER_READY);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
assertTrue(failTarget.calls.get(0).reason().contains("NEVER_READY"));
|
||||
}
|
||||
|
||||
@Test void stayingInATerminalStateProducesOneFailureNotN() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
}
|
||||
|
||||
@Test void aNonTerminalFaultStateDoesNotFailTheTarget() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.stop();
|
||||
assertEquals(0, failTarget.calls.size());
|
||||
}
|
||||
|
||||
@Test void failTargetRetryIsBounded() {
|
||||
AlwaysThrowingFailTarget failTarget = new AlwaysThrowingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(FleetHealthMonitor.MAX_FAIL_TARGET_ATTEMPTS, failTarget.calls);
|
||||
}
|
||||
|
||||
@Test void exhaustedRetryStillDoesNotRefireOnAnUnchangedTick() {
|
||||
AlwaysThrowingFailTarget failTarget = new AlwaysThrowingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
int afterFirstTransition = failTarget.calls;
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(afterFirstTransition, failTarget.calls);
|
||||
}
|
||||
|
||||
private record RecordedCall(String target, String reason) { }
|
||||
|
||||
private static final class RecordingFailTarget implements BiConsumer<String, String> {
|
||||
final java.util.List<RecordedCall> calls = new java.util.ArrayList<>();
|
||||
|
||||
@Override public void accept(String target, String reason) {
|
||||
calls.add(new RecordedCall(target, reason));
|
||||
}
|
||||
}
|
||||
|
||||
private static final class AlwaysThrowingFailTarget implements BiConsumer<String, String> {
|
||||
int calls = 0;
|
||||
|
||||
@Override public void accept(String target, String reason) {
|
||||
calls++;
|
||||
throw new RuntimeException("boom");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,27 +0,0 @@
|
||||
package dev.ltms.bridged.herdr;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** Unit tests for PID → pane resolution (the herdr half of connection-based MCP identity). */
|
||||
class PaneLocatorTest {
|
||||
|
||||
private final PaneLocator loc = new PaneLocator(new FakeHerdr());
|
||||
|
||||
@Test
|
||||
void resolvesTerminalForAForegroundPid() {
|
||||
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForAPidInNoPane() {
|
||||
assertNull(loc.terminalForPid(999_999));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForNonPositivePid() {
|
||||
assertNull(loc.terminalForPid(0));
|
||||
assertNull(loc.terminalForPid(-1));
|
||||
}
|
||||
}
|
||||
@@ -1,501 +0,0 @@
|
||||
package dev.ltms.bridged.inject;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.msg.TestTurnTokens;
|
||||
import dev.ltms.bridged.msg.TurnToken;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
||||
class CompletionResolverTest {
|
||||
|
||||
@Test
|
||||
void skipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
resolver.resolve("term_a", null); // no in-flight turn captured for this target
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a turn nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failSkipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
resolver.fail("term_a", null); // no in-flight turn, and no registered waiter to fall back to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a wedge nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void captureBaselineSkipsTheReadWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ X\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
resolver.captureBaseline("term_a", TestTurnTokens.inert("term_a")); // no send to attribute a later completion to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"with no waiting send there is no turn to baseline — skip the scrape");
|
||||
}
|
||||
|
||||
// --- CB-115 clean scrape: extract the last assistant block ----------------
|
||||
|
||||
@Test
|
||||
void extractsTheLastAssistantBlockStrippingChrome() {
|
||||
String raw = """
|
||||
⏺ Reading the file…
|
||||
|
||||
⏺ Done. The bug was an off-by-one in the loop bound.
|
||||
|
||||
╭──────────────────────────────────────╮
|
||||
│ > │
|
||||
╰──────────────────────────────────────╯
|
||||
⏵⏵ auto mode on · ? for shortcuts
|
||||
""";
|
||||
assertEquals("Done. The bug was an off-by-one in the loop bound.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void keepsMultiLineAssistantContent() {
|
||||
String raw = "⏺ Line one.\nLine two.\n❯ ";
|
||||
assertEquals("Line one.\nLine two.", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void fallsBackToRawTextWhenThereIsNoMarker() {
|
||||
String raw = "plain worker output with no glyph";
|
||||
assertEquals("plain worker output with no glyph", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void blankScrapeYieldsEmpty() {
|
||||
assertTrue(CompletionResolver.lastAssistantBlock("").isEmpty());
|
||||
assertTrue(CompletionResolver.lastAssistantBlock(null).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void stripsSpinnerAndRuleChrome() {
|
||||
String raw = """
|
||||
⏺ Channel check confirmed — your message got through.
|
||||
|
||||
✻ Brewed for 11s
|
||||
|
||||
─────────────────────────────────────
|
||||
""";
|
||||
assertEquals("Channel check confirmed — your message got through.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void cutsANextTurnPromptEchoAndTrailingTipsFromTheBlock() {
|
||||
// The exact turn-2 leak: the scrape captured the settled answer, then a "✻ Cooked" spinner,
|
||||
// then the NEXT turn's echoed prompt, then a "✶ Forming…" spinner and trailing tips/warnings
|
||||
// whose lines (⎿, ⚠) are not themselves chrome-terminated. Stopping at the first boundary
|
||||
// (the ✻ spinner) is what keeps every one of those interface lines out of the reply.
|
||||
String raw = """
|
||||
⏺ Channel confirmed — the bridge reply delivered successfully.
|
||||
|
||||
✻ Cooked for 9s
|
||||
|
||||
❯ Thanks. Now a small task: what is 17 * 23? Show just the number.
|
||||
|
||||
|
||||
|
||||
✶ Forming…
|
||||
⎿ Tip: Name your conversations with /rename
|
||||
⚠ claude.ai connectors are disabled because ANTHROPIC_API_KEY is set
|
||||
""";
|
||||
assertEquals("Channel confirmed — the bridge reply delivered successfully.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
// --- CB-115 misattribution guard: suppress a stale (unchanged) completion -------
|
||||
|
||||
@Test
|
||||
void suppressesACompletionWhoseScrapeIsUnchangedFromDelivery() {
|
||||
// Rapid back-to-back turn: the pane still shows the PREVIOUS turn's answer when this turn's
|
||||
// (misattributed) completion boundary fires. The scrape == the delivery baseline, so the
|
||||
// send must NOT be resolved with the stale answer.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ 391\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
// The turn as captured at delivery: its waiter, and the previous turn's answer still on screen.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn); // scrape still "391" == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(), "a completion with no output change must not resolve the send");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesACompletionWhoseScrapeChangedSinceDelivery() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ No, 391 = 17 × 23.\n❯ "); // the worker's real answer
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
// Delivery baseline was the previous turn's "391"; the scrape now differs → resolve.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(), "a completion with new output must resolve the send");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("No, 391 = 17 × 23.", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void marksAClippedCompletionPaneTail() {
|
||||
String block = "⏺ " + "x".repeat(CompletionResolver.MAX_SCRAPE_CHARS + 1) + "\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals("x".repeat(CompletionResolver.MAX_SCRAPE_CHARS)
|
||||
+ "\n[Pane tail clipped: member did not call bridge_reply.]",
|
||||
waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void leavesAnUnclippedCompletionPaneTailUnmarked() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ complete report\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals("complete report", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesSynchronouslyBeforePostTurnContextClearing() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ previous answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.captureBaseline("term_a", new TurnToken("term_a", waiter));
|
||||
herdr.readText("⏺ answer that /clear would erase\n❯ ");
|
||||
|
||||
resolver.resolveBeforePostAction("term_a");
|
||||
|
||||
assertTrue(waiter.isDone(), "the answer is captured before the adapter sends /clear");
|
||||
assertEquals("answer that /clear would erase", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void suppressesAnUnchangedCompletionEvenWhenTheBlockExceedsTheScrapeCap() {
|
||||
// The fan-out issue-hunt finding: captureBaseline once stored the RAW (unclipped) assistant
|
||||
// block while resolve compares against a clip()'d tail. For a block longer than MAX_SCRAPE_CHARS
|
||||
// the two capped representations differ even when the pane never changed, so the CB-115
|
||||
// byte-identical guard failed to fire and a stale completion could resolve the send. Both sides
|
||||
// must clip identically. The returned-text marker is added only after this comparison, so an
|
||||
// unchanged >cap block on rapid back-to-back turns still stays suppressed.
|
||||
String longBlock = "⏺ " + "x".repeat(CompletionResolver.MAX_SCRAPE_CHARS + 500) + "\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(longBlock);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
resolver.captureBaseline("term_a", new TurnToken("term_a", waiter)); // baseline is the clipped >cap block
|
||||
var turn = resolver.inFlight("term_a");
|
||||
assertEquals(CompletionResolver.MAX_SCRAPE_CHARS, turn.baseline().length(),
|
||||
"the delivery baseline is clipped to the same cap resolve() applies to the tail");
|
||||
|
||||
resolver.resolve("term_a", turn); // scrape unchanged → clipped tail == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(),
|
||||
"an unchanged >cap block must still be recognised as stale and suppressed");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenThereIsNoBaseline() {
|
||||
// No delivery baseline (e.g. the pre-turn read failed) ⇒ never suppress; the completion resolves.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ hello\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "with no baseline a completion resolves as before");
|
||||
assertEquals("hello", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenTheScrapeItselfFailsEvenWithABaselinePresent() {
|
||||
// The most important branch of the CB-115 guard: a failed read means the resolver could not
|
||||
// SEE the screen — "couldn't see", not "no change". It must still resolve the send (an empty
|
||||
// tail beats hanging until the caller's timeout), even though a baseline was captured. The
|
||||
// baseline here is "" (an empty pane at delivery), so without the !scrapeFailed clause the
|
||||
// byte-identical guard would wrongly match the empty tail and suppress.
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false); // agent.read throws HerdrException
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, ""); // empty pane baselined at delivery
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(),
|
||||
"a failed scrape must still resolve the send, not hang until the caller's timeout");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("", waiter.getNow(null).text(), "the tail is empty because the screen was unreadable");
|
||||
}
|
||||
|
||||
// --- CB-115/CB-116 fail guard: an already-done or absent waiter is left alone ---------
|
||||
|
||||
@Test
|
||||
void failLeavesAnAlreadyResolvedWaiterUntouchedAndSkipsTheScrape() {
|
||||
// The send was already resolved (e.g. by the worker's explicit reply) before fail fired.
|
||||
// fail must not overwrite that value, and must not even scrape the worker — nobody needs it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied"));
|
||||
|
||||
resolver.fail("term_a", turn);
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"fail must not scrape a waiter that is already done");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"fail must not overwrite the existing resolution");
|
||||
assertEquals("already replied", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void failFallsBackToTheRegisteredWaiterWhenThereIsNoInFlightTurn() {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fail falls back to the waiter currently registered on the Rendezvous and fails it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // send registered, but no captureBaseline ever ran
|
||||
resolver.fail("term_a", null); // no in-flight turn → fall back to the registered waiter
|
||||
|
||||
assertTrue(waiter.isDone(), "fail falls back to the registered waiter when no turn is in flight");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals("stuck on an error screen", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void failIsLoggedAtWarnWithTheReason() {
|
||||
// CB-564: this used to be a bare DEBUG "failed send to X via turn-stall fallback" — a symptom
|
||||
// with no cause, and below the level anyone watching for member health would see. A fail that
|
||||
// resolves a caller's blocked send is at least WARN and must carry the reason.
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger resolverLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(CompletionResolver.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
resolverLog.addAppender(appender);
|
||||
resolverLog.setLevel(Level.WARN);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
var waiter = rendezvous.open("term_a");
|
||||
|
||||
resolver.fail("term_a", null);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no turn-stall WARN logged");
|
||||
assertTrue(warn.contains("term_a"), "the log names the target: " + warn);
|
||||
assertTrue(warn.contains("stuck on an error screen"), "the log carries the reason: " + warn);
|
||||
assertTrue(waiter.isDone());
|
||||
} finally {
|
||||
resolverLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- CB-116 waiter identity: a late completion never crosses into the next turn ---------
|
||||
|
||||
@Test
|
||||
void aLateCompletionForOneTurnNeverResolvesTheNextTurnsWaiter() {
|
||||
// The cross-turn stale reply the conversation test surfaced: turn N's completion fallback
|
||||
// fires AFTER turn N was resolved by an explicit bridge_reply and turn N+1 has opened its own
|
||||
// waiter on the same session. Resolving "whatever is waiting now" would hand turn N's stale
|
||||
// scrape to turn N+1; targeting turn N's captured waiter makes the late completion a no-op.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ turn N answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiterN = rendezvous.open("term_a"); // turn N's send
|
||||
// The turn as the injector captured it at delivery (waiter + pre-turn baseline).
|
||||
var turnN = new CompletionResolver.InFlight(waiterN, "an earlier answer");
|
||||
|
||||
// Turn N is resolved by the worker's explicit reply, and its send deregisters the waiter.
|
||||
assertTrue(rendezvous.resolve("term_a", "N replied"));
|
||||
rendezvous.close("term_a", waiterN); // the sender's finally, before the next turn opens
|
||||
|
||||
// Turn N+1's send opens its own waiter on the same session (CB-548: open fails if the
|
||||
// previous waiter is still registered, so a clean turn deregisters it first as above).
|
||||
var waiterN1 = rendezvous.open("term_a");
|
||||
|
||||
resolver.resolve("term_a", turnN); // turn N's completion fallback finally fires
|
||||
|
||||
assertFalse(waiterN1.isDone(), "turn N's late completion must not resolve turn N+1's waiter");
|
||||
assertEquals(Rendezvous.Kind.REPLY, waiterN.getNow(null).kind(),
|
||||
"turn N stays resolved by its own reply");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "turn N+1 is still awaiting its own resolution");
|
||||
}
|
||||
|
||||
// --- CB-578 stage A: backend-exhausted classification ---------------------------------
|
||||
|
||||
@Test
|
||||
void classifiesAMatchingScrapeAsBackendExhaustedInsteadOfACompletedReply() {
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a matching scrape still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"not reported as a completed reply — the classification is distinct");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theExhaustedReasonCarriesTheMatchedLine() {
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals("backend exhausted (usage limit): The usage limit has been reached. Try again later.",
|
||||
waiter.getNow(null).text(), "the reason names the real cause and carries the matched line");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWinningBackendExhaustedClassificationNotifiesTheExhaustionSink() {
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(1, notified.size(), "the sink is notified exactly once for the winning classification");
|
||||
assertTrue(notified.get(0).startsWith("term_a: "), "the sink is told which target exhausted");
|
||||
assertTrue(notified.get(0).contains("The usage limit has been reached"),
|
||||
"the sink is told the matched reason: " + notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLosingBackendExhaustedClassificationNeverNotifiesTheExhaustionSink() {
|
||||
// The waiter was already resolved (e.g. by the worker's own reply) before this scrape landed —
|
||||
// resolveExhausted loses the race and must return false, so the sink must not fire either.
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied"));
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(), "a classification that loses the race must not quarantine anything");
|
||||
assertEquals("already replied", waiter.getNow(null).text(), "the earlier resolution stands untouched");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingScrapeResolvesAsAnOrdinaryCompletion() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ complete report\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"a scrape that does not match the pattern is an ordinary completion");
|
||||
assertEquals("complete report", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithNoConfiguredPatternKeepsTodaysCompletionFallbackUnchanged() {
|
||||
// Even a scrape that WOULD have matched some other profile's pattern must resolve as a
|
||||
// plain completion when this target's own profile has none configured (CB-578 criterion 4).
|
||||
String block = "⏺ The usage limit has been reached.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver =
|
||||
new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"no pattern configured for this target's profile ⇒ unchanged completion-fallback behaviour");
|
||||
assertEquals("The usage limit has been reached.", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsOffWhenNoProfileHasAPatternConfigured() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [terra])",
|
||||
CompletionResolver.coverage(Set.of("terra"), Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsFullWhenEveryProfileHasAPatternConfigured() {
|
||||
assertEquals("full (all profiles configured: [gx10, terra])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsPartialAndNamesWhichProfilesAreConfigured() {
|
||||
assertEquals("partial (configured: [terra]; not configured: [gx10])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra")));
|
||||
}
|
||||
}
|
||||
@@ -1,258 +0,0 @@
|
||||
package dev.ltms.bridged.lead;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-558 — the daemon starts a declared lead when none is running.
|
||||
*
|
||||
* <p>Two properties carry the whole feature. It must not double-spawn (a second orchestrator is
|
||||
* worse than none), and what it starts must be a <em>lead</em> and not a member: no reply charter,
|
||||
* no off-subscription env, and never registered with the session lifecycle.
|
||||
*/
|
||||
class LeadLauncherTest {
|
||||
|
||||
/** The lead's backend: a subscription profile with the bridge mounted, as `opus` really is. */
|
||||
private static BridgedConfig.Profile opusProfile() {
|
||||
return new BridgedConfig.Profile(
|
||||
"opus", null, "claude-opus-5", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms"), "tab", "bridged-workers", null,
|
||||
"http://127.0.0.1:8765/mcp", null, null,
|
||||
null, null, null,
|
||||
Map.of("CLAUDE_CODE_AUTO_COMPACT_WINDOW", "300000"), null, null, true, null);
|
||||
}
|
||||
|
||||
private static BridgedConfig configWith(BridgedConfig.Leader lead) {
|
||||
Map<String, BridgedConfig.Leader> leaders = new LinkedHashMap<>();
|
||||
leaders.put("opus", lead);
|
||||
BridgedConfig.Fleet fleet =
|
||||
new BridgedConfig.Fleet(leaders, Map.of(), Map.of(), Map.of(), null);
|
||||
return new BridgedConfig(
|
||||
null, null, Map.of("opus", opusProfile()), null, null, null, null, null,
|
||||
null, null, fleet, null, "fixed", null).withDefaults();
|
||||
}
|
||||
|
||||
private static BridgedConfig.Leader lead(String profile, String tab, int instances) {
|
||||
return new BridgedConfig.Leader(profile, tab, instances, "lead:", 10, null, null,
|
||||
"leads", "/repo");
|
||||
}
|
||||
|
||||
private static LeadLauncher launcher(FakeHerdr herdr, BridgedConfig cfg) {
|
||||
return new LeadLauncher(new AgentControl(herdr), new WorkspaceControl(herdr), cfg);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static List<String> startedArgs(FakeHerdr herdr) {
|
||||
return (List<String>) ((Map<String, Object>) herdr.lastCall("agent.start").params()).get("args");
|
||||
}
|
||||
|
||||
private static String startedName(FakeHerdr herdr) {
|
||||
return (String) ((Map<?, ?>) herdr.lastCall("agent.start").params()).get("name");
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, String> tabEnv(FakeHerdr herdr) {
|
||||
return (Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||
}
|
||||
|
||||
// ── it starts a lead when none is live ────────────────────────────────────────────────────
|
||||
|
||||
@Test
|
||||
void startsTheDeclaredLeadWhenNoneIsRunning() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
assertEquals(1, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
assertTrue(herdr.called("agent.start"), "a lead must actually be started");
|
||||
assertEquals("lead-opus", startedName(herdr));
|
||||
}
|
||||
|
||||
/** The tab is labelled with the configured `tab:` so the scanner finds the lead on the next resolve. */
|
||||
@Test
|
||||
void labelsTheTabWithTheConfiguredTabValue() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads();
|
||||
|
||||
assertEquals("lead: opus",
|
||||
((Map<?, ?>) herdr.lastCall("tab.rename").params()).get("label"));
|
||||
}
|
||||
|
||||
/** `instances: 2` with none live means two starts, not one. */
|
||||
@Test
|
||||
void startsAsManyInstancesAsAreDeclared() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
assertEquals(2, launcher(herdr, configWith(lead("opus", "lead: opus", 2))).ensureLeads());
|
||||
assertEquals(2, herdr.calls.stream().filter(c -> c.method().equals("agent.start")).count());
|
||||
}
|
||||
|
||||
// ── it must not double-spawn ──────────────────────────────────────────────────────────────
|
||||
|
||||
/** A labelled tab WITH a running agent in it is a live lead — leave it alone. */
|
||||
@Test
|
||||
void doesNotStartASecondLeadWhenOneIsAlreadyRunning() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wL", "leads")
|
||||
.withTab("wL", "wL:t1", "lead: opus")
|
||||
.withAgent("lead-opus", "term_lead", "wL:p1", "wL:t1");
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
assertFalse(herdr.called("agent.start"), "the live lead must not be duplicated");
|
||||
}
|
||||
|
||||
/**
|
||||
* The reason liveness is not "does the label exist". A tab left labelled by a session that has
|
||||
* since died must not block the relaunch, or one crash disables auto-launch permanently.
|
||||
*/
|
||||
@Test
|
||||
void aLabelledTabWithNoRunningAgentIsNotALiveLead() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wL", "leads")
|
||||
.withTab("wL", "wL:t1", "lead: opus"); // label only — nothing running in it
|
||||
|
||||
assertEquals(1, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads(),
|
||||
"a stale label is not a lead; the lead must be relaunched");
|
||||
}
|
||||
|
||||
/**
|
||||
* A lead the operator opened by hand is live once its tab carries the configured `tab:` label —
|
||||
* CB-579 retired the `terminal:` pin, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is, by its tab, not by a terminal id nobody wrote down in advance.
|
||||
*/
|
||||
@Test
|
||||
void aHandOpenedLeadWithTheConfiguredTabLabelCountsAsLive() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wX", "main")
|
||||
.withTab("wX", "wX:t1", "lead: opus")
|
||||
.withAgent("hand-opened", "term_hand", "wX:p1", "wX:t1");
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
assertFalse(herdr.called("agent.start"));
|
||||
}
|
||||
|
||||
/** A member sitting in a matching tab must never be counted — or spawn a lead — as one. */
|
||||
@Test
|
||||
void aMemberWorkspaceIsNeverScannedForLeads() {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withWorkspace("wM", "bridged-workers") // a configured member space
|
||||
.withTab("wM", "wM:t1", "lead: opus") // a member tab that looks like a lead
|
||||
.withAgent("claude-opus-x", "term_m", "wM:p1", "wM:t1");
|
||||
|
||||
assertEquals(1, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads(),
|
||||
"a member in a lead-labelled tab is not a lead, so the real lead is still missing");
|
||||
}
|
||||
|
||||
/** If herdr cannot be counted, start nothing: guessing risks a second orchestrator. */
|
||||
@Test
|
||||
void anUncountableHerdrStartsNothing() {
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false);
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads());
|
||||
assertFalse(herdr.called("agent.start"));
|
||||
}
|
||||
|
||||
// ── what it starts is a LEAD, not a member ────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* The single most important assertion here. The worker charter tells its reader it is an
|
||||
* off-subscription worker that must end every turn with bridge_reply — the opposite of what an
|
||||
* orchestrator is. A lead must never receive it.
|
||||
*/
|
||||
@Test
|
||||
void theLeadNeverReceivesTheWorkerReplyCharter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads();
|
||||
|
||||
List<String> args = startedArgs(herdr);
|
||||
assertFalse(args.contains("--append-system-prompt"),
|
||||
"the reply charter is a worker contract and must not be injected into a lead");
|
||||
assertTrue(args.stream().noneMatch(a -> a.contains("bridge_reply")), args.toString());
|
||||
}
|
||||
|
||||
/** It still mounts the bridge — a lead that cannot orchestrate is pointless. */
|
||||
@Test
|
||||
void theLeadMountsTheBridgeMcpAndPinsItsModel() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads();
|
||||
|
||||
List<String> args = startedArgs(herdr);
|
||||
assertTrue(args.contains("--mcp-config"));
|
||||
assertTrue(args.stream().anyMatch(a -> a.contains("http://127.0.0.1:8765/mcp")), args.toString());
|
||||
assertEquals("claude-opus-5", args.get(args.indexOf("--model") + 1));
|
||||
assertTrue(args.indexOf("--model") > args.indexOf("--mcp-config"),
|
||||
"--model is appended last so it outranks the ccs wrapper (CB-533)");
|
||||
}
|
||||
|
||||
/** A lead runs on the operator's subscription. Nothing may move it off. */
|
||||
@Test
|
||||
void theLeadEnvCarriesNoAnthropicBinding() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads();
|
||||
|
||||
Map<String, String> env = tabEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"));
|
||||
assertNull(env.get("ANTHROPIC_AUTH_TOKEN"));
|
||||
assertEquals("300000", env.get("CLAUDE_CODE_AUTO_COMPACT_WINDOW"),
|
||||
"the profile's own env: still applies");
|
||||
}
|
||||
|
||||
/** The lead's tab goes in its own workspace, never a member one — the scanner skips those. */
|
||||
@Test
|
||||
void theLeadTabIsCreatedOutsideEveryMemberWorkspace() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
launcher(herdr, configWith(lead("opus", "lead: opus", 1))).ensureLeads();
|
||||
|
||||
String label = (String) ((Map<?, ?>) herdr.lastCall("workspace.create").params()).get("label");
|
||||
assertEquals("leads", label);
|
||||
assertNotEquals("bridged-workers", label);
|
||||
}
|
||||
|
||||
// ── recognise-only and misconfiguration ───────────────────────────────────────────────────
|
||||
|
||||
/** A lead with a tab but no profile is recognise-only by design — not an error, not a launch. */
|
||||
@Test
|
||||
void aLeadThatNamesNoProfileIsRecognisedButNeverLaunched() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead(null, "lead: dead", 1))).ensureLeads());
|
||||
assertFalse(herdr.called("agent.start"));
|
||||
}
|
||||
|
||||
/** `instances: 0` is a deliberate off switch. */
|
||||
@Test
|
||||
void zeroInstancesLaunchesNothing() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead("opus", "lead: opus", 0))).ensureLeads());
|
||||
assertFalse(herdr.called("agent.start"));
|
||||
}
|
||||
|
||||
/** A profile name with no matching profile is logged and skipped, never a daemon crash. */
|
||||
@Test
|
||||
void anUnknownProfileIsSkippedRatherThanThrown() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
assertEquals(0, launcher(herdr, configWith(lead("nope", "lead: opus", 1))).ensureLeads());
|
||||
assertFalse(herdr.called("agent.start"));
|
||||
}
|
||||
|
||||
/** No leads declared at all: not a herdr call in sight. */
|
||||
@Test
|
||||
void noLeadersConfiguredTouchesHerdrNotAtAll() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig cfg = new BridgedConfig(
|
||||
null, null, Map.of("opus", opusProfile()), null, null, null, null, null,
|
||||
null, null, null, null, "fixed", null).withDefaults();
|
||||
|
||||
assertEquals(0, launcher(herdr, cfg).ensureLeads());
|
||||
assertTrue(herdr.calls.isEmpty(), "nothing declared ⇒ nothing scanned");
|
||||
}
|
||||
}
|
||||
@@ -1,204 +0,0 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.auth.Authz;
|
||||
import dev.ltms.bridged.auth.CallerResolver;
|
||||
import dev.ltms.bridged.auth.MemberRegistry;
|
||||
import dev.ltms.bridged.auth.Principal;
|
||||
import dev.ltms.bridged.auth.Role;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.inject.Injector;
|
||||
import dev.ltms.bridged.metrics.BridgedMetrics;
|
||||
import dev.ltms.bridged.metrics.Metrics;
|
||||
import dev.ltms.bridged.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.bridged.msg.MessageService;
|
||||
import dev.ltms.bridged.msg.Rendezvous;
|
||||
import dev.ltms.bridged.session.FakeWorktrees;
|
||||
import dev.ltms.bridged.session.SessionManager;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-513 — the CB-505 authorization gate on the <strong>MCP</strong> entry path.
|
||||
*
|
||||
* <p>Why this file exists: CB-505 claimed authorization is "enforced on both entry paths", and it
|
||||
* is — but only REST was ever tested ({@code BridgedAppAuthTest}). Coverage showed
|
||||
* {@code BridgeMcp.deny()}, {@code principal()} and every tool-registration lambda at <em>zero</em>
|
||||
* executed lines, because no test had ever constructed a {@code BridgeMcp} — the existing
|
||||
* {@code BridgeMcpTest} calls only the static handler methods. An unexercised security control is
|
||||
* a claim, not a control.
|
||||
*
|
||||
* <p>These tests construct a real {@code BridgeMcp} (which also exercises the constructor and the
|
||||
* tool wiring) and drive the policy half of the gate directly.
|
||||
*/
|
||||
class BridgeMcpAuthzTest {
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private Metrics metrics;
|
||||
private BridgeMcp mcp;
|
||||
|
||||
@AfterEach
|
||||
void close() {
|
||||
if (mcp != null) mcp.close();
|
||||
}
|
||||
|
||||
/** A fully wired BridgeMcp on fakes — constructing it is itself part of what is under test. */
|
||||
private BridgeMcp mcp(boolean enforce) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN", null,
|
||||
"tab", "bridged-workers", "worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(agents, new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "tok");
|
||||
SessionManager sessions = new SessionManager(workers, new FakeWorktrees());
|
||||
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(),
|
||||
new InMemoryReplyInbox());
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> 999_999);
|
||||
metrics = BridgedMetrics.create(sessions, new InMemoryReplyInbox());
|
||||
|
||||
mcp = new BridgeMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||
new PrimaryRegistry(null),
|
||||
enforce ? CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
Map::of, new MemberRegistry(null)) : null,
|
||||
metrics, BridgeMcp.CapacitySource.none(), new BridgeMcp.HealthCoverageSource(() -> "off"),
|
||||
BridgeMcp.QuarantineSource.none());
|
||||
return mcp;
|
||||
}
|
||||
|
||||
private static final Principal PRIMARY = Principal.primary(100);
|
||||
private static final Principal WORKER_A = Principal.worker("term_a", 200);
|
||||
private static final Principal ANON = Principal.anonymous();
|
||||
private static final Principal ARCH_DESIGN = Principal.architect("lead-designer", "term_design", 400);
|
||||
|
||||
// --- the table, enforced on THIS path too ---------------------------------------------------
|
||||
|
||||
@Test
|
||||
void primaryMayOrchestrate() {
|
||||
BridgeMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.SEND, Authz.Action.DRAIN, Authz.Action.READ}) {
|
||||
assertNull(m.denyFor(PRIMARY, a, "term_a"), a + " is the primary's to perform");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayNotOrchestrateOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.SEND, Authz.Action.DRAIN}) {
|
||||
McpSchema.CallToolResult denied = m.denyFor(WORKER_A, a, "term_a");
|
||||
assertNotNull(denied, a + " must be refused to a worker");
|
||||
assertTrue(denied.isError(), "a refusal is returned as an MCP tool error");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayReplyAndAskOnlyAsItself() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.REPLY, "term_a"), "its own session is allowed");
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.ASK, "term_a"));
|
||||
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.REPLY, "term_b"),
|
||||
"worker A must not reply on worker B's session");
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.ASK, "term_b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void thePrimaryMayNotForgeAWorkerReplyOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
// A forged reply would resolve the very rendezvous the primary is blocked on.
|
||||
assertNotNull(m.denyFor(PRIMARY, Authz.Action.REPLY, "term_a"));
|
||||
assertNotNull(m.denyFor(PRIMARY, Authz.Action.ASK, "term_a"));
|
||||
}
|
||||
|
||||
// --- CB-548: the architect on this path ------------------------------------------------
|
||||
|
||||
@Test
|
||||
void anArchitectMaySendAndReadButNotOrchestrateOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.SEND, "term_a"),
|
||||
"delegating a turn is the architect's job");
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.READ, null));
|
||||
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.DRAIN}) {
|
||||
McpSchema.CallToolResult denied = m.denyFor(ARCH_DESIGN, a, null);
|
||||
assertNotNull(denied, a + " must be refused to an architect");
|
||||
assertTrue(denied.isError(), "a refusal is returned as an MCP tool error");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReplyAndAskOnlyAsItsOwnPaneOverMcp() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.REPLY, "term_design"));
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.ASK, "term_design"));
|
||||
|
||||
assertNotNull(m.denyFor(ARCH_DESIGN, Authz.Action.REPLY, "term_a"),
|
||||
"architect 'lead-designer' must not reply on worker term_a's session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anonymousIsRefusedEverythingAndCountedAsUnauthenticated() {
|
||||
BridgeMcp m = mcp(true);
|
||||
McpSchema.CallToolResult denied = m.denyFor(ANON, Authz.Action.READ, null);
|
||||
|
||||
assertNotNull(denied, "authenticated as nothing ⇒ authorized for nothing");
|
||||
assertEquals(1, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "unauthenticated"));
|
||||
assertEquals(0, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "forbidden"),
|
||||
"a missing credential is 401-shaped, not 403-shaped");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWrongRoleIsCountedAsForbiddenNotUnauthenticated() {
|
||||
BridgeMcp m = mcp(true);
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.SPAWN, null));
|
||||
|
||||
assertEquals(1, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "forbidden"));
|
||||
assertEquals(0, metrics.count(BridgedMetrics.AUTH_FAILURES, "reason", "unauthenticated"),
|
||||
"the caller IS authenticated — it is just not the right role");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theLegacyConstructorLeavesTheGateOpen() {
|
||||
// The 22 pre-existing BridgeMcpTest cases rely on no authorization being enforced.
|
||||
BridgeMcp m = mcp(false);
|
||||
assertNull(m.denyFor(ANON, Authz.Action.SPAWN, null),
|
||||
"no CallerResolver supplied ⇒ authorization not enforced (legacy behaviour)");
|
||||
}
|
||||
|
||||
// --- identity reconstruction from the transport context ------------------------------------
|
||||
|
||||
@Test
|
||||
void principalIsRebuiltFromTheStashedRole() {
|
||||
assertEquals(Role.WORKER, BridgeMcp.principalFrom("WORKER", "term_a", 7).role());
|
||||
assertEquals("term_a", BridgeMcp.principalFrom("WORKER", "term_a", 7).terminal());
|
||||
assertEquals(Role.PRIMARY, BridgeMcp.principalFrom("PRIMARY", null, 7).role());
|
||||
assertEquals(Role.ANONYMOUS, BridgeMcp.principalFrom("ANONYMOUS", null, -1).role());
|
||||
// CB-548: an architect round-trips through the same stash, carrying its slot name.
|
||||
Principal arch = BridgeMcp.principalFrom("ARCHITECT", "term_design", 7, "lead-designer");
|
||||
assertEquals(Role.ARCHITECT, arch.role());
|
||||
assertEquals("lead-designer", arch.name());
|
||||
assertEquals("term_design", arch.terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingRoleFallsBackToTheHistoricalInterpretation() {
|
||||
// Legacy path: no role stashed. A terminal means worker; its absence meant "the primary",
|
||||
// which is exactly the pre-CB-501 default CB-501 inverted — preserved only here.
|
||||
assertEquals(Role.WORKER, BridgeMcp.principalFrom(null, "term_a", 7).role());
|
||||
assertEquals(Role.PRIMARY, BridgeMcp.principalFrom(null, null, 7).role());
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,46 +0,0 @@
|
||||
package dev.ltms.bridged.mcp;
|
||||
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.PaneLocator;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** Connection → caller-identity resolution, with the OS peer-PID lookup faked. */
|
||||
class ConnectionIdentityTest {
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
private ConnectionIdentity with(PeerPidLookup pids) {
|
||||
return new ConnectionIdentity(new PaneLocator(herdr), pids);
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWorkerFromLoopbackPeerPid() {
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.0.0.1", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForOffHostCaller() {
|
||||
// A non-loopback peer can't be an on-host worker → treat as primary/unknown.
|
||||
assertNull(with(_ -> FakeHerdr.WORKER_PID).callerTerminal("10.0.0.9", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullWhenPidOwnsNoPane() {
|
||||
// e.g. the primary — its PID maps to no worker pane.
|
||||
assertNull(with(_ -> 999_999).callerTerminal("127.0.0.1", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesTheCallersPidAndCwd() {
|
||||
// CB-112: the primary maps to no pane, but its PID and cwd are still readable.
|
||||
ConnectionIdentity id = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), _ -> 999_999, pid -> pid == 999_999 ? "/main/project" : null);
|
||||
ConnectionIdentity.Caller c = id.resolve("127.0.0.1", 55555);
|
||||
assertNull(c.terminal(), "the primary owns no worker pane");
|
||||
assertEquals(999_999, c.pid());
|
||||
assertEquals("/main/project", id.cwdForPid(c.pid()), "the primary's cwd is resolvable from its PID");
|
||||
assertNull(id.cwdForPid(-1), "no cwd for an unresolved PID");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,726 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.Agent;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.CharterReceipt;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import dev.ltms.bridged.placement.BackendQuarantine;
|
||||
import dev.ltms.bridged.placement.PlacementException;
|
||||
import dev.ltms.bridged.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The composite router: profile → owning adapter for spawn/cwd/parity, pane id → owner for stop,
|
||||
* and fleet-wide union/dedup for list/reap/caps/profiles. Exercised through two real adapters —
|
||||
* claude-code + opencode — over one FakeHerdr, so each call is observed reaching the right adapter
|
||||
* (the started herdr agent name carries that adapter's {@code claude-}/{@code opencode-} prefix).
|
||||
*/
|
||||
class CompositePeerLauncherTest {
|
||||
|
||||
private ClaudeCodeLauncher claudeAdapter(FakeHerdr herdr) {
|
||||
// 12-arg back-compat Worker ctor → kind defaults to claude-code.
|
||||
BridgedConfig.Profile claude = new BridgedConfig.Profile("claude", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("claude", claude), "claude", _ -> null);
|
||||
}
|
||||
|
||||
private OpenCodeLauncher opencodeAdapter(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile gemini = new BridgedConfig.Profile("gemini", null, "google/gemini-2.5-pro",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("opencode"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null, "GITEA_ACCESS_TOKEN", null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", gemini), "gemini", _ -> "tok");
|
||||
}
|
||||
|
||||
private CompositePeerLauncher composite(FakeHerdr herdr) {
|
||||
return new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(herdr), opencodeAdapter(herdr)), "claude");
|
||||
}
|
||||
|
||||
/**
|
||||
* A minimal concrete HerdrPeerLauncher for policy tests. It either returns a fake handle for the
|
||||
* requested profile or throws, depending on {@code failProfiles}. buildLaunch is a stub; only
|
||||
* spawn/stop/list/caps/reap are exercised by the composite.
|
||||
*/
|
||||
private static final class StubLauncher extends HerdrPeerLauncher {
|
||||
private final Set<String> failProfiles;
|
||||
private final Map<String, Integer> spawnCounts = new HashMap<>();
|
||||
|
||||
StubLauncher(String prefix, FakeHerdr herdr,
|
||||
Map<String, BridgedConfig.Profile> profiles, String defaultProfile,
|
||||
Set<String> failProfiles) {
|
||||
super(prefix, new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
profiles, defaultProfile, _ -> null, 0L, System::currentTimeMillis, () -> { });
|
||||
this.failProfiles = Set.copyOf(failProfiles);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected Launch buildLaunch(BridgedConfig.Profile cfg, LaunchSpec spec) {
|
||||
return new Launch(Map.of(), List.of());
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String p = (req.profileName() == null || req.profileName().isBlank())
|
||||
? defaultProfile() : req.profileName();
|
||||
spawnCounts.merge(p, 1, Integer::sum);
|
||||
if (failProfiles.contains(p)) {
|
||||
throw new PeerUnreachableException(p + " is down");
|
||||
}
|
||||
return new PeerHandle() {
|
||||
@Override public String id() { return "pane-" + p; }
|
||||
@Override public String terminalId() { return "term-" + p; }
|
||||
@Override public String profile() { return p; }
|
||||
@Override public String agentSessionId() { return null; }
|
||||
@Override public CharterReceipt charterReceipt() { return null; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) { }
|
||||
|
||||
@Override
|
||||
public List<Agent> list() { return List.of(); }
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() { return EnumSet.noneOf(Capability.class); }
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() { return 0; }
|
||||
|
||||
int spawnCount(String profile) {
|
||||
return spawnCounts.getOrDefault(profile, 0);
|
||||
}
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile stubWorker(String profile) {
|
||||
return new BridgedConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null);
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile stubWorker(String profile, float weight, Integer maxLoad) {
|
||||
return new BridgedConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null,
|
||||
weight, maxLoad);
|
||||
}
|
||||
|
||||
private static BridgedConfig.Profile stubWorker(String profile, String credentialId) {
|
||||
return new BridgedConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null,
|
||||
null, null, credentialId);
|
||||
}
|
||||
|
||||
/**
|
||||
* An <em>order-preserving</em> profile map. Never {@code Map.of} here: its iteration order is
|
||||
* salted per JVM run, and the weighted policy breaks an exact-weight tie on candidate order —
|
||||
* so a {@code Map.of} would make "which profile is tried first" a coin flip per run and any
|
||||
* assertion about the first attempt intermittently false.
|
||||
*/
|
||||
private static Map<String, BridgedConfig.Profile> ordered(String first, BridgedConfig.Profile a,
|
||||
String second, BridgedConfig.Profile b) {
|
||||
Map<String, BridgedConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put(first, a);
|
||||
m.put(second, b);
|
||||
return m;
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static String startedName(FakeHerdr herdr) {
|
||||
return (String) ((Map<String, Object>) herdr.lastCall("agent.start").params()).get("name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRoutesEachProfileToItsOwningAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
|
||||
composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("opencode-"),
|
||||
"the gemini profile is spawned by the opencode adapter: " + startedName(herdr));
|
||||
|
||||
composite.spawn(new SpawnRequest("claude", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"the claude profile is spawned by the claude-code adapter: " + startedName(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullProfileResolvesTheDefaultAndRoutesToItsOwner() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
composite(herdr).spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"a no-profile spawn resolves the default (claude) and routes to its adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unknownProfileIsRejected() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> composite.spawn(new SpawnRequest("nope", null, null)),
|
||||
"a profile no adapter declares is an error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesAndDefaultAreExposedAcrossAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(Set.of("claude", "gemini"), composite.profiles(),
|
||||
"profiles are the union of every adapter's profiles");
|
||||
assertEquals("claude", composite.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesAreTheUnionOfEveryAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher claude = claudeAdapter(herdr);
|
||||
OpenCodeLauncher opencode = opencodeAdapter(herdr);
|
||||
PeerLauncher composite = new CompositePeerLauncher(List.of(claude, opencode), "claude");
|
||||
|
||||
assertTrue(composite.capabilities().containsAll(claude.capabilities()),
|
||||
"the fleet offers every claude-code capability");
|
||||
assertTrue(composite.capabilities().containsAll(opencode.capabilities()),
|
||||
"the fleet offers every opencode capability (incl. SELF_PR from its git-token profile)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void listIsDeduplicatedByPaneIdAcrossAdaptersSharingHerdr() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
// Both adapters wrap the same herdr, so each list() returns the same global agent set;
|
||||
// the composite must return each pane once, not once per adapter.
|
||||
assertEquals(1, composite.list().size(),
|
||||
"the single herdr-tracked pane appears once, not duplicated per adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapSumsAcrossAdaptersAndEachAdapterReapsOnlyItsOwnPrefix() {
|
||||
// One foreign opencode orphan + one foreign claude orphan, from a prior daemon (different nonce).
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withAgent("opencode-gemini-ffffff-1", "term_o", "wQ:pO", "wQ:tO")
|
||||
.withAgent("claude-claude-eeeeee-1", "term_c", "wQ:pC", "wQ:tC");
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(2, composite.reapOrphanWorkers(),
|
||||
"both orphans are reaped — one by each adapter, summed by the composite");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopTearsDownAPaneSpawnedThroughTheComposite() {
|
||||
// CB-519: handle.id() is a host-unique opaque UUID, not the herdr pane — stop(id) must
|
||||
// resolve it through the owning adapter down to the actual pane coordinate it spawned.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertNotEquals("w9:pRoot_1", handle.id(), "the id is decoupled from the pane coordinate");
|
||||
|
||||
composite.stop(handle.id());
|
||||
assertTrue(herdr.calls.stream()
|
||||
.anyMatch(c -> c.method().equals("pane.close")
|
||||
&& "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id"))),
|
||||
"stop routes to the spawning adapter and closes exactly that worker's pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void opencodeContextResetIsANoOpAndWarnsOnlyOnce() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertFalse(opencodeAdapter(herdr).capabilities().contains(Capability.CONTEXT_RESET));
|
||||
assertTrue(herdr.calls.stream().noneMatch(c -> "agent.prompt".equals(c.method())),
|
||||
"never type Claude's /clear into an opencode prompt");
|
||||
assertEquals(1, appender.list.stream()
|
||||
.filter(e -> e.getFormattedMessage().contains("context reset is unsupported"))
|
||||
.count(), "unsupported reset is logged once per adapter, not once per turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAProfileClaimedByTwoAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// Two opencode adapters both declaring "gemini" — a profile-name collision.
|
||||
OpenCodeLauncher a = opencodeAdapter(herdr);
|
||||
OpenCodeLauncher b = opencodeAdapter(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(a, b), "gemini"),
|
||||
"a profile two adapters both claim is a configuration error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAnEmptyAdapterList() {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(), "claude"),
|
||||
"at least one adapter must be configured");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedDefaultIsNoOpForUnqualifiedSpawns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"fixed placement still routes an unqualified spawn to the default profile");
|
||||
assertEquals("claude", h.profile(), "the returned handle carries the resolved default profile");
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyGatesProfileAtMaxLoad() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
for (int i = 0; i < 5; i++) {
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "profile a is at maxLoad, so every spawn must land on b");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyDistributesAccordingToWeightRatio() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.75f, null),
|
||||
"b", stubWorker("b", 0.25f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 40; i++) {
|
||||
String p = composite.spawn(new SpawnRequest(null, null, null)).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(30, a, "weighted distribution should hold the 3:1 ratio");
|
||||
assertEquals(10, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicySkipsWeightZeroProfileOnUnqualifiedSpawn() {
|
||||
// CB-554: weight: 0 must exclude a profile from automatic placement, not coerce to 1.0.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.0f, null),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
for (int i = 0; i < 5; i++) {
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "profile a has weight 0, so every unqualified spawn must land on b");
|
||||
}
|
||||
assertEquals(0, adapter.spawnCount("a"), "a is never chosen automatically");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnStillSucceedsOnWeightZeroProfile() {
|
||||
// CB-554: weight: 0 excludes a profile from AUTOMATIC selection only — an explicit
|
||||
// bridge_spawn{profile:"a"} must still work exactly as today (e.g. `opus` on the
|
||||
// operator's own subscription, kept weight-0 so it is never picked automatically).
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.0f, null),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "b", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "b", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "naming a weight-0 profile explicitly bypasses placement and still spawns it");
|
||||
assertEquals(1, adapter.spawnCount("a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverRetriesNextCandidateWhenProfileIsUnreachable() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "the spawn must fail over from unreachable a to b");
|
||||
assertEquals(1, adapter.spawnCount("a"), "a was tried once and failed");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b was tried once and succeeded");
|
||||
}
|
||||
|
||||
/**
|
||||
* Definition order — not hash order — decides an exact-weight tie. Paired with the test above
|
||||
* (same two profiles, opposite declaration order, opposite expected first attempt) this pins the
|
||||
* ordering contract from both sides: under a salted map one of the two must fail on every run.
|
||||
*/
|
||||
@Test
|
||||
void reversingDefinitionOrderReversesWhichProfileIsTriedFirst() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"b", stubWorker("b"),
|
||||
"a", stubWorker("a"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("a", h.profile(), "b is declared first and unreachable, so the spawn lands on a");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b, declared first, is the one tried first");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverBoundedByCandidateCount() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a", "b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerUnreachableException e = assertThrows(PeerUnreachableException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("no reachable worker profile"), e.getMessage());
|
||||
assertEquals(1, adapter.spawnCount("a"));
|
||||
assertEquals(1, adapter.spawnCount("b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnAtMaxLoadThrowsPlacementExceptionNamingProfileLiveAndCap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 2 : 0);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("a", null, null)));
|
||||
assertTrue(e.getMessage().contains("'a'"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 live"), "message names the live count: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 cap"), "message names the cap: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("a"), "at cap, the spawn is refused before any delegation");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnUnderMaxLoadStillSucceeds() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile under its cap accepts an explicit spawn");
|
||||
assertEquals(1, adapter.spawnCount("a"), "the under-cap spawn is delegated");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnWithNullMaxLoadIsNeverCapped() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, null),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
// A deliberately absurd live count: an unset maxLoad means unlimited, so it must never refuse.
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 1000);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile with no maxLoad is never capped, however many live workers");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnOnMaxLoadZeroProfileIsRefusedEvenWithZeroLiveWorkers() {
|
||||
// CB-585: before the fix, maxLoad: 0 normalised to null (unlimited) in the compact
|
||||
// constructor, so this exact case — naming a zero-cap profile explicitly, with nothing
|
||||
// live on it yet — would have spawned instead of refusing. "at most zero members" must
|
||||
// hold even when the profile is named directly, not only against automatic placement.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 0),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 0);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("a", null, null)));
|
||||
assertTrue(e.getMessage().contains("'a'"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("0 cap"), "message names the zero cap: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("a"), "a maxLoad: 0 profile accepts no explicit spawn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void emptyCandidateSetThrowsClearException() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, 1));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 1);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
// ── CB-557: an unqualified spawn is placed inside its role's pool ─────────────────────────
|
||||
|
||||
/** Three profiles in definition order — pools are carved out of this set. */
|
||||
private static Map<String, BridgedConfig.Profile> threeProfiles() {
|
||||
Map<String, BridgedConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("opus", stubWorker("opus", 1.0f, null));
|
||||
m.put("sonnet", stubWorker("sonnet", 1.0f, null));
|
||||
m.put("terra", stubWorker("terra", 1.0f, null));
|
||||
return m;
|
||||
}
|
||||
|
||||
private static Map<String, BridgedConfig.Slot> pool(String... names) {
|
||||
Map<String, BridgedConfig.Slot> m = new LinkedHashMap<>();
|
||||
for (String n : names) {
|
||||
m.put(n, new BridgedConfig.Slot(n));
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
private static CompositePeerLauncher withPools(FakeHerdr herdr, BridgedConfig.Fleet fleet) {
|
||||
Map<String, BridgedConfig.Profile> profiles = threeProfiles();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
return new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* The point of the pools: a role is placed only on a backend its pool names. Before CB-557 an
|
||||
* unqualified spawn ranged over every configured profile, so a reviewer could land on the
|
||||
* architect-only one.
|
||||
*/
|
||||
@Test
|
||||
void anUnqualifiedSpawnIsPlacedInsideItsRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), pool("sonnet"), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.ARCHITECT)).profile());
|
||||
assertEquals("terra", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile());
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.REVIEWER)).profile());
|
||||
}
|
||||
|
||||
/** Under `fixed`, the pool's first entry wins — not the global defaultProfile. */
|
||||
@Test
|
||||
void theRolePoolOutranksTheGlobalDefaultProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), Map.of(), pool("sonnet", "terra"), Map.of(), null));
|
||||
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"the dev pool starts at sonnet, so the global default 'opus' must not win");
|
||||
}
|
||||
|
||||
/**
|
||||
* A role with no pool is unconstrained, not blocked. A config that declares pools for some roles
|
||||
* and not others must keep spawning the rest.
|
||||
*/
|
||||
@Test
|
||||
void aRoleWithNoPoolFallsBackToEveryProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("sonnet"), Map.of(), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"no dev pool ⇒ all profiles are candidates, so `fixed` takes the first one");
|
||||
}
|
||||
|
||||
/** No fleet at all is the pre-CB-557 wiring, and must behave exactly as it did. */
|
||||
@Test
|
||||
void noFleetConfiguredKeepsTheOldWholeProfileListBehaviour() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, null);
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest(null, null, null)).profile());
|
||||
}
|
||||
|
||||
/**
|
||||
* An explicit profile is the operator overriding and is NOT judged against the pool. It must
|
||||
* stay that way: an unrolled `bridge_spawn{profile:"opus"}` carries no role, so it defaults to
|
||||
* DEV, and enforcing the pool here would refuse a spawn the operator asked for by name.
|
||||
*/
|
||||
@Test
|
||||
void anExplicitProfileIsNotConfinedToTheRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new BridgedConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest("opus", null, null)).profile(),
|
||||
"naming opus explicitly must work even though the dev pool holds only terra");
|
||||
}
|
||||
|
||||
/** Placement still respects maxLoad, but only across the pool — never by escaping it. */
|
||||
@Test
|
||||
void aFullPoolIsRefusedRatherThanSpilledOntoAnotherRolesProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("opus", stubWorker("opus", 1.0f, null)); // architect-only, uncapped
|
||||
profiles.put("terra", stubWorker("terra", 1.0f, 1)); // the sole dev, capped at 1
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.weighted(), name -> "terra".equals(name) ? 1 : 0,
|
||||
new BridgedConfig.Fleet(Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
// ── CB-578 stage B: a BACKEND_EXHAUSTED classification quarantines the credential ──────────
|
||||
|
||||
@Test
|
||||
void explicitSpawnOntoAQuarantinedProfileIsRefusedNamingTheCredentialAndRemainingTime() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("sol", null, null)));
|
||||
assertTrue(e.getMessage().contains("sol"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("shared-openai"), "message names the credential: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("1800"), "message names roughly when it lifts: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("sol"), "the quarantined profile is never delegated to");
|
||||
}
|
||||
|
||||
/**
|
||||
* The part CB-578 stage B calls out as easy to get wrong: sol and terra are two different
|
||||
* profiles sharing one OpenAI credential. Quarantining because of an exhaustion classified on
|
||||
* ONE of them must lock out the other too, or the fleet just walks onto the same dead account
|
||||
* under the sibling's name.
|
||||
*/
|
||||
@Test
|
||||
void twoProfilesSharingACredentialAreBothQuarantinedByOneExhaustionEvent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
// Only "sol" was classified BACKEND_EXHAUSTED — but the two profiles share one credential.
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("sol", null, null)),
|
||||
"sol was the one classified exhausted");
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("terra", null, null)),
|
||||
"terra shares sol's credential, so it must be locked out too");
|
||||
assertEquals(0, adapter.spawnCount("sol"));
|
||||
assertEquals(0, adapter.spawnCount("terra"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void placementSkipsAQuarantinedProfileAndRoutesToAnUnquarantinedOne() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, quarantine);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "sol is quarantined, so an unqualified spawn must land on b");
|
||||
assertEquals(0, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aQuarantineLiftsOnTheInjectedClockAndTheProfileBecomesSpawnableAgain() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, BridgedConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
AtomicLong nowNanos = new AtomicLong(0L);
|
||||
BackendQuarantine quarantine = new BackendQuarantine(nowNanos::get, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("sol", null, null)),
|
||||
"still inside the cooldown");
|
||||
|
||||
nowNanos.set(TimeUnit.MINUTES.toNanos(31));
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("sol", null, null));
|
||||
assertEquals("sol", h.profile(), "the cooldown expired on the injected clock — sol is spawnable again");
|
||||
assertEquals(1, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFleetWithNoExhaustedPatternAnywhereBehavesExactlyAsBeforeQuarantineExisted() {
|
||||
// BackendQuarantine.none() is the inert stand-in every constructor already defaults to when
|
||||
// no quarantine is wired — the 2/5/6-arg constructors used throughout this file all exercise
|
||||
// it. This test pins that an explicit .none() also never refuses a spawn, for any profile.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("claude", null, null)));
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("gemini", null, null)));
|
||||
}
|
||||
}
|
||||
@@ -1,465 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.CharterReceipt;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.Future;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The opencode adapter's launch build: a file-based MCP mount + reply-charter instructions (no
|
||||
* inline flags, no {@code ANTHROPIC_*}, no guard), the {@code -m} model flag, and the shared base
|
||||
* transport (naming, reap, readiness gate) proving the {@link HerdrPeerLauncher} SPI is neutral.
|
||||
*/
|
||||
class OpenCodeLauncherTest {
|
||||
|
||||
private static BridgedConfig.Profile opencodeCfg(String model, String mcpUrl, String gitTokenEnv) {
|
||||
return new BridgedConfig.Profile("gemini", null, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, gitTokenEnv, null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
/** Gate-disabled launcher whose per-spawn config dirs land under an inspectable temp root. */
|
||||
private static OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, BridgedConfig.Profile cfg) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), k -> "GITEA_ACCESS_TOKEN".equals(k) ? "tok" : null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot);
|
||||
}
|
||||
|
||||
private static OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, BridgedConfig.Profile cfg,
|
||||
Supplier<BridgedConfig.Fleet> fleet) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot, fleet);
|
||||
}
|
||||
|
||||
private static OpenCodeLauncher serviceWithCredentials(FakeHerdr herdr, Path configRoot,
|
||||
BridgedConfig.Profile cfg,
|
||||
BridgedConfig.MemberCredentials creds) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), k -> "GITEA_ACCESS_TOKEN".equals(k) ? "tok" : null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot, null, () -> creds);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, Object> lastStart(FakeHerdr herdr) {
|
||||
return (Map<String, Object>) herdr.lastCall("agent.start").params();
|
||||
}
|
||||
|
||||
/** Protocol 19: the worker's env is injected at pane creation (tab.create), not agent.start. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, String> startEnv(FakeHerdr herdr) {
|
||||
Map<String, String> env =
|
||||
(Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||
return env == null ? Map.of() : env;
|
||||
}
|
||||
|
||||
/** Protocol 19: agent.start carries only the args after the kind-resolved executable. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static List<String> startArgs(FakeHerdr herdr) {
|
||||
return (List<String>) lastStart(herdr).get("args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void writesRemoteMcpConfigAndCharterInstructionsWhenMcpUrlSet(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Fleet fleet = new BridgedConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null),
|
||||
() -> fleet).spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"), "opencode carries no ANTHROPIC_* / subscription boundary");
|
||||
String cfgPath = env.get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "OPENCODE_CONFIG points the worker at the generated config file");
|
||||
assertTrue(Path.of(cfgPath).startsWith(root), "config file is generated under the injected root");
|
||||
|
||||
// Assert on parsed structure, not substrings: the generated config is real JSON and its
|
||||
// whitespace is the formatter's business, not the contract's.
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
assertTrue(json.path("compaction").path("auto").asBoolean(),
|
||||
"spawned opencode peers explicitly enable automatic compaction");
|
||||
JsonNode bridge = json.path("mcp").path("bridge");
|
||||
assertEquals("remote", bridge.path("type").asText(), "bridge is mounted as a remote MCP server");
|
||||
assertEquals("http://127.0.0.1:8765/mcp", bridge.path("url").asText(),
|
||||
"the profile's bridge MCP url is present");
|
||||
assertTrue(bridge.path("enabled").asBoolean(), "the bridge server is enabled");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the member charter is mounted via instructions");
|
||||
|
||||
// The instructions entry is a real file path holding the composed member charter.
|
||||
Path charter = Path.of(cfgPath).resolveSibling("member-charter.md");
|
||||
assertTrue(Files.exists(charter), "the charter file the config references was written");
|
||||
assertEquals("role rule\n\n" + HerdrPeerLauncher.REPLY_CHARTER, Files.readString(charter),
|
||||
"the composed charter keeps the role rule first and the reply rule last");
|
||||
assertEquals(charter.toAbsolutePath().toString(), json.path("instructions").get(0).asText(),
|
||||
"instructions names the charter file by its absolute path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noConfigFileWhenMcpUrlAbsent(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
assertNull(startEnv(herdr).get("OPENCODE_CONFIG"),
|
||||
"no bridge MCP url → no config file and no OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void roleCharterWithoutMcpOrCustomProviderStillWritesAConfig(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Fleet fleet = new BridgedConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
Path configRoot = Files.createDirectory(root.resolve("configs"));
|
||||
Path checkout = Files.createDirectory(root.resolve("checkout"));
|
||||
service(herdr, configRoot, opencodeCfg("google/gemini-2.5-pro", null, null), () -> fleet).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a role charter needs a config even without MCP or custom provider");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
Path charter = Path.of(json.path("instructions").get(0).asText());
|
||||
assertEquals("role rule", Files.readString(charter), "the base-composed role charter is unchanged");
|
||||
assertTrue(json.path("mcp").isMissingNode(), "a charter does not add an MCP mount");
|
||||
assertTrue(charter.startsWith(configRoot), "the charter is written under the temp config root");
|
||||
try (var files = Files.walk(checkout)) {
|
||||
assertFalse(files.anyMatch(path -> path.getFileName().toString().equals("member-charter.md")),
|
||||
"the worker checkout receives no charter file");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullCharterWritesNoCharterFileOrInstructions(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/model", "http://127.0.0.1:8000", null),
|
||||
() -> new BridgedConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(), Map.of(), null)).spawn();
|
||||
|
||||
String config = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(config, "the custom provider still needs a config");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(config).toFile());
|
||||
assertTrue(json.path("instructions").isMissingNode(), "a null charter adds no instructions entry");
|
||||
try (var files = Files.walk(root)) {
|
||||
assertFalse(files.anyMatch(path -> path.getFileName().toString().equals("member-charter.md")),
|
||||
"a null charter creates no charter file");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void concurrentSpawnsWriteSeparateCharterDirectories(@TempDir Path root) throws Exception {
|
||||
BridgedConfig.Profile cfg = opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null);
|
||||
ExecutorService executor = Executors.newFixedThreadPool(2);
|
||||
try {
|
||||
Future<String> first = executor.submit(() -> spawnConfigPath(root, cfg));
|
||||
Future<String> second = executor.submit(() -> spawnConfigPath(root, cfg));
|
||||
|
||||
Path firstCharter = Path.of(first.get()).resolveSibling("member-charter.md");
|
||||
Path secondCharter = Path.of(second.get()).resolveSibling("member-charter.md");
|
||||
assertNotEquals(firstCharter.getParent(), secondCharter.getParent(),
|
||||
"each concurrent spawn owns a separate config directory");
|
||||
assertTrue(Files.exists(firstCharter));
|
||||
assertTrue(Files.exists(secondCharter));
|
||||
} finally {
|
||||
executor.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
private static String spawnConfigPath(Path root, BridgedConfig.Profile cfg) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, cfg).spawn();
|
||||
return startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void passesTheModelAsDashMFlagAlongsideAutoApprove(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
assertTrue(args.contains("--auto"),
|
||||
"--auto is present alongside -m so a spawned peer never blocks on approval");
|
||||
int m = args.indexOf("-m");
|
||||
assertTrue(m >= 0, "model is selected with -m");
|
||||
assertEquals("google/gemini-2.5-pro", args.get(m + 1), "the provider/model selector follows -m");
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoApproveIsUnconditionalWhenModelBlank(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null)).spawn();
|
||||
assertEquals(List.of("--auto"), startArgs(herdr),
|
||||
"--auto is unconditional: a model-less worker still must never block on approval");
|
||||
}
|
||||
|
||||
@Test
|
||||
void injectsForgeTokenWhenProfileGrantsIt(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN")).spawn();
|
||||
assertEquals("tok", startEnv(herdr).get("GITEA_TOKEN"),
|
||||
"a git-token profile gets the peer-neutral GITEA_TOKEN grant, same as Claude");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: the config-driven shadow lives in {@link HerdrPeerLauncher#baseEnv}, shared by every
|
||||
* adapter — this pins that the opencode path gets it too, not just Claude's. See the matching
|
||||
* tests in {@code ClaudeCodeLauncherTest} for the full rationale (gitea #82, superseding CB-592's
|
||||
* single hardcoded name).
|
||||
*/
|
||||
@Test
|
||||
void aKnownNameNotAllowedIsShadowedWithTheSentinel(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.MemberCredentials creds = new BridgedConfig.MemberCredentials(
|
||||
null, List.of("AI_GATEWAY_TOKEN"), List.of("AI_GATEWAY_TOKEN", "GITEA_ACCESS_TOKEN"));
|
||||
serviceWithCredentials(herdr, root, opencodeCfg(null, null, null), creds).spawn();
|
||||
|
||||
String shadowed = startEnv(herdr).get("GITEA_ACCESS_TOKEN");
|
||||
assertNotNull(shadowed, "GITEA_ACCESS_TOKEN must be explicitly overlaid, not left unmentioned");
|
||||
assertFalse(shadowed.isBlank(), "a blank overlay value's override behaviour is unverified — must be non-blank");
|
||||
assertFalse(startEnv(herdr).containsKey("AI_GATEWAY_TOKEN"),
|
||||
"an allow-listed name must get no overlay entry at all");
|
||||
}
|
||||
|
||||
/** No {@code memberCredentials} configured (the pre-CB-596 constructor overloads) blocks nothing. */
|
||||
@Test
|
||||
void noMemberCredentialsConfiguredBlocksNothing(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null)).spawn();
|
||||
|
||||
assertNull(startEnv(herdr).get("GITEA_ACCESS_TOKEN"),
|
||||
"with no memberCredentials configured, nothing is shadowed — config must supply the policy");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesDeclareOrphanReapAndMcpAskAndConditionalSelfPr(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
assertEquals(java.util.Set.of(Capability.MID_TURN_ASK, Capability.WORKTREE, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_RESUME),
|
||||
service(herdr, root, opencodeCfg(null, null, null)).capabilities(),
|
||||
"opencode can be resumed by its own session id, so SESSION_RESUME is always declared");
|
||||
assertFalse(service(herdr, root, opencodeCfg(null, null, null))
|
||||
.capabilities().contains(Capability.SESSION_NAME),
|
||||
"opencode has no display-name flag, so SESSION_NAME must NOT be declared");
|
||||
assertTrue(service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN"))
|
||||
.capabilities().contains(Capability.SELF_PR),
|
||||
"a git-token profile adds SELF_PR");
|
||||
}
|
||||
|
||||
// --- CB-547: resume + post-hoc session discovery --------------------------------------------
|
||||
|
||||
@Test
|
||||
void aResumeSpawnPassesTheSessionIdAsDashS(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, "ses_41b79fc90ffeI9E8uZv6VprUn2"));
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
int s = args.indexOf("-s");
|
||||
assertTrue(s >= 0, "a resumed spawn carries opencode's -s flag");
|
||||
assertEquals("ses_41b79fc90ffeI9E8uZv6VprUn2", args.get(s + 1),
|
||||
"the resume target id follows -s");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFreshSpawnCarriesNoSessionFlag(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, null));
|
||||
|
||||
assertFalse(startArgs(herdr).contains("-s"),
|
||||
"no resume target → a fresh session with no -s flag");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theHandleDiscoversTheSessionIdForTheWorkersCwdOnlyAfterItAppears(@TempDir Path root,
|
||||
@TempDir Path discRoot)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = new OpenCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), Map.of("gemini", opencodeCfg(null, null, null)),
|
||||
"gemini", _ -> null, 0, System::currentTimeMillis, () -> { }, root, discRoot);
|
||||
|
||||
PeerHandle handle = launcher.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
|
||||
// opencode writes the record only when the session is first persisted — the instant the
|
||||
// pane is ready it does not exist, so agentSessionId() is null (never a spawn failure).
|
||||
assertNull(handle.agentSessionId(), "no record yet → null, not a spawn-time block");
|
||||
// Once the record appears (here: same cwd), lazy discovery resolves it — the handle's
|
||||
// session id matches its own worktree, not another's.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "p1", "ses_a.json",
|
||||
"ses_resolved", "/work/dir", 1000L);
|
||||
assertEquals("ses_resolved", handle.agentSessionId(),
|
||||
"agentSessionId() re-scans and picks up a record that has since been written");
|
||||
}
|
||||
|
||||
@Test
|
||||
void foreignWorkerMatchesOpencodePrefixButNotClaude() {
|
||||
String nonce = "abc123";
|
||||
assertTrue(OpenCodeLauncher.isForeignWorker("opencode-gemini-def456-1", nonce),
|
||||
"an opencode pane from another process is foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("opencode-gemini-" + nonce + "-1", nonce),
|
||||
"our own opencode pane (same nonce) is not foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("claude-ltms-local-def456-1", nonce),
|
||||
"a claude pane is never reaped by the opencode adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionConstructorsWireThroughToTheBase() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Profile cfg = opencodeCfg(null, null, null);
|
||||
// 5-arg (gate disabled) and 7-arg (gate enabled) production constructors both expose the profile.
|
||||
OpenCodeLauncher disabled = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
OpenCodeLauncher gated = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null, 5000, 100);
|
||||
assertEquals(java.util.Set.of("gemini"), disabled.profiles());
|
||||
assertEquals("gemini", gated.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnGateThrowsPeerUnreachableWhenNeverInjectable(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never injectable
|
||||
long[] clock = {0};
|
||||
OpenCodeLauncher svc = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", opencodeCfg(null, null, null)), "gemini", _ -> null,
|
||||
1000, () -> clock[0], () -> clock[0] += 50, root, root);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(clock[0] >= 1000, "the fake clock advanced past the timeout: " + clock[0]);
|
||||
long closes = herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("pane.close"))
|
||||
.filter(c -> "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
assertEquals(1, closes, "the worker pane was reaped on timeout (no orphan)");
|
||||
assertNotNull(ex.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsHandleWhenGateDisabled(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerHandle handle = service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null));
|
||||
assertNotNull(handle, "spawn returns a handle when the gate is disabled");
|
||||
assertFalse(herdr.called("agent.get"), "no polling when the gate is disabled");
|
||||
}
|
||||
|
||||
@Test
|
||||
void handleCarriesTheRealCharterReceiptNotTheInterfaceDefault(@TempDir Path root) {
|
||||
// The base's WorkerHandle computes a real CharterReceipt (CB-571), but the opencode adapter
|
||||
// wraps it in SessionAwareHandle for lazy session discovery. Before this fix that decorator
|
||||
// did not override charterReceipt(), so it silently inherited PeerHandle's `null` default
|
||||
// and the real receipt sitting on its delegate was lost.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
BridgedConfig.Fleet fleet = new BridgedConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
PeerHandle handle = service(herdr, root,
|
||||
opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null), () -> fleet)
|
||||
.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle.charterReceipt(),
|
||||
"an opencode spawn's charterReceipt() must not silently be null");
|
||||
String composed = "role rule\n\n" + HerdrPeerLauncher.REPLY_CHARTER;
|
||||
assertEquals(CharterReceipt.digestOf(composed), handle.charterReceipt().charterSha256(),
|
||||
"the receipt on the wrapped handle must match the exact composed charter bytes");
|
||||
}
|
||||
|
||||
// --- CB-508: pinned OpenAI-compatible endpoint (e.g. a local vLLM) ---------------------------
|
||||
|
||||
/** A profile with a baseUrl but no model provider prefix cannot be resolved — fail loudly. */
|
||||
private static BridgedConfig.Profile pinnedCfg(String model, String baseUrl, String mcpUrl) {
|
||||
return new BridgedConfig.Profile("local", baseUrl, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, null, null, BridgedConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
@Test
|
||||
void baseUrlDeclaresACustomOpenAiCompatibleProvider(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash", "http://127.0.0.1:8000", null))
|
||||
.spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a pinned endpoint needs a config file even with no bridge MCP url");
|
||||
JsonNode provider = new ObjectMapper().readTree(Path.of(cfgPath).toFile())
|
||||
.path("provider").path("local-vllm");
|
||||
|
||||
assertFalse(provider.isMissingNode(), "the provider id comes from the model selector");
|
||||
assertEquals("@ai-sdk/openai-compatible", provider.path("npm").asText());
|
||||
assertEquals("http://127.0.0.1:8000/v1", provider.path("options").path("baseURL").asText(),
|
||||
"a bare host:port gets /v1 appended — that is where these servers mount the API");
|
||||
assertFalse(provider.path("options").path("apiKey").asText().isBlank(),
|
||||
"the AI SDK requires a non-empty key even when the server ignores it");
|
||||
assertFalse(provider.path("models").path("deepseek-v4-flash").isMissingNode(),
|
||||
"the model half of the selector is declared under the provider");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBaseUrlThatAlreadyCarriesAPathIsUsedVerbatim(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/m", "http://127.0.0.1:8000/openai/v1", null)).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("http://127.0.0.1:8000/openai/v1",
|
||||
json.path("provider").path("local-vllm").path("options").path("baseURL").asText(),
|
||||
"an endpoint mounted on a custom path must not have /v1 bolted on");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointRejectsAModelWithNoProviderPrefix(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher =
|
||||
service(herdr, root, pinnedCfg("deepseek-v4-flash", "http://127.0.0.1:8000", null));
|
||||
|
||||
// Silently falling back to the default gateway would point the worker at the wrong LLM
|
||||
// while looking healthy — the one failure mode worth being loud about.
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, launcher::spawn);
|
||||
assertTrue(e.getMessage().contains("<provider>/<model>"), "the error says how to fix it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointAndTheBridgeMcpCoexistInOneConfig(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash",
|
||||
"http://127.0.0.1:8000", "http://127.0.0.1:8766/mcp")).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("remote", json.path("mcp").path("bridge").path("type").asText(),
|
||||
"pinning an endpoint must not drop the bridge MCP mount");
|
||||
assertFalse(json.path("provider").path("local-vllm").isMissingNode(),
|
||||
"and the provider block is still declared alongside it");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the reply charter survives too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noBaseUrlDeclaresNoProviderSoTheDefaultGatewayIsUsed(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("opencode/some-free-model", "http://127.0.0.1:8766/mcp", null))
|
||||
.spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertTrue(json.path("provider").isMissingNode(),
|
||||
"without a baseUrl opencode resolves its own provider as before");
|
||||
}
|
||||
}
|
||||
@@ -1,90 +0,0 @@
|
||||
package dev.ltms.bridged.member;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.FileTime;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* {@link OpenCodeSessionDiscovery} matches an opencode session record by the worker's cwd (its
|
||||
* {@code directory}) against opencode's on-disk storage. These tests populate a TEMP storage root
|
||||
* themselves — never the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
class OpenCodeSessionDiscoveryTest {
|
||||
|
||||
/**
|
||||
* Write a session record {@code {"id":..., "directory":...}} under
|
||||
* {@code <root>/session/<projectID>/<fileName>} and stamp it with a known last-modified time,
|
||||
* so "most recently modified wins" is deterministic. Static so the launcher test can reuse it.
|
||||
*/
|
||||
static void writeRecord(Path root, String projectId, String fileName, String id,
|
||||
String directory, long lastModifiedEpochMillis) throws Exception {
|
||||
Path dir = root.resolve("session").resolve(projectId);
|
||||
Files.createDirectories(dir);
|
||||
Path file = dir.resolve(fileName);
|
||||
Files.writeString(file, "{\"id\":\"" + id + "\",\"directory\":\"" + directory
|
||||
+ "\",\"projectID\":\"" + projectId + "\",\"version\":\"1.1.31\"}");
|
||||
Files.setLastModifiedTime(file, FileTime.fromMillis(lastModifiedEpochMillis));
|
||||
}
|
||||
|
||||
@Test
|
||||
void findsTheRecordWhoseDirectoryEqualsTheCwd(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "ses_b.json", "ses_bbb", "/w/b", 2000L);
|
||||
|
||||
assertEquals("ses_bbb", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/b"),
|
||||
"the record whose directory equals the cwd is the one found");
|
||||
assertEquals("ses_aaa", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingDirectoryYieldsNullRatherThanAMismatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/other"),
|
||||
"no record for this cwd yet → null, not a wrong session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void prefersTheMostRecentlyModifiedRecordWhenSeveralMatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "old.json", "ses_old", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "new.json", "ses_new", "/w/a", 5000L);
|
||||
|
||||
assertEquals("ses_new", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"the freshest record for the cwd wins");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingOrEmptyStorageRootYieldsNullWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
// Missing: no session dir at all under the root.
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
|
||||
// Present but empty: a session dir with nothing in it produces no match, not a throw.
|
||||
Path emptyRoot = root.resolve("empty");
|
||||
Files.createDirectories(emptyRoot.resolve("session"));
|
||||
assertNull(new OpenCodeSessionDiscovery(emptyRoot).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankOrNullDirectoryYieldsNull(@TempDir Path root) {
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertNull(discovery.sessionIdForDirectory(null));
|
||||
assertNull(discovery.sessionIdForDirectory(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMalformedRecordIsSkippedRatherThanFatal(@TempDir Path root) throws Exception {
|
||||
// A record that fails to parse must not abort the scan of its siblings.
|
||||
Path dir = root.resolve("session").resolve("p1");
|
||||
Files.createDirectories(dir);
|
||||
Files.writeString(dir.resolve("broken.json"), "{not valid json");
|
||||
writeRecord(root, "p1", "good.json", "ses_good", "/w/a", 1000L);
|
||||
|
||||
assertEquals("ses_good", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"an unreadable record is skipped; a later valid one still matches");
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,625 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-525 acceptance test for tool-surface isolation. This is one of the few tests that drives real
|
||||
* {@code git} — the behaviour under test is precisely what {@link GitWorktrees} does to a checkout,
|
||||
* so a fake would assert nothing. Everything happens inside a {@link TempDir} throwaway repo.
|
||||
*/
|
||||
class GitWorktreesTest {
|
||||
|
||||
/** A project MCP config with servers in it — what this repo actually commits. */
|
||||
private static final String WITH_SERVERS = """
|
||||
{
|
||||
"mcpServers": {
|
||||
"jetbrains": { "type": "sse", "url": "http://localhost:64342/sse" }
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** An opencode config carrying a {@code {file:.secrets/...}} reference — CB-543's crash repro. */
|
||||
private static final String OPENCODE_WITH_FILE_REF = """
|
||||
{
|
||||
"env": {
|
||||
"CONTEXT7_TOKEN": "{file:.secrets/context7-token}"
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** A non-empty autoenv file — the form that would prompt for authorization in a worktree. */
|
||||
private static final String AUTOENV_WITH_DIRECTIVE = "export HELLO=world\n";
|
||||
|
||||
private static Path initRepo(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
git(dir, "init", "-q", "-b", "main");
|
||||
git(dir, "config", "user.email", "test@example.invalid");
|
||||
git(dir, "config", "user.name", "Test");
|
||||
Files.writeString(dir.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||
git(dir, "add", ".mcp.json", "README.md");
|
||||
git(dir, "commit", "-q", "-m", "seed");
|
||||
return dir;
|
||||
}
|
||||
|
||||
private static void git(Path cwd, String... args) throws Exception {
|
||||
List<String> cmd = new java.util.ArrayList<>(List.of("git"));
|
||||
cmd.addAll(List.of(args));
|
||||
Process p = new ProcessBuilder(cmd).directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git timed out: " + String.join(" ", cmd));
|
||||
assertEquals(0, p.exitValue(), "git " + String.join(" ", args) + " failed:\n" + out);
|
||||
}
|
||||
|
||||
/** Pending changes to {@code file} in {@code cwd}, empty when git considers it unmodified. */
|
||||
private static String status(Path cwd, String file) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "status", "--porcelain", "--", file)
|
||||
.directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git status timed out");
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Every pending change in {@code cwd} — the whole-tree porcelain status, unlike {@link #status}. */
|
||||
private static String fullStatus(Path cwd) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "status", "--porcelain")
|
||||
.directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git status timed out");
|
||||
return out;
|
||||
}
|
||||
|
||||
private static String revParse(Path cwd, String ref) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "rev-parse", ref)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git rev-parse timed out");
|
||||
assertEquals(0, p.exitValue(), "git rev-parse " + ref + " failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The recursive file list of a commit's tree — used to check what a snapshot actually committed. */
|
||||
private static String lsTree(Path cwd, String ref) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "ls-tree", "-r", "--name-only", ref)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git ls-tree timed out");
|
||||
assertEquals(0, p.exitValue(), "git ls-tree " + ref + " failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The set of paths in {@code git diff --name-only from..to} — used to check exactly what a
|
||||
* snapshot's tree changed relative to its parent, the same shape {@code git status --porcelain}
|
||||
* reports for the worktree it was taken from. */
|
||||
private static Set<String> diffNameOnly(Path cwd, String from, String to) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "diff", "--name-only", from, to)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git diff timed out");
|
||||
assertEquals(0, p.exitValue(), "git diff " + from + ".." + to + " failed:\n" + out);
|
||||
Set<String> paths = new HashSet<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (!line.isBlank()) {
|
||||
paths.add(line.trim());
|
||||
}
|
||||
}
|
||||
return paths;
|
||||
}
|
||||
|
||||
/** The set of paths a {@code git status --porcelain} listing names, stripping the two-char status
|
||||
* code prefix each line carries. */
|
||||
private static Set<String> porcelainPaths(String porcelain) {
|
||||
Set<String> paths = new HashSet<>();
|
||||
for (String line : porcelain.split("\\R")) {
|
||||
if (!line.isBlank()) {
|
||||
paths.add(line.substring(3).trim());
|
||||
}
|
||||
}
|
||||
return paths;
|
||||
}
|
||||
|
||||
private static String forEachRef(Path cwd, String pattern) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "for-each-ref", pattern)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git for-each-ref timed out");
|
||||
assertEquals(0, p.exitValue(), "git for-each-ref " + pattern + " failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Write {@code content} as a blob into the object database; returns its sha. */
|
||||
private static String blobOf(Path cwd, String content) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "hash-object", "-w", "--stdin")
|
||||
.redirectErrorStream(true).start();
|
||||
p.getOutputStream().write(content.getBytes(StandardCharsets.UTF_8));
|
||||
p.getOutputStream().close();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git hash-object timed out");
|
||||
assertEquals(0, p.exitValue(), "git hash-object failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Build a single-file tree object from {@code blob}; returns the tree's sha. */
|
||||
private static String treeOf(Path cwd, String path, String blob) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "mktree")
|
||||
.redirectErrorStream(true).start();
|
||||
p.getOutputStream().write(("100644 blob " + blob + "\t" + path + "\n").getBytes(StandardCharsets.UTF_8));
|
||||
p.getOutputStream().close();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git mktree timed out");
|
||||
assertEquals(0, p.exitValue(), "git mktree failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** {@code git commit-tree} rooted at {@code tree} with a chosen committer date; returns the sha. */
|
||||
private static String commitTree(Path cwd, String tree, String parent, String committerDate,
|
||||
String message) throws Exception {
|
||||
ProcessBuilder pb = new ProcessBuilder("git", "-C", cwd.toString(), "commit-tree",
|
||||
tree, "-p", parent, "-m", message);
|
||||
pb.environment().put("GIT_COMMITTER_DATE", committerDate);
|
||||
Process p = pb.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git commit-tree timed out");
|
||||
assertEquals(0, p.exitValue(), "git commit-tree failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** {@code git update-ref <ref> <sha>} — create the snapshot ref directly. */
|
||||
private static void updateRef(Path cwd, String ref, String sha) throws Exception {
|
||||
git(cwd, "update-ref", ref, sha);
|
||||
}
|
||||
|
||||
/** True when {@code ref} exists in the repo (for-each-ref on a missing ref is empty, not an error). */
|
||||
private static boolean refExists(Path cwd, String ref) throws Exception {
|
||||
return !forEachRef(cwd, ref).trim().isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* The heart of CB-525: a provisioned worktree must not inherit the primary's MCP servers. Without
|
||||
* the isolation step the checked-out {@code .mcp.json} carries them in, and a worker navigating
|
||||
* through the primary's IDE servers edits the primary's tree while building its own.
|
||||
*/
|
||||
@Test
|
||||
void aProvisionedWorktreeInheritsNoMcpServers(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-a", "HEAD");
|
||||
|
||||
Path mcp = Path.of(wt).resolve(".mcp.json");
|
||||
assertTrue(Files.exists(mcp), ".mcp.json must still exist — present and explicitly empty");
|
||||
String body = Files.readString(mcp);
|
||||
assertFalse(body.contains("jetbrains"), "worktree inherited the primary's MCP servers:\n" + body);
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
"expected an explicitly empty server map, got:\n" + body);
|
||||
}
|
||||
|
||||
/** Neutralizing must not look like work in progress, or a worker would commit it into its PR. */
|
||||
@Test
|
||||
void theNeutralizedConfigIsNotAPendingLocalModification(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-b", "HEAD");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"),
|
||||
"the neutralized .mcp.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** Isolation is the worktree's business only; the primary's own checkout must be untouched. */
|
||||
@Test
|
||||
void thePrimaryCheckoutIsLeftAlone(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-525-c", "HEAD");
|
||||
|
||||
assertEquals(WITH_SERVERS, Files.readString(repo.resolve(".mcp.json")),
|
||||
"the primary's .mcp.json was rewritten — isolation reached out of the worktree");
|
||||
}
|
||||
|
||||
/** A repo that commits no {@code .mcp.json} still gets one, so nothing can be inherited later. */
|
||||
@Test
|
||||
void aRepoWithoutAnMcpConfigStillGetsANeutralOne(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-d", "HEAD");
|
||||
|
||||
// Untracked is the normal case here, so the --skip-worktree branch must be skipped rather
|
||||
// than run and fail: `update-index --skip-worktree` on an unknown path exits non-zero.
|
||||
String body = Files.readString(Path.of(wt).resolve(".mcp.json"));
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"), body);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-543's repro: a tracked {@code opencode.json} carries a {@code {file:.secrets/...}} reference
|
||||
* to a gitignored secret that never reaches a worktree, and opencode refuses to start on it. The
|
||||
* worktree's copy must be neutralized and hidden like {@code .mcp.json}.
|
||||
*/
|
||||
@Test
|
||||
void aTrackedOpencodeConfigIsNeutralizedAndHidden(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-a", "HEAD");
|
||||
|
||||
String body = Files.readString(Path.of(wt).resolve("opencode.json"));
|
||||
assertFalse(body.contains(".secrets"),
|
||||
"worktree kept a dangling {file:...} secret reference:\n" + body);
|
||||
assertEquals("{}", body.replaceAll("\\s+", ""),
|
||||
"expected an empty JSON object stub, got:\n" + body);
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"),
|
||||
"the neutralized opencode.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** A config the repo does not carry must be skipped — no stub invented, provisioning still succeeds. */
|
||||
@Test
|
||||
void anAbsentConfigIsSkippedWithoutError(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo")); // only .mcp.json + README are committed
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-b", "HEAD");
|
||||
|
||||
assertFalse(Files.exists(Path.of(wt).resolve("opencode.json")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
assertFalse(Files.exists(Path.of(wt).resolve(".autoenv")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
// .mcp.json's long-standing create-always behaviour must be unchanged.
|
||||
assertTrue(Files.exists(Path.of(wt).resolve(".mcp.json")), ".mcp.json stub was dropped");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-576. {@code hasUncommitted} must treat a freshly-provisioned worktree as clean, but a
|
||||
* worktree holding a brand-new, never-added file as dirty. The untracked-file-only shape is
|
||||
* exactly the work lost in the incident — a worker's draft that compiled but was never
|
||||
* committed because it stopped to ask its lead a question.
|
||||
*/
|
||||
@Test
|
||||
void anUntrackedOnlyWorktreeCountsAsDirty(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String wt = gitWorktrees.add(repo.toString(), "cb-576-u", "HEAD");
|
||||
|
||||
assertFalse(gitWorktrees.hasUncommitted(wt),
|
||||
"a freshly provisioned worktree must read as clean");
|
||||
|
||||
Files.writeString(Path.of(wt).resolve("brand-new.txt"), "draft that was never added\n");
|
||||
|
||||
assertTrue(gitWorktrees.hasUncommitted(wt),
|
||||
"an untracked-only file must count as dirty");
|
||||
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
assertTrue(gitWorktrees.hasUncommitted(wt),
|
||||
"a tracked modification must also count as dirty");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-576 review. {@code hasUncommitted} must tolerate a missing worktree exactly like
|
||||
* {@code remove}: an already-gone directory holds no work to lose, and throwing here would
|
||||
* break teardown — SessionManager.release() calls it before stopping the pane, so an
|
||||
* exception would orphan a live pane and skip the release notification (CB-516).
|
||||
*/
|
||||
@Test
|
||||
void hasUncommittedOnAMissingWorktreeReturnsFalseWithoutThrowing(@TempDir Path tmp) {
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String gone = tmp.resolve("wts").resolve("does-not-exist").toString();
|
||||
|
||||
assertFalse(gitWorktrees.hasUncommitted(gone),
|
||||
"a missing worktree is reported clean, not an error");
|
||||
}
|
||||
|
||||
/** All three protected configs are covered: each one present in a worktree is neutralized and hidden. */
|
||||
@Test
|
||||
void allThreeConfigsAreNeutralizedWhenPresent(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve(".autoenv"), AUTOENV_WITH_DIRECTIVE);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", ".autoenv", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-c", "HEAD");
|
||||
|
||||
assertTrue(Files.readString(Path.of(wt).resolve(".mcp.json"))
|
||||
.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
".mcp.json was not neutralized");
|
||||
assertEquals("{}", Files.readString(Path.of(wt).resolve("opencode.json")).replaceAll("\\s+", ""),
|
||||
"opencode.json was not neutralized");
|
||||
assertEquals("", Files.readString(Path.of(wt).resolve(".autoenv")),
|
||||
".autoenv was not neutralized");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"), ".mcp.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"), "opencode.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), ".autoenv"), ".autoenv still shows as modified");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, acceptance criterion 1. A dirty worktree — a tracked edit plus a brand-new
|
||||
* untracked file, exactly the shape lost in CB-576 — must land in {@code refs/wip/<branch>}'s
|
||||
* tree, and that ref must live outside {@code refs/heads} so it never shows up in
|
||||
* {@code git branch} or gets swept by a branch cleanup.
|
||||
*/
|
||||
@Test
|
||||
void dirtySnapshotCreatesARefWhoseTreeContainsUntrackedAndTrackedChanges(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-578-c-a";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertTrue(ref.isPresent(), "a dirty worktree snapshot returns a commit sha");
|
||||
String tree = lsTree(repo, "refs/wip/" + branch);
|
||||
assertTrue(tree.contains("untracked.txt"), "snapshot tree must include the untracked file:\n" + tree);
|
||||
assertTrue(tree.contains("README.md"), "snapshot tree must include the tracked edit:\n" + tree);
|
||||
String heads = forEachRef(repo, "refs/heads");
|
||||
assertFalse(heads.contains("refs/wip/"), "the snapshot ref must not live under refs/heads:\n" + heads);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, acceptance criterion 2. Building the commit through a temporary
|
||||
* {@code GIT_INDEX_FILE} must leave the worker's own index, working tree, and HEAD exactly as
|
||||
* they were — the worker may still be mid-write, and staging into the real index would corrupt
|
||||
* that.
|
||||
*/
|
||||
@Test
|
||||
void snapshotDoesNotTouchTheWorkersOwnIndexWorkingTreeOrHead(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-578-c-b";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
String headBefore = revParse(Path.of(wt), "HEAD");
|
||||
String statusBefore = fullStatus(Path.of(wt));
|
||||
|
||||
gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertEquals(headBefore, revParse(Path.of(wt), "HEAD"), "snapshot must not move HEAD");
|
||||
assertEquals(statusBefore, fullStatus(Path.of(wt)),
|
||||
"snapshot must not change the worker's own index or working tree status");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, acceptance criterion 3. {@code add -A} (never {@code -f}) respects
|
||||
* {@code .gitignore}, and that is the only thing keeping a gitignored file (secrets, local
|
||||
* config) out of a snapshot commit — an untracked file that is NOT ignored must still be
|
||||
* included, so this isn't just "untracked files are dropped".
|
||||
*/
|
||||
@Test
|
||||
void gitignoredFileIsExcludedButOtherUntrackedFilesAreNot(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".gitignore"), ".env\n");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".gitignore", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-578-c-c";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve(".env"), "SECRET=shh\n");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertTrue(ref.isPresent());
|
||||
String tree = lsTree(repo, "refs/wip/" + branch);
|
||||
assertFalse(tree.contains(".env"), "a gitignored file must never enter the snapshot:\n" + tree);
|
||||
assertTrue(tree.contains("untracked.txt"),
|
||||
"a non-ignored untracked file must still be included:\n" + tree);
|
||||
}
|
||||
|
||||
/** CB-578 stage C. A clean worktree still produces a valid, if tree-identical, commit — the caller
|
||||
* (SessionManager) is the one that decides not to call this on a clean worktree. */
|
||||
@Test
|
||||
void snapshotOfAMissingWorktreeReturnsEmptyWithoutThrowing(@TempDir Path tmp) {
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String gone = tmp.resolve("wts").resolve("does-not-exist").toString();
|
||||
|
||||
assertTrue(gitWorktrees.snapshot(gone, "some-branch", "msg").isEmpty(),
|
||||
"a missing worktree has nothing to snapshot, and must not throw");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-587, acceptance criteria 1 and 2. A file with a REAL {@code --skip-worktree} bit set in the
|
||||
* worktree's own index must never enter the snapshot, even though its on-disk content has locally
|
||||
* diverged from what is committed — that is exactly the divergence {@code --skip-worktree} exists
|
||||
* to hide from {@code git status}, and a snapshot built from a fresh empty temp index (the bug)
|
||||
* stages that local content anyway because the fresh index carries none of the real index's flags.
|
||||
* The snapshot's diff against its parent must list exactly what {@code git status --porcelain}
|
||||
* reports for the worktree — no more, no less.
|
||||
*/
|
||||
@Test
|
||||
void dirtySnapshotHonoursARealSkipWorktreeBit(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve("protected.cfg"), "committed-value\n");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", "protected.cfg", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-587-a";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
|
||||
git(Path.of(wt), "update-index", "--skip-worktree", "protected.cfg");
|
||||
Files.writeString(Path.of(wt).resolve("protected.cfg"), "locally-diverged-never-commit\n");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
|
||||
String porcelain = fullStatus(Path.of(wt));
|
||||
assertFalse(porcelain.contains("protected.cfg"),
|
||||
"test setup invalid — protected.cfg must not show in git status once skip-worktree is set:\n"
|
||||
+ porcelain);
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertTrue(ref.isPresent(), "a dirty worktree snapshot returns a commit sha");
|
||||
Set<String> diffPaths = diffNameOnly(repo, "HEAD", "refs/wip/" + branch);
|
||||
assertFalse(diffPaths.contains("protected.cfg"),
|
||||
"a --skip-worktree file's local drift leaked into the snapshot:\n" + diffPaths);
|
||||
assertEquals(porcelainPaths(porcelain), diffPaths,
|
||||
"snapshot diff must list exactly what git status --porcelain reports, no more, no less");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-587, acceptance criterion 6. If the worktree's real index cannot be resolved/read, snapshot
|
||||
* must fail loudly (throw) rather than silently falling back to an empty temp index and producing
|
||||
* a wrong snapshot. SessionManager's caller already catches and WARNs on any exception here — this
|
||||
* only needs to confirm the failure is not swallowed inside snapshot() itself.
|
||||
*/
|
||||
@Test
|
||||
void snapshotThrowsWhenTheWorktreesRealIndexCannotBeResolved(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-587-b";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
|
||||
// Break git's ability to resolve the worktree's real index by removing the linked worktree's
|
||||
// `.git` file (which normally points at the main repo's worktrees/<name>/ directory).
|
||||
Files.delete(Path.of(wt).resolve(".git"));
|
||||
|
||||
assertThrows(WorktreeException.class, () -> gitWorktrees.snapshot(wt, branch, "test snapshot"),
|
||||
"an unresolvable real index must fail loudly, not silently snapshot from an empty index");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, criterion 2. A snapshot whose content is NOT reachable from {@code main} is the last
|
||||
* copy of a worker's work, and must never be deleted automatically — even when it is old and
|
||||
* even when the caller passes a zero age floor. Uses the real snapshot path on a dirty worktree,
|
||||
* so the unreachable tree is exactly the shape CB-576/CB-578 stage C exist to protect.
|
||||
*/
|
||||
@Test
|
||||
void anUnreachableSnapshotIsNeverPruned(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-586-unreachable";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve("worker-draft.txt"), "work that exists nowhere else\n");
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "snapshot with unreachable content");
|
||||
assertTrue(ref.isPresent());
|
||||
|
||||
// Age floor 0 makes age a non-issue: only reachability can save it — and it must.
|
||||
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), 0),
|
||||
"the unreachable snapshot is the last copy and must not be pruned");
|
||||
assertTrue(refExists(repo, "refs/wip/" + branch),
|
||||
"an unreachable snapshot must survive the sweep");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, criterion 1 (the reachable half). A snapshot whose tree content IS already reachable
|
||||
* from {@code main} and which is older than the age floor is pure duplication — the work is
|
||||
* recovered — so it must be pruned.
|
||||
*/
|
||||
@Test
|
||||
void aReachableSnapshotOlderThanTheFloorIsPruned(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
// A snapshot whose tree is exactly main's current tree: fully reachable from main.
|
||||
String mainTree = revParse(repo, "main^{tree}");
|
||||
String old = commitTree(repo, mainTree, revParse(repo, "HEAD"), "2020-01-01T00:00:00", "snapshot");
|
||||
updateRef(repo, "refs/wip/recovered", old);
|
||||
|
||||
assertEquals(1, gitWorktrees.pruneWipRefs(repo.toString(), TimeUnit.HOURS.toMillis(24)),
|
||||
"an old, main-reachable snapshot must be pruned");
|
||||
assertFalse(refExists(repo, "refs/wip/recovered"),
|
||||
"the reachable snapshot's ref must be gone after the sweep");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, the age floor. A snapshot whose content IS reachable from {@code main} but which is
|
||||
* younger than the age floor must not be swept — a lead may still be looking at it.
|
||||
*/
|
||||
@Test
|
||||
void aReachableButRecentSnapshotIsNotPruned(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
// Reachable from main, but committed "now" — a fresh snapshot. The 24h floor must protect it.
|
||||
String mainTree = revParse(repo, "main^{tree}");
|
||||
String fresh = commitTree(repo, mainTree, revParse(repo, "HEAD"),
|
||||
"2038-01-01T00:00:00", "snapshot just taken");
|
||||
updateRef(repo, "refs/wip/fresh", fresh);
|
||||
|
||||
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), TimeUnit.HOURS.toMillis(24)),
|
||||
"a recent snapshot must be kept even when reachable");
|
||||
assertTrue(refExists(repo, "refs/wip/fresh"),
|
||||
"the recent reachable snapshot must survive the sweep");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, criterion 5. A fleet that has never snapshotted anything has no {@code refs/wip/*},
|
||||
* so a sweep is a no-op and the census reports none — identical to before CB-586 existed.
|
||||
*/
|
||||
@Test
|
||||
void aFleetWithNoSnapshotsPrunesNothingAndReportsNothing(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), 0),
|
||||
"no snapshot refs means nothing to prune");
|
||||
Worktrees.WipRefStats stats = gitWorktrees.wipRefs(repo.toString());
|
||||
assertEquals(0, stats.count(), "a never-snapshotted fleet has zero refs/wip refs");
|
||||
assertEquals(0L, stats.costBytes(), "a never-snapshotted fleet costs zero bytes");
|
||||
}
|
||||
|
||||
/** CB-586, criterion 4: the census reports how many refs exist and roughly what they cost. */
|
||||
@Test
|
||||
void wipRefsReportsCountAndCost(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
String blob = blobOf(repo, "a recoverable snapshot's worth of content");
|
||||
String tree = treeOf(repo, "snapshot.txt", blob);
|
||||
updateRef(repo, "refs/wip/one", commitTree(repo, tree, revParse(repo, "HEAD"),
|
||||
"2020-01-01T00:00:00", "snapshot"));
|
||||
updateRef(repo, "refs/wip/two", commitTree(repo, tree, revParse(repo, "HEAD"),
|
||||
"2020-01-02T00:00:00", "snapshot"));
|
||||
|
||||
Worktrees.WipRefStats stats = gitWorktrees.wipRefs(repo.toString());
|
||||
assertEquals(2, stats.count(), "two snapshot refs are reported");
|
||||
assertTrue(stats.costBytes() > 0, "the cost of the snapshots is a positive byte count");
|
||||
}
|
||||
}
|
||||
@@ -1,930 +0,0 @@
|
||||
package dev.ltms.bridged.session;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.guard.SubscriptionGuard;
|
||||
import dev.ltms.bridged.herdr.AgentControl;
|
||||
import dev.ltms.bridged.herdr.FakeHerdr;
|
||||
import dev.ltms.bridged.herdr.WorkspaceControl;
|
||||
import dev.ltms.bridged.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.bridged.msg.TestTurnTokens;
|
||||
import dev.ltms.bridged.peer.Capability;
|
||||
import dev.ltms.bridged.peer.CharterReceipt;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.bridged.peer.PeerHandle;
|
||||
import dev.ltms.bridged.peer.PeerLauncher;
|
||||
import dev.ltms.bridged.peer.PeerUnreachableException;
|
||||
import dev.ltms.bridged.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-301 / CB-303 acceptance tests for the authoritative session registry, one-shot lifecycle FSM,
|
||||
* and configurable lifecycle limits (idle TTL, context cap, drain).
|
||||
* No live herdr — everything runs against the same {@link FakeHerdr} the rest of the project uses.
|
||||
*/
|
||||
class SessionManagerTest {
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock) {
|
||||
return sessionManager(herdr, clock, 0);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, Worktrees worktrees) {
|
||||
return sessionManager(herdr, worktrees, System::nanoTime);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, Worktrees worktrees, LongSupplier clock) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers, worktrees, clock);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-581: a {@link Worktrees} test double whose {@code hasUncommitted} and {@code remove} can
|
||||
* be told to throw, so {@link SessionManager#release} can be exercised against exactly the
|
||||
* failure {@code GitWorktrees} produces when {@code git status}/{@code git worktree remove}
|
||||
* exits non-zero.
|
||||
*/
|
||||
private static final class RecordingWorktrees implements Worktrees {
|
||||
private final List<String> removeCalls = new java.util.ArrayList<>();
|
||||
private final List<String> snapshotCalls = new java.util.ArrayList<>();
|
||||
private final java.util.Set<String> failRemoveFor = new java.util.HashSet<>();
|
||||
private volatile boolean dirty = false;
|
||||
private volatile RuntimeException hasUncommittedFailure;
|
||||
private volatile RuntimeException snapshotFailure;
|
||||
private final java.util.concurrent.atomic.AtomicLong snapshotSeq = new java.util.concurrent.atomic.AtomicLong();
|
||||
|
||||
RecordingWorktrees dirty(boolean dirty) {
|
||||
this.dirty = dirty;
|
||||
return this;
|
||||
}
|
||||
|
||||
RecordingWorktrees failHasUncommittedWith(RuntimeException e) {
|
||||
this.hasUncommittedFailure = e;
|
||||
return this;
|
||||
}
|
||||
|
||||
RecordingWorktrees failRemoveFor(String worktreePath) {
|
||||
failRemoveFor.add(worktreePath);
|
||||
return this;
|
||||
}
|
||||
|
||||
RecordingWorktrees failSnapshotWith(RuntimeException e) {
|
||||
this.snapshotFailure = e;
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
return "/wt/" + branch.replace('/', '_');
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
if (failRemoveFor.contains(worktreePath)) {
|
||||
throw new WorktreeException("simulated remove failure for " + worktreePath);
|
||||
}
|
||||
removeCalls.add(worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
if (hasUncommittedFailure != null) {
|
||||
throw hasUncommittedFailure;
|
||||
}
|
||||
return dirty;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
return "/repo";
|
||||
}
|
||||
|
||||
@Override
|
||||
public java.util.Optional<String> snapshot(String worktreePath, String branch, String message) {
|
||||
snapshotCalls.add(worktreePath);
|
||||
if (snapshotFailure != null) {
|
||||
throw snapshotFailure;
|
||||
}
|
||||
return java.util.Optional.of("wip" + snapshotSeq.incrementAndGet());
|
||||
}
|
||||
|
||||
@Override
|
||||
public WipRefStats wipRefs(String repoRoot) {
|
||||
return new WipRefStats(0, 0L);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
List<String> removeCalls() {
|
||||
return List.copyOf(removeCalls);
|
||||
}
|
||||
|
||||
List<String> snapshotCalls() {
|
||||
return List.copyOf(snapshotCalls);
|
||||
}
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap) {
|
||||
return sessionManager(herdr, clock, contextCap, false);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap,
|
||||
boolean clearAfterTurn) {
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers, new GitWorktrees(), clock, contextCap, clearAfterTurn);
|
||||
}
|
||||
|
||||
@Test
|
||||
void primaryContactWithNoTerminalIsNotAReadinessSignal() {
|
||||
// The MCP context extractor calls presence.markPresent(p.terminal()) on EVERY request,
|
||||
// and the primary's terminal is null — the presence bridge must treat that as a no-op,
|
||||
// not feed it into the READY transition (which NPEd on the first real primary contact).
|
||||
SessionManager sessions = sessionManager(new FakeHerdr());
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null));
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireRegistersSpawningSessionWithDistinctPaneId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession a = sessions.acquire("ltms-local", "/work/a", "/caller/a", "term_primary");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/work/b", "/caller/b", "term_primary");
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, a.state(), "fresh session starts spawning");
|
||||
assertEquals("ltms-local", a.profile());
|
||||
assertEquals("/work/a", a.cwd(), "explicit requested cwd is recorded");
|
||||
assertEquals("term_primary", a.ownerTerminal());
|
||||
assertTrue(a.spawnedAtNanos() > 0);
|
||||
assertNotNull(a.paneId());
|
||||
assertNotNull(a.terminalId());
|
||||
|
||||
assertNotEquals(a.paneId(), b.paneId(), "no pane reuse");
|
||||
assertNotEquals(a.terminalId(), b.terminalId(), "no terminal reuse");
|
||||
assertEquals(2, sessions.roster().size(), "both sessions are registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void rosterViewExposesTheCharterReceiptButNeverTheCharterText() {
|
||||
// The roster (bridge_list and GET /members both render through rosterView) must let a lead
|
||||
// see which charter a member got, without ever carrying the charter prose itself (CB-571).
|
||||
MemberSession s = new MemberSession("p1", "term1", "prof", MemberRole.DEV, "/cwd", null,
|
||||
0, 0, 0, MemberSession.State.READY, null, null,
|
||||
CharterReceipt.compose(MemberRole.DEV, "prof", "role charter", "role charter\n\nreply"), null);
|
||||
|
||||
Map<String, Object> view = SessionManager.rosterView(s, null);
|
||||
|
||||
assertEquals("fleet.charters.dev", view.get("charterSource"),
|
||||
"the config key that supplied the role charter is reported");
|
||||
assertEquals(CharterReceipt.digestOf("role charter\n\nreply"), view.get("charterSha256"),
|
||||
"the digest of the exact composed charter bytes is reported");
|
||||
assertFalse(view.values().toString().contains("role charter"),
|
||||
"the roster row must not embed the charter text itself");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullTerminalFromThePrimaryIsANoOpEvenWithSessionsRegistered() {
|
||||
// The primary resolves to a Principal with no terminal, and BridgeMcp's context extractor
|
||||
// forwards that null into markPresent on EVERY MCP call. It only reached the registry scan
|
||||
// once a session existed, so this NPE'd the primary's second spawn while the first passed.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null),
|
||||
"the primary's null terminal must not blow up an unrelated tool call");
|
||||
assertDoesNotThrow(() -> sessions.onDelivered(null, TestTurnTokens.inert(null)));
|
||||
assertDoesNotThrow(() -> sessions.onTurnComplete(null));
|
||||
assertDoesNotThrow(() -> sessions.onTurnFailed(null));
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"and must not transition any registered session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void presenceMovesSpawningToReadyAndDeliveredTurnMovesToDone() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
assertEquals(MemberSession.State.READY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"MCP presence moves SPAWNING → READY");
|
||||
assertTrue(sessions.asPresence().isPresent(terminal), "presence is also recorded");
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
assertEquals(MemberSession.State.BUSY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"delivery moves READY → BUSY");
|
||||
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"turn completion moves BUSY → DONE");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseTearsDownWorkerAndRemovesFromRosterAndIsIdempotent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
String paneId = session.paneId();
|
||||
|
||||
sessions.release(paneId);
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release tears the worker pane down");
|
||||
assertTrue(sessions.get(paneId).isEmpty(), "released session is no longer retrievable");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session is no longer in the roster");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.release(paneId), "a second release is harmless");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedMovesSessionToFailed() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
|
||||
sessions.onTurnFailed(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.FAILED, updated.state(), "turn failure moves to FAILED");
|
||||
assertTrue(sessions.roster().contains(updated), "FAILED is still in acquired-minus-released roster");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedIsLoggedAtWarnWithThePriorState() {
|
||||
// CB-564: this transition used to be a bare DEBUG "session marked failed" — a symptom with no
|
||||
// cause. A member that can no longer be delegated to must be at least WARN, and should name
|
||||
// what stage it failed at (here: BUSY, i.e. a turn was in flight and never resolved).
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
|
||||
sessions.onTurnFailed(terminal);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no turn-failed WARN logged");
|
||||
assertTrue(warn.contains(terminal), "the log names the member: " + warn);
|
||||
assertTrue(warn.contains("BUSY"), "the log names the stage it failed at: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void rosterReflectsAcquiredMinusReleased() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession a = sessions.acquire("ltms-local", "/a", "/caller", "ownerA");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/b", "/caller", "ownerB");
|
||||
|
||||
assertEquals(2, sessions.roster().size());
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(a.paneId())));
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(b.paneId())));
|
||||
|
||||
sessions.release(a.paneId());
|
||||
|
||||
assertEquals(1, sessions.roster().size());
|
||||
assertEquals(b.paneId(), sessions.roster().getFirst().paneId());
|
||||
}
|
||||
|
||||
// --- CB-303 lifecycle limits ----------------------------------------------------
|
||||
|
||||
@Test
|
||||
void reapIdleDoesNothingWhenNoSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L);
|
||||
|
||||
assertEquals(0, sessions.reapIdle(10));
|
||||
assertTrue(sessions.roster().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 11;
|
||||
assertEquals(1, sessions.reapIdle(10), "READY session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "reaped session is removed from registry");
|
||||
assertTrue(herdr.called("pane.close"), "reaped session tears the pane down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionWithinIdleTtlSurvives() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 5;
|
||||
assertEquals(0, sessions.reapIdle(10), "READY session within TTL is not reaped");
|
||||
assertEquals(MemberSession.State.READY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"READY session survives");
|
||||
}
|
||||
|
||||
@Test
|
||||
void busySessionPastIdleTtlIsNotReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
|
||||
clock[0] = 100;
|
||||
assertEquals(0, sessions.reapIdle(10), "BUSY session past TTL is never reaped");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"BUSY session remains");
|
||||
}
|
||||
|
||||
@Test
|
||||
void doneSessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
clock[0] = 21;
|
||||
assertEquals(1, sessions.reapIdle(20), "DONE session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "DONE session is removed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIdleReturnsCorrectCountAndSkipsBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "owner1");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "owner2");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId(), TestTurnTokens.inert(busy.terminalId()));
|
||||
|
||||
clock[0] = 50;
|
||||
assertEquals(1, sessions.reapIdle(30), "only READY past TTL is reaped");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "READY session is gone");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(busy.paneId()).orElseThrow().state(),
|
||||
"BUSY session is still registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapDisabledSessionSurvivesMultipleTurns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.DONE, updated.state(), "session finishes second turn");
|
||||
assertEquals(2, updated.turnCount(), "turn count tracks both deliveries");
|
||||
long releaseCloseCount = paneCloseCallsFor(herdr, "w9:pRoot_1"); // the real pane coordinate
|
||||
assertEquals(0, releaseCloseCount, "cap disabled — no forced release of the worker pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapTwoReleasesAfterSecondComplete() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 2);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"first turn completes without release");
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "session released after cap reached");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session leaves roster");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"forced release tears the worker pane down exactly once");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnResetsContextWithoutDoubleCountingTheTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
|
||||
sessions.onDelivered(session.terminalId(), TestTurnTokens.inert(session.terminalId()));
|
||||
assertTrue(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(1, updated.turnCount(), "the reset is housekeeping, not a second delegation");
|
||||
assertEquals(List.of("/clear"), promptTexts(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapReleaseWinsOverClearAfterTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 1, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId(), TestTurnTokens.inert(session.terminalId()));
|
||||
|
||||
assertFalse(sessions.hasPostTurnAction(session.terminalId()),
|
||||
"a session at its cap will be released, not reset for reuse");
|
||||
assertFalse(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty());
|
||||
assertTrue(promptTexts(herdr).isEmpty(), "never send /clear into a worker being torn down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnFalsePreservesCompletionWithoutAControlPrompt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, false);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId(), TestTurnTokens.inert(session.terminalId()));
|
||||
|
||||
sessions.onTurnComplete(session.terminalId());
|
||||
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state());
|
||||
assertTrue(promptTexts(herdr).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllReleasesBusyAndReadySessionsAndWaitsForBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "ownerR");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "ownerB");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId(), TestTurnTokens.inert(busy.terminalId()));
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(sessions.roster().isEmpty(), "drain clears the roster");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "ready session is released");
|
||||
assertTrue(sessions.get(busy.paneId()).isEmpty(), "busy session is released after timeout");
|
||||
// ready is the first spawn → pane w9:pRoot_1, busy the second → w9:pRoot_2 (FakeHerdr order).
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"ready worker pane is torn down");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_2"),
|
||||
"busy worker pane is torn down");
|
||||
}
|
||||
|
||||
private static long paneCloseCallsFor(FakeHerdr herdr, String paneId) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "pane.close".equals(c.method()))
|
||||
.filter(c -> paneId.equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
}
|
||||
|
||||
private static List<String> promptTexts(FakeHerdr herdr) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "agent.prompt".equals(c.method()))
|
||||
.map(c -> String.valueOf(((Map<?, ?>) c.params()).get("text")))
|
||||
.toList();
|
||||
}
|
||||
|
||||
// --- CB-306 spawn-readiness gate: no half-registered session on timeout ----------------
|
||||
|
||||
@Test
|
||||
void acquireThrowsPeerUnreachableWhenGateTimesOutAndRegistersNoSession() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never becomes injectable
|
||||
long[] clock = {0};
|
||||
|
||||
// Gate-enabled launcher (1 ms timeout + no-op sleeper that advances clock past deadline)
|
||||
BridgedConfig.Profile cfg = new BridgedConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
1, () -> clock[0], () -> clock[0] += 10);
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(), () -> 0L, 0);
|
||||
|
||||
assertThrows(PeerUnreachableException.class,
|
||||
() -> sessions.acquire("ltms-local", null, "/caller", "term_primary"),
|
||||
"acquire must throw PeerUnreachableException when spawn times out");
|
||||
|
||||
// No half-registered session — the error happened inside spawn, before
|
||||
// SessionManager could put() anything into the registry.
|
||||
assertTrue(sessions.roster().isEmpty(),
|
||||
"no session is registered when spawn times out (roster empty)");
|
||||
}
|
||||
|
||||
// --- CB-516: release must notify, so a blocked send can be failed --------------------------
|
||||
|
||||
@Test
|
||||
void releaseNotifiesTheListenerWithTheReleasedTerminal() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(detail -> released.add(detail.terminalId()));
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(java.util.List.of(s.terminalId()), released,
|
||||
"every teardown path funnels through release, so one hook must see the terminal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releasingAnUnknownPaneNotifiesNobody() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(detail -> released.add(detail.terminalId()));
|
||||
|
||||
sessions.release("w9:p404"); // idempotent teardown of something already gone
|
||||
|
||||
assertTrue(released.isEmpty(), "no session removed ⇒ no send was waiting on it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aThrowingReleaseListenerDoesNotBlockTheTeardown() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
sessions.onRelease(_ -> {
|
||||
throw new IllegalStateException("listener blew up");
|
||||
});
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||
"a listener failure must never prevent the teardown it is reacting to");
|
||||
assertTrue(sessions.get(s.paneId()).isEmpty(), "and the session is still deregistered");
|
||||
}
|
||||
|
||||
// --- CB-581: a throw inside release() must not orphan the pane or abort reapIdle -----------
|
||||
|
||||
@Test
|
||||
void releasePreservesWorktreeWhenDirtyCheckThrows() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581a", null));
|
||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
try {
|
||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||
"a throwing dirty check must not abort the release");
|
||||
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"the worktree is preserved when its dirty state cannot be determined");
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains(s.worktree()))
|
||||
.findFirst()
|
||||
.orElse("no warn logged naming the worktree");
|
||||
assertTrue(warn.contains(s.paneId()), "the WARN names the pane: " + warn);
|
||||
assertTrue(warn.contains(s.terminalId()), "the WARN names the terminal: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseStillStopsThePaneWhenDirtyCheckThrows() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581b", null));
|
||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"the pane is stopped exactly once even though the dirty check threw");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseStillNotifiesTheListenerWhenDirtyCheckThrows() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(detail -> released.add(detail.terminalId()));
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581c", null));
|
||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(java.util.List.of(s.terminalId()), released,
|
||||
"a blocked caller must still be told the terminal was released, even though the "
|
||||
+ "dirty check threw");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIdleSurvivesOneSessionThatFailsToRelease() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees, () -> clock[0]);
|
||||
MemberSession a = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581d", null));
|
||||
MemberSession b = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581e", null));
|
||||
MemberSession c = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581f", null));
|
||||
sessions.asPresence().markPresent(a.terminalId());
|
||||
sessions.asPresence().markPresent(b.terminalId());
|
||||
sessions.asPresence().markPresent(c.terminalId());
|
||||
// The middle session's worktree removal fails — release() propagates that, so this is the
|
||||
// one call reapIdle's per-session guard must survive without skipping the rest of the pass.
|
||||
worktrees.failRemoveFor(b.worktree());
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
int reaped;
|
||||
try {
|
||||
clock[0] = 100;
|
||||
reaped = sessions.reapIdle(10);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains(b.paneId()))
|
||||
.findFirst()
|
||||
.orElse("no reap-failure WARN logged");
|
||||
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
||||
assertTrue(warn.contains(b.worktree()), "the WARN names the failed session's worktree: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertEquals(2, reaped, "the middle session's failure is logged, not counted as reaped");
|
||||
assertTrue(sessions.get(a.paneId()).isEmpty(), "the first session is still released");
|
||||
assertTrue(sessions.get(c.paneId()).isEmpty(), "the third session is still released");
|
||||
assertTrue(sessions.get(b.paneId()).isEmpty(),
|
||||
"the middle session is still deregistered even though its worktree removal threw");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"), "the first pane is stopped");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_2"),
|
||||
"the middle pane is still stopped even though its worktree removal failed");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_3"), "the third pane is stopped");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unchangedRegressionCleanCompletedReleaseStillRemovesTheWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581g", null));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(List.of(s.worktree()), worktrees.removeCalls(),
|
||||
"COMPLETED release of a clean worktree still removes it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unchangedRegressionDirtyCompletedReleaseStillPreservesTheWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees().dirty(true);
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581h", null));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"COMPLETED release of a dirty worktree still preserves it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unchangedRegressionShutdownDrainStillPreservesTheWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581i", null));
|
||||
sessions.asPresence().markPresent(s.terminalId());
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(worktrees.removeCalls().isEmpty(), "SHUTDOWN drain still preserves the worktree");
|
||||
}
|
||||
|
||||
// ── CB-584: agentSessionId recorded at acquire, and gated by Capability.SESSION_RESUME ─────
|
||||
|
||||
/**
|
||||
* A minimal {@link PeerLauncher} whose capabilities exclude SESSION_RESUME — the "does not
|
||||
* support it" case. {@link #requireResumeCapability} throws before ever reaching {@link #spawn},
|
||||
* so every method beyond {@link #capabilitiesFor} is unreachable in these tests and left unimplemented.
|
||||
*/
|
||||
private static final class NoResumeLauncher implements PeerLauncher {
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
return Set.of(Capability.WORKTREE); // deliberately no SESSION_RESUME
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
return capabilities();
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — the capability check refuses first");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return Set.of("stub-profile");
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return "stub-profile";
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — the capability check refuses first");
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<?> list() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireRecordsTheAgentSessionIdFromTheHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", MemberRole.DEV, null, null, null, null,
|
||||
"my-session-name", null);
|
||||
|
||||
assertNotNull(s.agentSessionId(), "a spawn that asked for session identity gets one back");
|
||||
Map<String, Object> view = SessionManager.rosterView(s, null);
|
||||
assertEquals(s.agentSessionId(), view.get("agentSessionId"),
|
||||
"the roster exposes the same id the session recorded");
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireWithNeitherSessionFieldLeavesAgentSessionIdNull() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, null, null);
|
||||
|
||||
assertNull(s.agentSessionId(), "no identity requested — unchanged from before CB-584");
|
||||
assertFalse(SessionManager.rosterView(s, null).containsKey("agentSessionId"),
|
||||
"a null id is omitted from the roster, like charterSha256 for a receipt-less session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSessionIdResumesTheSameConversationOnASupportingAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", MemberRole.DEV, null, null, null, null,
|
||||
null, "cb-resume-77");
|
||||
|
||||
assertEquals("cb-resume-77", s.agentSessionId(),
|
||||
"a resume adopts the prior id as its own agentSessionId");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSessionIdWithoutAnExplicitProfileIsRefused() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, () ->
|
||||
sessions.acquire(null, MemberRole.DEV, null, null, null, null, null, "cb-resume-1"));
|
||||
|
||||
assertTrue(e.getMessage().contains("explicit profile"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSessionIdOnAnAdapterWithoutTheCapabilityIsRefusedNamingIt() {
|
||||
SessionManager sessions = new SessionManager(new NoResumeLauncher());
|
||||
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, () ->
|
||||
sessions.acquire("stub-profile", MemberRole.DEV, null, null, null, null, null, "cb-resume-1"));
|
||||
|
||||
assertTrue(e.getMessage().contains("SESSION_RESUME"), e.getMessage());
|
||||
assertTrue(e.getMessage().contains("stub-profile"), e.getMessage());
|
||||
}
|
||||
}
|
||||
@@ -1,61 +0,0 @@
|
||||
# CB-504 — systemd unit for bridged (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.bridged.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
#
|
||||
# Install (user service — bridged drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/bridged.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now bridged
|
||||
# journalctl --user -u bridged -f
|
||||
|
||||
[Unit]
|
||||
Description=bridged — claude-bridge message server
|
||||
Documentation=https://git.ltms.dev/lms/claude-bridge/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# bridged retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take bridged down with it.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/bridged
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/bridged.jar bridged.yaml
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit bridged → [Service] / Environment=BRIDGED_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/bridged/env
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes bridged fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=bridged
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -1,36 +1,36 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<!--
|
||||
CB-504 / CB-594 — launchd agent for bridged (macOS).
|
||||
CB-504 / CB-594 — launchd agent for fleetd (macOS).
|
||||
|
||||
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
|
||||
no systemd. A systemd unit ships alongside (deploy/bridged.service) for the Linux gateways
|
||||
no systemd. A systemd unit ships alongside (deploy/fleetd.service) for the Linux gateways
|
||||
CB-308 introduces.
|
||||
|
||||
Install:
|
||||
cp deploy/dev.ltms.bridged.plist ~/Library/LaunchAgents/
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.bridged.plist
|
||||
launchctl list | grep bridged
|
||||
cp deploy/dev.ltms.fleetd.plist ~/Library/LaunchAgents/
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist
|
||||
launchctl list | grep fleetd
|
||||
|
||||
The paths below are already filled in for this host (resolved 2026-08-16 from
|
||||
`/usr/libexec/java_home`... except that reported the system Applet-plugin JVM, not the jenv-
|
||||
managed JDK 25 actually used to build/run bridged, so JAVA_HOME here is the real one:
|
||||
managed JDK 25 actually used to build/run fleetd, so JAVA_HOME here is the real one:
|
||||
`JENV_VERSION=25.0.3 java -XshowSettings:properties -version 2>&1 | grep java.home`; `which mvn`;
|
||||
`echo $HOME`). If this file is copied to a different host, re-resolve all three paths and check
|
||||
no placeholder path is left behind; scripts/redeploy-bridged.sh's check mode does not (and
|
||||
no placeholder path is left behind; scripts/redeploy-fleetd.sh's check mode does not (and
|
||||
cannot) check this file for you.
|
||||
|
||||
CB-594 — launchd cannot run a login shell (see the PATH comment on EnvironmentVariables below,
|
||||
and scripts/bridged-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
|
||||
and scripts/fleetd-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
|
||||
not java directly, so WORKER_GITEA_TOKEN and AI_GATEWAY_TOKEN still get sourced from
|
||||
${SHARED_ENV}/tools/secrets.sh even though launchd itself never sources anything.
|
||||
|
||||
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
|
||||
does systemd in a way that survives a socket appearing late. bridged retries the herdr socket
|
||||
does systemd in a way that survives a socket appearing late. fleetd retries the herdr socket
|
||||
on startup instead, so an agent that comes up before herdr converges rather than dying — that
|
||||
retry is the actual fix; KeepAlive below is the backstop.
|
||||
|
||||
CB-594 — KeepAlive vs. scripts/redeploy-bridged.sh: a bare SIGTERM makes this JVM exit 143 even
|
||||
CB-594 — KeepAlive vs. scripts/redeploy-fleetd.sh: a bare SIGTERM makes this JVM exit 143 even
|
||||
with its shutdown hook running to completion (measured, see the CB-594 report), which
|
||||
SuccessfulExit:false below reads as a crash and races to restart the OLD jar. The redeploy
|
||||
script now detects a loaded agent and uses `launchctl unload`/`load` instead of a raw kill, so
|
||||
@@ -40,20 +40,20 @@
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>dev.ltms.bridged</string>
|
||||
<string>dev.ltms.fleetd</string>
|
||||
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/bridged-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/target/bridged.jar</string>
|
||||
<string>bridged.yaml</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/target/fleetd.jar</string>
|
||||
<string>fleetd.yaml</string>
|
||||
</array>
|
||||
|
||||
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
|
||||
<key>WorkingDirectory</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd</string>
|
||||
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
@@ -62,7 +62,7 @@
|
||||
<key>HERDR_SOCKET_PATH</key>
|
||||
<string>/Users/dai.ha/.config/herdr/herdr.sock</string>
|
||||
<!--
|
||||
PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker
|
||||
PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker
|
||||
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
|
||||
source .zprofile/.zshrc, so without this the daemon (and therefore every worker) gets a
|
||||
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
|
||||
@@ -71,10 +71,10 @@
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin:/Users/dai.ha/Softwares/apache-maven/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
|
||||
<!--
|
||||
Worker/API tokens are NOT set here: this file is committed. CB-594 —
|
||||
scripts/bridged-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
|
||||
scripts/fleetd-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
|
||||
them, by execing a login shell that sources ${SHARED_ENV}/tools/secrets.sh before the
|
||||
daemon itself starts. bridged also reads the API token from the env var named by
|
||||
auth.tokenEnv (default BRIDGED_API_TOKEN) and only in auth.mode: token — the wrapper
|
||||
daemon itself starts. fleetd also reads the API token from the env var named by
|
||||
auth.tokenEnv (default FLEETD_API_TOKEN) and only in auth.mode: token — the wrapper
|
||||
covers that one too, since it is the same login shell.
|
||||
-->
|
||||
</dict>
|
||||
@@ -84,21 +84,21 @@
|
||||
|
||||
<!--
|
||||
CB-600 — read this before assuming ThrottleInterval bounds anything. It paces restarts to at
|
||||
most one per 10s; it does NOT cap how many times launchd retries. If bridged fails fast on
|
||||
every start — a bad bridged.yaml, for example auth.mode: token with the token env var unset,
|
||||
most one per 10s; it does NOT cap how many times launchd retries. If fleetd fails fast on
|
||||
every start — a bad fleetd.yaml, for example auth.mode: token with the token env var unset,
|
||||
which throws in main() before the daemon ever binds a port — launchd restarts it forever,
|
||||
once every 10s, until a human intervenes. LaunchAgents have no "give up after N attempts"
|
||||
primitive, so this is not something a config change here can fix.
|
||||
|
||||
That loop stops only two ways: (1) `launchctl unload -w ~/Library/LaunchAgents/dev.ltms.bridged.plist`,
|
||||
That loop stops only two ways: (1) `launchctl unload -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist`,
|
||||
or (2) the underlying cause gets fixed, so the process starts successfully and stays up (no
|
||||
more exits to restart). scripts/redeploy-bridged.sh does not add a third way — it does not
|
||||
make bridged self-disable on a config error, on purpose: a fail-fast exit path that
|
||||
more exits to restart). scripts/redeploy-fleetd.sh does not add a third way — it does not
|
||||
make fleetd self-disable on a config error, on purpose: a fail-fast exit path that
|
||||
sometimes decides "this is unrecoverable, stop trying" is one more thing that can misfire,
|
||||
and a wrongly self-disabled daemon needs the exact same manual `launchctl load -w` recovery
|
||||
this comment already names — so it buys nothing an operator watching for the crash loop
|
||||
doesn't already have, at the cost of a new way to be silently down. Watch for it with
|
||||
`launchctl list dev.ltms.bridged` (a high restart count) or by tailing bridged.out for the
|
||||
`launchctl list dev.ltms.fleetd` (a high restart count) or by tailing fleetd.out for the
|
||||
same startup error repeating every ~10s.
|
||||
-->
|
||||
<key>KeepAlive</key>
|
||||
@@ -110,16 +110,16 @@
|
||||
<integer>10</integer>
|
||||
|
||||
<!--
|
||||
CB-594 — same file scripts/redeploy-bridged.sh already tails ($BRIDGED/bridged.out), and both
|
||||
CB-594 — same file scripts/redeploy-fleetd.sh already tails ($BRIDGED/fleetd.out), and both
|
||||
streams point at it, not two separate log files: the script's fresh-line / ERROR-count checks
|
||||
after a restart read this one path regardless of whether launchd or the script started the
|
||||
process, and a stdout/stderr split would make half of what happens during a launchd-driven
|
||||
restart invisible to it.
|
||||
-->
|
||||
<key>StandardOutPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/bridged.out</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/bridged.out</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
|
||||
<key>ProcessType</key>
|
||||
<string>Background</string>
|
||||
@@ -0,0 +1,85 @@
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# fleetd #360: the previous version of this file started clean and broke the daemon in three ways
|
||||
# that nothing logs (see the DO NOT block and the ExecStart/PrivateTmp comments below for what and
|
||||
# why). The unit below, plus its companion deploy/herdr.service, is the version that has actually
|
||||
# run on fleet01 without those failures. Do not "improve" it back toward the old shape without
|
||||
# re-reading why each line is the way it is.
|
||||
#
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/fleetd.service deploy/herdr.service ~/.config/systemd/user/
|
||||
# # edit WorkingDirectory / ExecStart below for your host's paths and java location
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now herdr fleetd
|
||||
# loginctl enable-linger $USER # REQUIRED -- see below
|
||||
# journalctl --user -u fleetd -f
|
||||
#
|
||||
# `loginctl enable-linger` is not optional and is easy to miss, because leaving it out looks like
|
||||
# success: `systemctl --user enable` reports "enabled" and both units run for as long as you stay
|
||||
# logged in. A user manager without lingering starts at your first login and stops at your last
|
||||
# logout, so the fleet simply does not come back after a reboot -- which is the whole reason to
|
||||
# use systemd here rather than the setsid scripts these units replaced. Check it with
|
||||
# `loginctl show-user $USER -p Linger`; the answer must be `Linger=yes`.
|
||||
#
|
||||
# Secrets (AI_GATEWAY_TOKEN, WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI, ...) are not set
|
||||
# here and need no systemd drop-in: ExecStart runs a login shell, so they come from wherever your
|
||||
# login shell already sources them (this host: ~/.fleet/secrets.sh via ~/.zprofile). If a token is
|
||||
# missing there, fleetd still starts — the daemon reports every secret a configured profile
|
||||
# references, by name, never by value:
|
||||
# journalctl --user -u fleetd | grep 'startup secret'
|
||||
# A resolved one logs "startup secret NAME: set (profile 'x' tokenEnv)"; a missing one logs
|
||||
# "startup secret NAME: MISSING" at WARN and the daemon starts anyway — the first visible symptom
|
||||
# is a member that cannot open a pull request, hours later and in a different component.
|
||||
|
||||
[Unit]
|
||||
Description=fleetd — fleet message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only. fleetd retries the herdr socket rather than exiting, which is what actually makes
|
||||
# a late socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down too.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/LTMS/fleetd/fleetd
|
||||
|
||||
# A LOGIN shell, not java directly. Every secret this daemon needs (AI_GATEWAY_TOKEN,
|
||||
# WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI) lives in ~/.fleet/secrets.sh, which only
|
||||
# ~/.zprofile sources. systemd runs no login shell. Started any other way the daemon boots fine
|
||||
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
||||
# request. exec keeps it one process, so systemd tracks the right PID.
|
||||
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
||||
ExecStart=/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"
|
||||
|
||||
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
||||
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
||||
# unit) has to read them. A private /tmp turns the credential scrub into a silent no-op.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config makes fleetd fail fast by design. Give up rather than restart-loop forever.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# DO NOT add ProtectSystem=, ProtectHome=, ProtectKernelTunables= or ProtectControlGroups=.
|
||||
# Measured on fleet01 2026-09-05: each of those gives the unit its own mount namespace, and
|
||||
# fleetd resolves a caller role by running lsof to find the loopback peer PID
|
||||
# (mcp/LsofPeerPidLookup). Inside such a namespace lsof returns nothing, every caller falls back
|
||||
# to ANONYMOUS, and the primary is refused every orchestration call with
|
||||
# "unauthenticated: anonymous may not SPAWN".
|
||||
# The daemon still starts, healthz still returns ok and the secrets still resolve - the only
|
||||
# symptom is that the fleet cannot be driven at all. Verified by bisecting the directives:
|
||||
# no sandbox 3 lsof lines | ProtectSystem=strict 0 | ProtectHome=read-only 0
|
||||
# ProtectKernelTunables 0 | ProtectControlGroups 0 | RestrictSUIDSGID 3 | NoNewPrivileges 3
|
||||
# The two below add no mount namespace and are safe.
|
||||
NoNewPrivileges=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=fleetd
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/bin/zsh
|
||||
# fleetd #360 — template for the script deploy/herdr.service's ExecStart wraps in a pty.
|
||||
#
|
||||
# `script -qfec <this> /dev/null` needs a real command to run, and that command has to be a LOGIN
|
||||
# shell script: herdr itself needs the same secrets fleetd.service's login shell picks up (this
|
||||
# host: ~/.fleet/secrets.sh via ~/.zprofile), because members it spawns inherit its environment.
|
||||
# systemd's own Environment= lines in herdr.service are not enough for that -- they set TERM and a
|
||||
# bare PATH so the pty starts at all, nothing more.
|
||||
#
|
||||
# Copy this file to the path deploy/herdr.service's ExecStart names
|
||||
# (%h/LTMS/fleetd/fleetd-run/herdr-inner.sh by default) and `chmod +x` it. Not committed under
|
||||
# that path itself because the session name below is host-specific.
|
||||
|
||||
# A 0x0 pty makes every pane spawn fail with "ghostty error -2" (see herdr-multi-instance-facts /
|
||||
# fleet01-headless-herdr-standup) -- give it a real size before herdr ever touches it.
|
||||
stty rows 50 cols 200
|
||||
|
||||
# -l: login shell, so herdr and everything it spawns gets the real secrets and PATH.
|
||||
exec zsh -lc 'exec herdr --session <name>'
|
||||
@@ -0,0 +1,40 @@
|
||||
# fleetd #360 — systemd unit for herdr (Linux), the terminal multiplexer fleetd drives.
|
||||
#
|
||||
# This is fleetd.service's companion: fleetd.service's After=/Wants=herdr.service assumes this
|
||||
# unit exists. Before this ticket it did not, so on a fresh host fleetd started against a herdr
|
||||
# that systemd never supervised at all.
|
||||
#
|
||||
# Install: see deploy/fleetd.service's header comment (both units install the same way).
|
||||
#
|
||||
# ExecStart below runs deploy/herdr-inner.sh (copy the template of that name from this directory
|
||||
# to the path in ExecStart, or point ExecStart at wherever you keep it, and make it executable).
|
||||
# It is a separate file rather than an inline command because it must itself be a login shell (see
|
||||
# its own header for why) and systemd's ExecStart does not run one.
|
||||
|
||||
[Unit]
|
||||
Description=herdr terminal multiplexer (fleet session)
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
# script(1) gives herdr a real pty. Without it the client reports a 0x0 window and every pane
|
||||
# spawn fails with "ghostty error -2" -- which surfaces as a fleetd spawn failure, not a herdr one.
|
||||
ExecStart=/usr/bin/script -qfec %h/LTMS/fleetd/fleetd-run/herdr-inner.sh /dev/null
|
||||
StandardInput=null
|
||||
Environment=TERM=xterm-256color
|
||||
Environment=PATH=%h/.local/bin:/usr/local/bin:/usr/bin:/bin
|
||||
|
||||
# PrivateTmp MUST stay false. fleetd writes the member ZDOTDIR scrub dir and the opencode config
|
||||
# dir under its own java.io.tmpdir, and the member pane -- a child of THIS process -- has to read
|
||||
# them. A private /tmp here silently breaks the credential scrub instead of failing loudly.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=5s
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=herdr
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -1,8 +1,8 @@
|
||||
# LavinMQ — the AMQP broker behind bridged's durable ReplyInbox (CB-307 Stage 2).
|
||||
# LavinMQ — the AMQP broker behind fleetd's durable ReplyInbox (CB-307 Stage 2).
|
||||
#
|
||||
# Why this file exists: the broker was previously run ad hoc and simply vanished from the host,
|
||||
# which takes bridged down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Bridged.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# which takes fleetd down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Fleetd.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# degraded mode. This pins the version, keeps the data, and brings itself back after a reboot.
|
||||
#
|
||||
# Usage:
|
||||
@@ -14,27 +14,27 @@
|
||||
#
|
||||
# Management UI: http://127.0.0.1:15672 (guest / guest)
|
||||
#
|
||||
# This is bridged's OWN broker. Do not point bridged at any other AMQP server on this host —
|
||||
# This is fleetd's OWN broker. Do not point fleetd at any other AMQP server on this host —
|
||||
# notably not the `local-rabbitmq` container, which belongs to a different project and would end
|
||||
# up carrying this project's queues.
|
||||
|
||||
name: bridged-broker
|
||||
name: fleetd-broker
|
||||
|
||||
services:
|
||||
lavinmq:
|
||||
# Pinned deliberately: :latest silently moves the broker under a running daemon.
|
||||
image: cloudamqp/lavinmq:2.9.1
|
||||
container_name: bridged-lavinmq
|
||||
container_name: fleetd-lavinmq
|
||||
|
||||
# The failure this deployment exists to prevent — survive reboots and Docker restarts, but
|
||||
# stay down if it was stopped on purpose.
|
||||
restart: unless-stopped
|
||||
|
||||
# Loopback-bound on purpose. LavinMQ ships a default guest/guest account, which is only
|
||||
# acceptable because nothing off-host can reach it. bridged connects over 127.0.0.1, and
|
||||
# acceptable because nothing off-host can reach it. fleetd connects over 127.0.0.1, and
|
||||
# binding 0.0.0.0 here would expose a broker with default credentials to the network.
|
||||
ports:
|
||||
- "127.0.0.1:5672:5672" # AMQP — bridged.yaml broker.uri points here
|
||||
- "127.0.0.1:5672:5672" # AMQP — fleetd.yaml broker.uri points here
|
||||
- "127.0.0.1:15672:15672" # HTTP management API + UI
|
||||
|
||||
# The whole point of Stage 2. Held-but-unacked replies live here; without a named volume a
|
||||
@@ -57,4 +57,4 @@ services:
|
||||
|
||||
volumes:
|
||||
lavinmq-data:
|
||||
name: bridged-lavinmq-data
|
||||
name: fleetd-lavinmq-data
|
||||
|
||||
@@ -0,0 +1,531 @@
|
||||
# CB-201 and CB-227 refinement
|
||||
|
||||
Date: 2026-09-03
|
||||
|
||||
## Decision
|
||||
|
||||
#201 and #227 are one delivery program, but they are not one implementation unit.
|
||||
|
||||
#201 has a real seam: `CompletionResolver` can publish a typed backend-error event only after its
|
||||
waiter resolution wins. #227 can consume that event without knowing any pane text. The classifier
|
||||
must land before the final #227 wiring. However, the policy engine, roster state, and lead nudge can
|
||||
be built in parallel with the classifier.
|
||||
|
||||
I propose five units. Units 1 to 4 own separate files and can run in parallel. Unit 5 owns all
|
||||
composition files and lands after them. It also depends on the #234 defect 2 fix named in the task.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
U1["Unit 1: typed backend-error classification"] --> U5["Unit 5: wire policy, spawn gate, and fleet views"]
|
||||
U2["Unit 2: credential outage policy"] --> U5
|
||||
U3["Unit 3: lead outage nudge"] --> U5
|
||||
U4["Unit 4: durable member outcome"] --> U5
|
||||
D234["#234 defect 2: fail-loud target resolution"] --> U5
|
||||
```
|
||||
|
||||
*Figure 1. Four file-disjoint foundations feed one composition unit.*
|
||||
|
||||
This split keeps `Fleetd.java` under one owner. It also keeps every other production file under one
|
||||
unit in this plan.
|
||||
|
||||
## Evidence checked in the current branch
|
||||
|
||||
I read both issue pages in full. Each page reports zero comments.
|
||||
|
||||
| Evidence | What the code says now |
|
||||
|---|---|
|
||||
| `inject/CompletionResolver.java:229-237` | A turn below two seconds fails before normal scrape classification. A matching fast backend error is therefore only a generic failure today. |
|
||||
| `inject/CompletionResolver.java:260-276` and `:332-367` | #211 already added raw-screen classification when `lastAssistantBlock` is empty. The “dead-code question” in #201 is stale on this branch. |
|
||||
| `inject/CompletionResolver.java:288-317` | Exhaustion wins before the hard-coded `API Error:` match. A backend error then goes through generic `fail(...)`. |
|
||||
| `inject/CompletionResolver.java:311-316` | The code admits that the pattern is a heuristic. A member report which quotes an API error may match it. |
|
||||
| `inject/CompletionResolver.java:449-467` | Startup coverage exists only for `exhaustedPattern`. |
|
||||
| `inject/ExhaustedPatternLookup.java:13-25` | The current lookup and explicit `none()` value are a good shape for the new classifier seam. |
|
||||
| `Fleetd.java:196-207` | One `BackendQuarantine` is shared by placement and the exhaustion sink. Its cooldown comes from `quarantineCooldownSeconds`. |
|
||||
| `Fleetd.java:322-363` | Pattern compilation, target-to-profile lookup, and the live `ExhaustionSink` are composed in `Fleetd.main`. The sink on this branch still ends in `.ifPresent(...)`. This plan assumes #234 replaces that silent path. |
|
||||
| `placement/BackendQuarantine.java:60-87` | A repeated exhaustion restarts one long quarantine. The store is credential-keyed and uses an injected monotonic clock. |
|
||||
| `member/CompositePeerLauncher.java:260-317` | Explicit and policy-selected spawns have separate gates. Both paths must learn about outage cool-off. |
|
||||
| `member/CompositePeerLauncher.java:347-379` | Exhaustion refusal already checks a credential for explicit spawns and filters policy candidates. Its error text says “exhausted”. |
|
||||
| `placement/PlacementContext.java:10-22` and `PlacementPolicyUtil.java:14-83` | Automatic placement has only one transient exclusion set named `quarantined`. Reusing it would make outage errors say “backend exhausted”. |
|
||||
| `mcp/FleetMcp.java:913-1025` | `fleet_list` sets `free: 0` and adds `credentialId` plus `quarantinedForSeconds` when quarantine is active. |
|
||||
| `session/MemberSession.java:51-59` | The roster has `DONE` and generic `FAILED`, but no backend-error state or stored reason. |
|
||||
| `session/SessionManager.java:695-773` | A normal boundary moves `BUSY` to `DONE`. A failure moves any non-released session to `FAILED`. The async completion resolver can race the `DONE` update. |
|
||||
| `session/SessionManager.java:648-687` | `rosterView` reports the session state, but it reports no terminal reason. |
|
||||
| `msg/MessageService.java:922-940` | CB-588 already nudges for every terminal async ticket, including failures. Current code would report failed tickets, but it would not report one correlated outage. |
|
||||
| `msg/ReplyPushLoop.java:20-48` | Replies, terminal tickets, and questions share one per-lead schedule. This prevents two push sources from injecting competing turns. |
|
||||
| `msg/ReplyPushLoop.java:305-395` | Each push entry point resolves the owning lead through `PrimaryRegistry`. Missing ownership is logged and the durable or pending item remains the backstop. |
|
||||
| `msg/ReplyPushLoop.java:496-547` | One tick builds one combined nudge. Pending items have separate reminder counts. |
|
||||
| `health/FleetHealthMonitor.java:91-143` | Health is a slow periodic observer of members and message-layer facts. It does not receive completion classifications. |
|
||||
| `health/FleetHealthMonitor.java:206-208` | `healthCoverage` means health enabled plus webhook configured. It does not describe lead-pane alerts. |
|
||||
| `Fleetd.java:465-486` | Health stays `detection-only` without the webhook notification setting. |
|
||||
|
||||
I also read the related unit tests for `CompletionResolver`, `ReplyPushLoop`, `BackendQuarantine`,
|
||||
`CompositePeerLauncher`, `PlacementPolicyUtil`, `SessionManager`, `MessageService`, and `FleetMcp`.
|
||||
|
||||
I did not inspect the in-progress #234 branch. I only used the two measured facts in the task. No
|
||||
peer architect was named, so I did not exchange a design with one.
|
||||
|
||||
## Required behaviour
|
||||
|
||||
The policy should use these first values:
|
||||
|
||||
- Threshold: **2** classified backend errors.
|
||||
- Window: **60 seconds**, measured from the first error to the second.
|
||||
- Cool-off: **60 seconds**, starting when the threshold is reached.
|
||||
- Correlation key: `credentialId`, never profile name and never error text.
|
||||
- Incident rule: one active incident per credential. Errors during its cool-off do not extend it and
|
||||
do not create more lead notices.
|
||||
- Rearm rule: after cool-off ends, two fresh errors are needed for another incident.
|
||||
|
||||
Two errors are the smallest threshold which protects the honest one-turn failure. A 60-second window
|
||||
fits the measured two-member outage. A 60-second cool-off blocks immediate repeat spawns without
|
||||
turning a short backend fault into the default 1,800-second exhaustion quarantine.
|
||||
|
||||
A single classified error still fails its send and marks its member `backend_error`. It does not
|
||||
cool a credential and does not send an outage notice. This is what “a single error changes nothing”
|
||||
must mean at the credential level. It cannot mean that the failed member still looks successful.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant R1 as Resolver for member A
|
||||
participant R2 as Resolver for member B
|
||||
participant P as Outage policy
|
||||
participant S as Spawn gate
|
||||
participant N as Lead push loop
|
||||
participant L as Lead pane
|
||||
|
||||
R1->>P: backend error for credential C
|
||||
Note over P: Count 1, no cool-off
|
||||
R2->>P: backend error for credential C within 60s
|
||||
P->>P: Start one 60s incident
|
||||
P->>S: Credential C is cooling off
|
||||
P->>N: Queue one incident notice
|
||||
N->>L: Inject when lead is idle, blocked, or done
|
||||
L->>S: Request another spawn on credential C
|
||||
S-->>L: Refuse and report remaining cool-off
|
||||
```
|
||||
|
||||
*Figure 2. The second independent classification creates the fleet-level event.*
|
||||
|
||||
Against the 2026-09-01 case, the second failed member would start cool-off. `fleet_list` would show
|
||||
zero free capacity and both members as `backend_error`. The push loop would inject one outage notice
|
||||
even if the lead had not polled either ticket yet. The design reports the outage. It does not recover
|
||||
uncommitted work from the members.
|
||||
|
||||
## Unit 1 — Typed backend-error classification
|
||||
|
||||
### Scope
|
||||
|
||||
Replace the direct hard-coded check inside `CompletionResolver` with a lookup and a sink. Keep the
|
||||
public send result as a failed send. The typed internal event is the seam #227 consumes.
|
||||
|
||||
The lookup returns the pattern for a target. The sink receives the target, matched line, and full
|
||||
failure reason. It fires only after `Rendezvous.resolveFailure(...)` wins for that exact captured
|
||||
waiter. This copies the race rule already used by `ExhaustionSink`.
|
||||
|
||||
The classifier must run in all three current paths:
|
||||
|
||||
1. a normal non-empty assistant block;
|
||||
2. the #211 raw scrape fallback;
|
||||
3. a turn inside `MIN_TURN_NANOS`, before it becomes a generic too-fast failure.
|
||||
|
||||
In every path, the order stays: stale-baseline guard, exhaustion, backend error, then generic
|
||||
failure or completion. A fast turn still fails when no configured pattern matches.
|
||||
|
||||
Keep `(?i)\bAPI Error\s*:` as a compatibility pattern for profiles without `errorPattern` until the
|
||||
operator config is updated. Do not call this full coverage. Startup reporting in Unit 5 must name
|
||||
profiles using this weaker legacy default.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorPatternLookup.java`.
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorSink.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/inject/CompletionResolverTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. A target-specific error pattern matches a normal assistant block and resolves the send as failed.
|
||||
2. The same match calls `BackendErrorSink` exactly once after the waiter resolution wins.
|
||||
3. A late classification which loses to `fleet_reply` does not call the sink.
|
||||
4. An exhausted line that also matches the generic error pattern stays `BACKEND_EXHAUSTED`. It calls
|
||||
only `ExhaustionSink`.
|
||||
5. A raw pane with leading Terminal User Interface (TUI) chrome and no assistant marker still uses
|
||||
the #211 fallback and calls the backend-error sink.
|
||||
6. A matching error inside the two-second floor is typed and sent to the sink. A non-matching fast
|
||||
turn stays a generic failure.
|
||||
7. An unchanged delivery baseline which contains old backend-error text is suppressed. It never
|
||||
increments outage evidence.
|
||||
8. A non-match keeps the existing completion result and text.
|
||||
9. Constructors used by current callers keep compiling. They use the legacy default lookup and an
|
||||
explicit inert sink until Unit 5 supplies the production objects.
|
||||
10. Unit tests pass. The developer runs the focused test first, then `mvn clean install` from
|
||||
`fleetd/`.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 1 can run with Units 2, 3, and 4.
|
||||
|
||||
Unit 5 depends on its new lookup, sink, and constructor.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact classifier order in all three paths.
|
||||
- The focused test command and result.
|
||||
- The test which proves a losing waiter race does not publish an event.
|
||||
- The test which proves a fast matching failure is typed.
|
||||
- The final `mvn clean install` result.
|
||||
- Any constructor kept only for transition and where Unit 5 replaces it.
|
||||
|
||||
## Unit 2 — Credential outage policy
|
||||
|
||||
### Scope
|
||||
|
||||
Build a small credential-keyed state machine. It accepts already-classified backend-error events.
|
||||
It does not read pane text, profiles, sessions, or lead state.
|
||||
|
||||
Use an injected monotonic clock. A call records `credentialId`, target, and reason. It returns a new
|
||||
incident only on the threshold crossing. The incident contains a stable event id, credential id,
|
||||
the distinct affected targets, evidence count, window, and remaining cool-off.
|
||||
|
||||
This class owns both correlation and short cool-off. Keeping them together makes threshold crossing
|
||||
and the cool-off deadline one atomic state change.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/placement/BackendOutagePolicy.java`.
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/placement/BackendOutagePolicyTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One error creates no incident and no cool-off.
|
||||
2. Two errors for one credential within 60 seconds create exactly one incident and a 60-second
|
||||
cool-off.
|
||||
3. Two errors more than 60 seconds apart do not create an incident.
|
||||
4. The exact 60-second boundary has a pinned result. Use inclusive `<= 60s` so scheduler delay does
|
||||
not discard evidence at the boundary.
|
||||
5. Different credentials never share evidence.
|
||||
6. Different profiles which supply the same credential id do share evidence. The policy itself only
|
||||
sees the credential id.
|
||||
7. More errors during active cool-off do not extend its deadline and do not return another incident.
|
||||
8. After expiry, old evidence is cleared. Two fresh errors are needed to create the next incident.
|
||||
9. Remaining seconds round up, matching `BackendQuarantine` reporting.
|
||||
10. Concurrent second and third errors cannot return two incidents.
|
||||
11. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 2 can run with Units 1, 3, and 4.
|
||||
|
||||
Unit 5 depends on the policy API.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The state transition table and locking method.
|
||||
- The exact threshold, window, cool-off, and boundary rule.
|
||||
- The test which proves one incident under concurrent calls.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 3 — Lead outage nudge
|
||||
|
||||
### Scope
|
||||
|
||||
Add backend incidents as a fourth pending source in `ReplyPushLoop`. Do not create another scheduler
|
||||
or call `AgentControl.send` from `Fleetd`. The existing combined per-lead schedule is the control
|
||||
which prevents competing injected turns.
|
||||
|
||||
The entry point takes an incident id, affected worker targets, credential id, affected profile
|
||||
names, and remaining cool-off. It resolves distinct owning leads through `PrimaryRegistry`.
|
||||
|
||||
Each `(incidentId, lead)` item is one-shot. It waits while the lead is not injectable. After one
|
||||
successful `agents.send`, remove it. A send exception keeps it pending for a bounded retry. It never
|
||||
uses the repeated reminder behaviour of an uncollected ticket.
|
||||
|
||||
Also add a fail-loud entry point for a classified target that Unit 5 cannot map to a credential. It
|
||||
uses `PrimaryRegistry.nudgeTargetFor(target)` and says that correlation could not run. If no lead is
|
||||
known, log at `WARN`, not `DEBUG`.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/msg/ReplyPushLoop.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/msg/ReplyPushLoopTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One incident affecting two workers owned by one lead causes one successful pane injection.
|
||||
2. Two affected workers owned by two leads cause one successful injection per affected lead. This
|
||||
is one notice per event per lead, not one notice per member.
|
||||
3. Repeating the same incident id is idempotent.
|
||||
4. A busy or unknown lead is not injected. The item stays pending until the lead becomes injectable
|
||||
or its attempt cap is reached.
|
||||
5. After one successful injection, later ticks do not mention that incident again.
|
||||
6. A failed `agents.send` is retried within the existing bound. A successful retry still gives only
|
||||
one successful send.
|
||||
7. A pending ticket and an outage incident for one lead appear in one combined nudge, not two
|
||||
competing turns.
|
||||
8. The text names the credential, profiles, affected workers, and remaining cool-off. It tells the
|
||||
lead to run `fleet_list`.
|
||||
9. An unmapped target produces a direct warning notice when a lead is known. If no lead is known,
|
||||
the code logs a `WARN` naming the target and reason.
|
||||
10. `stop()` clears incident state as it clears other push state.
|
||||
11. Existing reply, ticket, and question tests stay green. The focused tests and
|
||||
`mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. The API uses plain values, not the Unit 2 incident class. This lets Unit 3 run in parallel.
|
||||
|
||||
Unit 5 adapts the Unit 2 incident into this entry point.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact one-shot and retry rules.
|
||||
- The test showing one combined nudge with a failed ticket.
|
||||
- The test showing one successful send for two affected workers.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 4 — Durable member backend-error outcome
|
||||
|
||||
### Scope
|
||||
|
||||
Make a classified backend failure remain visible after its ticket is collected or expires.
|
||||
|
||||
Add `BACKEND_ERROR` to `MemberSession.State`. Add a nullable failure detail to `MemberSession` and
|
||||
render it as `failureReason` in `SessionManager.rosterView`. Add
|
||||
`SessionManager.onBackendError(target, reason)`.
|
||||
|
||||
The transition must handle both completion orderings:
|
||||
|
||||
- `BUSY -> BACKEND_ERROR` when classification wins before the normal completion state update;
|
||||
- `DONE -> BACKEND_ERROR` when the async resolver runs after `SessionManager.onTurnComplete`.
|
||||
|
||||
It must use a compare-and-set retry or another atomic update. `RELEASED` must never return to the
|
||||
roster. A backend-error member is terminal and cannot accept another delivery.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/MemberSession.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/SessionManager.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/session/SessionManagerTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `onBackendError` moves a `BUSY` member to `BACKEND_ERROR` and stores the reason.
|
||||
2. It also moves `DONE` to `BACKEND_ERROR`, covering the resolver race.
|
||||
3. A later normal `onTurnComplete` cannot change `BACKEND_ERROR` back to `DONE`.
|
||||
4. A released or unknown member is not recreated. The unknown case logs at `WARN` and returns an
|
||||
explicit false result to its caller.
|
||||
5. `onDelivered` refuses a `BACKEND_ERROR` member, just as it refuses generic `FAILED`.
|
||||
6. `rosterView` reports `state: backend_error` and `failureReason` after the send ticket is gone.
|
||||
7. Ordinary members do not gain a blank or invented `failureReason` field.
|
||||
8. Existing constructors keep source compatibility for tests and adapters.
|
||||
9. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 4 can run with Units 1, 2, and 3.
|
||||
|
||||
Unit 5 calls the new session method from the production sink.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The two race orderings and the tests for both.
|
||||
- The exact roster JSON shape.
|
||||
- The unknown-target result and log level.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 5 — Production wiring, spawn gate, and fleet views
|
||||
|
||||
### Scope
|
||||
|
||||
Compose Units 1 to 4 in production. This is the only unit which edits `Fleetd.java`.
|
||||
|
||||
Add per-profile `errorPattern` config beside `exhaustedPattern`. Compile both once at startup. A
|
||||
configured pattern wins over the legacy default. Report configured profiles and legacy-default
|
||||
profiles separately at startup. A bad regex must stop startup with the profile and key in the
|
||||
message.
|
||||
|
||||
Wire one production `BackendErrorSink` with this order:
|
||||
|
||||
1. mark the member `backend_error` with its reason;
|
||||
2. resolve the profile and its current `effectiveCredentialId()` through the fail-loud #234 seam;
|
||||
3. record the error in `BackendOutagePolicy`;
|
||||
4. on a new incident, submit one event to `ReplyPushLoop`.
|
||||
|
||||
If target metadata cannot be resolved, do not end in `Optional.ifPresent`. Log an error and call the
|
||||
Unit 3 unmapped-target notice. The failed send still reaches its ticket through CB-588.
|
||||
|
||||
Teach both spawn paths about a separate cool-off source. Exhaustion quarantine has priority when
|
||||
both states are active. Automatic placement needs a distinct `coolingOff` set so its refusal does
|
||||
not say “exhausted”.
|
||||
|
||||
Extend the MCP (Model Context Protocol) views:
|
||||
|
||||
- A cooling profile has `free: 0`, `credentialId`, and `coolingOffForSeconds` in `fleet_list`.
|
||||
- It does not have `quarantinedForSeconds` unless exhaustion quarantine is also active.
|
||||
- `fleet_profiles` has a separate `coolingOff` map, not an entry in `quarantined`.
|
||||
- A direct spawn refusal says the credential is cooling off after repeated backend errors and gives
|
||||
the remaining seconds.
|
||||
|
||||
Do not change `FleetHealthMonitor.coverage`. It still describes the periodic health webhook path.
|
||||
Lead-pane outage delivery is a separate capability.
|
||||
|
||||
### Files owned
|
||||
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/Fleetd.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/ConfigRef.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/member/CompositePeerLauncher.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementContext.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementPolicyUtil.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/FleetConfigTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/ConfigRefTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/member/CompositePeerLauncherTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/placement/PlacementPolicyTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/mcp/FleetMcpTest.java`
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/BackendOutageFlowTest.java`.
|
||||
- `fleetd/fleetd.example.yaml`
|
||||
- `CLAUDE.md`
|
||||
|
||||
No earlier unit edits these files.
|
||||
|
||||
The lead, not a worker, must update `wiki/7-Use-Cases.md`, `wiki/9-Implementation.md`, and
|
||||
`wiki/11-Features.md`. Project rules forbid workers from committing `wiki/`. The portable block in
|
||||
`CLAUDE.md` and `wiki/7-Use-Cases.md` must remain byte-identical.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `errorPattern` binds per profile. Blank uses the legacy default and is reported as degraded
|
||||
coverage. A config reload which changes it is reported as deferred because patterns are compiled
|
||||
at startup.
|
||||
2. A malformed `errorPattern` stops startup and names `profiles.<name>.errorPattern`.
|
||||
3. The real production sink never silently drops an unknown target. A test captures its error log
|
||||
and the fallback notice call.
|
||||
4. One real backend-error classification through `CompletionResolver` marks only its member. It does
|
||||
not cool the credential and does not send an outage notice.
|
||||
5. Two real classifications for one credential within 60 seconds start one incident.
|
||||
6. The integration test then calls the real explicit-profile spawn gate. It is refused before any
|
||||
adapter spawn call, with “cooling off” and remaining seconds in the message.
|
||||
7. The same test calls an automatic placement path. A cooling candidate is skipped. If every
|
||||
candidate is cooling, the error names cool-off rather than exhaustion.
|
||||
8. Profiles sharing the credential are all blocked. A profile on another credential stays usable.
|
||||
9. `fleet_list` from the same fixture shows both members as `backend_error`, preserves each failure
|
||||
reason, and reports `free: 0`, the credential, and `coolingOffForSeconds`.
|
||||
10. `fleet_profiles` reports cool-off separately from quarantine.
|
||||
11. The real `ReplyPushLoop` receives one incident and makes one successful lead-pane send. Existing
|
||||
failed-ticket notice content may share that same combined send.
|
||||
12. Exhaustion still wins when a line matches both patterns. A simultaneous exhaustion quarantine
|
||||
also wins in spawn errors and fleet views.
|
||||
13. After the 60-second cool-off, spawn is allowed again. A new incident needs two fresh errors.
|
||||
14. `healthCoverage` has the same value before and after this change for the same health config.
|
||||
15. `fleetd.example.yaml` explains `errorPattern`, the legacy fallback, 2/60/60 policy, and the
|
||||
difference between cool-off and exhaustion quarantine.
|
||||
16. `CLAUDE.md` tells leads how `fleet_profiles` and `fleet_list` report cool-off. The lead later
|
||||
applies the matching wiki updates and runs the documented byte-sync check.
|
||||
17. The developer records the new end-to-end test failing before implementation, then passing. The
|
||||
focused suites and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
Unit 5 starts only after Units 1 to 4 are merged or rebased into its branch. It also starts after the
|
||||
#234 defect 2 fix lands, because both areas touch the same target-resolution control path.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact commits used for Units 1 to 4 and #234.
|
||||
- The startup coverage line with one configured and one legacy-default profile.
|
||||
- The red test output before the implementation and its green result after.
|
||||
- The explicit and automatic spawn refusal text.
|
||||
- Sample `fleet_list` and `fleet_profiles` JSON for cool-off and exhaustion.
|
||||
- The number and text of lead-pane sends in the real-path test.
|
||||
- The focused test commands and final `mvn clean install` result.
|
||||
- The exact `CLAUDE.md` change and the wiki edits the lead must apply.
|
||||
|
||||
## File ownership summary
|
||||
|
||||
| Area | Unit | Shared edit risk |
|
||||
|---|---:|---|
|
||||
| Completion classification | 1 | Only Unit 1 edits `CompletionResolver` and its test. |
|
||||
| Correlation and cool-off state | 2 | New files only. |
|
||||
| Lead push scheduling | 3 | Only Unit 3 edits `ReplyPushLoop` and its test. |
|
||||
| Member terminal state | 4 | Only Unit 4 edits `MemberSession`, `SessionManager`, and their test. |
|
||||
| Main composition, config, placement, MCP views, shipped prompt | 5 | Only Unit 5 edits `Fleetd`, `FleetConfig`, `CompositePeerLauncher`, placement context, `FleetMcp`, and `CLAUDE.md`. |
|
||||
| Wiki propagation | Lead after Unit 5 | Workers do not commit the wiki submodule. |
|
||||
|
||||
## What I would not build
|
||||
|
||||
1. **Do not reuse `BackendQuarantine` for outages.** Its repeat call restarts a long credential
|
||||
quarantine. Its fields and errors say “exhausted”. That is wrong for a short outage.
|
||||
2. **Do not merge the exhaustion and generic error patterns.** Exhaustion must win because it has a
|
||||
different policy and duration.
|
||||
3. **Do not group by error string.** One outage can produce different text. The shared operational
|
||||
limit is the credential.
|
||||
4. **Do not mark a profile unusable until config changes.** The current classifier cannot safely
|
||||
tell a permanent malformed request from a transient service fault. A permanent state would need
|
||||
a stronger error taxonomy first.
|
||||
5. **Do not quarantine on the first generic backend error.** That would turn one bad request or one
|
||||
false pattern match into a fleet-wide capacity loss.
|
||||
6. **Do not add this to `FleetHealthMonitor`.** The monitor samples slow member health. The exact
|
||||
backend event already exists at completion resolution, and moving it to polling would lose type
|
||||
and time.
|
||||
7. **Do not add another direct lead injector.** `ReplyPushLoop` already owns status gating,
|
||||
per-lead coalescing, retry bounds, and heartbeat stand-down.
|
||||
8. **Do not change `healthCoverage` to `full`.** That field still means a webhook notification sink
|
||||
exists for periodic health. A backend outage nudge does not make every health event visible.
|
||||
9. **Do not persist incident history across daemon restart in this work.** Existing exhaustion
|
||||
quarantine is also in memory. A 60-second state does not justify a new durable store.
|
||||
10. **Do not build work recovery.** The PR-body survival story proves why checkpoint-first work is
|
||||
useful, but these tickets are about detection, capacity, and signalling.
|
||||
11. **Do not remove the legacy `API Error:` fallback in the first release.** Doing so would turn an
|
||||
unedited config back into a false successful completion. Report it as degraded coverage instead.
|
||||
12. **Do not reorder or add the old `visibleTurn` fallback from #201.** #211 already implemented the
|
||||
narrow raw-scrape fallback at `CompletionResolver.classifyRawScrapeFallback`.
|
||||
|
||||
## Riskiest assumption and cheapest experiment
|
||||
|
||||
The riskiest assumption is that a configured error regex means “the backend failed this turn”. The
|
||||
current code and test already show the counterexample: a worker may quote `API Error:` while writing
|
||||
a valid report. Two such false matches on one credential would now remove capacity for 60 seconds.
|
||||
|
||||
The cheapest experiment is a replay corpus before Unit 5 ships:
|
||||
|
||||
1. Save the full pane text from the measured 2026-09-01 outage.
|
||||
2. Produce one safe failure per backend with a disposable invalid endpoint or request.
|
||||
3. Save one valid member report which quotes each error line.
|
||||
4. Replay all samples through the real `CompletionResolver` test fixture.
|
||||
5. Require outage samples to match and quoted-report samples not to match after assistant-block
|
||||
extraction and baseline checks.
|
||||
|
||||
This costs no outage deployment and no real sleep. If quoted reports still match, narrow the profile
|
||||
patterns before enabling correlation. Do not raise the threshold to hide a bad classifier.
|
||||
|
||||
## Sequencing with three developers
|
||||
|
||||
First wave:
|
||||
|
||||
1. Developer A: Unit 1, typed classification.
|
||||
2. Developer B: Unit 2, credential outage policy.
|
||||
3. Developer C: Unit 3, lead outage nudge.
|
||||
|
||||
As soon as one slot is free, start Unit 4. It is file-disjoint from every first-wave unit. Merge and
|
||||
review Units 1 to 4 independently.
|
||||
|
||||
Start Unit 5 only after all four foundations and #234 are available. Unit 5 is the only high-conflict
|
||||
integration branch, so no other active unit should touch its file list.
|
||||
|
||||
## Checks performed for this refinement
|
||||
|
||||
- Read issue #201 and issue #227 through their Gitea pages. Both showed zero comments.
|
||||
- Read the source and tests named in the evidence section.
|
||||
- Ran `git status --short --branch`; the branch was clean before this document was added.
|
||||
- Ran `git log --oneline -12` to identify the branch base.
|
||||
- I did not run Maven because this change adds only a design document.
|
||||
- Rendered both Mermaid blocks with `npx @mermaid-js/mermaid-cli`; both commands succeeded.
|
||||
@@ -2,7 +2,7 @@
|
||||
|
||||
**Status:** design spec for review → delegate implementation.
|
||||
**Grounded in:** `WorkerService`, `Injector`/`StatusPoller`/`TurnListener`, `MessageService`,
|
||||
`BridgeMcp`, `BridgedApp` (see [wiki 9. Implementation](../wiki/9-Implementation.md)).
|
||||
`FleetMcp`, `FleetApp` (see [wiki 9. Implementation](../wiki/9-Implementation.md)).
|
||||
|
||||
## Problem
|
||||
|
||||
@@ -15,7 +15,7 @@ Consequences today:
|
||||
since when?" without shelling to herdr for a raw agent list (no state, no ownership, no age).
|
||||
- Cleanup of a worker that outlived its owning process depends entirely on the boot-time
|
||||
name-nonce **reaper** (CB-117) — there is no live, authoritative roster during a run.
|
||||
- `bridge_list` (CB-304) can only surface herdr's view, not a bridge-owned roster.
|
||||
- `fleet_list` (CB-304) can only surface herdr's view, not a bridge-owned roster.
|
||||
- There is no seam for per-session policy (checkpoint on teardown → CB-302; idle_ttl /
|
||||
context_cap / drain → CB-303).
|
||||
|
||||
@@ -43,7 +43,7 @@ Under no-reuse, a released session is terminal. A new `acquire` always creates a
|
||||
subscription-guarded spawn/teardown mechanics; `SessionManager` adds the registry, lifecycle, and
|
||||
ownership on top.
|
||||
|
||||
**Package:** new `dev.ltms.bridged.session` — keeps the registry/lifecycle concern separate from
|
||||
**Package:** new `dev.ltms.fleet.session` — keeps the registry/lifecycle concern separate from
|
||||
the `worker` spawn mechanics. Holds `SessionManager` + `WorkerSession`.
|
||||
|
||||
### `WorkerSession` (record or small mutable holder)
|
||||
@@ -96,13 +96,13 @@ final class SessionManager {
|
||||
|
||||
### Integration points
|
||||
|
||||
- **`Bridged.main`** — construct `SessionManager(workerService, ...)`; wire it as/decorating the
|
||||
- **`Fleetd.main`** — construct `SessionManager(workerService, ...)`; wire it as/decorating the
|
||||
`TurnListener` alongside `CompletionResolver` so it sees turn boundaries, and give it the
|
||||
`WorkerPresence` signal for `READY`.
|
||||
- **`BridgeMcp.spawn` / `BridgedApp.spawnWorker`** — route spawn through `SessionManager.acquire`
|
||||
(carry `callerTerminal` as `ownerTerminal`). **`bridge_stop` / `DELETE /workers/{paneId}`** →
|
||||
- **`FleetMcp.spawn` / `FleetApp.spawnWorker`** — route spawn through `SessionManager.acquire`
|
||||
(carry `callerTerminal` as `ownerTerminal`). **`fleet_stop` / `DELETE /workers/{paneId}`** →
|
||||
`SessionManager.release`.
|
||||
- **`bridge_list` / `GET /sessions` (CB-304 later)** — read `SessionManager.roster()`.
|
||||
- **`fleet_list` / `GET /sessions` (CB-304 later)** — read `SessionManager.roster()`.
|
||||
- **`MessageService`** — no change required for one-shot; a later CB-303 auto-release hook can call
|
||||
`release` from `onTurnComplete` under policy.
|
||||
|
||||
@@ -121,4 +121,4 @@ final class SessionManager {
|
||||
- **CB-302** — attach a checkpoint step (`STATE.md` + commit) to the `release` path.
|
||||
- **CB-303** — a policy loop over `roster()` using `spawnedAtNanos`/state to auto-`release` on
|
||||
`idle_ttl`, or drain on `context_cap`.
|
||||
- **CB-304** — `bridge_list` reads `roster()` for a bridge-owned roster + live join.
|
||||
- **CB-304** — `fleet_list` reads `roster()` for a bridge-owned roster + live join.
|
||||
|
||||
@@ -2,12 +2,12 @@
|
||||
|
||||
**Status:** ✅ shipped — implemented at commit `97ecc71` (per-worker git worktree + config-parity
|
||||
overlay). As-built: `session/GitWorktrees.java` behind the `Worktrees` port, wired in
|
||||
`Bridged.main` and configurable via `worktreeRoot` / per-profile `parityOverlay`
|
||||
(see `bridged.example.yaml`). Branch/worktree surface in `bridge_list` landed with CB-304
|
||||
`Fleetd.main` and configurable via `worktreeRoot` / per-profile `parityOverlay`
|
||||
(see `fleetd.example.yaml`). Branch/worktree surface in `fleet_list` landed with CB-304
|
||||
(`9fe04bf`); the worker-opened-PR checkpoint landed as CB-302 (`64e70ef`).
|
||||
**Extends:** [CB-301 Session Manager](CB-301-Session-Manager.md) (shipped, commit `54d907c`).
|
||||
**Realizes:** the config-parity requirement in [Worker Git Workflow](Worker-Git-Workflow.md).
|
||||
**Grounded in:** `SessionManager`, `WorkerService.spawn/effectiveCwd`, `BridgedConfig.Worker`,
|
||||
**Grounded in:** `SessionManager`, `WorkerService.spawn/effectiveCwd`, `FleetConfig.Worker`,
|
||||
`inject/…LsofPeerPidLookup` (the `ProcessBuilder` exec pattern).
|
||||
|
||||
## Problem
|
||||
@@ -43,7 +43,7 @@ provider.
|
||||
### `WorktreeRequest` (new, nullable = "no worktree")
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.session;
|
||||
package dev.ltms.fleet.session;
|
||||
/** Ask acquire() to provision an isolated worktree. null ⇒ run in the shared primary tree. */
|
||||
public record WorktreeRequest(String ticketSlug, String baseRef) {
|
||||
// ticketSlug seeds the branch name; baseRef null/blank ⇒ current HEAD of the repo.
|
||||
@@ -63,7 +63,7 @@ untouched.
|
||||
### `Worktrees` seam (new)
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.session;
|
||||
package dev.ltms.fleet.session;
|
||||
public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree add <path> -b <branch> <baseRef|HEAD>. Returns the worktree path. */
|
||||
String add(String repoRoot, String branch, String baseRef);
|
||||
@@ -112,7 +112,7 @@ public void release(String paneId) {
|
||||
}
|
||||
```
|
||||
|
||||
### Config — `BridgedConfig.Worker.parityOverlay` + a `worktreeRoot`
|
||||
### Config — `FleetConfig.Worker.parityOverlay` + a `worktreeRoot`
|
||||
|
||||
- Add `List<String> parityOverlay` to the `Worker` record (12th field). Compact-constructor default
|
||||
when null/empty: `[".mcp.json", ".claude/settings.local.json", ".env", ".envrc"]` (missing paths are
|
||||
@@ -122,7 +122,7 @@ public void release(String paneId) {
|
||||
|
||||
### Surface: MCP + REST
|
||||
|
||||
- `bridge_spawn` gains an optional `worktree` arg: `true`, or a ticket slug string. Truthy ⇒ build a
|
||||
- `fleet_spawn` gains an optional `worktree` arg: `true`, or a ticket slug string. Truthy ⇒ build a
|
||||
`WorktreeRequest(slug, null)` and call the 5-arg `acquire`.
|
||||
- `POST /workers` gains `worktree` (+ optional `ticket`) in the body/query, same mapping.
|
||||
- `workerView`/`view(WorkerSession)` include `worktree` and `branch` **when non-null** (omit for
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
# CB-306 — Spawn-Readiness Gate (launcher-owned terminal readiness)
|
||||
|
||||
**Status:** design note / delegation spec (branch `worker/cb-306-readiness`)
|
||||
**Issue:** gitea `lms/claude-bridge` #4
|
||||
**Issue:** gitea `fleet/fleetd` #4
|
||||
**Owner of the behaviour:** `ClaudeCodeLauncher` (the `PeerLauncher` adapter) — NOT core.
|
||||
|
||||
## 1. Problem
|
||||
|
||||
`bridge_spawn` today returns a session the instant the herdr pane is started. The pane is not
|
||||
`fleet_spawn` today returns a session the instant the herdr pane is started. The pane is not
|
||||
yet a usable Claude REPL — it may still be sitting at the folder-trust prompt, or the CLI may
|
||||
never come up at all. Nothing blocks or times out on that. Consequences:
|
||||
|
||||
- A `bridge_send` to a not-yet-ready worker surfaces as a **~60 s MCP-client timeout** (the send
|
||||
- A `fleet_send` to a not-yet-ready worker surfaces as a **~60 s MCP-client timeout** (the send
|
||||
blocks waiting for a turn that can't start) instead of a fast, explicit spawn failure.
|
||||
- A worker stuck at the folder-trust prompt lingers in `SPAWNING` forever; nothing fails it.
|
||||
|
||||
@@ -55,7 +55,7 @@ a crash, and a slow start all present as "never becomes injectable" and all corr
|
||||
or `spawnReadyTimeoutMs` elapses.
|
||||
3. **Ready** → return the `WorkerHandle(paneId, terminalId)` as today.
|
||||
4. **Timeout** → the launcher **closes the pane it started** (and its tab, via the same path
|
||||
`release`/`stop` uses) and throws **`PeerUnreachableException`** (new, in `dev.ltms.bridged.peer`).
|
||||
`release`/`stop` uses) and throws **`PeerUnreachableException`** (new, in `dev.ltms.fleet.peer`).
|
||||
No orphan pane is left behind — the launcher cleans up its own failed birth.
|
||||
|
||||
`spawnReadyTimeoutMs == 0` (or unset) **disables** the gate = legacy non-blocking behaviour, so the
|
||||
@@ -79,8 +79,8 @@ Unit tests (add to the existing `ClaudeCodeLauncher` test):
|
||||
|
||||
## 5. Config
|
||||
|
||||
Add to the launcher-level config (a bridged-level knob, not per-profile) in `bridged.yaml` +
|
||||
`BridgedConfig`:
|
||||
Add to the launcher-level config (a fleetd-level knob, not per-profile) in `fleetd.yaml` +
|
||||
`FleetConfig`:
|
||||
|
||||
```yaml
|
||||
spawn_ready_timeout_ms: 20000 # 0 disables the gate (legacy non-blocking spawn)
|
||||
@@ -88,7 +88,7 @@ spawn_ready_poll_ms: 300
|
||||
```
|
||||
|
||||
Jackson ignores unknown keys, so omitting them in existing YAML is safe; pick sane defaults in code
|
||||
(`20000` / `300`). Keep the names consistent with existing config field style in `BridgedConfig`.
|
||||
(`20000` / `300`). Keep the names consistent with existing config field style in `FleetConfig`.
|
||||
|
||||
## 6. Core / MCP propagation
|
||||
|
||||
@@ -99,11 +99,11 @@ Jackson ignores unknown keys, so omitting them in existing YAML is safe; pick sa
|
||||
`spawn` throws — verify the new exception flows through it (worktree removed, nothing registered).
|
||||
- The **non-worktree** path registers the session only *after* `spawn` returns, so a throw means no
|
||||
half-live `SPAWNING` session is ever registered — confirm this and add a test.
|
||||
- `bridge_spawn` (MCP verb) must return an **error result** carrying the exception message, not a
|
||||
success with a dead session. Trace `BridgeMcp`/`BridgedApp` spawn handlers and make sure the
|
||||
- `fleet_spawn` (MCP verb) must return an **error result** carrying the exception message, not a
|
||||
success with a dead session. Trace `FleetMcp`/`FleetApp` spawn handlers and make sure the
|
||||
exception becomes a clean tool error, not an uncaught 500 with a stack trace.
|
||||
|
||||
**Out of scope (do NOT do here):** gating `bridge_send` on session `READY` (existing status-gate +
|
||||
**Out of scope (do NOT do here):** gating `fleet_send` on session `READY` (existing status-gate +
|
||||
this spawn gate already close the window), MCP-handshake-as-readiness signal, the CB-307 broker,
|
||||
any config `kind:` discriminator, any second adapter.
|
||||
|
||||
@@ -111,7 +111,7 @@ any config `kind:` discriminator, any second adapter.
|
||||
|
||||
- `ClaudeCodeLauncher.spawn` blocks until injectable or throws `PeerUnreachableException` +
|
||||
self-reaps the pane; gate disabled when timeout is 0.
|
||||
- New `PeerUnreachableException` in `dev.ltms.bridged.peer`.
|
||||
- New `PeerUnreachableException` in `dev.ltms.fleet.peer`.
|
||||
- Config knobs wired (`spawn_ready_timeout_ms`, `spawn_ready_poll_ms`) with safe defaults.
|
||||
- Existing `SPAWNING→READY` MCP-contact transition untouched.
|
||||
- New unit tests (ready / timeout+reap / disabled) green; **all existing tests still pass unchanged**.
|
||||
|
||||
@@ -13,28 +13,28 @@ Stage 2 (the AMQP/LavinMQ adapter behind the same port) is explicitly **out of s
|
||||
## 1. The bug this fixes (grounded in current code)
|
||||
|
||||
The reverse (worker→primary) path is `Rendezvous` — a `ConcurrentHashMap<session, CompletableFuture<Resolution>>`
|
||||
of **live blocking waiters only**. No queue, no store. When a worker calls `bridge_reply` and **no send
|
||||
of **live blocking waiters only**. No queue, no store. When a worker calls `fleet_reply` and **no send
|
||||
is currently open** for that worker:
|
||||
|
||||
- `Rendezvous.resolve(session, content)` → `complete(...)` → `waiters.get(session) == null` →
|
||||
returns `false` (`msg/Rendezvous.java:212-215`).
|
||||
- The `content` string is **never retained** — it is dropped. The worker is told it failed:
|
||||
`BridgeMcp.reply` returns `error("no send is awaiting a reply for this worker")` (`mcp/BridgeMcp.java:270-272`);
|
||||
REST returns `409 no_pending_send` (`rest/BridgedApp.java:339-345`).
|
||||
`FleetMcp.reply` returns `error("no send is awaiting a reply for this worker")` (`mcp/FleetMcp.java:270-272`);
|
||||
REST returns `409 no_pending_send` (`rest/FleetApp.java:339-345`).
|
||||
|
||||
This is the observed "communication break": a worker that finishes just after its `bridge_send` timed
|
||||
This is the observed "communication break": a worker that finishes just after its `fleet_send` timed
|
||||
out (the ~60s sync window) replies into the void. There is **no message-id, dedup, or ack** anywhere in
|
||||
the message path today.
|
||||
|
||||
## 2. What to build
|
||||
|
||||
### 2.1 The port — `dev.ltms.bridged.msg.ReplyInbox`
|
||||
### 2.1 The port — `dev.ltms.fleet.msg.ReplyInbox`
|
||||
|
||||
A thin interface owned by the `msg` layer. The in-memory adapter is Stage 1; the AMQP adapter (Stage 2)
|
||||
implements the **same** interface, so keep it broker-agnostic.
|
||||
|
||||
```java
|
||||
package dev.ltms.bridged.msg;
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
@@ -70,7 +70,7 @@ public interface ReplyInbox {
|
||||
seen-set — your call; preserve insertion order).
|
||||
- `peek` returns an immutable copy; `ack` removes by `msgId`. Thread-safe (concurrent publish vs. drain).
|
||||
- **This is soft-state, NOT persistence.** Lost on a `java -jar` bounce — that is correct and consistent
|
||||
with "bridged stays soft-state." Do **not** add any file/DB backing.
|
||||
with "fleetd stays soft-state." Do **not** add any file/DB backing.
|
||||
|
||||
### 2.3 Publish seam — route reply through the service layer
|
||||
|
||||
@@ -79,7 +79,7 @@ in `MessageService`, which already owns the `Rendezvous` and will own the `Reply
|
||||
|
||||
- Add `MessageService.reply(String session, String content)`:
|
||||
```java
|
||||
/** Route a worker's explicit bridge_reply: resolve an open send, or queue it in the inbox if none. */
|
||||
/** Route a worker's explicit fleet_reply: resolve an open send, or queue it in the inbox if none. */
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
return true; // a live send took it — unchanged fast path
|
||||
@@ -89,12 +89,12 @@ in `MessageService`, which already owns the `Rendezvous` and will own the `Reply
|
||||
}
|
||||
```
|
||||
- Repoint the two callers off the bare `rendezvous.resolve(...)` onto `messages.reply(...)`:
|
||||
- `BridgeMcp.reply` (`mcp/BridgeMcp.java:262-273`) — on success return a normal ack; **remove** the
|
||||
- `FleetMcp.reply` (`mcp/FleetMcp.java:262-273`) — on success return a normal ack; **remove** the
|
||||
`error("no send is awaiting a reply…")` branch (that case is now a successful queue).
|
||||
- `BridgedApp.replyMessage` (`rest/BridgedApp.java:330-346`) — return `200` (queued) instead of
|
||||
- `FleetApp.replyMessage` (`rest/FleetApp.java:330-346`) — return `200` (queued) instead of
|
||||
`409 no_pending_send`.
|
||||
|
||||
**DO NOT touch the QUESTION path.** `bridge_ask` / `rendezvous.resolveQuestion` must keep today's
|
||||
**DO NOT touch the QUESTION path.** `fleet_ask` / `rendezvous.resolveQuestion` must keep today's
|
||||
`NO_WAITER` behaviour — a mid-turn question is **interactive** (the worker blocks synchronously and cannot
|
||||
consume a late answer), so it must **never** be queued. Only terminal `REPLY`s go to the inbox.
|
||||
|
||||
@@ -110,27 +110,27 @@ The primary re-checks a worker it delegated to. Expose a drain keyed by **worker
|
||||
- Add `MessageService.drainReplies(String target)`: `peek` the inbox, `ack` each returned `msgId`, hand
|
||||
back the `List<InboxMessage>` (or just the contents). At-least-once: peek→deliver→ack (ack only after
|
||||
the caller has them, so an in-flight failure re-surfaces them).
|
||||
- **DECISION (required default): expose via the existing poll verb, keyed by target.** Extend `bridge_poll`
|
||||
- **DECISION (required default): expose via the existing poll verb, keyed by target.** Extend `fleet_poll`
|
||||
to accept an optional `target` (worker session) and, when present, return that worker's drained replies —
|
||||
alongside a matching REST route `GET /sessions/{id}/replies`. Do **not** change `send`/`answer` semantics
|
||||
(do not drain inside `send` — that conflates "deliver to worker" with "collect its mail"). Keep the
|
||||
existing ticket-based `bridge_poll(ticket)` path working unchanged. If you see a cleaner surface, still
|
||||
existing ticket-based `fleet_poll(ticket)` path working unchanged. If you see a cleaner surface, still
|
||||
ship this default and note the alternative for review.
|
||||
|
||||
## 3. Config
|
||||
|
||||
**None for Stage 1.** The in-memory adapter is the unconditional default — wire `new InMemoryReplyInbox()`
|
||||
into `MessageService` in `Bridged.main`. Do **not** add a `broker:` config block (that arrives with the
|
||||
into `MessageService` in `Fleetd.main`. Do **not** add a `broker:` config block (that arrives with the
|
||||
Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 4. Acceptance criteria (what the primary will verify)
|
||||
|
||||
1. New `ReplyInbox` + `InboxMessage` + `InMemoryReplyInbox` in `dev.ltms.bridged.msg`.
|
||||
2. `bridge_reply` with **no open send** now **succeeds and queues** (no more `error` / `409`); the reply is
|
||||
1. New `ReplyInbox` + `InboxMessage` + `InMemoryReplyInbox` in `dev.ltms.fleet.msg`.
|
||||
2. `fleet_reply` with **no open send** now **succeeds and queues** (no more `error` / `409`); the reply is
|
||||
later retrievable and identical.
|
||||
3. The queued reply is drainable by the primary keyed by target; draining **acks** it (a second drain
|
||||
returns nothing); dedup by `msgId` (re-publishing the same id does not double-queue).
|
||||
4. **QUESTION path unchanged** — `bridge_ask` with no open send still returns `NO_WAITER` (add/keep a test
|
||||
4. **QUESTION path unchanged** — `fleet_ask` with no open send still returns `NO_WAITER` (add/keep a test
|
||||
proving a question is never queued).
|
||||
5. Completion/failure fallbacks unchanged.
|
||||
6. Unit tests covering: `InMemoryReplyInbox` publish/peek/ack/dedup/FIFO/concurrency; `MessageService.reply`
|
||||
@@ -140,7 +140,7 @@ Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 5. Build & verification (worker side)
|
||||
|
||||
- Build with Maven from the worktree's `bridged/` dir. **Capture the exit code without a masking pipe**
|
||||
- Build with Maven from the worktree's `fleetd/` dir. **Capture the exit code without a masking pipe**
|
||||
(`mvn clean install; echo "MVN_EXIT=$?"` — never `mvn … | tail`, which hides failures).
|
||||
- Read the real test totals from `target/surefire-reports/TEST-*.xml`, not from stdout scroll.
|
||||
- You do **not** have IDE MCP access — do not claim `ide_diagnostics` results. The **primary** runs the
|
||||
@@ -152,7 +152,7 @@ Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
- **`.mcp.json` is `--skip-worktree` in your worktree — never edit, `git add`, or commit it.**
|
||||
- **`wiki/` is a submodule — never run git in it; never touch it.**
|
||||
- Commit only your feature changes (the new port/adapter, the `msg`/`mcp`/`rest` wiring, tests, and if
|
||||
you add config wiring in `Bridged.java`). Nothing else.
|
||||
you add config wiring in `Fleetd.java`). Nothing else.
|
||||
- Work only inside your assigned worktree on your feature branch. The primary fast-forwards `main` after
|
||||
re-gating — do not touch `main`.
|
||||
- Java 25 idioms are welcome (unnamed `_` params, records). Keep the diff minimal and match surrounding style.
|
||||
@@ -171,7 +171,7 @@ removal, msgId dedup, cross-restart redelivery). It is `@Tag("contract")`, so th
|
||||
untouched. Run it explicitly when Docker (or a broker) is available:
|
||||
|
||||
```bash
|
||||
cd bridged
|
||||
cd fleetd
|
||||
mvn -Pcontract test -Dtest=AmqpReplyInboxContractTest # local: spins a RabbitMQ Testcontainers fixture
|
||||
```
|
||||
|
||||
|
||||
@@ -18,7 +18,7 @@ The design rests on three pieces (the shape this ticket proposes):
|
||||
1. **Dedicated per-agent channels** — every agent has its own addressable inbox on the broker.
|
||||
2. **A federated agent directory** — a global "who/where/status" lookup, assembled from per-host
|
||||
presence, not a central database.
|
||||
3. **A per-host gateway** — each host runs a `bridged` that owns its local herdr, registers/manages
|
||||
3. **A per-host gateway** — each host runs a `fleetd` that owns its local herdr, registers/manages
|
||||
its own sessions, and proxies messages to/from other hosts over the broker.
|
||||
|
||||
## 2. What is single-host today (the assumptions to break)
|
||||
@@ -27,7 +27,7 @@ The design rests on three pieces (the shape this ticket proposes):
|
||||
flowchart TB
|
||||
subgraph host["Single host (today)"]
|
||||
primary["primary<br/>(MCP client)"]
|
||||
daemon["bridged daemon<br/>127.0.0.1:8765"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765"]
|
||||
reg["in-process registry<br/>keyed by PeerHandle.id() == paneId"]
|
||||
herdr["herdr<br/>(local unix-socket PTY mux)"]
|
||||
w1["worker pane wQ:p1"]
|
||||
@@ -48,21 +48,21 @@ Three concrete bake-ins assume one host:
|
||||
|---|---|---|
|
||||
| **herdr is local** | `herdr/` unix socket `~/.config/herdr/herdr.sock` | You cannot drive another host's PTYs → each host **must** own its herdr. This is why a per-host gateway is mandatory. |
|
||||
| **registry is in-process, keyed by `paneId`** | `session/SessionManager` | `paneId` (e.g. `wQ:p2B`) is a herdr-local coordinate — meaningless off-host. Routing needs a host-unique id. |
|
||||
| **loopback, no authn** | `rest/BridgedApp` binds `127.0.0.1:8765` | Fine on one host; the moment a second host can talk to a gateway, that link is a trust boundary. |
|
||||
| **loopback, no authn** | `rest/FleetApp` binds `127.0.0.1:8765` | Fine on one host; the moment a second host can talk to a gateway, that link is a trust boundary. |
|
||||
|
||||
## 3. Target architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A"]
|
||||
gA["gateway = bridged A"]
|
||||
gA["gateway = fleetd A"]
|
||||
regA["local registry + herdr"]
|
||||
primary["primary (MCP client)"]
|
||||
gA --- regA
|
||||
primary --- gA
|
||||
end
|
||||
subgraph hostB["HOST B"]
|
||||
gB["gateway = bridged B"]
|
||||
gB["gateway = fleetd B"]
|
||||
regB["local registry + herdr"]
|
||||
wb["worker panes"]
|
||||
gB --- regB
|
||||
@@ -96,13 +96,13 @@ host's terminals.*
|
||||
delayed-message exchange (the remind/backoff loop for free) — the same reasons CB-307 picked it.
|
||||
|
||||
- **Federated agent directory** = a **soft-state, bridge-owned** roster, *not* a broker-stored
|
||||
database. Per the persistence-boundary decision (bridged is soft-state; the broker owns *message*
|
||||
database. Per the persistence-boundary decision (fleetd is soft-state; the broker owns *message*
|
||||
durability, not *who/where/status*), each gateway announces its local agents `(globalId, host,
|
||||
status, capabilities)` on a `roster.*` presence topic with periodic heartbeats. Every gateway
|
||||
builds an eventually-consistent **union view** — literally CB-304's `rosterView`, federated. A
|
||||
stale entry expires by missed heartbeat (reuses CB-303's idle/TTL thinking).
|
||||
|
||||
- **Per-host gateway** = today's `bridged` daemon, evolved. It already registers/manages sessions
|
||||
- **Per-host gateway** = today's `fleetd` daemon, evolved. It already registers/manages sessions
|
||||
and controls its local herdr; multi-host adds exactly two responsibilities: (a) a broker client
|
||||
that consumes its agents' inboxes and injects into local herdr, and (b) presence announce +
|
||||
union-roster assembly. Evolution, not rewrite.
|
||||
@@ -111,7 +111,7 @@ host's terminals.*
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
send["bridge_send(globalId, msg)"] --> lookup{"directory:<br/>is globalId local?"}
|
||||
send["fleet_send(globalId, msg)"] --> lookup{"directory:<br/>is globalId local?"}
|
||||
lookup -->|"yes"| local["inject via local herdr<br/>(today's Injector path)"]
|
||||
lookup -->|"no"| pub["publish agent.<id>.inbox<br/>(broker routes to owning gateway)"]
|
||||
pub --> consume["owning gateway consumes<br/>→ injects into its local herdr"]
|
||||
@@ -123,7 +123,7 @@ broker. A sender is oblivious to which branch it took.*
|
||||
## 4. What CB-307 already provides vs. what is net-new
|
||||
|
||||
**CB-307 delivers the transport half** and is independently valuable on a single host: the AMQP
|
||||
broker fabric, the `bridged → broker` client/adapter, at-least-once + idempotent (dedup-by-id)
|
||||
broker fabric, the `fleetd → broker` client/adapter, at-least-once + idempotent (dedup-by-id)
|
||||
delivery, DLQ, and delayed-retry (remind). That *is* the "proxy cross-host message" backbone;
|
||||
extending the same broker from "worker→primary reliability" to "gateway↔gateway" is incremental.
|
||||
|
||||
@@ -158,17 +158,17 @@ sequenceDiagram
|
||||
participant BR as broker
|
||||
participant GA as gateway A
|
||||
participant P as primary (host A, MCP client)
|
||||
W->>GB: bridge_reply
|
||||
W->>GB: fleet_reply
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route to A's primary inbox
|
||||
Note over GA: held durably until the primary pulls
|
||||
P->>GA: blocking bridge_send resolves / bridge_poll
|
||||
P->>GA: blocking fleet_send resolves / fleet_poll
|
||||
GA-->>P: reply (then ACK to broker)
|
||||
```
|
||||
|
||||
*Figure 4 — the broker makes the middle hop lossless, ordered, and idempotent; the **final** hop
|
||||
into the primary is still a **pull** (gateway A holds the message until the primary's blocking
|
||||
`bridge_send` or `bridge_poll`). Cross-host neither improves nor worsens this — it just spans hosts.
|
||||
`fleet_send` or `fleet_poll`). Cross-host neither improves nor worsens this — it just spans hosts.
|
||||
This is precisely the gap CB-307 closes on one host and CB-308 stretches across hosts.*
|
||||
|
||||
## 6. Staging & dependencies
|
||||
@@ -207,8 +207,8 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
gateway's host. This extends the single-host invariant — *identity comes from the connection,
|
||||
never an argument* — across the broker: cross-host, identity comes from the key. Complements
|
||||
(not replaces) per-gateway broker logins over TLS.
|
||||
2. **Profiles are owned by the worker's host.** `bridge_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `bridged.yaml`. Gateways advertise their profile names in presence
|
||||
2. **Profiles are owned by the worker's host.** `fleet_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `fleetd.yaml`. Gateways advertise their profile names in presence
|
||||
heartbeats, so a leader sees what each host offers before spawning; an unknown name is a clear
|
||||
error from the target. Secrets (base URLs, tokens) never leave the host that uses them.
|
||||
3. **Repo provisioning — clone from the forge, pinned.** A cross-host spawn names the repo URL and
|
||||
@@ -259,7 +259,7 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
serialize acks fleet-wide. Ordering caveat: a *return* (unroutable) arrives **before** the
|
||||
confirm, so "confirmed" ≠ "routed"; the sender checks the returned-set at confirm time.
|
||||
`mandatory` is false only for `BROADCAST`, where an empty group is legal silence.
|
||||
10. **Queue lifecycle is session lifecycle.** `bridge_stop`/reap deletes the worker's inbox queue
|
||||
10. **Queue lifecycle is session lifecycle.** `fleet_stop`/reap deletes the worker's inbox queue
|
||||
(its `broadcast.*` bindings die with it — no broadcasts to the dead); `x-expires` collects
|
||||
queues orphaned by a crashed gateway (long for main/orchestrator inboxes, short for workers).
|
||||
Queue names carry a version suffix (`.v2`): AMQP refuses to redeclare an existing durable
|
||||
@@ -275,11 +275,11 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
- **Gateway discovery:** how gateways find the broker and each other (static config vs. discovery).
|
||||
- **Control authorization — THE GATE ON U4.** Signing (§7.1) settles *who sent it*; authorization
|
||||
is *who may do what*. **Cross-host spawn must not land before the minimal version exists**: a
|
||||
per-host allowlist in `bridged.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
per-host allowlist in `fleetd.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
publish control to this host, checked against the verified signature. A few lines of config and
|
||||
check; without them, any principal holding broker credentials can start processes on every host
|
||||
in the fleet.
|
||||
- **Key distribution & rotation:** static config (host → public key in each `bridged.yaml`) is
|
||||
- **Key distribution & rotation:** static config (host → public key in each `fleetd.yaml`) is
|
||||
fine at the current 2–3 host scale; rotation is manual. A refinement, not a blocker.
|
||||
- **Gateway death mid-turn:** the roster reaps it by missed heartbeat, and in-flight primary-bound
|
||||
messages survive by broker durability; still open is reconciling *worker* state when the dead
|
||||
|
||||
@@ -7,7 +7,7 @@ the core learning any one peer's environment.
|
||||
## 1. Why
|
||||
|
||||
`claude-bridge` is a **communication bus between heterogeneous AI agents** — its stable surface is
|
||||
the protocol (`bridge_spawn / send / poll / reply / ask / list / stop / status`), and that surface
|
||||
the protocol (`fleet_spawn / send / poll / reply / ask / list / stop / status`), and that surface
|
||||
should stay provider-neutral. Today the daemon can only materialize one kind of peer: an
|
||||
off-subscription Claude Code CLI over herdr. Everything specific to *how that peer is set up*
|
||||
(`ANTHROPIC_BASE_URL`, the subscription guard, `--mcp-config`/system-prompt flags, `claude-*`
|
||||
@@ -60,7 +60,7 @@ Where Claude/herdr specifics actually live today:
|
||||
| `claude-<profile>-<nonce>-<seq>` naming, `WORKER_NAME` regex, orphan reap (CB-117) | `WorkerService` | **→ adapter** (naming is a herdr-label detail) |
|
||||
| `GITEA_TOKEN` / `GITEA_HOST` injection (CB-302 checkpoint) | `WorkerService.spawn` | **→ adapter** + a **capability** (§6) |
|
||||
| tab/pane placement, worker space, tab labels | `WorkerService.spawnInTab/spawnAsPane` via herdr `WorkspaceControl` | **→ adapter** (herdr transport detail) |
|
||||
| `BridgedConfig.Worker` profile shape (`baseUrl`, `model`, `configDir`, …) | `config` | **mostly adapter-shaped** — see §5 |
|
||||
| `FleetConfig.Worker` profile shape (`baseUrl`, `model`, `configDir`, …) | `config` | **mostly adapter-shaped** — see §5 |
|
||||
| FSM, registry, roster, `reapIdle`/`drainAll`/`contextCap`, `rosterView` | `SessionManager` | **stays core** |
|
||||
| turn/completion detection (`TurnListener`, `CompletionResolver`, `StatusPoller`, `WorkerPresence`) | `inject/` | **stays core**, but reads herdr terminal output → transport-coupled (§4b) |
|
||||
| message store & routing | `msg/` | **stays core** |
|
||||
@@ -118,7 +118,7 @@ hard-wire "turns come from herdr".
|
||||
|
||||
## 5. Config shape
|
||||
|
||||
`BridgedConfig.Worker` is Claude-shaped (`baseUrl`, `model`, `configDir`, `tokenEnv`). Rather than
|
||||
`FleetConfig.Worker` is Claude-shaped (`baseUrl`, `model`, `configDir`, `tokenEnv`). Rather than
|
||||
break existing YAML, CB-401 keeps `workers:` exactly as-is and treats those fields as the
|
||||
**ClaudeCodeLauncher's** profile schema. A future peer kind adds a `kind:` discriminator
|
||||
(default `"claude-code"`) selecting the launcher; unknown-kind → clear config error. No migration of
|
||||
@@ -126,7 +126,7 @@ existing configs. (Jackson already ignores unknown keys, so adding `kind` is bac
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MCP as bridge_spawn (MCP/REST)
|
||||
participant MCP as fleet_spawn (MCP/REST)
|
||||
participant SM as SessionManager
|
||||
participant L as PeerLauncher (by profile.kind)
|
||||
participant T as transport (herdr)
|
||||
@@ -150,7 +150,7 @@ degrades gracefully when a launcher lacks one:
|
||||
|
||||
| Capability | Meaning | Claude Code | Codex (likely) | Human |
|
||||
|---|---|---|---|---|
|
||||
| `MID_TURN_ASK` | supports `bridge_ask` rendezvous | ✓ | ? | ✗ |
|
||||
| `MID_TURN_ASK` | supports `fleet_ask` rendezvous | ✓ | ? | ✗ |
|
||||
| `SELF_PR` | can open its own PR at checkpoint (CB-302) | ✓ (opt-in token) | ? | ✗ |
|
||||
| `WORKTREE` | can run in a provisioned git worktree | ✓ | ✓ | ✗ |
|
||||
| `ORPHAN_REAP` | spawner can reconcile orphaned peers on boot | ✓ | ? | ✗ |
|
||||
@@ -185,13 +185,13 @@ messages, not verified facts.
|
||||
Deliverable for CB-401 Stage A — mechanical, behaviour-preserving:
|
||||
|
||||
1. `PeerLauncher` interface + `PeerHandle` (opaque id) + `SpawnRequest` (profile, requestedCwd,
|
||||
callerCwd) + `Capability` enum, new package `dev.ltms.bridged.peer`.
|
||||
callerCwd) + `Capability` enum, new package `dev.ltms.fleet.peer`.
|
||||
2. `ClaudeCodeLauncher implements PeerLauncher` = today's `WorkerService`, adapted: `spawn(...)`
|
||||
returns a `PeerHandle` (id = paneId), `capabilities()` declares
|
||||
`MID_TURN_ASK, SELF_PR(when token), WORKTREE, ORPHAN_REAP`.
|
||||
3. `SessionManager` depends on `PeerLauncher`, not `WorkerService` concretely; routing keys on
|
||||
`PeerHandle.id()` (== paneId today, so zero value change).
|
||||
4. `Bridged.main` wires the concrete `ClaudeCodeLauncher` behind the interface.
|
||||
4. `Fleetd.main` wires the concrete `ClaudeCodeLauncher` behind the interface.
|
||||
5. **No behaviour change, no config change.** Full green gate: `ide_sync` → `ide_diagnostics`
|
||||
(0 errors/0 warnings) → `mvn clean install` with `MVN_EXIT` captured (no masking pipe). All
|
||||
existing tests pass unchanged; add tests only for the new `PeerHandle` indirection.
|
||||
|
||||
@@ -11,7 +11,7 @@ All five increments of §4 are done, including increment 5 (the §5 live checkli
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Prove the [`PeerLauncher`](../bridged/src/main/java/dev/ltms/bridged/peer/PeerLauncher.java) SPI
|
||||
Prove the [`PeerLauncher`](../fleetd/src/main/java/dev/ltms/fleet/peer/PeerLauncher.java) SPI
|
||||
actually holds for a **non-Claude** coding agent by shipping a second, first-class in-tree
|
||||
adapter: **opencode** (`opencode` 1.1.31, a provider-agnostic terminal coding agent).
|
||||
|
||||
@@ -56,8 +56,8 @@ flowchart TB
|
||||
|
||||
*Figure 1 — the two concerns tangled inside today's single launcher; CB-402 splits them.*
|
||||
|
||||
There is also a **Stage-A deferral** to finish: `Bridged.main` still casts
|
||||
`(ClaudeCodeLauncher) workers` at the `BridgeMcp` and `BridgedApp` constructors. Those two
|
||||
There is also a **Stage-A deferral** to finish: `Fleetd.main` still casts
|
||||
`(ClaudeCodeLauncher) workers` at the `FleetMcp` and `FleetApp` constructors. Those two
|
||||
callers only invoke `profiles()`, `defaultProfile()`, and `list()` — **all already on the
|
||||
`PeerLauncher` interface**. The cast survives for one reason only: `PeerLauncher.list()`
|
||||
returns `List<?>` (element type erased) while the callers use `Agent` element methods in their
|
||||
@@ -68,7 +68,7 @@ roster join. Finishing the migration is therefore small and contained (§4.D).
|
||||
## 3. Target design
|
||||
|
||||
Template-Method base + two thin adapters + a routing composite that keeps the Stage-A seam
|
||||
(one `PeerLauncher` reference held by `SessionManager` / `BridgeMcp` / `BridgedApp`) intact.
|
||||
(one `PeerLauncher` reference held by `SessionManager` / `FleetMcp` / `FleetApp`) intact.
|
||||
|
||||
```mermaid
|
||||
flowchart TB
|
||||
@@ -114,7 +114,7 @@ Two adapter **hooks** (abstract):
|
||||
protected abstract String namePrefix(); // "claude" | "opencode"
|
||||
|
||||
/** Build the peer-specific launch: env map + argv. Runs any pre-spawn guard here. */
|
||||
protected abstract Launch buildLaunch(BridgedConfig.Worker cfg, SpawnRequest req);
|
||||
protected abstract Launch buildLaunch(FleetConfig.Worker cfg, SpawnRequest req);
|
||||
record Launch(Map<String,String> env, List<String> argv) {}
|
||||
```
|
||||
|
||||
@@ -126,7 +126,7 @@ so the opencode adapter never reaps a `claude-*` pane and vice-versa. The compos
|
||||
|
||||
### B. `kind:` config discriminator
|
||||
|
||||
Add one field to `BridgedConfig.Worker`:
|
||||
Add one field to `FleetConfig.Worker`:
|
||||
|
||||
```java
|
||||
String kind // "claude-code" (default) | "opencode"
|
||||
@@ -138,7 +138,7 @@ String kind // "claude-code" (default) | "opencode"
|
||||
so the record stays declarative.)
|
||||
- Keep the existing back-compat constructors; `kind` is additive and optional.
|
||||
|
||||
`bridged.example.yaml` documents a two-kind `workers:` block.
|
||||
`fleetd.example.yaml` documents a two-kind `workers:` block.
|
||||
|
||||
### C. `OpenCodeLauncher` — the adapter hooks for opencode
|
||||
|
||||
@@ -164,13 +164,13 @@ String kind // "claude-code" (default) | "opencode"
|
||||
|
||||
### D. `CompositePeerLauncher` + finish the Stage-A migration
|
||||
|
||||
- `Bridged.main` groups configured profiles by `kind`, instantiates one launcher per kind
|
||||
- `Fleetd.main` groups configured profiles by `kind`, instantiates one launcher per kind
|
||||
present, and wraps them in `CompositePeerLauncher implements PeerLauncher`.
|
||||
- Routing methods (`spawn(req)`, `effectiveCwd(req)`, `parityOverlay(name)`) dispatch by the
|
||||
profile's kind. Fan-out methods (`list()`, `reapOrphanWorkers()`, `capabilities()`,
|
||||
`profiles()`, `defaultProfile()`) merge across sub-launchers. `stop(id)` tries each (teardown
|
||||
only knows the pane id) — already best-effort/idempotent.
|
||||
- **Migrate `BridgeMcp` + `BridgedApp` to the `PeerLauncher` interface**, dropping both
|
||||
- **Migrate `FleetMcp` + `FleetApp` to the `PeerLauncher` interface**, dropping both
|
||||
`(ClaudeCodeLauncher)` casts. Only friction is `list()`'s `List<?>`; resolve by giving the SPI
|
||||
a typed roster element (small neutral `PeerAgent` view exposing `id()`/`name()`/status) that
|
||||
the CB-304 roster join consumes — or, minimally, narrow at the callsite. Prefer the typed view.
|
||||
@@ -179,12 +179,12 @@ String kind // "claude-code" (default) | "opencode"
|
||||
sequenceDiagram
|
||||
autonumber
|
||||
participant P as Primary
|
||||
participant M as BridgeMcp / REST
|
||||
participant M as FleetMcp / REST
|
||||
participant C as CompositePeerLauncher
|
||||
participant O as OpenCodeLauncher
|
||||
participant B as HerdrPeerLauncher (base)
|
||||
participant H as herdr
|
||||
P->>M: bridge_spawn(profile="oc-impl")
|
||||
P->>M: fleet_spawn(profile="oc-impl")
|
||||
M->>C: spawn(SpawnRequest)
|
||||
C->>C: kind(profile)=="opencode"
|
||||
C->>O: spawn(req)
|
||||
@@ -208,7 +208,7 @@ sequenceDiagram
|
||||
`ClaudeCodeLauncher` extend it with `namePrefix()="claude"` and `buildLaunch()` wrapping
|
||||
today's guard+env+argv logic. Green build, identical tests — pure refactor. *(IDE
|
||||
`refactor` where possible; the primary re-runs the gate workers can't.)*
|
||||
2. **`kind:` discriminator.** Add the field + normalization + `bridged.example.yaml`. Default
|
||||
2. **`kind:` discriminator.** Add the field + normalization + `fleetd.example.yaml`. Default
|
||||
path unchanged (`kind=claude-code`).
|
||||
3. **`OpenCodeLauncher`.** Implement the three hooks; unit-test `buildLaunch` (env has no
|
||||
`ANTHROPIC_BASE_URL`; `OPENCODE_CONFIG` points at a file carrying the bridge MCP block +
|
||||
@@ -226,11 +226,11 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
- **opencode TUI ⇄ herdr injection.** herdr drives a pane by typing into a TUI. Must confirm
|
||||
opencode's TUI accepts injected keystrokes/submit the way `claude` does, and reaches an
|
||||
`injectable` status the CB-306 gate recognizes. *Validation:* spawn one opencode worker,
|
||||
watch the readiness gate pass, `bridge_send` a trivial task.
|
||||
watch the readiness gate pass, `fleet_send` a trivial task.
|
||||
- **Bridge MCP visibility in opencode.** Confirm `OPENCODE_CONFIG` (or `opencode mcp add`)
|
||||
actually surfaces the `bridge_*` tools inside the opencode session, and that `bridge_reply`
|
||||
actually surfaces the `fleet_*` tools inside the opencode session, and that `fleet_reply`
|
||||
is callable — the reply-charter is worthless if the tool isn't mounted. *Validation:* the
|
||||
worker completes a task by calling `bridge_reply`; the reply lands via the CB-307 path.
|
||||
worker completes a task by calling `fleet_reply`; the reply lands via the CB-307 path.
|
||||
- **opencode MCP/config schema drift.** opencode is fast-moving (1.1.31 today). Pin the config
|
||||
schema we generate against the installed version; treat the exact keys (`type: "remote"` vs
|
||||
`"http"`, `instructions` shape) as a dogfood-verified fact, not an assumption.
|
||||
@@ -244,7 +244,7 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
|
||||
- **Unit (hermetic):** base-extraction regression (existing `ClaudeCodeLauncher` tests pass
|
||||
unchanged); `OpenCodeLauncher.buildLaunch` env/argv/config assertions; `kind` normalization
|
||||
in `BridgedConfigTest`; `CompositePeerLauncher` routing + fan-out (merge of `profiles()`,
|
||||
in `FleetConfigTest`; `CompositePeerLauncher` routing + fan-out (merge of `profiles()`,
|
||||
summed `reapOrphanWorkers()`, per-kind reap isolation) with fake sub-launchers.
|
||||
- **Live (dogfood, manual):** the §5 checklist on the running daemon.
|
||||
- **Gate (primary):** IDE diagnostics 0/0 on every changed file, `mvn clean install` green with
|
||||
@@ -269,7 +269,7 @@ until it is "major" (Stage-B whole), per the CB-401 bar.
|
||||
|
||||
## 8. As-built — live dogfood (2026-07-29)
|
||||
|
||||
Run against `bridged` on `127.0.0.1:8766` at main `19cdf8d`, with opencode **1.18.5** installed
|
||||
Run against `fleetd` on `127.0.0.1:8766` at main `19cdf8d`, with opencode **1.18.5** installed
|
||||
via Homebrew. Every §5 risk is now a verified fact rather than an assumption.
|
||||
|
||||
**The version-drift risk was the real one, and it did not bite.** This adapter was designed against
|
||||
@@ -281,7 +281,7 @@ as a dogfood-verified fact for 1.18.5.
|
||||
| §5 risk | Result |
|
||||
|---|---|
|
||||
| opencode TUI ⇄ herdr injection; CB-306 gate | ✅ `peer pane=wD:p3 reached injectable state` ~0.6s after `agent.start` |
|
||||
| Bridge MCP visible + `bridge_reply` callable | ✅ MCP `initialize` from `Implementation[name=opencode, version=1.18.5]`; worker replied through the tool |
|
||||
| Bridge MCP visible + `fleet_reply` callable | ✅ MCP `initialize` from `Implementation[name=opencode, version=1.18.5]`; worker replied through the tool |
|
||||
| Config schema drift (1.1.31 → 1.18.5) | ✅ unchanged, see above |
|
||||
| Provider credentials | ✅ free tier, zero credentials |
|
||||
|
||||
@@ -291,7 +291,7 @@ Full lifecycle exercised through the REST surface:
|
||||
to `OpenCodeLauncher` (`spawning opencode profile=opencode-free`), pane `wD:p3`.
|
||||
2. Readiness: `{"ready":true,"status":"idle"}`, roster state `ready`.
|
||||
3. `POST /sessions/{id}/message` → **`{"replySource":"reply","reply":"391"}`** — a *structured*
|
||||
`bridge_reply`, not the CB-115 completion-fallback transcript scrape. The clean path.
|
||||
`fleet_reply`, not the CB-115 completion-fallback transcript scrape. The clean path.
|
||||
4. `DELETE /workers/wD:p3` → `204`, roster empty, tolerant teardown (`tab_not_found` ignored —
|
||||
opencode had already closed its own tab).
|
||||
|
||||
@@ -301,5 +301,5 @@ identity (loopback peer PID → herdr pane) classified an **opencode** process a
|
||||
opencode-specific handling — confirming the identity model is peer-kind-agnostic, which is exactly
|
||||
what CB-308 needs when it stretches the roster across hosts.
|
||||
|
||||
CB-502 counters for the same run: `bridged_sends_total{outcome="replied"} 1`,
|
||||
`bridged_replies_total{path="rendezvous"} 1`, `bridged_inbox_depth{...} 0`.
|
||||
CB-502 counters for the same run: `fleet_sends_total{outcome="replied"} 1`,
|
||||
`fleet_replies_total{path="rendezvous"} 1`, `fleet_inbox_depth{...} 0`.
|
||||
|
||||
@@ -38,14 +38,14 @@ never toolchain ownership (§7).
|
||||
flowchart TB
|
||||
human["human (types)"]
|
||||
primary["PRIMARY (Opus)<br/>MCP client — pull-only"]
|
||||
daemon["bridged daemon<br/>127.0.0.1:8765 (single host)"]
|
||||
daemon["fleetd daemon<br/>127.0.0.1:8765 (single host)"]
|
||||
comp["CompositePeerLauncher<br/>routes by kind"]
|
||||
cc["ClaudeCodeLauncher"]
|
||||
oc["OpenCodeLauncher"]
|
||||
w1["worker pane (gx00 vLLM)"]
|
||||
w2["worker pane (ollama)"]
|
||||
human --> primary
|
||||
primary -->|"bridge_send / spawn / ask"| daemon
|
||||
primary -->|"fleet_send / spawn / ask"| daemon
|
||||
daemon --> comp
|
||||
comp --> cc
|
||||
comp --> oc
|
||||
@@ -80,7 +80,7 @@ flowchart TB
|
||||
m1["main A: Opus<br/>MCP client"]
|
||||
m2["main B: cloud module<br/>MCP client"]
|
||||
end
|
||||
subgraph bus["bridged fabric (CB-307/308 substrate)"]
|
||||
subgraph bus["fleetd fabric (CB-307/308 substrate)"]
|
||||
chan["per-agent inbox channels<br/>agent.<globalId>.inbox"]
|
||||
roster["federated roster (union view)"]
|
||||
end
|
||||
@@ -146,11 +146,11 @@ env-manager" rule (§7).*
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant M as main (delegator)
|
||||
participant D as bridged
|
||||
participant D as fleetd
|
||||
participant SL as SandboxLauncher
|
||||
participant SB as sandbox (peer-owned)
|
||||
participant A as agent in sandbox
|
||||
M->>D: bridge_spawn(profile=backend, role=backend)
|
||||
M->>D: fleet_spawn(profile=backend, role=backend)
|
||||
D->>SL: spawn(SpawnRequest)
|
||||
SL->>SB: start entrypoint (image = backend role)
|
||||
Note over SL,SB: bridge injects guarded ANTHROPIC_BASE_URL,<br/>mounts bridge MCP url + reply charter
|
||||
@@ -189,7 +189,7 @@ flowchart TB
|
||||
m2["main B: cloud module<br/>MCP client (pull-only)"]
|
||||
end
|
||||
reg["PrimaryRegistry → multi-slot<br/>(terminal per main)"]
|
||||
subgraph fabric["bridged"]
|
||||
subgraph fabric["fleetd"]
|
||||
ca["agent.A.inbox"]
|
||||
cb["agent.B.inbox"]
|
||||
push["ReplyPushLoop → N terminals"]
|
||||
@@ -211,14 +211,14 @@ transport is already peer-neutral — what was missing is N pull endpoints).*
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant MA as main A (Opus)
|
||||
participant BR as bridged / broker
|
||||
participant BR as fleetd / broker
|
||||
participant MB as main B (cloud)
|
||||
MA->>BR: bridge_send(to = main B, msg)
|
||||
MA->>BR: fleet_send(to = main B, msg)
|
||||
BR->>BR: publish agent.B.inbox (durable, msg id)
|
||||
Note over BR: held until B pulls (B is a client too)
|
||||
MB->>BR: blocking bridge_send / poll resolves
|
||||
MB->>BR: blocking fleet_send / poll resolves
|
||||
BR-->>MB: msg (then ACK)
|
||||
MB->>BR: bridge_reply(to = main A)
|
||||
MB->>BR: fleet_reply(to = main A)
|
||||
BR->>BR: publish agent.A.inbox
|
||||
MA->>BR: poll resolves
|
||||
BR-->>MA: reply
|
||||
@@ -233,7 +233,7 @@ push-loop fan-out; relax "orchestration tools only the primary calls" to "any re
|
||||
## 6. Development C — Orchestrator tier
|
||||
|
||||
> **SUPERSEDED — do not implement this model.** The operator rejected a supervisor above the lead.
|
||||
> The human continues to drive the pre-existing lead directly; bridged neither spawns nor resumes that
|
||||
> The human continues to drive the pre-existing lead directly; fleetd neither spawns nor resumes that
|
||||
> lead. What replaced this proposal is **one lead, two short-lived advisory architects, and N workers**:
|
||||
> the lead engages architects sideways for a strong-model assessment, then discards them. Architect
|
||||
> slots are declared in `architects:` (see Gitea issue #16), rather than making leads managed sessions.
|
||||
@@ -243,7 +243,7 @@ push-loop fan-out; relax "orchestration tools only the primary calls" to "any re
|
||||
|
||||
> **Historical alternative retained.** The text and figures below record the considered model and why it
|
||||
> was rejected: it re-rooted the human-facing session above the lead, violating the still-true premise
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by bridged.
|
||||
> that configured leaders pre-exist, are recognised, and cannot be resumed by fleetd.
|
||||
|
||||
The orchestrator is **`SessionManager` recursed one tier up**: today it spawns/names/reaps *worker*
|
||||
sessions; the orchestrator does the same for *main* sessions, and adds **context scoping**.
|
||||
@@ -287,7 +287,7 @@ sequenceDiagram
|
||||
H->>O: high-level goal (large context)
|
||||
O->>O: name/resume main A session
|
||||
O->>MA: task + SCOPED context slice (not the whole history)
|
||||
MA->>W: bridge_send(delegation, carrying only the relevant slice)
|
||||
MA->>W: fleet_send(delegation, carrying only the relevant slice)
|
||||
W-->>MA: result
|
||||
MA-->>O: rollup
|
||||
O->>O: fold into orchestrator context, pick next main/turn
|
||||
@@ -385,7 +385,7 @@ ticket is an extension of an existing pattern (CB-402 for A, CB-308 for B/C).
|
||||
|
||||
The follow-up question — *"clarify the architecture when we have distributed agents in sandboxes"* —
|
||||
resolves the fork left open in §4 and §9. **Decision: Development A (sandbox launcher) and CB-308
|
||||
(per-host federation) *compose*, not compete — each host runs a `bridged` gateway whose launcher
|
||||
(per-host federation) *compose*, not compete — each host runs a `fleetd` gateway whose launcher
|
||||
spawns agents into that host's *local* sandboxes.** A sandbox is never reached across the network; it
|
||||
is reached by the gateway sitting next to it.
|
||||
|
||||
@@ -393,7 +393,7 @@ is reached by the gateway sitting next to it.
|
||||
|
||||
The bus delivers a turn by **herdr keystroke-injection** — `Injector → AgentControl.send` writes into
|
||||
a PTY that its **local** herdr owns. The broker moves *messages and presence*, **never keystrokes**.
|
||||
So an agent's PTY must live in a herdr that *some* `bridged` instance drives locally: a remote
|
||||
So an agent's PTY must live in a herdr that *some* `fleetd` instance drives locally: a remote
|
||||
container with no local herdr **cannot be injected into**. That rules out a central daemon reaching
|
||||
remote PTYs, and collapses the design to a single identity:
|
||||
|
||||
@@ -403,7 +403,7 @@ remote PTYs, and collapses the design to a single identity:
|
||||
flowchart TB
|
||||
subgraph hostA["HOST A — gateway"]
|
||||
mA["main / orchestrator<br/>MCP client → LOCAL gateway"]
|
||||
gA["bridged A<br/>herdr + CompositePeerLauncher<br/>(incl. SandboxLauncher)"]
|
||||
gA["fleetd A<br/>herdr + CompositePeerLauncher<br/>(incl. SandboxLauncher)"]
|
||||
cBEa["sandbox: backend<br/>(local container)"]
|
||||
cFEa["sandbox: frontend<br/>(local container)"]
|
||||
mA --- gA
|
||||
@@ -415,7 +415,7 @@ flowchart TB
|
||||
roster["roster.* (federated presence)"]
|
||||
end
|
||||
subgraph hostB["HOST B — gateway"]
|
||||
gB["bridged B<br/>herdr + SandboxLauncher"]
|
||||
gB["fleetd B<br/>herdr + SandboxLauncher"]
|
||||
cBEb["sandbox: backend<br/>(local container)"]
|
||||
gB -->|"spawn → PTY in B's herdr"| cBEb
|
||||
end
|
||||
@@ -440,13 +440,13 @@ sequenceDiagram
|
||||
participant BR as broker
|
||||
participant GB as gateway B
|
||||
participant SB as sandbox agent (host B, container)
|
||||
MA->>GA: bridge_send(globalId on B, msg)
|
||||
MA->>GA: fleet_send(globalId on B, msg)
|
||||
GA->>GA: directory lookup - is globalId local? NO
|
||||
GA->>BR: publish agent.ID.inbox (durable)
|
||||
BR->>GB: route to the owning gateway
|
||||
GB->>SB: inject via B's LOCAL herdr (keystrokes)
|
||||
Note over GB,SB: SandboxLauncher already spawned the container -<br/>its PTY is in B's herdr, CB-306 readiness passed
|
||||
SB-->>GB: bridge_reply (to B's LOCAL MCP endpoint)
|
||||
SB-->>GB: fleet_reply (to B's LOCAL MCP endpoint)
|
||||
GB->>BR: publish primary-bound (durable, msg id)
|
||||
BR->>GA: route back to A
|
||||
Note over GA: held until the main pulls (the main is a client)
|
||||
@@ -470,7 +470,7 @@ unchanged — the sandbox is transparent to it.*
|
||||
|
||||
| Concern | Provided by |
|
||||
|---|---|
|
||||
| Per-host gateway (owns local herdr + sessions) | **CB-308** (today's `bridged`, evolved) |
|
||||
| Per-host gateway (owns local herdr + sessions) | **CB-308** (today's `fleetd`, evolved) |
|
||||
| Spawn into a local sandbox / role→image | **Development A** `SandboxLauncher` (§4), routed by `CompositePeerLauncher` |
|
||||
| Addressing a remote sandboxed agent | **CB-308** global id + federated roster (host + role as metadata) |
|
||||
| Orphan reap after a gateway restart | **CB-117** per-gateway, summed by the composite — each reaps only its **local** herdr |
|
||||
|
||||
@@ -9,7 +9,7 @@ with no terminator, and our third-party members cannot detect it (§7.2).
|
||||
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
||||
|
||||
The gateway went live on 2026-08-15 and replaced Bifrost. This plan says what that means for a
|
||||
**member definition** in `bridged.yaml`, because that is the part of this repo the change actually
|
||||
**member definition** in `fleetd.yaml`, because that is the part of this repo the change actually
|
||||
touches.
|
||||
|
||||
---
|
||||
@@ -51,7 +51,7 @@ Three facts about our side that decide the shape of this work:
|
||||
`ANTHROPIC_AUTH_TOKEN` (the value is read from a host env var and never stored in config).
|
||||
`local` sets no `tokenEnv` today, because a direct vLLM needs no token.
|
||||
2. **`SubscriptionGuard` refuses any host not on an allowlist**, and that allowlist is
|
||||
`guard.offSubscriptionHosts: [gx00.gw]`. It is built once in `Bridged.java:93` and handed to the
|
||||
`guard.offSubscriptionHosts: [gx00.gw]`. It is built once in `Fleetd.java:93` and handed to the
|
||||
launcher, so **it is a restart-required key**, not a hot one. Changing `baseUrl` without changing
|
||||
this makes every `local` spawn throw.
|
||||
3. **The wiki names us as a blocker.** Under *Not done yet*: retiring the shared `legacy` token is
|
||||
@@ -132,18 +132,18 @@ own provider credentials. So 3b needs **no allowlist change**; only 3a does.
|
||||
|
||||
**Both still need a restart, for a different reason.** `tokenEnv` is resolved by
|
||||
`HerdrPeerLauncher.resolveEnv` → `env.apply(name)`, which reads the **daemon's own process
|
||||
environment**. The running `bridged` inherited its environment when it started, so a variable added to
|
||||
environment**. The running `fleetd` inherited its environment when it started, so a variable added to
|
||||
`secrets.sh` afterwards is simply not there — the launcher would inject an empty token and the
|
||||
gateway would answer 401. This is the same failure as trap 1 in `scripts/redeploy-bridged.sh`
|
||||
gateway would answer 401. This is the same failure as trap 1 in `scripts/redeploy-fleetd.sh`
|
||||
(`WORKER_GITEA_TOKEN`), and it has the same fix: **restart from a login shell**, and use
|
||||
`scripts/redeploy-bridged.sh --check` to confirm the name resolves before restarting anything.
|
||||
`scripts/redeploy-fleetd.sh --check` to confirm the name resolves before restarting anything.
|
||||
|
||||
### 3c. What this does to `ccs`
|
||||
|
||||
Once a profile carries `baseUrl`, `tokenEnv` and `model` itself, the ccs instance stops being what
|
||||
routes a member. Be precise about what is left, though: `configDir` still supplies **folder trust**
|
||||
and `settings.json`, and dropping it is what produced the trust dialog and the wrong-model error
|
||||
recorded in `bridged.yaml`. So ccs goes from *deciding where the tokens go* to *holding client-side
|
||||
recorded in `fleetd.yaml`. So ccs goes from *deciding where the tokens go* to *holding client-side
|
||||
state*. Less load-bearing, not removable.
|
||||
|
||||
### Why `/anthropic` and never `/v1/chat/completions`
|
||||
@@ -186,7 +186,7 @@ Switching buys four things we do not have:
|
||||
- **Free opencode capacity, off the shared credential.** The largest single win. See §3b — it retires
|
||||
a real single point of failure, not just a cost line.
|
||||
- **Per-consumer usage figures.** The cockpit counts requests per consumer. That is the first real
|
||||
measurement of what the fleet consumes, and it feeds [CB-589](https://git.ltms.dev/lms/claude-bridge/issues/74) Gap 2 directly.
|
||||
measurement of what the fleet consumes, and it feeds [CB-589](https://git.ltms.dev/fleet/fleetd/issues/74) Gap 2 directly.
|
||||
- **Our own revocable token.** One consumer to revoke if a worker ever leaks it, instead of a shared
|
||||
`legacy` token used by four systems.
|
||||
- **It works off-LAN.** `gx00.gw` resolves on the LAN only.
|
||||
@@ -196,7 +196,7 @@ every member spawn. The wiki keeps the direct route open precisely because "if t
|
||||
nothing that matters is blocked."
|
||||
|
||||
So keep it. Add a second profile `local-direct` pointing at `http://gx00.gw:8000` with **`weight: 0`**
|
||||
— never auto-selected, still spawnable with an explicit `bridge_spawn{profile: "local-direct"}`.
|
||||
— never auto-selected, still spawnable with an explicit `fleet_spawn{profile: "local-direct"}`.
|
||||
That is exactly what CB-554 made `weight: 0` mean, and it turns the escape hatch into something the
|
||||
lead can actually reach during an incident.
|
||||
|
||||
@@ -207,7 +207,7 @@ worker's honesty rule leans on it ("never claim the result of a check you had no
|
||||
|
||||
- **Keep bridge-only.** The invariant stays true and simple. Workers stay cheap and narrow.
|
||||
- **Add the gateway MCP.** Implementers get context7 documentation lookups, which is genuinely useful
|
||||
for library work. But `mcpUrl` in `BridgedConfig.Profile` is a **single `String`**, so a
|
||||
for library work. But `mcpUrl` in `FleetConfig.Profile` is a **single `String`**, so a
|
||||
claude-code member can mount exactly one MCP — this needs a code change, not a config edit.
|
||||
|
||||
Note the invariant is **already inaccurate**: `opencode.json` gives `sol` and `terra` both context7
|
||||
@@ -217,7 +217,7 @@ defensible; picking one is not mine to do.
|
||||
### D3 — token scope
|
||||
|
||||
One consumer, `claude-bridge`, its token in `${SHARED_ENV}/tools/secrets.sh` as `AI_GATEWAY_TOKEN`,
|
||||
referenced by name only. Never the literal value in `bridged.yaml` — `tokenEnv` exists for this.
|
||||
referenced by name only. Never the literal value in `fleetd.yaml` — `tokenEnv` exists for this.
|
||||
|
||||
---
|
||||
|
||||
@@ -226,7 +226,7 @@ referenced by name only. Never the literal value in `bridged.yaml` — `tokenEnv
|
||||
```mermaid
|
||||
flowchart TB
|
||||
U1["U1 · consumer token<br/>issue via cockpit, add to secrets.sh"]
|
||||
U2["U2 · profile + guard<br/>bridged.yaml, restart"]
|
||||
U2["U2 · profile + guard<br/>fleetd.yaml, restart"]
|
||||
U3["U3 · verify live<br/>spawn, prove thinking survives"]
|
||||
U4["U4 · context7 via gateway<br/>.mcp.json + opencode.json"]
|
||||
U5["U5 · docs<br/>CLAUDE.md, wiki 11-Features"]
|
||||
@@ -238,7 +238,7 @@ flowchart TB
|
||||
| # | Scope | Who | Why |
|
||||
|---|---|---|---|
|
||||
| U1 | Issue the `claude-bridge` consumer at `auth.ltms.dev`; store as `AI_GATEWAY_TOKEN` | **operator** | touches secrets and a host we do not own |
|
||||
| U2a | New `gx` opencode profile on `/v1` — pure config, no guard change | **lead** | `bridged.yaml` is gitignored, so a worker cannot see or edit it |
|
||||
| U2a | New `gx` opencode profile on `/v1` — pure config, no guard change | **lead** | `fleetd.yaml` is gitignored, so a worker cannot see or edit it |
|
||||
| U2b | `local` → `/anthropic`; add `local-direct` weight 0; add `llm.ltms.dev` to the guard allowlist | **lead** | same |
|
||||
| U2c | One restart from a **login shell**, after U2a and U2b | **lead** | picks up `AI_GATEWAY_TOKEN` into the daemon env *and* the guard allowlist, in one stop |
|
||||
| U3 | Live spawn on both new profiles; confirm reasoning survives on each surface | **lead** | needs real spawns and the running daemon |
|
||||
@@ -266,7 +266,7 @@ Each of these cost someone real debugging time upstream. They apply to us.
|
||||
1. **Rotating a token restarts the auth proxy, which drops in-flight streaming responses.** For us
|
||||
that means rotating `AI_GATEWAY_TOKEN` kills every live member mid-turn, and an async ticket's
|
||||
report goes with it. This is the same rule as a daemon redeploy: **drain the fleet first**
|
||||
(`bridge_list` → `bridge_poll` anything wanted → `bridge_stop`), then rotate.
|
||||
(`fleet_list` → `fleet_poll` anything wanted → `fleet_stop`), then rotate.
|
||||
2. **The gateway's own `SecurityPolicy` fails open.** Standalone `aigw run` accepts it and silently
|
||||
ignores it — an unauthenticated request returned **200**. Auth is the Caddy proxy in front, and
|
||||
nothing else. Never reason as if the gateway authenticates.
|
||||
@@ -283,10 +283,10 @@ Each of these cost someone real debugging time upstream. They apply to us.
|
||||
|
||||
Merging config is not proving it. The checks, in order:
|
||||
|
||||
1. `bridge_spawn{profile: "gx"}` succeeds and the member completes a real turn ending in
|
||||
`bridge_reply`. This is the first proof of the token, the URL and the model name, and it risks
|
||||
1. `fleet_spawn{profile: "gx"}` succeeds and the member completes a real turn ending in
|
||||
`fleet_reply`. This is the first proof of the token, the URL and the model name, and it risks
|
||||
nothing the fleet depends on.
|
||||
2. `bridge_spawn{profile: "local"}` succeeds. If the guard allowlist was missed, this **throws** — a
|
||||
2. `fleet_spawn{profile: "local"}` succeeds. If the guard allowlist was missed, this **throws** — a
|
||||
loud, self-correcting failure, which is the good kind. If the restart was missed, it also throws,
|
||||
for the same reason.
|
||||
3. A `local` member completes a turn. That exercises streaming through two TLS edges, the auth proxy
|
||||
@@ -297,9 +297,9 @@ Merging config is not proving it. The checks, in order:
|
||||
deltas. For `gx` on `/v1`, this answers the open question in §3 rather than assuming it.
|
||||
5. The cockpit at `auth.ltms.dev` shows requests counted against the `claude-bridge` consumer, not
|
||||
`legacy`. That is the whole point of taking our own token.
|
||||
6. `bridge_spawn{profile: "local-direct"}` still works, so the escape hatch is real rather than
|
||||
6. `fleet_spawn{profile: "local-direct"}` still works, so the escape hatch is real rather than
|
||||
theoretical.
|
||||
7. `bridge_list` shows `gx` carrying no `credentialId`, so a `sol`/`terra` exhaustion cannot
|
||||
7. `fleet_list` shows `gx` carrying no `credentialId`, so a `sol`/`terra` exhaustion cannot
|
||||
quarantine it. This is the single-point-of-failure claim in §3b, checked rather than asserted.
|
||||
|
||||
---
|
||||
@@ -372,7 +372,7 @@ one turn; this one costs the whole task and is indistinguishable from a slow wor
|
||||
|
||||
> **Diagnosing a stuck opencode member.** Do not read its pane. The launcher writes its config to a
|
||||
> temp dir and passes it as `OPENCODE_CONFIG` — find it with
|
||||
> `ls -dt /var/folders/*/*/T/bridged-opencode-* | head -1`, check the provider block and the key's
|
||||
> `ls -dt /var/folders/*/*/T/fleetd-opencode-* | head -1`, check the provider block and the key's
|
||||
> length and prefix (never its value), then reproduce with `opencode run --auto -m <provider>/<model>`
|
||||
> using the same `OPENCODE_CONFIG`. That is what turned "it hangs" into a one-line error.
|
||||
|
||||
@@ -383,7 +383,7 @@ one turn; this one costs the whole task and is indistinguishable from a slow wor
|
||||
- `/v1/models` returns exactly `["deepseek-v4-flash"]`, so trap 3 is clear.
|
||||
- **Reasoning survives both surfaces** — see §3b above.
|
||||
- The launcher's generated opencode provider block is correct, carrying a real 48-character `llmk-`
|
||||
key rather than the `bridged-local-noauth` placeholder.
|
||||
key rather than the `fleetd-local-noauth` placeholder.
|
||||
- `SubscriptionGuard` accepted `llm.ltms.dev` after the allowlist edit and the restart: `local`
|
||||
spawned without throwing, which is the check that catches a missed restart.
|
||||
|
||||
@@ -460,7 +460,7 @@ code.** That is the whole reason this section exists.
|
||||
|
||||
## 8. Related
|
||||
|
||||
- [CB-589 / #74](https://git.ltms.dev/lms/claude-bridge/issues/74) — cost-first placement and a
|
||||
- [CB-589 / #74](https://git.ltms.dev/fleet/fleetd/issues/74) — cost-first placement and a
|
||||
gateway that reports live capacity. The per-consumer figures this migration unlocks are the first
|
||||
input that ticket actually needs.
|
||||
- `docs/CB-500-Multi-Tier-Coordination.md` §11 — the distributed-sandbox topology this gateway is
|
||||
|
||||
+18
-18
@@ -12,12 +12,12 @@ not after it.
|
||||
|
||||
## 1. Why this stage is not optional bookkeeping
|
||||
|
||||
`bridged` today has **exactly one security control: the loopback bind**. Every other guarantee
|
||||
`fleetd` today has **exactly one security control: the loopback bind**. Every other guarantee
|
||||
rests on it.
|
||||
|
||||
The identity model (`mcp/ConnectionIdentity.java`) resolves a caller from the connection alone —
|
||||
the OS reports the connecting PID, herdr owns the PID→pane map, so a worker cannot forge another
|
||||
worker. Its own javadoc is explicit: *"Single-host only (the herd shares the `bridged` host); the
|
||||
worker. Its own javadoc is explicit: *"Single-host only (the herd shares the `fleetd` host); the
|
||||
token path is the split-host fallback."* The token path does not exist yet.
|
||||
|
||||
That leaves a seam that is **latent today and load-bearing the moment the bind moves**:
|
||||
@@ -67,7 +67,7 @@ it will be disabled and the stage is wasted. So:
|
||||
auth:
|
||||
mode: loopback-trust # default — behaves exactly like today: loopback ⇒ PRIMARY, no token needed
|
||||
# mode: token # every non-worker caller must present a valid bearer token
|
||||
# tokenEnv: BRIDGED_API_TOKEN # host env var holding the token; never the literal value
|
||||
# tokenEnv: FLEETD_API_TOKEN # host env var holding the token; never the literal value
|
||||
```
|
||||
|
||||
`mode: loopback-trust` is the current behaviour, named honestly and now *chosen* rather than
|
||||
@@ -112,9 +112,9 @@ The roadmap says "systemd unit". **This host is macOS — there is no systemd on
|
||||
not found), and the daemon that has been dogfooded for weeks runs as a bare foreground
|
||||
`java -jar`. Ship **both**:
|
||||
|
||||
- `deploy/dev.ltms.bridged.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
- `deploy/dev.ltms.fleet.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
ordered start after herdr.
|
||||
- `deploy/bridged.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
- `deploy/fleetd.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
|
||||
Ordering after herdr is advisory in both: the herdr socket may not exist at boot, so the daemon
|
||||
must **retry the socket rather than exit** — supervision ordering is a nicety, socket-retry is the
|
||||
@@ -158,10 +158,10 @@ configured. It already leaks nothing but herdr's version and up/down.
|
||||
### 3.1 There are TWO entry paths, and only one of them has identity today
|
||||
|
||||
The wiki describes MCP as "a thin adapter over the REST core". **At the code level that is not
|
||||
literally true, and the difference is security-relevant.** `BridgeMcp` calls `MessageService` /
|
||||
literally true, and the difference is security-relevant.** `FleetMcp` calls `MessageService` /
|
||||
`SessionManager` *directly*; it never issues an HTTP request against a Javalin route. And `/mcp` is
|
||||
mounted as a raw servlet on Jetty's `ServletContextHandler`
|
||||
(`BridgedApp.build → cfg.jetty.modifyServletContextHandler`), so it does **not** pass through
|
||||
(`FleetApp.build → cfg.jetty.modifyServletContextHandler`), so it does **not** pass through
|
||||
Javalin's `before` filters at all.
|
||||
|
||||
The current split is the mirror image of what you'd expect:
|
||||
@@ -172,7 +172,7 @@ The current split is the mirror image of what you'd expect:
|
||||
| REST routes | ❌ **none at all** — the session id is taken from the URL path and trusted | ❌ none |
|
||||
|
||||
So REST is the *more* exposed surface: `POST /sessions/{id}/reply` accepts any `{id}` from the
|
||||
path, whereas the MCP `bridge_reply` derives the worker from the connection and refuses to read it
|
||||
path, whereas the MCP `fleet_reply` derives the worker from the connection and refuses to read it
|
||||
from an argument. Loopback-only bind is what makes this safe today.
|
||||
|
||||
**Therefore CB-505 must enforce on both paths against one shared resolver** — not at a single
|
||||
@@ -188,15 +188,15 @@ Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
|
||||
| Metric | Type | Why it exists |
|
||||
|---|---|---|
|
||||
| `bridged_sends_total{outcome}` | counter | outcome ∈ replied\|completion_fallback\|timeout\|failed — the completion-fallback rate is the health signal for turn detection (CB-115/116/118) |
|
||||
| `bridged_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `bridged_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `bridged_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `bridged_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `bridged_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `bridged_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `bridged_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
| `bridged_auth_failures_total{reason}` | counter | only meaningful once CB-501 lands; catches misconfigured workers |
|
||||
| `fleet_sends_total{outcome}` | counter | outcome ∈ replied\|completion_fallback\|timeout\|failed — the completion-fallback rate is the health signal for turn detection (CB-115/116/118) |
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ sent\|exhausted — a rising `exhausted` means the primary is not draining. `sent` was called `delivered` until fleetd #365; it counts the herdr paste-and-submit call returning, never a confirmation the pane read it |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
| `fleet_auth_failures_total{reason}` | counter | only meaningful once CB-501 lands; catches misconfigured workers |
|
||||
|
||||
---
|
||||
|
||||
@@ -226,7 +226,7 @@ was answered from the forge itself. Kept here as the decision record.*
|
||||
|
||||
1. ✅ **TLS scope (D3) — confirmed.** Bearer auth + the fail-fast bind guard ship in the daemon;
|
||||
TLS terminates at a reverse proxy, documented with a worked example. AMQP gets TLS via an
|
||||
`amqps://` URI. No keystore handling in `bridged`.
|
||||
`amqps://` URI. No keystore handling in `fleetd`.
|
||||
2. ✅ **Micrometer (D4) — confirmed dropped.** Zero-dependency Prometheus text renderer, for the
|
||||
reasons in D4 (pom reconciliation burden + the mandated CVE gate being un-runnable this
|
||||
session). Revisit if a push-gateway or JVM-metrics requirement appears; the endpoint is the
|
||||
|
||||
+34
-34
@@ -1,7 +1,7 @@
|
||||
# M4 - Fleet health, recovery, routing, and capacity
|
||||
|
||||
**Status:** Design accepted on 2026-08-15. CB-573 part 1 has shipped the classification model and
|
||||
the `bridge_list` capacity view; the remaining M4 units are not yet shipped. See
|
||||
the `fleet_list` capacity view; the remaining M4 units are not yet shipped. See
|
||||
[Unit 2 — what has landed so far](#unit-2---what-has-landed-so-far) before planning Unit 2 work:
|
||||
some of its criteria were met by separate CB tickets, and one of them contradicts the unit text.
|
||||
**Scope:** Fleet evidence, safe mechanical repair, lead routing, capacity reporting, and optional
|
||||
@@ -72,7 +72,7 @@ post-teardown invariant so a later regression becomes `DELEGATION_ORPHANED`.
|
||||
|
||||
### 2.2 Corrections made during design
|
||||
|
||||
The first state table missed `BUSY` in bridged plus `IDLE` or `DONE` in herdr. It would have found
|
||||
The first state table missed `BUSY` in fleetd plus `IDLE` or `DONE` in herdr. It would have found
|
||||
the real trace only through a late, weak stall timer. The final model adds
|
||||
`TURN_BOUNDARY_LOST` as a strong disagreement state.
|
||||
|
||||
@@ -145,7 +145,7 @@ Idle may drive configured resource cleanup. It never opens an incident and never
|
||||
| `TURN_BOUNDARY_LOST` | Same session turn stays `BUSY`; same accepted task stays open; two raw snapshots show `IDLE` or `DONE` | Strong disagreement. Strict reconciliation may repair it. |
|
||||
| `ERROR_ON_SCREEN` | Suspicious non-working state survives grace; `detection` matches a tested adapter-specific fatal signature | Certain only for the matched signature. A bare word such as `Exception` is not enough. |
|
||||
| `STALL_SUSPECTED` | Open turn is older than the configured threshold; two normalised `recent_unwrapped` digests are unchanged; no boundary or reply occurs | Not certain. A long valid API call can look the same. Lead decides. |
|
||||
| `MUTE` | Turn resolves through completion fallback instead of `bridge_reply` | Certain that no structured reply won. It does not prove an MCP failure. A single event is a metric, not an incident. |
|
||||
| `MUTE` | Turn resolves through completion fallback instead of `fleet_reply` | Certain that no structured reply won. It does not prove an MCP failure. A single event is a metric, not an incident. |
|
||||
| `REPLY_STRANDED` | Typed reply or health message remains after owning-lead push reaches its cap | Collection failed. This does not explain whether the lead is busy, dead, or ignoring the nudge. |
|
||||
| `DELEGATION_ORPHANED` | Target is gone, failed, or released, but one or more tasks remain `PENDING` after reconciliation grace | Certain bridge invariant failure. This is not an inbox-drain fault. |
|
||||
| `WORK_PRODUCT_AT_RISK` | Provisioned branch has commits after its recorded base; member is `DONE`, `FAILED`, or preserved after release; no turn or inbox item remains; long-idle threshold passed | A warning, not proof of loss. Work may already have an open pull request or a squash merge. |
|
||||
@@ -161,7 +161,7 @@ request or merge state. If committed work appears with `REPLY_STRANDED` or
|
||||
|
||||
| State | Exact evidence | Certainty and action |
|
||||
|---|---|---|
|
||||
| `CONTROL_LINK_DOWN` | Two full-fleet `agent.list` calls fail across the grace, and herdr `ping` also fails | Certain for the bridged-to-herdr link. Retry calls, record the incident, and use human escalation if no lead can be reached. |
|
||||
| `CONTROL_LINK_DOWN` | Two full-fleet `agent.list` calls fail across the grace, and herdr `ping` also fails | Certain for the fleetd-to-herdr link. Retry calls, record the incident, and use human escalation if no lead can be reached. |
|
||||
|
||||
A failed fleet list alone is not a dead-member claim. A single `_not_found` with a healthy global
|
||||
link is a target fault, not a control-link fault.
|
||||
@@ -230,8 +230,8 @@ Only the lead may:
|
||||
- choose how to use partial work in a worktree;
|
||||
- restart herdr or change network, model, credentials, backend, or configuration.
|
||||
|
||||
Reports include literal safe tool calls such as `bridge_status(sessionId="...")`,
|
||||
`bridge_poll(ticket="...")`, `bridge_list()`, and optional `bridge_stop(paneId="...")`. A judgement
|
||||
Reports include literal safe tool calls such as `fleet_status(sessionId="...")`,
|
||||
`fleet_poll(ticket="...")`, `fleet_list()`, and optional `fleet_stop(paneId="...")`. A judgement
|
||||
state never presents stop as the only action.
|
||||
|
||||
### 5.3 Release causes and worktree safety
|
||||
@@ -251,7 +251,7 @@ preserving cause.
|
||||
|
||||
Before abnormal release removes the live session, M4 writes an atomic manifest under the worktree
|
||||
root. It records session identity, owner, role, profile, repository, path, branch, base commit,
|
||||
release cause, release time, state, and pending task ids. `bridge_list.preservedWorktrees` loads these
|
||||
release cause, release time, state, and pending task ids. `fleet_list.preservedWorktrees` loads these
|
||||
manifests after restart. Stop output and WARN logs also name the path and cause. M4 never
|
||||
auto-deletes a preserved worktree.
|
||||
|
||||
@@ -298,7 +298,7 @@ Compiled brakes apply even if config asks for more:
|
||||
- at most two pane reads occur in one fleet tick;
|
||||
- targets rotate fairly;
|
||||
- only a normalised digest and optional clipped local excerpt are stored;
|
||||
- no pane excerpt leaves bridged in a human webhook.
|
||||
- no pane excerpt leaves fleetd in a human webhook.
|
||||
|
||||
Use `detection` for tested screen signatures. Use normalised `recent_unwrapped` only for progress
|
||||
comparison.
|
||||
@@ -330,7 +330,7 @@ The reader uses AMQP `content_type`, never body sniffing:
|
||||
|
||||
```text
|
||||
Legacy v0: text/plain
|
||||
Typed family: application/vnd.ltms.bridged.inbox-message+json
|
||||
Typed family: application/vnd.ltms.fleet.inbox-message+json
|
||||
```
|
||||
|
||||
A legacy reply may begin with `{`. It remains plain text because its media type is `text/plain`.
|
||||
@@ -348,7 +348,7 @@ Invalid known-format data is copied byte-for-byte to durable queue
|
||||
before the original is acknowledged. A failed quarantine handoff leaves the original unacknowledged.
|
||||
The raw body never enters logs.
|
||||
|
||||
Decode failure creates a redacted WARN, metric, `bridge_list` summary, and routed health incident.
|
||||
Decode failure creates a redacted WARN, metric, `fleet_list` summary, and routed health incident.
|
||||
One bad entry never escapes the consumer callback and never stops later valid messages.
|
||||
|
||||
Safe downgrade is not supported. The previous build ignores `content_type` and would show typed JSON
|
||||
@@ -399,13 +399,13 @@ Choose the candidate with the fewest assigned foreign incidents. Break ties by s
|
||||
then terminal id. Pin the recipient. Reassign only if that peer becomes unhealthy or retires. A
|
||||
routing generation marks a reassignment, and old pending assignments become superseded.
|
||||
|
||||
`bridge_list` lead rows show health, health age, assigned foreign incident count, and a bounded list
|
||||
`fleet_list` lead rows show health, health age, assigned foreign incident count, and a bounded list
|
||||
of incident id, subject, state, severity, age, and routing generation. The top-level view also shows
|
||||
owner, recipient, and routing reason.
|
||||
|
||||
A peer incident is published under the recipient lead's inbox key, not the failed subject's key. Its
|
||||
status-gated nudge names the failed lead and gives the exact
|
||||
`bridge_poll(target="<recipient-terminal>")` call.
|
||||
`fleet_poll(target="<recipient-terminal>")` call.
|
||||
|
||||
### 7.5 Lead inbox ownership
|
||||
|
||||
@@ -427,7 +427,7 @@ An idle, reachable lead may receive the existing bounded nudge. An unreachable o
|
||||
has no safe in-loop recovery. The bridge must not restart or replace it. A new lead would not have the
|
||||
failed lead's plan or context, and an uncertain relaunch could create two orchestrators.
|
||||
|
||||
With no webhook, only `bridge_list`, `/healthz`, metrics, WARN logs, and the incident journal remain.
|
||||
With no webhook, only `fleet_list`, `/healthz`, metrics, WARN logs, and the incident journal remain.
|
||||
These are passive surfaces. They are not a human notification.
|
||||
|
||||
## 8. Lost-boundary reconciliation
|
||||
@@ -520,8 +520,8 @@ A repaired result uses distinct `RECONCILED_COMPLETION` values in `Rendezvous`,
|
||||
task poll source, and metrics. The lead sees:
|
||||
|
||||
```text
|
||||
[repaired completion - bridged detected a lost turn boundary. The member did not call
|
||||
bridge_reply; pane-derived text follows and may be partial]
|
||||
[repaired completion - fleetd detected a lost turn boundary. The member did not call
|
||||
fleet_reply; pane-derived text follows and may be partial]
|
||||
```
|
||||
|
||||
Clipped text also keeps the existing clipped-tail marker.
|
||||
@@ -551,7 +551,7 @@ failure operation once. It never recreates the task.
|
||||
|
||||
Capacity is a view, not a health state.
|
||||
|
||||
`bridge_list` adds one block per profile:
|
||||
`fleet_list` adds one block per profile:
|
||||
|
||||
```text
|
||||
profile, maxLoad, live, free, reclaimable
|
||||
@@ -612,7 +612,7 @@ incident feeds lead-health evidence. If the lead then becomes unhealthy, peer or
|
||||
Webhook mode requires a resolved environment variable. Turning notification off stops outbound
|
||||
attempts but keeps incidents. Turning it back on resumes still-open human incidents.
|
||||
|
||||
Without a sink, `bridge_list.healthCoverage` states that human escalation is unavailable. `/healthz`
|
||||
Without a sink, `fleet_list.healthCoverage` states that human escalation is unavailable. `/healthz`
|
||||
keeps its existing HTTP liveness result and adds a nested `fleetHealth.status=partial` component.
|
||||
Metrics and one startup or reload WARN expose the same limit.
|
||||
|
||||
@@ -687,14 +687,14 @@ The sink response body is ignored. A webhook cannot direct recovery. n8n remains
|
||||
M4 adds bounded-label series:
|
||||
|
||||
```text
|
||||
bridged_health_incidents{scope,state,severity}
|
||||
bridged_health_incidents_total{event}
|
||||
bridged_health_notifications_total{event,outcome}
|
||||
bridged_health_notification_queue_depth
|
||||
bridged_health_notification_last_success_seconds
|
||||
bridged_health_notification_capability{mode,status}
|
||||
bridged_lead_health{lead,state}
|
||||
bridged_lead_assigned_incidents{lead}
|
||||
fleet_health_incidents{scope,state,severity}
|
||||
fleet_health_incidents_total{event}
|
||||
fleet_health_notifications_total{event,outcome}
|
||||
fleet_health_notification_queue_depth
|
||||
fleet_health_notification_last_success_seconds
|
||||
fleet_health_notification_capability{mode,status}
|
||||
fleet_lead_health{lead,state}
|
||||
fleet_lead_assigned_incidents{lead}
|
||||
```
|
||||
|
||||
Metric labels never include terminal ids, incident ids, URLs, or error text.
|
||||
@@ -792,7 +792,7 @@ Acceptance criteria:
|
||||
them.
|
||||
16. Explicit stop is state-aware. Any pending task or non-terminal state preserves the worktree.
|
||||
17. Atomic preserved-worktree manifests reload after restart and appear in lead-only
|
||||
`bridge_list.preservedWorktrees`.
|
||||
`fleet_list.preservedWorktrees`.
|
||||
18. Manifest failure preserves the worktree and opens an operator-visible health failure.
|
||||
19. Provision records the base commit. Terminal, long-idle worktrees report
|
||||
`WORK_PRODUCT_AT_RISK` only under the evidence in Section 4.2 and never auto-delete work.
|
||||
@@ -807,7 +807,7 @@ Checked against `main` at `e09cac6` on 2026-08-15. Unit 2 was written as one blo
|
||||
have since been built by separate CB tickets. Read this before planning the rest, or that work gets
|
||||
done twice.
|
||||
|
||||
The check was a symbol survey of `bridged/src/main/java` plus the merge history. It tells you whether
|
||||
The check was a symbol survey of `fleetd/src/main/java` plus the merge history. It tells you whether
|
||||
the machinery exists at all. It is **not** a line-by-line audit of whether each criterion is fully
|
||||
met, and I did not run one.
|
||||
|
||||
@@ -820,7 +820,7 @@ met, and I did not run one.
|
||||
| 14 | `DELEGATION_ORPHANED` | 3 files | **Partial.** The health state exists. The teardown-invariant check that creates it, and the retry rule, do not. |
|
||||
| 15 | `SPAWN_ROLLBACK` | 0 files | **Contradicted — see below.** |
|
||||
| 16 | — | — | Partial at best. CB-576 made release preserve a dirty worktree; whether explicit stop is state-aware is not checked. |
|
||||
| 17, 18 | `preservedWorktrees` | 0 files | Not started. No manifest, and no lead-only `bridge_list` field. |
|
||||
| 17, 18 | `preservedWorktrees` | 0 files | Not started. No manifest, and no lead-only `fleet_list` field. |
|
||||
| 19 | `WORK_PRODUCT_AT_RISK` | 0 files | Not started. |
|
||||
|
||||
**Criterion 15 no longer matches the code, and the code is right.** It says "normal `COMPLETED`
|
||||
@@ -837,7 +837,7 @@ preserve it. `SPAWN_ROLLBACK` itself does not exist yet.
|
||||
### Unit 3 - Typed inbox and member routing
|
||||
|
||||
Scope: semantic record, AMQP migration, both adapters, member routing, polling, and member health in
|
||||
`bridge_list`.
|
||||
`fleet_list`.
|
||||
|
||||
Acceptance criteria:
|
||||
|
||||
@@ -852,7 +852,7 @@ Acceptance criteria:
|
||||
6. Invalid known data never escapes the callback, appears as a reply, or blocks later valid messages.
|
||||
7. Invalid data reaches durable per-target quarantine before original ack. Failed handoff leaves the
|
||||
original unacked.
|
||||
8. Decode failures create redacted WARN, metric, `bridge_list` summary, and routed incident without
|
||||
8. Decode failures create redacted WARN, metric, `fleet_list` summary, and routed incident without
|
||||
raw content.
|
||||
9. Both adapters pass one semantic contract for fields, FIFO, dedup, ownership, ack, and release.
|
||||
10. Lead keys require explicit ownership. Publication never claims a queue.
|
||||
@@ -865,7 +865,7 @@ Acceptance criteria:
|
||||
15. RabbitMQ contract tests pass with `mvn test -Pcontract`. The same cases run once on production
|
||||
LavinMQ, or the release states that LavinMQ was not checked.
|
||||
16. Member incidents route to the exact delegating lead and never resolve a task rendezvous.
|
||||
17. `bridge_list` shows compact member health and capacity without pane content. Member
|
||||
17. `fleet_list` shows compact member health and capacity without pane content. Member
|
||||
`idleForSeconds` is present only when no accepted turn or inbox item exists.
|
||||
|
||||
### Unit 4 - Lead health and peer routing
|
||||
@@ -887,7 +887,7 @@ Acceptance criteria:
|
||||
8. A selected working peer is not interrupted. Its push waits for an injectable window.
|
||||
9. Recipient assignment stays pinned. Reassignment increments generation and supersedes old pending
|
||||
assignment.
|
||||
10. `bridge_list` shows bounded foreign assignments, recipient, reason, and generation without pane
|
||||
10. `fleet_list` shows bounded foreign assignments, recipient, reason, and generation without pane
|
||||
content.
|
||||
11. `LeadInboxRegistry` owns configured and discovered lead keys before publication.
|
||||
12. Missing leads keep ownership. Retirement needs an empty queue and handled incidents.
|
||||
@@ -911,7 +911,7 @@ Acceptance criteria:
|
||||
3. Webhook mode requires a resolved environment value. Bad notification config does not disable an
|
||||
already valid detector.
|
||||
4. Mode changes keep open incidents. Re-enable resumes eligible incidents.
|
||||
5. `bridge_list`, `/healthz`, metrics, and one WARN show partial coverage without a sink. HTTP
|
||||
5. `fleet_list`, `/healthz`, metrics, and one WARN show partial coverage without a sink. HTTP
|
||||
liveness behavior stays unchanged.
|
||||
6. One-lead, no-sink coverage states that lead failure has no active notification or recovery.
|
||||
7. Incident and outbound dedupe use the stable keys in Section 10.3.
|
||||
@@ -927,7 +927,7 @@ Acceptance criteria:
|
||||
15. Webhook response bodies are ignored and cannot direct recovery.
|
||||
16. Tests cover disabled mode, one lead without sink, open/update/reminder/resolve, restart, dedup,
|
||||
reassignment, disable/re-enable, and sink failure while local health continues.
|
||||
17. `bridged.example.yaml` documents all hot keys and compiled floors.
|
||||
17. `fleetd.example.yaml` documents all hot keys and compiled floors.
|
||||
18. The operator Features wiki is updated separately. The portable `CLAUDE.md` block is checked and
|
||||
changed only if shipped tool or inbox semantics make it untrue.
|
||||
19. `mvn clean install` passes.
|
||||
|
||||
+129
-315
@@ -1,296 +1,162 @@
|
||||
# MCP Contract — `bridged`'s unified gateway
|
||||
# MCP flows and error model — `fleetd`
|
||||
|
||||
> **Status:** 🟡 Design (2026-07-14). Greenfield — no MCP code exists yet; the pom carries
|
||||
> only Javalin/Jackson. This page defines the tool surface that CB-104 and its followers
|
||||
> implement. It supersedes nothing; it fills the "MCP server face" left open by the
|
||||
> [Architecture](1-Architecture) page.
|
||||
|
||||
`bridged` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
> **What this page is.** The **flows**: how a delegation, a clarification, a detached task and a
|
||||
> silent member each travel through `fleetd`. These shapes are what shipped, and they are hard to
|
||||
> read off the code because they span the MCP face, the rendezvous registry, the `Injector` and
|
||||
> herdr.
|
||||
>
|
||||
> **What this page is NOT: a tool reference.** It deliberately holds no tool catalogue, no
|
||||
> parameter tables and no REST paths. **The live MCP schema is the authority** — each tool's own
|
||||
> description and parameters, as mounted — with the intent→tool table in `CLAUDE.md` as the short
|
||||
> form.
|
||||
>
|
||||
> That absence is the fix for fleetd #114 (CB-609), and it is worth stating why. This page used to
|
||||
> carry a full tool catalogue written in July 2026, before any MCP code existed. The code shipped;
|
||||
> the page did not follow. By August it named two tools that do not exist, omitted five that do,
|
||||
> had the wrong name for nearly every parameter, pointed at REST paths the daemon does not serve,
|
||||
> and — worst — still described an identity model (*"any connection that does not map to a known
|
||||
> worker is treated as a primary"*) that was a real privilege bug, fixed since by the ancestry
|
||||
> walk in fleetd #161. Every one of those errors is the same error: **a second, hand-maintained
|
||||
> copy of something the code already states**. So the second copy is gone rather than corrected.
|
||||
> Only the flows remain, because a flow is a shape rather than a name, and shapes are what this
|
||||
> page was ever good for.
|
||||
>
|
||||
> The names that do appear below are checked by `McpContractDocTest`, which fails if this page
|
||||
> names a `fleet_*` tool the server does not register. That test is the whole reason it is safe to
|
||||
> write a tool name here at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
## 1. Rendezvous flows
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
### 1.1 Delegation — happy path
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `bridged` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `bridged` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`bridged` owns policy; herdr owns PTYs.** MCP tools express *intent*; `bridged`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["bridged — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>bridge_send · bridge_reply<br/>bridge_ask · bridge_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"bridge_send (blocks)"| MCP
|
||||
W -.->|"bridge_reply / bridge_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `bridged` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `bridged` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `bridge_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `bridge_reply` / `bridge_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`bridged` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http bridged http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`bridge_send`](#bridge_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`bridge_reply`](#bridge_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`bridge_ask`](#bridge_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`bridge_status`](#bridge_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`bridge_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`bridge_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`bridge_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`bridge_read`](#bridge_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`bridge_cancel`](#bridge_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `bridge_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `bridge_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `bridge_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `bridge_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `bridge_status` on a split-host primary.
|
||||
|
||||
#### `bridge_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `bridged` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `bridge_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `bridge_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `bridge_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`bridge_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`bridge_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`bridge_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `bridge_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `bridge_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `bridge_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
One blocking call, zero polls. The lead's call is held open by `fleetd` until the member answers.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as bridged (MCP + Injector)
|
||||
participant P as "Lead (primary)"
|
||||
participant B as "fleetd (MCP + Injector)"
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: bridge_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B->>B: "register waiter(sessionId)"
|
||||
B->>H: "agent.send — only in an injectable window"
|
||||
H-->>W: "prompt injected"
|
||||
W->>W: "works the turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B->>B: "resolve waiter"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`bridge_ask`)
|
||||
**The cap that matters:** a blocking `fleet_send` is bounded by the *caller's own* MCP client
|
||||
timeout, about 60 seconds — not by the task. Anything slower than that must use the detached flow
|
||||
in §1.3, or the lead's call returns while the member is still working.
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
### 1.2 Clarification — reverse rendezvous
|
||||
|
||||
The member pauses mid-turn to ask, the lead answers, and the member resumes **the same turn** with
|
||||
its context intact.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant W as Worker
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: bridge_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: bridge_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve bridge_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: bridge_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B-->>W: "content injected"
|
||||
W->>B: "fleet_ask{question} — member blocks"
|
||||
B-->>P: "{ outcome: question, turnId }"
|
||||
P->>B: "fleet_send{turnId, content} — answers THIS turn"
|
||||
B-->>W: "fleet_ask returns the answer"
|
||||
W->>W: "resumes the same turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
**Answer with `turnId`, never `sessionId`.** A `sessionId` send starts a new turn; it does not
|
||||
resolve the waiting `fleet_ask`.
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
**The window is about 55 seconds and no nudge extends it.** So never brief a member to "ask me":
|
||||
decide before delegating, or give the member an explicit default to fall back on.
|
||||
|
||||
### 1.3 Detached delegation — the lead does not block
|
||||
|
||||
The lead gets a ticket immediately and collects the answer later. This is the flow for any real
|
||||
task, because of the ~60s cap in §1.1.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant W as Worker
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: bridge_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
P->>B: "fleet_send{sessionId, content, wait:false}"
|
||||
B-->>P: "accepted — ticket"
|
||||
P->>P: "continues its own work"
|
||||
W->>B: "fleet_reply{content}"
|
||||
Note over B: "no waiter is blocked — the reply is held"
|
||||
B->>B: "nudge the lead's own pane (status-gated)"
|
||||
P->>B: "fleet_poll{ticket}"
|
||||
B-->>P: "the member's report"
|
||||
P->>B: "fleet_ack{target, msgId}"
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
A terminal ticket nudges the lead's pane by itself, so a detached task does not need watching. The
|
||||
nudge needs an injectable lead pane and is capped, so it is a convenience rather than a guarantee.
|
||||
|
||||
A worker that never calls `bridge_reply` still returns a result: `bridged` reads its terminal
|
||||
tail when the turn completes.
|
||||
### 1.4 The member never replies — turn-done fallback
|
||||
|
||||
A member that ends its turn without `fleet_reply` still produces something: `fleetd` reads its
|
||||
pane tail. This is a **fallback, not a channel** — it is lossy in three separate ways, and every
|
||||
one of them has produced a wrong answer in practice.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant B as bridged
|
||||
participant W as Worker
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: bridge_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls bridge_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
P->>B: "fleet_send — blocks or detaches"
|
||||
B-->>W: "content injected"
|
||||
W->>W: "works, never calls fleet_reply"
|
||||
B->>B: "StatusPoller sees the turn end"
|
||||
B->>B: "read the pane tail"
|
||||
B->>B: "classify: exhausted? echoed brief? real report?"
|
||||
B-->>P: "{ outcome: turn_done } or a named failure"
|
||||
```
|
||||
|
||||
The three ways it goes wrong, and what each looks like now:
|
||||
|
||||
| What happened | What the lead used to get | What it gets today |
|
||||
|---|---|---|
|
||||
| The report is longer than the scrape window | The **end** silently cut off | Still clipped, but marked partial |
|
||||
| The member never started — spent credential | The lead's **own brief** echoed back as a report | A named failure: backend exhausted |
|
||||
| The member is simply slow | A tail of work in progress | Unchanged — read it as a hint, not a result |
|
||||
|
||||
The echoed-brief case is the one to remember: it reads as a long, on-topic report with nothing in
|
||||
it from the member. It is suppressed now, but the general rule stands — **check the member's
|
||||
worktree with `git log` before believing a report you did not watch arrive.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
## 2. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `bridge_send` is simply its producer.
|
||||
Delivery only happens in a safe window. `fleet_send` is a producer for the `Injector`, which
|
||||
already enforces this through `AgentStatus.injectable()`.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
IDLE --> WORKING: "message delivered, picked up"
|
||||
WORKING --> IDLE: "turn done"
|
||||
WORKING --> BLOCKED: "awaits input"
|
||||
BLOCKED --> WORKING: "input delivered"
|
||||
IDLE --> UNKNOWN: "detection glitch"
|
||||
BLOCKED --> UNKNOWN: "detection glitch"
|
||||
UNKNOWN --> IDLE: "re-detected"
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
@@ -306,69 +172,17 @@ stateDiagram-v2
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
**At most one message per turn.** After a send, the `Injector` waits for a `WORKING` pickup before
|
||||
delivering the next, with a grace-poll fallback for turns that finish faster than the poll
|
||||
interval.
|
||||
|
||||
---
|
||||
Two consequences a lead feels directly:
|
||||
|
||||
## 8. Error model
|
||||
- **A second send to a busy member never lands.** It reports as queued and times out. The member
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
|
||||
| Condition | `bridge_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `bridge_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`bridge_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `bridge_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `bridge_reply` / `bridge_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `bridge_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `bridge_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `bridge_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `BridgedApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`bridge_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `bridge_send` (keeps the catalog
|
||||
small) vs. a separate `bridge_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `bridge_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `bridge_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `bridge_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `bridge_reply` / `bridge_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`bridge_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `bridge_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
unreadable pane as a ready one.
|
||||
|
||||
@@ -1,8 +1,8 @@
|
||||
# v1.0.0 — One leader, one host, complete
|
||||
|
||||
This is the first release of **`bridged`**.
|
||||
This is the first release of **`fleetd`**.
|
||||
|
||||
`bridged` lets one main Claude Code session (the **leader**, on your Pro/Max subscription)
|
||||
`fleetd` lets one main Claude Code session (the **leader**, on your Pro/Max subscription)
|
||||
run a team of **workers** — extra Claude Code sessions on a cheaper or local model, and
|
||||
non-Claude agents too. The leader's own session is never touched: it stays on subscription,
|
||||
with a clean environment.
|
||||
@@ -19,12 +19,12 @@ of this one.
|
||||
## One gateway for all messages
|
||||
|
||||
- **Everyone talks through the same door.** The leader and every worker connect to the same
|
||||
MCP server and use only its tools: `bridge_whoami` · `bridge_profiles` · `bridge_spawn` ·
|
||||
`bridge_list` · `bridge_status` · `bridge_send` · `bridge_reply` · `bridge_ask` ·
|
||||
`bridge_poll` · `bridge_ack` · `bridge_stop`.
|
||||
MCP server and use only its tools: `fleet_whoami` · `fleet_profiles` · `fleet_spawn` ·
|
||||
`fleet_list` · `fleet_status` · `fleet_send` · `fleet_reply` · `fleet_ask` ·
|
||||
`fleet_poll` · `fleet_ack` · `fleet_stop`.
|
||||
- **You are who your connection says you are.** The bridge finds out who is calling from the
|
||||
connection itself, never from a name the caller sends. So a worker cannot pretend to be
|
||||
someone else, and `bridge_whoami` tells each agent its own role — no guessing.
|
||||
someone else, and `fleet_whoami` tells each agent its own role — no guessing.
|
||||
- **The subscription line cannot be crossed.** Only a spawned worker gets
|
||||
`ANTHROPIC_BASE_URL`; the leader never does. Each worker profile has a list of allowed
|
||||
model hosts, checked before anything starts.
|
||||
@@ -59,7 +59,7 @@ had to ask for its replies. That gap is now closed on a single machine:
|
||||
- **The leader gets a tap on the shoulder.** When a reply lands, the bridge nudges the
|
||||
leader's own pane — only when the leader is free, and only a few times. If the leader is on
|
||||
another machine, this quietly falls back to pick-up mode; the reply still waits.
|
||||
- **Workers can ask questions.** With `bridge_ask`, a worker can pause mid-task, ask the
|
||||
- **Workers can ask questions.** With `fleet_ask`, a worker can pause mid-task, ask the
|
||||
leader something, and continue the *same* task with the answer.
|
||||
|
||||
## More than one kind of worker
|
||||
|
||||
+21
-21
@@ -1,9 +1,9 @@
|
||||
# Team — lead orchestrating a mixed Claude + local-LLM fleet
|
||||
|
||||
The message server (`bridged`) delivers **one turn into one worker**. A **team** is the
|
||||
The message server (`fleetd`) delivers **one turn into one worker**. A **team** is the
|
||||
layer above it: a **Claude team-lead** that fans a job out across a **mixed fleet** of
|
||||
workers — some on Claude, some on the remote local LLM — and reduces their replies. Same
|
||||
`bridged` delivery, same subscription boundary; this doc is only about **orchestration** —
|
||||
`fleetd` delivery, same subscription boundary; this doc is only about **orchestration** —
|
||||
who the workers are, how the lead picks one, and how it runs many at once.
|
||||
|
||||
> Delivery mechanics (blocking `POST /message`, status-gated reply envelope) live in the
|
||||
@@ -12,12 +12,12 @@ who the workers are, how the lead picks one, and how it runs many at once.
|
||||
## The team
|
||||
|
||||
- **Team-lead** — the primary **Opus** (Claude Code, env **CLEAN**, on Pro/Max). Not a
|
||||
worker; a **thin client of `bridged`**. It plans, routes, dispatches, and integrates, and
|
||||
worker; a **thin client of `fleetd`**. It plans, routes, dispatches, and integrates, and
|
||||
never sets `ANTHROPIC_BASE_URL`.
|
||||
- **Workers** — a herd of `claude` panes in herdr, each an addressable `bridged` session
|
||||
- **Workers** — a herd of `claude` panes in herdr, each an addressable `fleetd` session
|
||||
with its **own model/env**:
|
||||
- **Claude workers** (clean env, e.g. Sonnet) — reasoning-heavy or high-accuracy subtasks.
|
||||
- **Local workers** (`ANTHROPIC_BASE_URL=https://ollama.ltms.dev`) — bulk, cheap, or
|
||||
- **Local workers** (`ANTHROPIC_BASE_URL=https://llm.ltms.dev/anthropic`) — bulk, cheap, or
|
||||
embarrassingly parallel subtasks.
|
||||
|
||||
Every worker is still a *real Claude Code process* (inherits `CLAUDE.md`, hooks, skills,
|
||||
@@ -28,14 +28,14 @@ MCP) — only its model differs. Scale each kind horizontally by adding panes.
|
||||
```mermaid
|
||||
flowchart TB
|
||||
LEAD["lead — Opus<br/>(Claude Code, env CLEAN)"]
|
||||
BD["bridged<br/>message server + router"]
|
||||
BD["fleetd<br/>message server + router"]
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
WC1["w-claude-1<br/>Sonnet · CLEAN"]
|
||||
WC2["w-claude-2<br/>Sonnet · CLEAN"]
|
||||
WL1["w-local-1<br/>ANTHROPIC_BASE_URL set"]
|
||||
WL2["w-local-2<br/>ANTHROPIC_BASE_URL set"]
|
||||
ANT["api.anthropic.com<br/>(Pro/Max)"]
|
||||
OLL["ollama.ltms.dev<br/>(local model)"]
|
||||
OLL["llm.ltms.dev<br/>(gateway to the local model)"]
|
||||
|
||||
LEAD -->|"blocking POST /message (target role)"| BD
|
||||
BD -->|"Unix socket · send_text · events.subscribe"| HERDR
|
||||
@@ -62,14 +62,14 @@ flowchart TB
|
||||
| `w-local-*` | `ANTHROPIC_BASE_URL` set | local LLM | task is bulk / cheap / embarrassingly parallel |
|
||||
|
||||
The lead applies this rubric itself, guided by its `CLAUDE.md` team charter (below). Worker
|
||||
selection is **policy in the lead**, not a `bridged` concern — `bridged` just delivers to
|
||||
selection is **policy in the lead**, not a `fleetd` concern — `fleetd` just delivers to
|
||||
the session the lead names.
|
||||
|
||||
## Subscription boundary in a team
|
||||
|
||||
Unchanged from the base architecture, and it scales with the fleet: **only local-worker
|
||||
panes** launch with `ANTHROPIC_BASE_URL`. The lead and every Claude worker stay env-clean on
|
||||
the subscription. `bridged` enforces which panes may carry the off-subscription env, so
|
||||
the subscription. `fleetd` enforces which panes may carry the off-subscription env, so
|
||||
adding workers never widens the boundary.
|
||||
|
||||
## Parallel fan-out (map / reduce)
|
||||
@@ -80,7 +80,7 @@ different workers at once, then results are gathered.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant L as lead (Opus)
|
||||
participant B as bridged
|
||||
participant B as fleetd
|
||||
participant WC as w-claude-1
|
||||
participant WL as w-local-1
|
||||
|
||||
@@ -100,26 +100,26 @@ sequenceDiagram
|
||||
```
|
||||
|
||||
- **Map:** the lead issues N concurrent blocking `POST /message` calls (one per subtask → its
|
||||
chosen worker). Each call blocks only *that* request; `bridged` holds it open until the
|
||||
chosen worker). Each call blocks only *that* request; `fleetd` holds it open until the
|
||||
worker's turn completes (status-gated) and returns the reply envelope.
|
||||
- **Reduce:** the lead collects the N envelopes and integrates. A slow local worker never
|
||||
blocks a fast Claude worker — wall-clock ≈ the slowest single subtask, not the sum.
|
||||
- **Detached / long jobs** use the async broker path instead of a held request (Channel 2 in
|
||||
the base architecture), so the lead never busy-polls across turns.
|
||||
|
||||
Fan-out is bounded by the herd size (pane count) and `bridged`'s concurrency policy, not by
|
||||
Fan-out is bounded by the herd size (pane count) and `fleetd`'s concurrency policy, not by
|
||||
the lead.
|
||||
|
||||
## Knowing the roster
|
||||
|
||||
The lead discovers its team from `bridged` (session list / roles) rather than hard-coding
|
||||
The lead discovers its team from `fleetd` (session list / roles) rather than hard-coding
|
||||
pane ids, so workers can be added or restarted without editing the lead. A minimal charter
|
||||
in the lead's `CLAUDE.md` turns Opus into the orchestrator:
|
||||
|
||||
```markdown
|
||||
## Your team (via bridged)
|
||||
You are the team-lead. Delegate through the bridged client — never launch workers yourself.
|
||||
Roster: ask bridged for current sessions/roles.
|
||||
## Your team (via fleetd)
|
||||
You are the team-lead. Delegate through the fleetd client — never launch workers yourself.
|
||||
Roster: ask fleetd for current sessions/roles.
|
||||
- w-claude-* — Claude Sonnet. Reasoning-heavy / high-accuracy subtasks.
|
||||
- w-local-* — remote local LLM. Bulk, cheap, or parallelizable subtasks.
|
||||
|
||||
@@ -133,22 +133,22 @@ tool instead of hand-rolling the HTTP request.
|
||||
|
||||
## What this layer does NOT change
|
||||
|
||||
- **Delivery** is still `bridged` → herdr `pane.send_text` + status events (Message-Server).
|
||||
- **Delivery** is still `fleetd` → herdr `pane.send_text` + status events (Message-Server).
|
||||
- **Completion timing** is still the worker status event; **reply content** still rides the
|
||||
worker `Stop`-hook envelope.
|
||||
- **Single-host** still applies: herdr's socket is local, so the whole herd lives on the
|
||||
`bridged` host. The lead may be remote — it only needs HTTP to `bridged`.
|
||||
`fleetd` host. The lead may be remote — it only needs HTTP to `fleetd`.
|
||||
|
||||
## Open questions
|
||||
|
||||
- **Routing intelligence:** rubric-in-`CLAUDE.md` (lead decides) vs. a `bridged` role-router
|
||||
- **Routing intelligence:** rubric-in-`CLAUDE.md` (lead decides) vs. a `fleetd` role-router
|
||||
(label-based). Start with the former; promote to the latter if routing logic grows.
|
||||
- **Backpressure:** per-role concurrency caps in `bridged` so a fan-out can't exhaust the
|
||||
- **Backpressure:** per-role concurrency caps in `fleetd` so a fan-out can't exhaust the
|
||||
local gateway.
|
||||
- **Result schema:** whether reply envelopes should carry structured metadata (worker, model,
|
||||
tokens) to help the lead's reduce step.
|
||||
|
||||
## Status
|
||||
|
||||
🟡 Design (2026-07-11). Orchestration layer over the selected `bridged` server; inherits
|
||||
🟡 Design (2026-07-11). Orchestration layer over the selected `fleetd` server; inherits
|
||||
herdr (chosen) + AgentAPI (fallback). Delivery unchanged — see the Message-Server design.
|
||||
|
||||
+10
-10
@@ -20,7 +20,7 @@ requirement, not a nice-to-have.**
|
||||
- **Worktree provisioned by the daemon** — `SessionManager` creates a dedicated git worktree +
|
||||
branch per session, **hydrates it to full config parity** (below), and tears it down on release.
|
||||
- **Worker opens its own PR** — the worker commits, pushes its branch, and opens the PR/MR itself,
|
||||
returning the PR URL in its `bridge_reply`.
|
||||
returning the PR URL in its `fleet_reply`.
|
||||
|
||||
## Why worktrees (the hazard being fixed)
|
||||
|
||||
@@ -76,7 +76,7 @@ worktree checks out anyway. Amber is the real gap — untracked local config the
|
||||
3. **Never overlay the git plumbing** — the worktree's own `.git` file/branch is what gives
|
||||
isolation; that's the *one* thing that must differ from the main tree.
|
||||
|
||||
The overlay set lives in config (`BridgedConfig.Worker.parityOverlay` — a list of repo-relative
|
||||
The overlay set lives in config (`FleetConfig.Worker.parityOverlay` — a list of repo-relative
|
||||
paths, with sane defaults) so it's auditable and per-repo tunable.
|
||||
|
||||
> **Trust note (deliberate).** Hydrating local config means the primary's local secrets/tokens
|
||||
@@ -103,7 +103,7 @@ sequenceDiagram
|
||||
Note over W: implement in the isolated worktree
|
||||
W->>G: git commit + git push (SSH, same user)
|
||||
W->>G: open PR (branch to main)
|
||||
W-->>P: bridge_reply (prUrl, branch, summary, tests)
|
||||
W-->>P: fleet_reply (prUrl, branch, summary, tests)
|
||||
P->>SM: release(paneId)
|
||||
SM->>G: git worktree remove wt
|
||||
Note over G: branch + PR persist for review/merge
|
||||
@@ -115,9 +115,9 @@ earlier `STATE.md` idea — a PR is reviewable, mergeable, and self-describing.*
|
||||
|
||||
## Infra facts (verified this session)
|
||||
|
||||
- **Remote:** `ssh://git@git.ltms.dev:2224/lms/claude-bridge.git` (gitea). Push is over **SSH** —
|
||||
- **Remote:** `ssh://git@git.ltms.dev:2224/fleet/fleetd.git` (gitea). Push is over **SSH** —
|
||||
a worker running as the same user with the same keys can `git push` **with no extra credential**.
|
||||
- **gitea is NOT in the project `.mcp.json`** (only `jetbrains`, `intellij-index`, `bridged`). The
|
||||
- **gitea is NOT in the project `.mcp.json`** (only `jetbrains`, `intellij-index`, `fleetd`). The
|
||||
primary's gitea MCP comes from a global/user config, so **workers do not inherit it**. A worker
|
||||
gets only the `bridge` MCP mounted (via `--mcp-config` launch flag).
|
||||
- **No gitea CLI** (`tea`) installed; `glab` is present but is the GitLab CLI (wrong backend).
|
||||
@@ -128,7 +128,7 @@ earlier `STATE.md` idea — a PR is reviewable, mergeable, and self-describing.*
|
||||
|
||||
| Option | Mechanism | Trade-off |
|
||||
|---|---|---|
|
||||
| **A. gitea REST + token** | Worker `curl`s `POST /api/v1/repos/lms/claude-bridge/pulls` with a scoped token injected by the daemon into the worker env | Minimal, no new server; token lives in the off-subscription worker's env (scope it tightly) |
|
||||
| **A. gitea REST + token** | Worker `curl`s `POST /api/v1/repos/fleet/fleetd/pulls` with a scoped token injected by the daemon into the worker env | Minimal, no new server; token lives in the off-subscription worker's env (scope it tightly) |
|
||||
| **B. mount gitea MCP into workers** | Add the gitea MCP to the worker's `--mcp-config` alongside `bridge` | Clean tool call, but the gitea MCP's own auth/token must be provisioned per worker; more moving parts |
|
||||
| **C. install `tea` CLI** | Worker runs `tea pr create` with a token | Another dependency to install + configure; same token question as A |
|
||||
|
||||
@@ -140,7 +140,7 @@ and the token is a single scoped secret the daemon injects like it already injec
|
||||
|
||||
- Off-subscription workers already *could* push (SSH, same user). The **incremental grant is
|
||||
PR-create**, i.e. a gitea API token.
|
||||
- Scope the token **minimally**: the `lms/claude-bridge` repo, `write:repository` (create branch +
|
||||
- Scope the token **minimally**: the `fleet/fleetd` repo, `write:repository` (create branch +
|
||||
PR), **not** merge/admin/org. A leaked token can open PRs, not merge them — the primary/human is
|
||||
still the merge gate.
|
||||
- Inject via the daemon (env var, e.g. `GITEA_TOKEN`), never written to the worker's config dir —
|
||||
@@ -153,11 +153,11 @@ and the token is a single scoped secret the daemon injects like it already injec
|
||||
|---|---|---|
|
||||
| Worktree provision/teardown | **CB-301 ext** — `SessionManager.acquire`/`release`; `WorkerSession` gains `worktree`, `branch` | daemon shells out to `git worktree add/remove` |
|
||||
| **Config-parity overlay** | **CB-301 ext** — `SessionManager.acquire`, after `git worktree add` | symlink/copy the `parityOverlay` set into the worktree so the worker is a full peer; **this is what makes worktrees viable, not a dead-end** |
|
||||
| Overlay config | `BridgedConfig.Worker.parityOverlay` — repo-relative paths, sane defaults | auditable, per-repo tunable; keep explicit + minimal (trust) |
|
||||
| Overlay config | `FleetConfig.Worker.parityOverlay` — repo-relative paths, sane defaults | auditable, per-repo tunable; keep explicit + minimal (trust) |
|
||||
| Branch naming | `worker/<ticket-slug>-<nonce>` off `main` (or a configured base) | one branch per session |
|
||||
| Commit + push + PR handoff | **CB-302** — worker-driven, guided by the skill | push = SSH; PR = option A |
|
||||
| Implementer skill | `.claude/skills/implementer/SKILL.md` | worktree-aware playbook (see below); mounts automatically since workers inherit repo cwd |
|
||||
| gitea token injection | `WorkerService` env + `BridgedConfig` | repo-scoped, minimal perms |
|
||||
| gitea token injection | `WorkerService` env + `FleetConfig` | repo-scoped, minimal perms |
|
||||
| PR review + merge | Primary (has gitea MCP + judgment) | merge on green; the human/primary gate stays |
|
||||
|
||||
## Implementer skill (outline)
|
||||
@@ -170,7 +170,7 @@ A worker-facing playbook (sibling to the existing `reviewer` skill):
|
||||
3. **Push** your branch (`git push -u origin HEAD`).
|
||||
4. **Open a PR** to `main` (option A `curl`, or the decided mechanism) with a title/body describing
|
||||
the change and referencing the ticket.
|
||||
5. **Reply** via `bridge_reply` with the **PR URL**, branch name, files changed, and test names —
|
||||
5. **Reply** via `fleet_reply` with the **PR URL**, branch name, files changed, and test names —
|
||||
that reply is the whole handoff.
|
||||
6. Do **not** merge; do **not** touch `.mcp.json` or `wiki/`.
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
# Worker startup: working directory & the folder-trust prompt
|
||||
|
||||
When `bridged` spawns a worker, the worker CLI may show an **interactive startup prompt** before it
|
||||
When `fleetd` spawns a worker, the worker CLI may show an **interactive startup prompt** before it
|
||||
is ready to accept a task — most importantly a *"Do you trust the files in this folder?"* dialog. An
|
||||
unattended worker parked on that prompt never becomes injectable: the status-gated injector waits for
|
||||
`idle`/`blocked`, the task is never delivered, and (worst case) a stray Enter answers the dialog
|
||||
@@ -14,11 +14,11 @@ rule that **a worker inherits the primary's directory** (never `$HOME`), and how
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A["bridge_spawn / POST /workers"] --> B{"explicit cwd?<br/>(profile cwd or spawn arg)"}
|
||||
A["fleet_spawn / POST /workers"] --> B{"explicit cwd?<br/>(profile cwd or spawn arg)"}
|
||||
B -->|"yes — told otherwise"| C["use that cwd"]
|
||||
B -->|"no"| D{"caller PID resolvable?<br/>(MCP peer PID)"}
|
||||
D -->|"yes"| E["cwd = the primary's cwd<br/>lsof -a -p PID -d cwd"]
|
||||
D -->|"no (REST / off-host)"| F["cwd = bridged daemon cwd<br/>(never $HOME by assumption)"]
|
||||
D -->|"no (REST / off-host)"| F["cwd = fleetd daemon cwd<br/>(never $HOME by assumption)"]
|
||||
C --> G["ensureWorkspace → tab.create → agent.start {cwd}"]
|
||||
E --> G
|
||||
F --> G
|
||||
@@ -51,10 +51,10 @@ only affect the seed shell, which the bridge closes).
|
||||
| # | Source | When |
|
||||
|---|--------|------|
|
||||
| 1 | Explicit `cwd` — a per-profile `cwd:` in config, or a spawn argument | "told otherwise" — pin a fixed workdir |
|
||||
| 2 | The **primary's cwd**, auto-detected from the `bridge_spawn` caller | normal MCP spawn from the primary |
|
||||
| 3 | The `bridged` daemon's own cwd | REST spawn / off-host caller — **never `$HOME`** |
|
||||
| 2 | The **primary's cwd**, auto-detected from the `fleet_spawn` caller | normal MCP spawn from the primary |
|
||||
| 3 | The `fleetd` daemon's own cwd | REST spawn / off-host caller — **never `$HOME`** |
|
||||
|
||||
The primary's cwd (source 2) is discoverable with no new plumbing: `bridged` already resolves the MCP
|
||||
The primary's cwd (source 2) is discoverable with no new plumbing: `fleetd` already resolves the MCP
|
||||
caller's loopback **peer PID** for connection identity (`ConnectionIdentity` → `LsofPeerPidLookup`);
|
||||
the same PID yields its cwd via `lsof -a -p <pid> -d cwd -Fn` (the `n…` line). The primary maps to no
|
||||
worker pane (it is not a worker), but its PID and cwd are still readable.
|
||||
@@ -62,10 +62,10 @@ worker pane (it is not a worker), but its PID and cwd are still readable.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as "Primary (main)"
|
||||
participant B as "bridged"
|
||||
participant B as "fleetd"
|
||||
participant O as "OS (lsof)"
|
||||
participant H as "herdr"
|
||||
P->>B: "bridge_spawn {profile} (no cwd)"
|
||||
P->>B: "fleet_spawn {profile} (no cwd)"
|
||||
B->>O: "peer PID for this connection's port"
|
||||
O-->>B: "pid"
|
||||
B->>O: "cwd of pid (lsof -d cwd)"
|
||||
@@ -77,9 +77,9 @@ sequenceDiagram
|
||||
|
||||
*Figure 2 — a no-cwd spawn inherits the primary's directory from the caller's PID.*
|
||||
|
||||
> **Status:** implemented (CB-112). `bridged` threads the resolved `cwd` onto **`agent.start {cwd}`**
|
||||
> **Status:** implemented (CB-112). `fleetd` threads the resolved `cwd` onto **`agent.start {cwd}`**
|
||||
> (verified: the worker process is rooted there), keeping the single shared worker space. On an MCP
|
||||
> `bridge_spawn` the primary's cwd is auto-detected from the caller's PID; over REST (no MCP caller)
|
||||
> `fleet_spawn` the primary's cwd is auto-detected from the caller's PID; over REST (no MCP caller)
|
||||
> it is the explicit `cwd` param else the daemon's cwd. Both placements (`tab` and legacy `pane`)
|
||||
> carry it, since it rides `agent.start`.
|
||||
|
||||
@@ -133,5 +133,5 @@ unattended.
|
||||
|
||||
## See also
|
||||
|
||||
- `docs/MCP-Contract.md` — the tool surface (`bridge_spawn`, `bridge_profiles`, …).
|
||||
- `docs/MCP-Contract.md` — the tool surface (`fleet_spawn`, `fleet_profiles`, …).
|
||||
- `wiki/2-Message-Server.md` — the herdr `agent.*` / `workspace.*` schema (`workspace.create {cwd}`).
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
# Wiki audit for #168
|
||||
|
||||
**Source checked:** `.wiki-snapshot/` at `68e32c6` (2026-08-31). I did not use
|
||||
`wiki/`. Code references below are from the current `fleetd` source tree. A quoted
|
||||
line is a concrete claim that needs correction, unless the table says `KEEP`.
|
||||
|
||||
| Page | Verdict | One-line reason |
|
||||
|---|---|---|
|
||||
| `Home.md` | REVISE | Good overview, but it still names the retired product. |
|
||||
| `_Sidebar.md` | REVISE | The heading still says `claude-bridge`. |
|
||||
| `1-Architecture.md` | REBUILD | Its component contract mixes current names with removed tools, routes, and planned backends. |
|
||||
| `2-Message-Server.md` | REBUILD | The claimed MCP schema, mount command, REST/SSE surface, and fallback paths are pre-build design. |
|
||||
| `3-Approaches.md` | REVISE | Useful research history, but it presents unbuilt AgentAPI as a selectable fallback. |
|
||||
| `4-Setup.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `5-Operations.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `6-Team.md` | REBUILD | It teaches role-addressed sends and a Claude-only team model that the shipped API does not have. |
|
||||
| `7-Use-Cases.md` | REBUILD | Its flagship flow depends on removed `ccs` profiles and removed send parameters. |
|
||||
| `8-Roadmap.md` | REBUILD | It is a historical plan, but it presents old implementation choices and planned work as the current stack. |
|
||||
| `9-Implementation.md` | REBUILD | Its package, class, endpoint, and outcome map has drifted from the source. |
|
||||
| `10-Cross-Host-Messaging.md` | REVISE | It labels most federation work proposed, but misses the shipped `coordinator:` lead channel. |
|
||||
| `11-Features.md` | REVISE | It is the right catalogue, but code-path names are old and it misses the second-herdr-daemon capability. |
|
||||
| `12-Claude-to-OpenCode.md` | REVISE | The porting guide is mostly current, but calls the product and spawned-member path a bridge. |
|
||||
| `13-User-Guide.md` | REVISE | It is the best operator page, but needs the product rename and the second-herdr-daemon setup. |
|
||||
|
||||
## Pages needing work
|
||||
|
||||
### `Home.md` — REVISE
|
||||
|
||||
- Quote: `# claude-bridge` (line 1) and `` `claude-bridge` keeps`` (line 11).
|
||||
The product is `fleet` / `fleetd`. The MCP server identifies itself as `fleet` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:313-315`.
|
||||
- Quote: `AgentAPI ... swappable fallback injector` (lines 73-76).
|
||||
There is no AgentAPI implementation under `fleetd/src/main/java`; the actual
|
||||
launchers are selected by `Profile.kind` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java:265-270`.
|
||||
|
||||
### `_Sidebar.md` — REVISE
|
||||
|
||||
- Quote: `### 📖 claude-bridge` (line 1).
|
||||
Rename it to `fleet`. `FleetMcp` registers the current product-facing tool set at
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:301-326`.
|
||||
|
||||
### `1-Architecture.md` — REBUILD
|
||||
|
||||
- Quote: `` `claude-bridge` lets`` (line 3). The product was renamed; the MCP
|
||||
server name is `fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: ``fleet_read`` in the tool list (line 102). No such tool is registered.
|
||||
The complete registered list is `fleet_send` through `fleet_whoami` at
|
||||
`FleetMcp.java:301-326`; `fleet_read` is absent.
|
||||
- Quote: `SSE (GET /events)` (line 143). `FleetApp.build()` registers no `/events`
|
||||
route; its routes are listed at `FleetApp.java:143-159`.
|
||||
- Quote: `Redis Streams / NATS JetStream, or an embedded queue` (line 106).
|
||||
The shipped durable inbox is AMQP, configured by `broker`, at
|
||||
`FleetConfig.java:49-50` and `FleetConfig.java:655-714`.
|
||||
- Quote: `AgentAPI (fallback)` (line 107). No AgentAPI adapter exists; shipped
|
||||
launcher kinds are `claude-code` and `opencode` (`FleetConfig.java:265-270`).
|
||||
|
||||
### `2-Message-Server.md` — REBUILD
|
||||
|
||||
- Quote: `claude mcp add --transport http bridge http://127.0.0.1:8080/mcp`
|
||||
(line 67). The daemon defaults to port `8765` in `FleetConfig.java:183-187`,
|
||||
and identifies its server as `fleet` at `FleetMcp.java:313-315`.
|
||||
- Quote: ``fleet_send(message, target?, {block, timeout_seconds, auto_spawn,
|
||||
turn_id})`` (line 80). The real parameters are `sessionId`, `content`,
|
||||
`timeoutMs`, `wait`, `turnId`, and `coordId` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: ``fleet_read(target, source)`` (line 85). It is not registered; see the
|
||||
complete registration at `FleetMcp.java:301-326`.
|
||||
- Quote: `docs/MCP-Contract.md ... normative` (lines 87-88). That is not a valid
|
||||
reference: only §6 is current, as the current operator guide itself says at
|
||||
`.wiki-snapshot/13-User-Guide.md:466`.
|
||||
- Quote: `SSE (GET /events)` (line 45). No route exists in the built REST surface,
|
||||
`FleetApp.java:143-159`.
|
||||
|
||||
### `3-Approaches.md` — REVISE
|
||||
|
||||
- Quote: `AgentAPI ... remains a swappable fallback injector` (lines 78-84).
|
||||
It was never built. The shipped adapter selection is only `claude-code` or
|
||||
`opencode` (`FleetConfig.java:265-270`). Keep it as discarded research, not an
|
||||
operational fallback.
|
||||
- Quote: `claude-bridge` (line 109). Rename the product to `fleet`; the runtime
|
||||
package is `dev.ltms.fleet`, for example `FleetMcp.java:1`.
|
||||
|
||||
### `4-Setup.md` — RETIRE
|
||||
|
||||
It is a 25-line redirect and says its procedure was never written (lines 3-9).
|
||||
Chapter 13 is the maintained install procedure. Keeping a second navigation page
|
||||
adds no working documentation.
|
||||
|
||||
### `5-Operations.md` — RETIRE
|
||||
|
||||
It is a 35-line redirect and says its runbook was never written (lines 3-14).
|
||||
Chapter 13 now owns run and recovery instructions.
|
||||
|
||||
### `6-Team.md` — REBUILD
|
||||
|
||||
- Quote: `fleet_send {role: w-claude, prompt: A}` (line 98). `fleet_send` accepts
|
||||
`sessionId` and `content`, not `role` or `prompt` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: `some on Claude, some on the remote local LLM` (lines 3-5) and `Every
|
||||
worker is ... Claude Code` (line 25). `opencode` is a first-class launcher kind,
|
||||
not a Claude worker (`FleetConfig.java:265-270`).
|
||||
- Quote: `fleetd's concurrency policy` (line 121). The configured capacity control
|
||||
is per-profile `maxLoad` (`FleetConfig.java:251-264`), not the role routing model
|
||||
described here.
|
||||
|
||||
### `7-Use-Cases.md` — REBUILD
|
||||
|
||||
- Quote: `ccs profile` (line 10), `ccs + herdr` (line 22), and `ccs-spawn`
|
||||
(line 45). The configuration has `profiles` and `fleet`, not `ccs`:
|
||||
`FleetConfig.java:34-58` and `FleetConfig.java:81-101`.
|
||||
- Quote: `fleet_send({"to", "kind", "body", "block"})` (lines 55-62).
|
||||
None of those are the shipped send parameters. The schema is
|
||||
`FleetMcp.java:1096-1108`.
|
||||
- Quote: `fleet_list() → { "profiles": ... }` (lines 74-80). `fleet_list` is a
|
||||
roster view; `fleet_profiles` is the configured-backend view, as registered at
|
||||
`FleetMcp.java:307-311` and described at `FleetMcp.java:1176-1182`.
|
||||
|
||||
### `8-Roadmap.md` — REBUILD
|
||||
|
||||
- Quote: `Java 21+` (line 43). The current project guidance and source use Java 25;
|
||||
the `FleetConfig` source itself uses Java 25 unnamed lambda parameters, for
|
||||
example `FleetConfig.java:102`.
|
||||
- Quote: `herdr 0.7.0 / protocol 14` (line 46). The current REST health endpoint
|
||||
reports the live protocol returned by herdr (`FleetApp.java:240-244`), while the
|
||||
current operator guide records protocol 19 at
|
||||
`.wiki-snapshot/13-User-Guide.md:76-85`.
|
||||
- Quote: `ccs <profile> claude` and `ccs env <profile>` (lines 47-48). Shipped
|
||||
configuration uses `Profile` records and launcher `kind`,
|
||||
`FleetConfig.java:313-330` and `FleetConfig.java:265-270`.
|
||||
- Quote: `Redis Streams via Lettuce` (line 50). The actual durable inbox is AMQP
|
||||
`broker`, `FleetConfig.java:655-714`.
|
||||
|
||||
### `9-Implementation.md` — REBUILD
|
||||
|
||||
- Quote: `rest.FleetdApp` and `mcp.BridgeMcp` (lines 29-30). The classes are
|
||||
`rest.FleetApp` and `mcp.FleetMcp` (`FleetApp.java:46`; `FleetMcp.java:67`).
|
||||
- Quote: `dev.ltms.fleetd` (line 67). The source package is `dev.ltms.fleet`
|
||||
(`FleetMcp.java:1`).
|
||||
- Quote: `WorkerPresence` (line 110). The current class is `MemberPresence`, as
|
||||
imported and used by `FleetMcp` at `FleetMcp.java:12` and `465-469`.
|
||||
- Quote: the outcome list ending in `STALE_TURN` (lines 128-131). The code also
|
||||
has `BACKEND_EXHAUSTED` (`FleetMcp.java:550-554`) and async `ASKING` handling
|
||||
(`FleetMcp.java:664-668`).
|
||||
- Quote: `FleetdApp` (line 207) and `FleetdConfig` (line 211). These names do not
|
||||
resolve; current classes are `FleetApp` and `FleetConfig`.
|
||||
|
||||
### `10-Cross-Host-Messaging.md` — REVISE
|
||||
|
||||
- Quote: the chapter says the cross-host fabric is proposed except for the
|
||||
single-host inbox (lines 3-8). Cross-host **lead-to-lead** delivery shipped:
|
||||
`fleet_send` accepts `coordId` (`FleetMcp.java:1094-1107`) and publishes it at
|
||||
`FleetMcp.java:616-641`; configuration has `coordinator` at
|
||||
`FleetConfig.java:74-78` and `99-101`.
|
||||
- Quote: `bridge.dlx` (line 90). This product name is stale. The shipped lead path
|
||||
uses `LeadChannel`, not the proposed exchange flow (`FleetMcp.java:95-96` and
|
||||
`616-641`). Keep the proposed federation design, but add a clear shipped/proposed
|
||||
boundary for CB-637.
|
||||
|
||||
### `11-Features.md` — REVISE
|
||||
|
||||
- Quote: `mcp/BridgeMcp` (line 22), `config/FleetdConfig` (lines 25-27), and other
|
||||
index references. These paths no longer resolve; the source classes are
|
||||
`mcp/FleetMcp` (`FleetMcp.java:67`) and `config/FleetConfig`
|
||||
(`FleetConfig.java:81`).
|
||||
- Quote: `fleet_whoami` returns only `primary` or `worker` (lines 99-100).
|
||||
It also returns `architect` (`FleetMcp.java:1235-1244`).
|
||||
- The page needs the missing separate member-herdr-daemon feature listed below.
|
||||
|
||||
### `12-Claude-to-OpenCode.md` — REVISE
|
||||
|
||||
- Quote: `same bridge mount` (line 5) and `a bridge-spawned worker` (line 94).
|
||||
Rename the product path to `fleet`. The daemon exposes the MCP server as `fleet`
|
||||
(`FleetMcp.java:313-315`), and profiles select OpenCode with `kind: opencode`
|
||||
(`FleetConfig.java:332-335`).
|
||||
- Quote: the sample mount name is `fleetd` (line 67). The server name is `fleet`;
|
||||
update the sample to avoid teaching a second product name.
|
||||
|
||||
### `13-User-Guide.md` — REVISE
|
||||
|
||||
- Quote: `The bridge is the only channel` (line 63). The invariant is correct, but
|
||||
the product term needs the `fleet` rename. The daemon's MCP server name is
|
||||
`fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: it describes one herdr socket (lines 72-85). It needs the optional
|
||||
`memberHerdrSocket` setup and two-daemon health meaning. The config key is in
|
||||
`FleetConfig.java:34-37`, and `/healthz` checks both daemons when configured at
|
||||
`FleetApp.java:210-245`.
|
||||
|
||||
## MISSING
|
||||
|
||||
`11-Features.md` has a body section for **routing members through a separate herdr daemon**
|
||||
(`## memberHerdrSocket`, line 2174), but **no row in the index table** at the top of the page
|
||||
(lines 20-95). That table is how the page is meant to be read, so a capability absent from it is
|
||||
effectively undiscoverable. Lead note: this is my own omission — I added the section on 2026-08-31
|
||||
and did not add the matching row. Fixed in the wiki at `68e32c6`'s successor.
|
||||
|
||||
The original audit stated the feature had no entry at all. That was wrong: the section exists. The
|
||||
gap is the index row. Recorded here rather than silently corrected, because the difference matters —
|
||||
"undocumented" and "documented but unindexed" are different jobs.
|
||||
|
||||
Evidence for the feature itself: `FleetConfig.java:34-37` and `FleetApp.java:103-115`, `210-245`,
|
||||
and `247-263`.
|
||||
|
||||
## Audit method and coverage
|
||||
|
||||
I checked all 15 pages. I checked concrete tool, route, config, class, file, and
|
||||
product-name claims claim-by-claim on 11 pages: Home, Sidebar, 1, 2, 4, 5, 6, 7, 9,
|
||||
11, and 13. I skimmed the remaining four long historical or research pages (3, 8, 10,
|
||||
12), then checked their concrete claims that affect the verdict. This is an audit of
|
||||
the supplied snapshot, not a wiki rewrite.
|
||||
+6
-6
@@ -2,11 +2,11 @@
|
||||
|
||||
A standard, repeatable **live** end-to-end test of the two-way channel: it drives a real
|
||||
multi-turn conversation between a primary and an off-subscription worker **through the
|
||||
running `bridged` daemon**, captures the full transcript, and grades the channel.
|
||||
running `fleetd` daemon**, captures the full transcript, and grades the channel.
|
||||
|
||||
This is the committed form of the ad-hoc channel test that discovered the CB-115 gaps
|
||||
(herdr `unknown` misclassification wedging delivery, dirty completion scrapes, and workers
|
||||
never calling `bridge_reply` in conversation). Run it after any change to the injector,
|
||||
never calling `fleet_reply` in conversation). Run it after any change to the injector,
|
||||
status handling, completion/failure paths, or the worker reply charter.
|
||||
|
||||
## What it exercises
|
||||
@@ -20,7 +20,7 @@ construction** — it only calls the bridge's loopback REST face.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant T as conversation_test.py
|
||||
participant B as bridged (REST)
|
||||
participant B as fleetd (REST)
|
||||
participant W as worker (off-sub)
|
||||
T->>B: POST /workers (spawn)
|
||||
T->>B: GET /sessions/{id}/status (await ready)
|
||||
@@ -28,7 +28,7 @@ sequenceDiagram
|
||||
T->>B: POST /sessions/{id}/message {wait:false}
|
||||
B-->>T: ticket
|
||||
B->>W: inject prompt (status-gated)
|
||||
W-->>B: bridge_reply
|
||||
W-->>B: fleet_reply
|
||||
T->>B: GET /tasks/{ticket} (poll)
|
||||
B-->>T: done + reply
|
||||
end
|
||||
@@ -37,7 +37,7 @@ sequenceDiagram
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- `bridged` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
- `fleetd` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
profile configured and its backend reachable.
|
||||
- herdr is up (the daemon needs it).
|
||||
- Python 3 (standard library only — no pip installs).
|
||||
@@ -68,7 +68,7 @@ Per-turn grade:
|
||||
|
||||
| Grade | Meaning |
|
||||
|------------|---------------------------------------------------------------------|
|
||||
| `OK` | delivered and resolved by an explicit `bridge_reply` (`source=reply`) |
|
||||
| `OK` | delivered and resolved by an explicit `fleet_reply` (`source=reply`) |
|
||||
| `DEGRADED` | delivered and answered, but resolved via completion-scrape fallback |
|
||||
| `EMPTY` | turn completed but the reply was empty |
|
||||
| `FAILED` | the worker's turn ended in failure (`phase=failed`) |
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Sustained back-and-forth bridge test — ONE primary, ONE worker, many dependent turns
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `bridged` daemon.
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `fleetd` daemon.
|
||||
|
||||
Where conversation_test.py proves a handful of turns work and issue_hunt_test.py proves
|
||||
fan-out isolation, this proves the channel stays healthy under a *sustained, stateful*
|
||||
@@ -24,7 +24,7 @@ Usage:
|
||||
--turn-timeout per-turn max wait, seconds (default 150)
|
||||
--keep-worker do not stop the worker at the end
|
||||
|
||||
Exit code: 0 if every turn in the window resolved via a clean bridge_reply with no channel
|
||||
Exit code: 0 if every turn in the window resolved via a clean fleet_reply with no channel
|
||||
break; 1 otherwise. A live per-turn log streams to stdout so the run can be watched.
|
||||
"""
|
||||
import argparse
|
||||
@@ -47,12 +47,12 @@ STEPS = [7, 3, 11, 5, 9, 4, 13, 6, 8, 2]
|
||||
RULES = (
|
||||
"Let's play a running-total game across several messages. The total starts at 0. "
|
||||
"In each message I'll tell you to add a number; keep the running total yourself and "
|
||||
"reply via bridge_reply with ONLY the current total as a plain integer — no words, no "
|
||||
"reply via fleet_reply with ONLY the current total as a plain integer — no words, no "
|
||||
"punctuation, just the number. Do not restate the arithmetic. First move: add {step}."
|
||||
)
|
||||
NEXT = ("Add {step}. Reply via bridge_reply with only the new running total.")
|
||||
NEXT = ("Add {step}. Reply via fleet_reply with only the new running total.")
|
||||
REANCHOR = ("Let's re-sync — the running total is {total}. Now add {step}. Reply via "
|
||||
"bridge_reply with only the new running total.")
|
||||
"fleet_reply with only the new running total.")
|
||||
|
||||
|
||||
def parse_int(reply):
|
||||
@@ -142,15 +142,15 @@ def main():
|
||||
print("=" * 72)
|
||||
print(f"SUSTAINED CONVERSATION SUMMARY — 1 primary <-> 1 worker over {dur}s (~{dur/60:.1f} min)")
|
||||
print(f" turns: {turns}")
|
||||
print(f" clean bridge_reply exchanges: {oks + drifts}/{turns} (channel breaks: {breaks})")
|
||||
print(f" clean fleet_reply exchanges: {oks + drifts}/{turns} (channel breaks: {breaks})")
|
||||
print(f" arithmetic correct (continuity held): {oks}/{turns} (drifts: {drifts})")
|
||||
print(f" latency: avg {avg}s over {turns} turns")
|
||||
ok = breaks == 0 and turns >= 2
|
||||
if ok and drifts == 0:
|
||||
print(" RESULT: PASS — every turn resolved via bridge_reply and the worker held the "
|
||||
print(" RESULT: PASS — every turn resolved via fleet_reply and the worker held the "
|
||||
"running total across the whole window.")
|
||||
elif ok:
|
||||
print(f" RESULT: PASS (channel) — every turn resolved via bridge_reply for the full "
|
||||
print(f" RESULT: PASS (channel) — every turn resolved via fleet_reply for the full "
|
||||
f"window; {drifts} arithmetic drift(s) (worker recovered after re-anchor).")
|
||||
else:
|
||||
print(" RESULT: FAIL — the channel broke on at least one turn (see CHANNEL BREAK above).")
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge conversation test — a multi-turn primary↔worker exchange through
|
||||
the running `bridged` daemon, fully captured, with automatic gap analysis.
|
||||
the running `fleetd` daemon, fully captured, with automatic gap analysis.
|
||||
|
||||
This is the repeatable form of the ad-hoc channel test that surfaced the CB-115 gaps
|
||||
(herdr `unknown` misclassification, dirty completion scrape, workers not calling
|
||||
bridge_reply). It drives a real off-subscription worker over the live gateway exactly
|
||||
fleet_reply). It drives a real off-subscription worker over the live gateway exactly
|
||||
as a primary Opus session would (async fire-and-poll), records every turn, and grades
|
||||
the channel.
|
||||
|
||||
@@ -130,9 +130,9 @@ def grade(rec):
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit bridge_reply"
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call bridge_reply)"
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
@@ -206,7 +206,7 @@ def main():
|
||||
ok = all(g in ("OK", "DEGRADED") for g in grades)
|
||||
reply_clean = all(g == "OK" for g in grades)
|
||||
if reply_clean:
|
||||
print(" RESULT: PASS — every turn delivered and got a clean bridge_reply.")
|
||||
print(" RESULT: PASS — every turn delivered and got a clean fleet_reply.")
|
||||
elif ok:
|
||||
print(" RESULT: PASS (with notes) — every turn delivered & replied, but some via fallback.")
|
||||
else:
|
||||
|
||||
@@ -1,16 +1,16 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Live bridge_ask test — the REVERSE rendezvous (CB-205), watched end to end.
|
||||
"""Live fleet_ask test — the REVERSE rendezvous (CB-205), watched end to end.
|
||||
|
||||
Every other harness drives the forward path: primary `bridge_send` → worker `bridge_reply`.
|
||||
Every other harness drives the forward path: primary `fleet_send` → worker `fleet_reply`.
|
||||
This drives the one that runs the other way. A worker is told to pause its delegated turn,
|
||||
ask the primary a question via `bridge_ask`, and only finish once it has the answer — so the
|
||||
ask the primary a question via `fleet_ask`, and only finish once it has the answer — so the
|
||||
turn round-trips primary→worker→primary→worker inside a SINGLE delegation.
|
||||
|
||||
The mechanics that only this path exercises:
|
||||
|
||||
• a worker's mid-turn question surfacing on the primary's *own* blocked send (Outcome.QUESTION),
|
||||
• the `turnId` correlation that lets the primary answer the exact paused turn,
|
||||
• the answer resuming that same turn and the worker's final `bridge_reply` landing on the
|
||||
• the answer resuming that same turn and the worker's final `fleet_reply` landing on the
|
||||
re-opened forward waiter (never a stale or cross-wired one).
|
||||
|
||||
It is two blocking REST calls, no polling:
|
||||
@@ -24,7 +24,7 @@ Like the rest of the suite it talks ONLY to the bridge's REST face on loopback
|
||||
ANTHROPIC_BASE_URL and never touches herdr, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 bridge_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
python3 fleet_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
[--send-timeout SECS] [--answer-timeout SECS] [--keep-worker]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
@@ -57,14 +57,14 @@ ANSWER_COLOR = "blue"
|
||||
|
||||
# A task that CANNOT be completed without asking: the worker is not told which color to choose,
|
||||
# only that the primary will name one when asked. So a correct final reply is only reachable by
|
||||
# actually calling bridge_ask and using the answer.
|
||||
# actually calling fleet_ask and using the answer.
|
||||
TASK_PROMPT = (
|
||||
"You are a bridge worker in a quick coordination game. You do NOT know which color to pick — "
|
||||
"only the primary does. Do exactly this, in order:\n"
|
||||
"1. Call the `bridge_ask` tool with EXACTLY this question: \"PICK A COLOR: red or blue?\"\n"
|
||||
"1. Call the `fleet_ask` tool with EXACTLY this question: \"PICK A COLOR: red or blue?\"\n"
|
||||
"2. The primary will answer with one color word. Take that color and uppercase it.\n"
|
||||
"3. Call `bridge_reply` with EXACTLY one line: CHOSEN=<COLOR> (e.g. CHOSEN=GREEN if told green).\n"
|
||||
"Do not guess a color. Do not call bridge_reply before bridge_ask has returned an answer. "
|
||||
"3. Call `fleet_reply` with EXACTLY one line: CHOSEN=<COLOR> (e.g. CHOSEN=GREEN if told green).\n"
|
||||
"Do not guess a color. Do not call fleet_reply before fleet_ask has returned an answer. "
|
||||
"Do nothing else — no file reads, no other tools."
|
||||
)
|
||||
|
||||
@@ -134,7 +134,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
rec.update(tid=tid, pane=pane, spawned=True)
|
||||
await_ready(base, tid)
|
||||
|
||||
# 1) Delegate the ask-forcing task. This blocks until the worker calls bridge_ask, at which
|
||||
# 1) Delegate the ask-forcing task. This blocks until the worker calls fleet_ask, at which
|
||||
# point our own send unblocks carrying the question and the turnId to answer on.
|
||||
print(f"[{now()}] delegating task (blocks until the worker asks; up to {send_timeout}s)…")
|
||||
t0 = time.time()
|
||||
@@ -159,7 +159,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
rec.update(phase="no_turnid", detail="question surfaced without a turnId to answer on")
|
||||
return rec
|
||||
|
||||
# 2) Answer on that exact turn. This blocks again until the resumed worker calls bridge_reply.
|
||||
# 2) Answer on that exact turn. This blocks again until the resumed worker calls fleet_reply.
|
||||
print(f"[{now()}] answering '{ANSWER_COLOR}' on turn {rec['turnId']} (blocks until reply; up to {answer_timeout}s)…")
|
||||
t1 = time.time()
|
||||
rec["phase"] = "awaiting_reply"
|
||||
@@ -188,7 +188,7 @@ def run(base, profile, repo, send_timeout, answer_timeout):
|
||||
def grade(rec):
|
||||
"""PASS only if the worker asked, the turn resumed, and the reply reflects the answer."""
|
||||
if rec["phase"] == "no_question":
|
||||
return "NO_ASK", "the worker finished/stalled without ever calling bridge_ask"
|
||||
return "NO_ASK", "the worker finished/stalled without ever calling fleet_ask"
|
||||
if rec["phase"] in ("send_error", "answer_error", "spawn"):
|
||||
return "ERROR", rec.get("detail") or "transport error before the round-trip completed"
|
||||
if rec["phase"] == "no_turnid":
|
||||
@@ -203,26 +203,26 @@ def grade(rec):
|
||||
return "OK", "asked, resumed the same turn, and the reply reflected the primary's answer"
|
||||
if reflected:
|
||||
return "DEGRADED", f"reply reflected the answer but resolved via {rec['replySource']} " \
|
||||
"(worker did not call bridge_reply cleanly)"
|
||||
"(worker did not call fleet_reply cleanly)"
|
||||
return "WRONG_ANSWER", f"the worker replied but did not reflect '{ANSWER_COLOR}' — " \
|
||||
f"the answer may not have reached the resumed turn: {rec['reply']!r}"
|
||||
return "WEDGE", f"unexpected terminal phase {rec['phase']}: {rec.get('detail')}"
|
||||
|
||||
|
||||
def write_transcript(out_dir, rec, meta):
|
||||
path = out_dir / "bridge_ask_transcript.md"
|
||||
path = out_dir / "fleet_ask_transcript.md"
|
||||
g, note = grade(rec)
|
||||
with path.open("w") as f:
|
||||
f.write(f"# Live bridge_ask — reverse rendezvous — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"# Live fleet_ask — reverse rendezvous — {datetime.now():%Y-%m-%d %H:%M}\n\n")
|
||||
f.write(f"One worker paused its delegated turn to ask the primary, then resumed with the "
|
||||
f"answer (profile `{meta['profile']}`). Result: **`{g}`**.\n\n")
|
||||
f.write("## Round-trip\n\n")
|
||||
f.write(f"1. **primary → worker** (delegation): the ask-forcing task.\n")
|
||||
f.write(f"2. **worker → primary** (`bridge_ask`, {rec.get('ask_latency')}s): "
|
||||
f.write(f"2. **worker → primary** (`fleet_ask`, {rec.get('ask_latency')}s): "
|
||||
f"{rec.get('question')!r} — surfaced on the primary's blocked send as a "
|
||||
f"`question` with `turnId={rec.get('turnId')}`.\n")
|
||||
f.write(f"3. **primary → worker** (answer on that turn): `{ANSWER_COLOR}`.\n")
|
||||
f.write(f"4. **worker → primary** (`bridge_reply`, {rec.get('answer_latency')}s, "
|
||||
f.write(f"4. **worker → primary** (`fleet_reply`, {rec.get('answer_latency')}s, "
|
||||
f"source={rec.get('replySource')}): {rec.get('reply')!r}\n\n")
|
||||
f.write(f"> **{g}:** {note}\n")
|
||||
if rec.get("detail"):
|
||||
@@ -231,7 +231,7 @@ def write_transcript(out_dir, rec, meta):
|
||||
|
||||
|
||||
def main():
|
||||
ap = argparse.ArgumentParser(description="Live bridge_ask reverse-rendezvous test (CB-205)")
|
||||
ap = argparse.ArgumentParser(description="Live fleet_ask reverse-rendezvous test (CB-205)")
|
||||
ap.add_argument("--base", default="http://127.0.0.1:8765")
|
||||
ap.add_argument("--profile", default=None)
|
||||
ap.add_argument("--repo", default=str(REPO_ROOT))
|
||||
@@ -241,7 +241,7 @@ def main():
|
||||
ap.add_argument("--keep-worker", action="store_true")
|
||||
args = ap.parse_args()
|
||||
|
||||
print(f"[{now()}] live bridge_ask: 1 primary, 1 worker "
|
||||
print(f"[{now()}] live fleet_ask: 1 primary, 1 worker "
|
||||
f"(profile={args.profile or 'default'}, repo={args.repo})\n")
|
||||
|
||||
rec = {"spawned": False, "pane": None}
|
||||
@@ -262,7 +262,7 @@ def main():
|
||||
|
||||
print()
|
||||
print("=" * 72)
|
||||
print("LIVE bridge_ask SUMMARY — reverse rendezvous (CB-205)")
|
||||
print("LIVE fleet_ask SUMMARY — reverse rendezvous (CB-205)")
|
||||
print(f" asked: {rec.get('question')!r} (turnId={rec.get('turnId')}, {rec.get('ask_latency')}s)")
|
||||
print(f" answered: {ANSWER_COLOR!r}")
|
||||
print(f" replied: {rec.get('reply')!r} (source={rec.get('replySource')}, {rec.get('answer_latency')}s)")
|
||||
@@ -1,12 +1,12 @@
|
||||
# Live bridge_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
# Live fleet_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
|
||||
One worker paused its delegated turn to ask the primary, then resumed with the answer (profile `default`). Result: **`OK`**.
|
||||
|
||||
## Round-trip
|
||||
|
||||
1. **primary → worker** (delegation): the ask-forcing task.
|
||||
2. **worker → primary** (`bridge_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
2. **worker → primary** (`fleet_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
3. **primary → worker** (answer on that turn): `blue`.
|
||||
4. **worker → primary** (`bridge_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
4. **worker → primary** (`fleet_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
|
||||
> **OK:** asked, resumed the same turn, and the reply reflected the primary's answer
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge fan-out test — ONE primary vs MANY workers, concurrently, for
|
||||
issue hunting through the running `bridged` daemon, fully captured, with gap analysis.
|
||||
issue hunting through the running `fleetd` daemon, fully captured, with gap analysis.
|
||||
|
||||
Where conversation_test.py exercises a single worker over multiple turns, this drives
|
||||
the path that only appears under fan-out: the primary spawns N workers, sends each a
|
||||
@@ -12,7 +12,7 @@ and collects every reply concurrently. That stresses what a single worker never
|
||||
• reply routing under concurrency (worker A's answer must never resolve worker B's send),
|
||||
|
||||
and, as the payload, whether a fleet of off-subscription workers can actually surface
|
||||
real issues in the repo and report them back structurally via bridge_reply.
|
||||
real issues in the repo and report them back structurally via fleet_reply.
|
||||
|
||||
It talks ONLY to the bridge's REST face on loopback — it never sets ANTHROPIC_BASE_URL
|
||||
and never touches herdr directly, so it is subscription-safe by construction.
|
||||
@@ -53,18 +53,18 @@ REPO_ROOT = HERE.parent
|
||||
# hot files this project has been iterating on, so a real issue is plausible to find.
|
||||
DEFAULT_ASSIGNMENTS = [
|
||||
{"id": "completion", "probe": "CompletionResolver",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/inject/CompletionResolver.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java"},
|
||||
{"id": "worker", "probe": "WorkerService",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/worker/WorkerService.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/worker/WorkerService.java"},
|
||||
{"id": "rendezvous", "probe": "Rendezvous",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/msg/Rendezvous.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/msg/Rendezvous.java"},
|
||||
]
|
||||
|
||||
PROMPT_TMPL = (
|
||||
"You are one of several issue-hunting workers in the claude-bridge repo (it is your "
|
||||
"current working directory). Your assignment: inspect the file `{target}` and find the "
|
||||
"SINGLE most important real bug, correctness gap, or risk in it. Read the file before "
|
||||
"answering. Reply via bridge_reply with EXACTLY these four lines:\n"
|
||||
"answering. Reply via fleet_reply with EXACTLY these four lines:\n"
|
||||
"1. {target}:<line>\n"
|
||||
"2. issue: <one sentence>\n"
|
||||
"3. fix: <one line>\n"
|
||||
@@ -147,9 +147,9 @@ def grade(rec):
|
||||
phase, source, reply = rec["phase"], rec["source"], rec["reply"]
|
||||
has_reply = bool(reply and reply.strip())
|
||||
if phase == "done" and source == "reply" and has_reply:
|
||||
return "OK", "clean explicit bridge_reply"
|
||||
return "OK", "clean explicit fleet_reply"
|
||||
if phase == "done" and has_reply:
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call bridge_reply)"
|
||||
return "DEGRADED", f"resolved via {source} (worker did not call fleet_reply)"
|
||||
if phase == "done" and not has_reply:
|
||||
return "EMPTY", "turn completed but reply was empty"
|
||||
if phase == "failed":
|
||||
|
||||
@@ -0,0 +1,16 @@
|
||||
# Build output
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
||||
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
||||
fleetd.yaml
|
||||
bridged.yaml
|
||||
|
||||
# CB-505 audit trail + daemon stdout/stderr — runtime records, never source
|
||||
logs/
|
||||
|
||||
# Editor / OS
|
||||
*.iml
|
||||
.idea/
|
||||
.DS_Store
|
||||
@@ -5,10 +5,10 @@
|
||||
|
||||
## Why this exists
|
||||
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `bridge_send` is
|
||||
Stage 2 gave a worker→primary reply a **durable place to wait** when no `fleet_send` is
|
||||
open: it lands in `agent.<target>.inbox` on the broker and survives a daemon bounce. But
|
||||
delivery is still **pull** — the primary only sees the reply if it happens to call
|
||||
`bridge_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
`fleet_poll(target)` / `GET /sessions/{id}/replies`. A reply can sit indefinitely while
|
||||
the primary works on something else.
|
||||
|
||||
This layer makes delivery **active**: the bridge *pushes* a nudge to the primary the moment
|
||||
@@ -26,11 +26,11 @@ pointed at the primary's pane instead.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
W["worker"] -->|"bridge_reply (no open send)"| MS["MessageService.reply"]
|
||||
W["worker"] -->|"fleet_reply (no open send)"| MS["MessageService.reply"]
|
||||
MS -->|"inbox.publish"| INBOX[("agent.<target>.inbox<br/>(durable, LavinMQ)")]
|
||||
MS -->|"notify"| LOOP["ReplyPushLoop"]
|
||||
LOOP -->|"status-gated inject"| PANE["primary's herdr pane"]
|
||||
PANE -->|"primary drains"| DRAIN["bridge_poll(target)<br/>= peek + ack"]
|
||||
PANE -->|"primary drains"| DRAIN["fleet_poll(target)<br/>= peek + ack"]
|
||||
DRAIN -->|"inbox now empty"| LOOP
|
||||
LOOP -.->|"still non-empty →<br/>re-inject on backoff"| PANE
|
||||
classDef store fill:#2c5282,stroke:#1a365d,color:#ffffff;
|
||||
@@ -51,13 +51,13 @@ the caller runs in a herdr pane on this host. Today it's discarded for the prima
|
||||
(`presence.markPresent` is a no-op on it).
|
||||
|
||||
**Plan:** a single-slot `PrimaryRegistry` (thread-safe) holding the primary's `terminal_id`.
|
||||
Populate it from the **orchestration-side** MCP tools — `bridge_send`, `bridge_spawn`,
|
||||
`bridge_poll`, `bridge_list`, `bridge_status`, `bridge_profiles` — capturing
|
||||
Populate it from the **orchestration-side** MCP tools — `fleet_send`, `fleet_spawn`,
|
||||
`fleet_poll`, `fleet_list`, `fleet_status`, `fleet_profiles` — capturing
|
||||
`callerTerminal(exchange)` when it is (a) non-null and (b) **not** a registered worker
|
||||
session in `SessionManager`. That caller is, by construction, the primary. Worker-side tools
|
||||
(`bridge_reply`, `bridge_ask`) never set it.
|
||||
(`fleet_reply`, `fleet_ask`) never set it.
|
||||
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `BridgedConfig`
|
||||
- **Config override / pin:** a `primary: { terminal: "<id>" }` block in `FleetConfig`
|
||||
(nested record, same shape as `Broker`). Lets an operator pin it, or supply it when
|
||||
derivation can't (see degrade case).
|
||||
- **Degrade:** if the primary is off-host or in a non-herdr terminal, `terminalForPid`
|
||||
@@ -71,13 +71,13 @@ A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
(`MessageService.reply` → the `inbox.publish` branch, `MessageService.java:192`).
|
||||
|
||||
- **Inject a nudge, not the payload.** The injected turn tells the primary *to drain*
|
||||
(e.g. "Worker `<target>` returned a reply — run `bridge_poll(target=<target>)` to collect
|
||||
(e.g. "Worker `<target>` returned a reply — run `fleet_poll(target=<target>)` to collect
|
||||
it"), it does **not** carry the reply text. Rationale: replies can be large/multiline and
|
||||
terminal injection would mangle them; the drain response is the clean transport. Keeps the
|
||||
push idempotent — re-nudging is harmless.
|
||||
- **Ack = drain.** The primary draining (`drainReplies` = peek + ack) is the acknowledgement.
|
||||
The loop's **stop condition is `inbox.peek(target).isEmpty()`** — the reply is gone from the
|
||||
inbox because it was acked. No new `bridge_ack` tool needed for v1 (see Increment 3).
|
||||
inbox because it was acked. No new `fleet_ack` tool needed for v1 (see Increment 3).
|
||||
- **Status-gated injection (mechanism (b), chosen).** A dedicated lightweight scheduled loop,
|
||||
**not** the worker `Injector`. It injects via `AgentControl.send(primaryTerminal, nudge)`
|
||||
(the same herdr `agent.send` = `pane send-text` + submit that delivers to workers) only when
|
||||
@@ -93,10 +93,10 @@ A `ReplyPushLoop` component, notified at the single no-waiter call site
|
||||
the reply remains in the durable inbox and the next natural poll (or a later worker reply's
|
||||
nudge) still surfaces it. Bounded so the bridge never spams the primary.
|
||||
|
||||
### Increment 3 — optional per-`msgId` `bridge_ack` tool (deferred)
|
||||
### Increment 3 — optional per-`msgId` `fleet_ack` tool (deferred)
|
||||
|
||||
Drain-as-ack is coarse: it clears *all* pending replies for a target at once. If finer
|
||||
control is ever needed (ack one reply, leave others held), add a `bridge_ack(msgId)` tool
|
||||
control is ever needed (ack one reply, leave others held), add a `fleet_ack(msgId)` tool
|
||||
mapping to `inbox.ack(target, msgId)` — the port already supports per-`msgId` ack. Not built
|
||||
in v1; the stop-on-empty loop is sufficient.
|
||||
|
||||
@@ -106,7 +106,7 @@ in v1; the stop-on-empty loop is sufficient.
|
||||
resolved terminal is non-null **and not a registered worker session**, seen on an
|
||||
orchestration-side tool. This never mislabels a worker (workers are in `SessionManager`)
|
||||
and needs no new env var or argument (identity stays connection-derived, per the existing
|
||||
`BridgeMcp` invariant).
|
||||
`FleetMcp` invariant).
|
||||
|
||||
2. **Readiness-gate mismatch → dedicated loop.** The existing `Injector` gates delivery on
|
||||
`ready.test(target)` = `WorkerPresence` (the *worker's* MCP connected). The primary is not
|
||||
@@ -132,7 +132,7 @@ boundary**. The bridge is signalling the primary that it has mail — not drivin
|
||||
non-null terminal AND not a registered session" predicate; the loop's stop-on-empty and
|
||||
bounded-reminder logic with an injected clock + a fake injector (no real herdr).
|
||||
- **Live dogfood (primary-side):** with the daemon on the broker jar + a real worker,
|
||||
delegate a task, let the worker reply after the `bridge_send` window closes, and observe the
|
||||
delegate a task, let the worker reply after the `fleet_send` window closes, and observe the
|
||||
bridge inject a drain nudge into *this* primary pane; confirm draining stops the reminders;
|
||||
confirm an unreachable primary (registry empty) degrades to pull with no loss.
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
# bridged configuration (example). Copy to bridged.yaml and adjust.
|
||||
# fleetd configuration (example). Copy to fleetd.yaml and adjust.
|
||||
#
|
||||
# bridged is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# fleetd is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — bridged REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
# below — fleetd REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
@@ -21,7 +21,7 @@ bind:
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# BRIDGED_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
# FLEETD_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
@@ -29,13 +29,13 @@ bind:
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: BRIDGED_API_TOKEN
|
||||
# tokenEnv: FLEETD_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
# the pane mapping is as unforgeable as a worker's), and reply nudges are pushed to it.
|
||||
# REQUIRED when the primary runs inside a herdr pane — without it the pane match reads the
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from bridge_whoami; re-pin if
|
||||
# primary as a worker and refuses spawn/send/stop. Get the id from fleet_whoami; re-pin if
|
||||
# the primary moves panes.
|
||||
# primary:
|
||||
# terminal: term_0123456789abcd
|
||||
@@ -48,13 +48,13 @@ bind:
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# tab → the ONLY field identity depends on (CB-579); the exact label of the tab hosting the lead.
|
||||
# Label the tab yourself, or let bridged label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by bridge_whoami
|
||||
# Label the tab yourself, or let fleetd label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by fleet_whoami
|
||||
#
|
||||
# A lead's tab must already carry its label (or be launched by bridged, which labels it) — there is
|
||||
# A lead's tab must already carry its label (or be launched by fleetd, which labels it) — there is
|
||||
# no terminal id to paste in and nothing to re-pin when the session restarts: the tab survives, so
|
||||
# the same label resolves the same lead again on the next scan.
|
||||
# `bridge_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# `fleet_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
# IS a primary for authorization, so nothing that keys on the role breaks.
|
||||
#
|
||||
# KEEP `primary:` when adding leads: it still addresses the CB-307 push loop, which needs a single
|
||||
@@ -63,13 +63,13 @@ bind:
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing bridged places can land in a matching tab;
|
||||
# member spaces are excluded from the scan, so nothing fleetd places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
|
||||
# CB-551: IDLE-LEAD HEARTBEAT — nudge the single lead back to work when it has been continuously
|
||||
# idle (no open bridge_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# idle (no open fleet_send driving it) past the quiet period. The fleet is one lead + architects +
|
||||
# workers, so a lead that stalls is a single point of failure; the ReplyPushLoop only nudges when a
|
||||
# reply lands, and this timer catches the gap where nothing lands and the lead just sits idle.
|
||||
#
|
||||
@@ -93,28 +93,47 @@ bind:
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
# Math.max(15, intervalSeconds), so a lower value is silently raised, not
|
||||
# rejected.
|
||||
# workingSuspectAfterSeconds, paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by
|
||||
# anything — the dormant monitor only consumes intervalSeconds today (CB-573
|
||||
# shipped ahead of the evidence publishers these two knobs are for). Setting
|
||||
# them changes nothing right now, and no minimum is enforced on either, because
|
||||
# nothing reads them to enforce one. They exist so a later build can start
|
||||
# honouring them without another config-shape change.
|
||||
# notifications.mode → "webhook" flips what bridge_list REPORTS (healthCoverage: "full" instead
|
||||
# of "detection-only") — it does NOT make bridged send any webhook call; no
|
||||
# workingSuspectAfterSeconds → age before a BUSY member is suspected of a stall (default 600).
|
||||
# ENFORCED floor of 300: a lower value is silently raised.
|
||||
# paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by anything. Setting it changes
|
||||
# nothing right now. It exists so a later build can start honouring it without
|
||||
# another config-shape change.
|
||||
# notifications.mode → "webhook" flips what fleet_list REPORTS (healthCoverage: "full" instead
|
||||
# of "detection-only") — it does NOT make fleetd send any webhook call; no
|
||||
# delivery mechanism is implemented yet. Any other value, or omitting the
|
||||
# block, reports "detection-only".
|
||||
# health:
|
||||
# enabled: true
|
||||
# intervalSeconds: 30
|
||||
# workingSuspectAfterSeconds: 600
|
||||
# paneProbeIntervalSeconds: 60
|
||||
# intervalSeconds: 30 # floor 15
|
||||
# workingSuspectAfterSeconds: 600 # floor 300 — how long BUSY with no activity means STALL_SUSPECTED
|
||||
# paneProbeIntervalSeconds: 60 # parsed, but nothing reads it yet — changing it changes nothing
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# Idle-sleep guard: while at least one member is live, hold an OS-level assertion against idle
|
||||
# sleep (macOS only — a `caffeinate -i` child; a no-op elsewhere or if caffeinate is missing), so
|
||||
# an unattended host does not idle-sleep out from under a member's long turn. Unlike health/
|
||||
# configReload above, this is ON BY DEFAULT — omitting the block entirely leaves it enabled, the
|
||||
# same as `enabled: true`. Uncomment only to turn it off:
|
||||
# idleSleepGuard:
|
||||
# enabled: false
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# Optional socket for member panes. Omit this to use herdrSocket for both leads and members.
|
||||
# memberHerdrSocket: /Users/member/.config/herdr/herdr.sock
|
||||
|
||||
# fleetd #213: the login shell the member OS user (memberHerdrSocket above) actually runs. ONLY
|
||||
# read when memberHerdrSocket is set — fleetd's own $SHELL says nothing about a pane running
|
||||
# under a different OS user, and there is no channel to ask herdr for that user's shell, so this
|
||||
# must be told rather than guessed. Absent, blank, or anything not ending in "zsh" is treated the
|
||||
# same as "not zsh": the memberCredentials.policy: allow-list ZDOTDIR scrub (see worktreeGroup
|
||||
# below) is skipped in favour of the weaker CB-596 sentinel overlay — a degraded control, never a
|
||||
# refusal to spawn. When memberHerdrSocket is absent this key is never consulted at all.
|
||||
# memberLoginShell: /bin/zsh
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
@@ -124,8 +143,34 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → bridged mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# mcpUrl → fleetd mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# ideMcpUrl → opt-in (CB-634), default off. When set, fleetd mounts the IDE Index MCP as a
|
||||
# second inline server named `intellij`, and adds an IDE charter that pins every
|
||||
# ide_* call to the member's own worktree. A URL, not a boolean — host and port
|
||||
# are host-specific. Set it only on a host where the IDE actually runs.
|
||||
# ideProjectDir → repo-relative module dir the IDE opens and the overlay pins (CB-634). Only read
|
||||
# when ideMcpUrl is set. This repo's Maven pom lives in `fleetd/`, not at the
|
||||
# worktree root, so opening the root imports no module and ide_* resolves nothing;
|
||||
# set this to `fleetd`. Omit for a repo whose project is the worktree root.
|
||||
# ideOpenCommand → host command that opens ideProjectDir in the IDE at spawn (CB-634 auto-open).
|
||||
# Only read when ideMcpUrl is set. `{dir}` is replaced with the absolute module
|
||||
# dir and the command runs through `/bin/sh -c`, so set env inline if needed —
|
||||
# e.g. `env DISPLAY=:10.0 idea {dir}`. Best-effort: a failure is logged, never
|
||||
# fails the spawn. Omit to open the member's module by hand. There is no close
|
||||
# half yet — an opened module stays open until the operator closes it.
|
||||
# autoCompactWindow → opt-in, default off. A bounded token window that forces a spawned member to
|
||||
# compact its context instead of running on the backend's own default and dying
|
||||
# mid-turn (losing its fleet_reply — the whole point of the turn — with it).
|
||||
# Validated at config load to [100000, 1000000] — the band Claude Code's own
|
||||
# --autocompact flag accepts.
|
||||
# CROSS-BACKEND SEMANTICS DIFFER: on claude-code this is a launch-time
|
||||
# `--autocompact <tokens>` flag — the member compacts AT this window. opencode
|
||||
# has no equivalent flag (it only forces `compaction.auto: true`, unconditionally,
|
||||
# already), so this is instead applied as the model's `limit.context` in the
|
||||
# generated opencode.json — the member compacts WITHIN this window, not exactly
|
||||
# at it — and only when this profile's `model:` is in `provider/model` form; if it
|
||||
# isn't, fleetd logs a WARN naming the profile rather than silently doing nothing.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
@@ -135,7 +180,12 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.claude/settings.local.json, .env, .envrc].
|
||||
# [.env] only (CB-148). .envrc is left out of the default on purpose: it is
|
||||
# executable shell that direnv runs on every cd, so copying it carries
|
||||
# behaviour into the worker, not just values, unlike .env. An operator who
|
||||
# wants it copied can still write parityOverlay: [.env, .envrc] explicitly.
|
||||
# (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
# mounts — the bridge, and nothing else. Replicating the primary's MCP config
|
||||
@@ -143,7 +193,7 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# bridged neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# fleetd neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
@@ -152,11 +202,11 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# gitHostEnv → host env var holding the forge host (default GITEA_HOST). Injected as
|
||||
# GITEA_HOST *only* alongside a resolved gitTokenEnv.
|
||||
# exhaustedPattern → regex matched against a completion-fallback scrape (CB-578 stage A) to
|
||||
# classify a turn that ended with no bridge_reply as the backend having
|
||||
# classify a turn that ended with no fleet_reply as the backend having
|
||||
# refused on a subscription usage limit, rather than a real answer. Opt-in —
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into bridged itself.
|
||||
# vendor string baked into fleetd itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
@@ -168,18 +218,51 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# errorPattern → fleetd #201 / #227: regex matched against a completion-fallback scrape to
|
||||
# classify a turn that ended with no fleet_reply as a BACKEND ERROR — a
|
||||
# credential outage or a provider 5xx — rather than a real answer or a
|
||||
# usage-limit exhaustion (exhaustedPattern above always wins when a line
|
||||
# matches both). Opt-in. Omit it and this profile falls back to fleetd's
|
||||
# built-in legacy pattern `(?i)\bAPI Error\s*:` — classification still
|
||||
# happens, just without a profile-specific match; every backend words its
|
||||
# failure differently, so a hardcoded sentence would only ever match one
|
||||
# of them.
|
||||
# DEFERRED: compiled once into a startup pattern map, same as exhaustedPattern
|
||||
# — editing it needs a daemon restart.
|
||||
# # errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage
|
||||
#
|
||||
# What happens once a match fires (BackendOutagePolicy, credentialId-keyed,
|
||||
# SEPARATE from the CB-578 stage B quarantine above and never merged with it):
|
||||
# - threshold 2 — TWO DISTINCT TARGETS (never raw events) on the same
|
||||
# effective credential inside a 60-second window start an "incident" and a
|
||||
# 60-second cool-off for that credential. One member repeating the same
|
||||
# classified line twice never cools anything off — a real outage hits
|
||||
# every target on that credential, so requiring a second, independent
|
||||
# target loses nothing against the case this guards against, while
|
||||
# protecting against a heuristic misfire on one flaky member.
|
||||
# - a fresh error while a credential is already cooling off is ignored
|
||||
# outright: it neither extends the 60s deadline nor starts a new incident.
|
||||
# - `fleet_list`/`fleet_profiles` report a cooling credential with
|
||||
# `coolingOffForSeconds` (never `quarantinedForSeconds`, unless CB-578
|
||||
# exhaustion quarantine is ALSO independently active for the same
|
||||
# credential — the two checks can both fire at once). A spawn onto a
|
||||
# cooling profile is refused with a message naming the credential and
|
||||
# remaining seconds — "cooling off", never "exhausted", so an operator can
|
||||
# tell a short transient fault from a spent subscription at a glance.
|
||||
# - the lead gets ONE nudge per incident (not one per affected target), via
|
||||
# the same push loop that already delivers ticket/question reminders.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. bridged hands herdr an
|
||||
# A worker's environment does NOT come from your shell. fleetd hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# bridged now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# fleetd now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.bridged.plist and deploy/bridged.service.
|
||||
# lines in deploy/dev.ltms.fleet.plist and deploy/fleetd.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
@@ -192,33 +275,96 @@ profiles:
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: BRIDGED_WORKER_TOKEN
|
||||
tokenEnv: FLEETD_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
# weight: relative selection weight for automatic placement (weighted, round-robin, and
|
||||
# fixed's fallback walk). Absent defaults to 1.0. An explicit 0 or negative value means
|
||||
# "never auto-select this profile" (CB-554) — it stays reachable via an explicit
|
||||
# `bridge_spawn{profile:"gx10"}`, which bypasses placement entirely; only automatic
|
||||
# `fleet_spawn{profile:"gx10"}`, which bypasses placement entirely; only automatic
|
||||
# selection skips it.
|
||||
weight: 0.5
|
||||
# maxLoad: max live workers on this profile. Omit for unlimited. An explicit 0 (CB-585) caps
|
||||
# the profile at zero live members — it is excluded from automatic placement and an explicit
|
||||
# `bridge_spawn{profile:"gx10"}` against it is refused too; a cap holds even when the profile
|
||||
# `fleet_spawn{profile:"gx10"}` against it is refused too; a cap holds even when the profile
|
||||
# is named directly. Negative is refused at config load — there is no sane meaning for it.
|
||||
maxLoad: 2
|
||||
# subscription: true
|
||||
# THE KNOB THAT DECIDES WHO PAYS (CB-539). Default false. When true, this profile's members
|
||||
# run on the OPERATOR'S OWN Claude subscription instead of a metered endpoint — every spawn
|
||||
# bills your plan and eats your usage limit. Off-subscription is the whole point of this
|
||||
# daemon, so treat `true` as a deliberate exception, not a convenience.
|
||||
#
|
||||
# What changes when it is set (ClaudeCodeLauncher):
|
||||
# - no ANTHROPIC_BASE_URL and no ANTHROPIC_AUTH_TOKEN are injected — the member inherits
|
||||
# the operator's own Claude Code auth, which is exactly why it bills the plan;
|
||||
# - SubscriptionGuard never vets it, because there is no baseUrl to vet;
|
||||
# - no token is required, so `tokenEnv` is irrelevant here.
|
||||
#
|
||||
# MUTUALLY EXCLUSIVE with `baseUrl` — setting both is refused at config load (CB-542). On the
|
||||
# subscription path no guard would vet the URL, so allowing both would be a way around the
|
||||
# guard rather than a configuration.
|
||||
#
|
||||
# GOTCHA 1 — it is invisible to the startup secret check. `Fleetd.reportRequiredSecrets`
|
||||
# skips subscription profiles on purpose (they need no token), so a boot log that reports
|
||||
# every secret as fine says nothing about these profiles.
|
||||
#
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
#
|
||||
# GOTCHA 3 (fleetd #176, corrected by fleetd #257) — `maxLoad` counts members, never the lead
|
||||
# itself. The lead is a live `claude` session on this SAME account (a lead is never moved
|
||||
# off-subscription, whatever its own profile says), so it already holds one seat before any
|
||||
# member spawns. If a lead's `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER
|
||||
# `subscription: true` profile that shares this one's account (see THE SENTINEL, just below,
|
||||
# next to `credentialId:`) — `fleet_list` reports that seat count under `leadSeats`; see
|
||||
# `profile:` under THE FLEET below. `free` itself is NEVER reduced by `leadSeats`: `free` means
|
||||
# "what the real placement gate (`CompositePeerLauncher#enforceMaxLoad`) will actually grant a
|
||||
# fresh `fleet_spawn` right now", and that gate only ever compares live members against
|
||||
# `maxLoad` — it has no notion of the lead's own seat. An earlier cut of this feature
|
||||
# subtracted `leadSeats` from `free` on the theory it made `free` describe the true ceiling on
|
||||
# the account, but no backend seat ceiling shared with the lead has ever actually been
|
||||
# measured, and the subtraction just made `free` disagree with the one thing it is supposed to
|
||||
# describe — the fleetd #257 fix. `maxLoad: 3` means 3 member slots, full stop; a lead sharing
|
||||
# the account is a fact you can see in `leadSeats`, not a reason `free` undercounts spawns that
|
||||
# will, in practice, succeed.
|
||||
#
|
||||
# THE SENTINEL (fleetd #176 stage 2, correcting an inert stage 1 fix): every `subscription:
|
||||
# true` profile that leaves `credentialId` unset shares ONE implicit account-wide credential
|
||||
# id with every other such profile on this host — because a subscription profile doesn't
|
||||
# authenticate with a credential of its own, it authenticates as the operator's own Claude
|
||||
# login, and there is exactly one of those. So on a typical host, `opus` (the lead's profile)
|
||||
# and `sonnet` (the members' profile) are linked automatically, with NOTHING to set here — that
|
||||
# is what makes GOTCHA 3 above work without also writing matching `credentialId:` values on
|
||||
# both. This linkage is not just cosmetic: it is the same key `BackendQuarantine`/cool-off use,
|
||||
# so a usage-limit hit on `opus` now quarantines `sonnet` too (and vice versa) — correct, since
|
||||
# they are one Claude account, but worth knowing before you wonder why an unrelated-looking
|
||||
# profile went quarantined.
|
||||
#
|
||||
# WHEN TO OVERRIDE — set explicit, DIFFERENT `credentialId:` values on two `subscription: true`
|
||||
# profiles only when they are genuinely two separate Claude logins on the same host (a real,
|
||||
# if unusual, setup). An explicit `credentialId` always wins over the sentinel, so this is the
|
||||
# one way to keep two subscription profiles from being treated as one account for lead-seat
|
||||
# counting AND for quarantine/cool-off grouping alike.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage (fleetd #201/#227) — see the key doc above
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".claude/settings.local.json", ".env", ".envrc"] # never add .mcp.json — see above
|
||||
# parityOverlay: [".env"] # the default; add ".envrc" explicitly if you want it copied too (CB-148) — never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
# autoCompactWindow: 250000 # opt-in: bound member context; claude-code compacts AT this, opencode within it (model limit.context)
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
@@ -236,15 +382,15 @@ profiles:
|
||||
# The bridge MCP + reply charter mount via a generated OPENCODE_CONFIG file, and the model is a
|
||||
# `provider/model` selector. Placement, tabs, cwd, and the readiness gate are shared with Claude.
|
||||
#
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → bridge_send →
|
||||
# structured bridge_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# Dogfood-verified 2026-07-29 against opencode 1.18.5 (spawn → readiness gate → fleet_send →
|
||||
# structured fleet_reply → teardown). The `opencode/*-free` models run on opencode's own gateway
|
||||
# and need NO credentials — check `opencode models` for the current free list, since the names
|
||||
# change. That also makes the worker off-subscription by construction.
|
||||
# opencode-free:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
@@ -269,7 +415,7 @@ profiles:
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
@@ -304,11 +450,17 @@ placement: weighted
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
#
|
||||
# This does NOT govern the fleetd #201 / #227 backend-error cool-off documented under errorPattern
|
||||
# above — that mechanism is a separate, shorter-lived, NOT-configurable policy (threshold 2 distinct
|
||||
# targets, 60-second window, 60-second cool-off), on purpose: it exists to survive a brief transient
|
||||
# fault, not to replace this 30-minute exhaustion quarantine. Do not conflate the two when reading
|
||||
# fleet_list/fleet_profiles — coolingOffForSeconds and quarantinedForSeconds are independent facts.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded bridged keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. bridged checks the file's modified time on a timer and
|
||||
# upgraded fleetd keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. fleetd checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
@@ -319,7 +471,7 @@ placement: weighted
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Bridged.main reads it once at startup to build the lead tab
|
||||
# EXCEPT `fleet.leaders`: Fleetd.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
# with nothing in the deferred list — but has NO effect until you restart. Treat it
|
||||
@@ -330,8 +482,10 @@ placement: weighted
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern, errorPattern (fleetd #201 / #227 —
|
||||
# compiled once into a startup pattern map the same way exhaustedPattern is). The
|
||||
# launcher takes a copy of `profiles:` at startup and resolves every spawn out of
|
||||
# that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
@@ -392,6 +546,23 @@ fleet:
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `profile:` has a SECOND job as of fleetd #176, even for a recognise-only lead you never want
|
||||
# auto-launched: it is also how fleetd learns which account this lead's own session shares. A
|
||||
# `subscription: true` profile bills the operator's Claude account, and the lead itself is always
|
||||
# a live `claude` session on that same account — `maxLoad` never counted that seat. If a lead
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list` reports the
|
||||
# lead's live seat(s) on that worker profile under `leadSeats` — informational only, as of fleetd
|
||||
# #257 it is NEVER subtracted from `free` (see GOTCHA 3, next to `maxLoad:`, in THE WORKERS above,
|
||||
# for why). "Shares the account" is decided by matching `effectiveCredentialId()`, which (fleetd
|
||||
# #176 stage 2 — see THE SENTINEL, next to `credentialId:`, in THE WORKERS above) means: an
|
||||
# explicit, matching `credentialId:` on both, OR — the common case, needing NO extra config — both
|
||||
# being `subscription: true` with `credentialId` left unset, since those all share one implicit
|
||||
# account-wide id. A lead on `opus` and workers on `sonnet` link automatically this way; they do
|
||||
# NOT need the same profile name. Setting `profile:` on an already-running, recognise-only lead is
|
||||
# safe — the daemon only launches the SHORTFALL below `instances`, so naming a profile here does
|
||||
# not, by itself, start anything. Omit it and fleetd has no way to derive the sharing — there is
|
||||
# no other reliable signal on the daemon's side — so that lead's seat never appears in `leadSeats`.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
@@ -423,8 +594,8 @@ fleet:
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: bridged's own)
|
||||
# kind: claude # descriptive; reported by bridge_whoami
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
||||
# kind: claude # descriptive; reported by fleet_whoami
|
||||
# gpt-sol-5.6:
|
||||
# tab: "lead: gpt-sol-5.6"
|
||||
# kind: opencode
|
||||
@@ -452,7 +623,7 @@ guard:
|
||||
|
||||
# Member credential policy (CB-596, gitea issue #82). A herdr pane runs a LOGIN shell, and that
|
||||
# shell re-sources the operator's own secret store — so a spawned member inherits every credential
|
||||
# the operator's shell holds, not just the ones bridged means to give it. Measured on this host:
|
||||
# the operator's shell holds, not just the ones fleetd means to give it. Measured on this host:
|
||||
# 31 credential names, all set, with only ONE (GITEA_ACCESS_TOKEN) blocked before this — and that
|
||||
# block was a single name hardcoded in HerdrPeerLauncher.java, not driven by this file. This block
|
||||
# replaces that hardcoded shadow with a config-driven list of names.
|
||||
@@ -482,20 +653,66 @@ guard:
|
||||
# through either: the daemon logs a WARN naming any credential-shaped env var it finds on neither
|
||||
# list (never its value), so a secret added to the store later does not go unnoticed forever.
|
||||
#
|
||||
# policy → only "deny-by-default" exists today (an operator-authored deny-list was deliberately
|
||||
# rejected — see above). An unrecognized value refuses to start, naming it.
|
||||
# allow → credential names a member legitimately needs. Left OUT of the pane's env overlay
|
||||
# entirely, so the value the pane's own (login) shell exports passes through untouched.
|
||||
# known → every credential name the operator's store is known to export. Every name here NOT
|
||||
# also in `allow` is overlaid with a non-secret sentinel value before the pane's login
|
||||
# shell runs — real protection only for names the login shell does not itself re-export
|
||||
# (see the ROUND-2 CORRECTION note above for the ones it does).
|
||||
# policy → "deny-by-default" (the default; also accepted spelled "deny-list") overlays each
|
||||
# known-but-not-allowed name BEFORE the pane's login shell runs — real protection only
|
||||
# where that shell does not re-export the name (see ROUND-2 CORRECTION above). An
|
||||
# unrecognized value refuses to start, naming it.
|
||||
# policy → "allow-list" (CB-633) moves the control to a per-spawn ZDOTDIR directory the daemon
|
||||
# generates and passes through tab.create's env map. Each generated startup file sources
|
||||
# its ~/ counterpart FIRST and then runs the scrub, so the scrub happens after the
|
||||
# operator's whole chain and no sourced file can undo it.
|
||||
# The scrub is sourced from BOTH the generated .zshrc and the generated .zlogin, because
|
||||
# herdr does not open the same kind of shell everywhere: macOS panes run a LOGIN zsh (so
|
||||
# .zlogin runs), Linux panes run a plain interactive zsh (so .zlogin never runs at all).
|
||||
# A scrub in .zlogin alone would be a control that silently does nothing on Linux.
|
||||
# The allow-list is DERIVED, never typed:
|
||||
# every profile's tokenEnv/gitTokenEnv/gitHostEnv values and env-map keys, plus an
|
||||
# infrastructure set (PATH HOME SHELL TERM LANG LC_* TMPDIR USER LOGNAME PWD SHLVL EDITOR
|
||||
# PAGER JAVA_HOME XDG_* ZDOTDIR), plus whatever keys this spawn's own env overlay carries.
|
||||
# Adding a profile can therefore only widen the list, never break another spawn's scrub.
|
||||
# Under this policy `known`/`allow` below become REPORTING ONLY — they feed the gap WARN,
|
||||
# they are no longer a control. If the member's login shell is NOT zsh, the daemon logs a
|
||||
# loud WARN saying protection is off and falls back to deny-by-default's overlay.
|
||||
# Each pane writes a scrub-report.txt naming how many variables it kept of how many it
|
||||
# saw; the daemon logs that "allowed N of M" line when the pane stops. If the report is
|
||||
# MISSING the daemon logs a WARN instead — the scrub cannot then be confirmed to have
|
||||
# run, and a silently-dead control is exactly what this policy exists to prevent.
|
||||
# allow → credential names a member legitimately needs. Under deny-by-default, left OUT of the
|
||||
# pane's env overlay entirely, so the value the pane's own (login) shell exports passes
|
||||
# through untouched. Under allow-list: reporting only.
|
||||
# known → every credential name the operator's store is known to export. Under deny-by-default,
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAgentEnv → whether SSH_AUTH_SOCK may pass through under allow-list ("inherit") or is omitted
|
||||
# from the member environment ("omit", the default). Omitting it only omits the
|
||||
# inherited ssh-agent path. It discourages automatic use of the operator's agent.
|
||||
# It does not deny same-user access to that socket. It also does not block SSH keys that
|
||||
# are readable on disk. Git over SSH may still work from inside a member. Keep the block:
|
||||
# it is correct and costs nothing, but it is not a control. A member runs as the same OS
|
||||
# user as the lead. Inside one uid, ordinary Unix permissions provide no meaningful
|
||||
# confidentiality boundary. A real boundary needs a different OS user or OS-level
|
||||
# confinement, such as a container or VM. That is the open question in fleetd #184.
|
||||
#
|
||||
# Still do not set this to "inherit" casually. SSH_AUTH_SOCK is a live handle to YOUR
|
||||
# ssh-agent, so a member holding it can sign with EVERY key the agent holds. It sits in
|
||||
# no secret file and looks like no credential, which is why it slipped past three
|
||||
# earlier tickets (gitea #110). Blocking it does not contain a member, but allowing it
|
||||
# hands one a signing capability for no gain — the block costs nothing, so keep it.
|
||||
#
|
||||
# Both halves of this are measured, not argued. 2026-08-28: a member with
|
||||
# SSH_AUTH_SOCK blanked pushed to the forge over SSH successfully, because `ssh -G`
|
||||
# resolves an IdentityFile outside ~/.ssh that is readable and has no passphrase. An
|
||||
# earlier version of this comment claimed blocking the socket BREAKS git over SSH. It
|
||||
# does not. That claim came from looking only in ~/.ssh, which holds nothing but four
|
||||
# `Include` lines — looking in one place and concluding about the whole host.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAgentEnv: omit # allow-list only; see the sshAgentEnv note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
@@ -547,6 +764,37 @@ guard:
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Worktree group sharing (fleetd #185 stage 3). OPTIONAL, off by default. Names an OS group
|
||||
# that a provisioned worktree's repo is made group-writable for (git config
|
||||
# core.sharedRepository group, plus a one-time chgrp/chmod/setgid fix-up), so a member spawned
|
||||
# under a DIFFERENT OS user (see memberHerdrSocket) can write its own worktree, its
|
||||
# per-worktree git metadata, and its own commit objects — without it, every file GitWorktrees
|
||||
# creates is owned by fleetd's own uid and unwritable by another user.
|
||||
# CAUTION: this isolates credentials, not the repository — a member in the group can still
|
||||
# write the operator's git objects and refs in the shared repo. The operator running fleetd
|
||||
# must already be a member of the named group, or every provisioning spawn fails loudly.
|
||||
#
|
||||
# fleetd #213: this is also the ONE group the memberCredentials.policy: allow-list ZDOTDIR scrub
|
||||
# reuses when memberHerdrSocket is set — deliberately not a second config key. Under
|
||||
# memberHerdrSocket, the scrub directory is generated under worktreeRoot (never java.io.tmpdir,
|
||||
# which the member OS user cannot reach) and shared read-only with this group. If worktreeGroup
|
||||
# is unset while memberHerdrSocket is set, the scrub cannot be guaranteed reachable by the member,
|
||||
# so fleetd falls back to the weaker CB-596 sentinel overlay instead (a WARN names the gap).
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# fleetd #362: a directory of skill folders (each a subdirectory holding a SKILL.md, the same
|
||||
# shape as this repo's own .claude/skills/) copied into every PROVISIONED worktree's
|
||||
# .claude/skills/, so a member spawned against ANY repo — not only one that already ships its own
|
||||
# copy — can load a bridge skill (e.g. implementer). Unset (the default): no worktree is touched
|
||||
# beyond today's behaviour. A skill folder the target repo already carries under
|
||||
# .claude/skills/<name> is never overwritten — the repo's own copy always wins. Claude Code
|
||||
# members only; an opencode member reads a different path (.opencode/agent) this key does not
|
||||
# touch. Best-effort like worktreeGroup above: a missing/unreadable directory here is logged and
|
||||
# skipped, never a failed spawn. Every non-hidden subdirectory of this directory is copied
|
||||
# wholesale, with no per-file allowlist — don't park scratch files or drafts alongside the real
|
||||
# skill folders, they will be copied into every provisioned worktree too.
|
||||
# memberSkills: /path/to/fleetd/checkout/.claude/skills
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
@@ -570,14 +818,43 @@ guard:
|
||||
# RabbitMQ speaks the same AMQP 0-9-1, so it is a URI-only swap.
|
||||
# uri → AMQP connection URI. No trailing slash ⇒ the default vhost "/"; an empty path ("/")
|
||||
# is vhost "" and will NOT connect. Encode a named vhost as .../%2Fmyvhost.
|
||||
# uriEnv → CB-151: name of a host env var holding the AMQP URI, preferred over `uri` (wins
|
||||
# whenever set). The URI carries `user:pass@` inline, so naming a variable keeps the
|
||||
# password out of fleetd.yaml — same pattern as auth.tokenEnv/Profile.tokenEnv. A
|
||||
# uriEnv that resolves to an unset or blank variable is treated as NOT configured and
|
||||
# the daemon falls back to the in-memory inbox, warning loudly.
|
||||
# prefetch → CB-527: consumer basicQos, capping how many unacked messages the inbox holds
|
||||
# in-heap per owned target (the rest sits on the broker's durable queue instead of
|
||||
# growing the JVM heap). Default 32 when omitted.
|
||||
# broker:
|
||||
# uri: amqp://guest:guest@127.0.0.1:5672
|
||||
# uriEnv: LAVINMQ_URI
|
||||
# prefetch: 32
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open bridge_send,
|
||||
# Shared cross-host LEADER coordination broker. OMIT this block to leave lead-to-lead messaging
|
||||
# off entirely (config-only in this ticket — nothing here wires it into a live LeadMailbox yet).
|
||||
# This is a SEPARATE AMQP vhost from `broker:` above: member/worker inboxes always stay on the
|
||||
# per-fleet `broker:` vhost, and this vhost carries only leader-to-leader traffic, so two fleets
|
||||
# whose members must never see each other can still share one coordination vhost for their leads.
|
||||
# uriEnv → name of a host env var holding the coordination AMQP URI, same convention as
|
||||
# broker.uriEnv (keeps the credential out of fleetd.yaml). Wins over `uri` when set.
|
||||
# selfId → this daemon's own lead coord-id — the name its mailbox is owned under
|
||||
# (lead.<selfId>.inbox), e.g. "mac-opus" or "fleet01-lead". Must be globally unique
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# peers → fleetd #361: the coord-ids of the OTHER daemons on this vhost, declared by the
|
||||
# operator (the daemon never guesses). fleet_list reports each one's live reachability
|
||||
# (a passive queue check, never a presence protocol) alongside this daemon's own
|
||||
# mailbox state. Omit, or leave empty, for a daemon with no known peers yet — an
|
||||
# undeclared peer can still reach you and be reached by fleet_send, it just will not
|
||||
# show up as a row in fleet_list.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
# peers: [fleet01-lead]
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
# pane — status-gated (only when injectable, never mid-turn) and bounded. Ack = drain: the loop
|
||||
# stops as soon as the primary's inbox is empty.
|
||||
@@ -588,11 +865,11 @@ guard:
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just bridged-spawned ones — so such a primary is otherwise
|
||||
# EVERY pane — not just fleetd-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
# id off bridge_whoami (it reports the current terminal even while
|
||||
# id off fleet_whoami (it reports the current terminal even while
|
||||
# misclassified) and re-pin whenever the primary moves panes.
|
||||
# pushReminders → max nudges before giving up (default 5)
|
||||
# pushBackoffMs → delay between nudges in ms (default 15000)
|
||||
@@ -5,17 +5,17 @@
|
||||
<modelVersion>4.0.0</modelVersion>
|
||||
|
||||
<groupId>dev.ltms</groupId>
|
||||
<artifactId>bridged</artifactId>
|
||||
<artifactId>fleetd</artifactId>
|
||||
<version>1.0.0</version>
|
||||
<packaging>jar</packaging>
|
||||
|
||||
<name>bridged</name>
|
||||
<description>claude-bridge message server: sole gateway between primary/worker Claude sessions and herdr</description>
|
||||
<name>fleetd</name>
|
||||
<description>fleet message server: sole gateway between a lead session, its members, and herdr</description>
|
||||
|
||||
<properties>
|
||||
<maven.compiler.release>25</maven.compiler.release>
|
||||
<project.build.sourceEncoding>UTF-8</project.build.sourceEncoding>
|
||||
<mainClass>dev.ltms.bridged.Bridged</mainClass>
|
||||
<mainClass>dev.ltms.fleet.Fleetd</mainClass>
|
||||
|
||||
<jackson.version>2.19.0</jackson.version>
|
||||
<javalin.version>6.7.0</javalin.version>
|
||||
@@ -28,6 +28,8 @@
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
<archunit.version>1.5.0</archunit.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -44,6 +46,12 @@
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
|
||||
fleetd #206: org.xerial:sqlite-jdbc 3.53.4.0 (added for OpenCodeSessionDiscovery) — the
|
||||
only known advisory against this artifact is CVE-2023-32697 (RCE via an attacker-controlled
|
||||
JDBC URL), fixed in 3.41.2.2; 3.53.4.0 is well past that fix and OSV.dev reports no open
|
||||
advisory against it. Checked via the OSV.dev API (no Mend.io/JetBrains IDE MCP mount
|
||||
available from this worktree) on 2026-08-31.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
@@ -106,7 +114,7 @@
|
||||
</dependency>
|
||||
|
||||
<!-- MCP server: the SERVER face. Streamable-HTTP servlet mounted on Javalin's Jetty at
|
||||
/mcp, exposing bridge_send/bridge_reply/bridge_status as thin adapters over REST. -->
|
||||
/mcp, exposing fleet_send/fleet_reply/fleet_status as thin adapters over REST. -->
|
||||
<dependency>
|
||||
<groupId>io.modelcontextprotocol.sdk</groupId>
|
||||
<artifactId>mcp</artifactId>
|
||||
@@ -123,6 +131,17 @@
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #206: opencode moved its session store from a JSON tree to SQLite
|
||||
(opencode.db). This is the JDBC driver OpenCodeSessionDiscovery uses to read it
|
||||
read-only. Ships bundled native libraries (linux/mac/windows, several archs), so it
|
||||
is a heavier jar than most deps here — see the pom's dependency-security note below
|
||||
for the size/CVE tradeoff actually measured. -->
|
||||
<dependency>
|
||||
<groupId>org.xerial</groupId>
|
||||
<artifactId>sqlite-jdbc</artifactId>
|
||||
<version>${sqlite-jdbc.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -158,10 +177,21 @@
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #131: package-boundary and cycle enforcement (PackageCyclesTest). -->
|
||||
<dependency>
|
||||
<groupId>com.tngtech.archunit</groupId>
|
||||
<artifactId>archunit-junit5</artifactId>
|
||||
<version>${archunit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
<finalName>bridged</finalName>
|
||||
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
||||
launchd plist together. The installed plist names fleetd/target/fleetd.jar and
|
||||
KeepAlive is armed, so this name, the plist, and the wrapper must move as one. -->
|
||||
<finalName>fleetd</finalName>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
@@ -203,7 +233,7 @@
|
||||
</configuration>
|
||||
</plugin>
|
||||
|
||||
<!-- Runnable fat jar: java -jar target/bridged.jar -->
|
||||
<!-- Runnable fat jar: java -jar target/fleetd.jar -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-shade-plugin</artifactId>
|
||||
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
+2
-2
@@ -1,9 +1,9 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* The authorization table (CB-505), stated once and enforced on both entry paths.
|
||||
*
|
||||
* <p>Most of these rules are already true de facto — {@code BridgeMcp} derives a worker's identity
|
||||
* <p>Most of these rules are already true de facto — {@code FleetMcp} derives a worker's identity
|
||||
* from the connection rather than reading it from an argument, so a worker has never been able to
|
||||
* reply <em>as</em> another worker over MCP. What was missing is that the REST surface trusted the
|
||||
* session id in the URL path, and neither surface checked role at all. This class makes the
|
||||
+24
-12
@@ -1,7 +1,7 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.bridged.mcp.ConnectionIdentity;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.security.MessageDigest;
|
||||
@@ -12,7 +12,7 @@ import java.util.function.Supplier;
|
||||
/**
|
||||
* Resolves every caller to a {@link Principal}, for both entry paths into the core (CB-501).
|
||||
*
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code BridgeMcp}
|
||||
* <p>There are two of them and they are not layered the way the docs suggest: {@code FleetMcp}
|
||||
* calls the service layer directly and is mounted as a raw servlet (so it never passes through a
|
||||
* Javalin filter), while the REST routes historically resolved no identity at all. Both now
|
||||
* delegate here, so the authorization rules are stated once instead of drifting apart.
|
||||
@@ -59,7 +59,7 @@ public final class CallerResolver {
|
||||
* <p>Like {@link #leadTerminals}, a supplier rather than a fixed map, so a binding injected
|
||||
* after startup — when the later spawn lifecycle establishes a live architect session, or an
|
||||
* operator pins one — takes effect without a restart. Consulted per resolve; today's wiring
|
||||
* in {@code Bridged} reads a constant from config, which is the degenerate live case.
|
||||
* in {@code Fleetd} reads a constant from config, which is the degenerate live case.
|
||||
*/
|
||||
private final Supplier<Map<String, String>> architectTerminals;
|
||||
private final Function<String, MemberRole> memberSlotRoles;
|
||||
@@ -235,8 +235,16 @@ public final class CallerResolver {
|
||||
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (BridgedConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
//
|
||||
// fleetd #317: "not a worker" must not be conflated with "identity unresolved". The real
|
||||
// primary is a real process — its pid resolves (c.resolved()), it just owns no herdr pane.
|
||||
// A caller whose peer-PID lookup failed (LsofPeerPidLookup's -1 sentinel — on any failure,
|
||||
// silently including "lsof found no match") has no such pid, and PaneLocator's own javadoc
|
||||
// already names what happens if that case is handed the primary role: a worker→primary
|
||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||
return isLoopback(remoteAddr) && c.resolved() ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
@@ -262,11 +270,15 @@ public final class CallerResolver {
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #305: delegates to {@link ConnectionIdentity#isLoopback}. This used to be a second,
|
||||
* independent copy of the same rule, and the two drifted: this one accepted all of
|
||||
* {@code 127.0.0.0/8}, {@code ConnectionIdentity}'s accepted only {@code 127.0.0.1}. A caller
|
||||
* from {@code 127.0.0.2} therefore had its identity skipped (so it had no terminal) and was
|
||||
* then read as loopback here — which under loopback-trust is the primary. Sharing the inputs
|
||||
* would not have prevented that; only sharing the computation does.
|
||||
*/
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
return ConnectionIdentity.isLoopback(remoteAddr);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
/** A slot held before a member process starts. */
|
||||
record SlotReservation(String slot, String profile) { }
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return role; // no registry configured — nothing to bind against, so the request stands
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
// no registry configured — nothing to validate against, so nothing is refused
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||
*
|
||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||
* normal operation this fallback should not happen once a reservation has been bound.
|
||||
* It remains the honest answer if a caller has no reservation, or if binding a
|
||||
* reservation unexpectedly fails.
|
||||
*/
|
||||
MemberRole acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
|
||||
/**
|
||||
* Refuse an acquire before anything spawns when {@code role} requires a live slot binding and
|
||||
* no configured slot carries {@code profile} (CB-619 / fleetd #123). A no-op for a role with
|
||||
* no slot-binding semantics.
|
||||
*
|
||||
* @throws IllegalArgumentException naming the role, the profile, and the pools that do carry it
|
||||
*/
|
||||
void requireSlotFor(MemberRole role, String profile);
|
||||
|
||||
/** Reserve a matching slot before launch, or refuse before a charter can be delivered. */
|
||||
SlotReservation reserve(MemberRole role, String profile);
|
||||
|
||||
/** Convert a reservation into a live terminal binding. */
|
||||
boolean bind(SlotReservation reservation, String terminal);
|
||||
|
||||
/** Return an unbound reservation after a failed launch. */
|
||||
void release(SlotReservation reservation);
|
||||
}
|
||||
+111
-11
@@ -1,13 +1,14 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.bridged.config.BridgedConfig;
|
||||
import dev.ltms.bridged.peer.MemberRole;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
@@ -55,11 +56,13 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(BridgedConfig.Fleet fleet) {
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
@@ -94,7 +97,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code bridge_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* {@code fleet_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
@@ -166,7 +169,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
if (terminalToSlot.containsValue(slot) || reservedSlots.contains(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
@@ -206,19 +209,116 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*
|
||||
* <p>CB-619 / fleetd #123: the return value is the role this session actually holds, and the
|
||||
* caller is required to record THAT — never the requested {@code role} — on the session. Before
|
||||
* this fix the caller kept the requested role regardless of whether the bind below succeeded, so
|
||||
* a demoted session's {@code GET /members} row still said {@code "architect"} while
|
||||
* {@code fleet_whoami} (which reads the live binding, not the request) correctly said
|
||||
* {@code "worker"} — three sources of truth that disagreed about one live member, silently.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
return role;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
return MemberRole.ARCHITECT;
|
||||
}
|
||||
}
|
||||
// fleetd #123: at least WARN — a role downgrade that the roster must now also reflect is
|
||||
// not routine bookkeeping. requireSlotFor already refuses the config-gap case (no slot at
|
||||
// all carries this profile) before a process ever spawns; reaching here means the config DID
|
||||
// carry a matching slot but every one of them was already bound to a different terminal — a
|
||||
// race this pre-spawn check cannot close on its own (see requireSlotFor's javadoc).
|
||||
log.warn("member slot: no free architect slot for profile={} terminal={}; holding the session "
|
||||
+ "as {} instead of the architect it asked for — every configured slot for this "
|
||||
+ "profile is already bound to a different terminal", profile, terminal,
|
||||
MemberRole.DEV.wireName());
|
||||
return MemberRole.DEV;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||
* documented operator override, not a defect.
|
||||
*
|
||||
* <p>This closes the config-gap case, not the live-capacity case: a profile that DOES carry a
|
||||
* slot can still lose the race to a concurrent spawn between this check and the actual
|
||||
* {@link #bind}, which is why {@link #acquired} must still answer honestly even after this
|
||||
* check has passed.
|
||||
*/
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return;
|
||||
}
|
||||
boolean hasSlot = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.anyMatch(e -> Objects.equals(profile, e.profile()));
|
||||
if (hasSlot) {
|
||||
return;
|
||||
}
|
||||
List<String> pools = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.map(Entry::profile)
|
||||
.distinct()
|
||||
.toList();
|
||||
throw new IllegalArgumentException(
|
||||
"no " + role.wireName() + " slot for profile '" + profile + "' — an architect's "
|
||||
+ "identity IS the slot it is bound to, so there is nothing to bind this "
|
||||
+ "session's identity to. fleet." + role.configKey() + " carries profiles: "
|
||||
+ (pools.isEmpty() ? "(none configured)" : String.join(", ", pools))
|
||||
+ "; add profile '" + profile + "' there, or spawn " + role.wireName()
|
||||
+ " on one of those profiles instead");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if ((profile == null || profile.isBlank() || Objects.equals(profile, entry.profile()))
|
||||
&& !terminalToSlot.containsValue(entry.key()) && reservedSlots.add(entry.key())) {
|
||||
return new SlotReservation(entry.key(), entry.profile());
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("no free architect slot for profile '" + profile
|
||||
+ "' — every matching slot is already bound or reserved");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
if (reservation == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!reservedSlots.remove(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
if (!isSlot(reservation.slot()) || terminalToSlot.containsKey(terminal)
|
||||
|| terminalToSlot.containsValue(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
terminalToSlot.put(terminal, reservation.slot());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
if (reservation != null) {
|
||||
synchronized (terminalToSlot) {
|
||||
reservedSlots.remove(reservation.slot());
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
+4
-4
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* A resolved caller: its {@link Role}, and — for a worker — the herdr {@code terminal_id} that
|
||||
@@ -39,14 +39,14 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
*
|
||||
* <p>Carries {@link Role#PRIMARY}: a lead <em>is</em> a primary as far as authorization goes,
|
||||
* so every existing {@code isPrimary()} gate keeps working unchanged and the role table needed
|
||||
* no new entry. The name is reporting only — it lets {@code bridge_whoami} say <em>which</em>
|
||||
* no new entry. The name is reporting only — it lets {@code fleet_whoami} say <em>which</em>
|
||||
* lead is asking once more than one is configured.
|
||||
*
|
||||
* <p><strong>CB-532: a lead now carries the terminal it was matched by.</strong> Under CB-530 it
|
||||
* deliberately did not, because {@code terminal} meant "which worker pane" everywhere and a
|
||||
* non-null one would have enrolled the lead in the worker presence map. That reading was what
|
||||
* made a lead unaddressable: {@link #ownsSession} could never be true for it, so
|
||||
* {@code bridge_reply} was refused and one lead could send to another but never be answered.
|
||||
* {@code fleet_reply} was refused and one lead could send to another but never be answered.
|
||||
* The terminal now means "which pane is this caller", the presence map keys on
|
||||
* {@link #isSpawnedMember()} instead, and a lead is a peer that can both send and receive.
|
||||
*/
|
||||
@@ -63,7 +63,7 @@ public record Principal(Role role, String terminal, long pid, String name) {
|
||||
* An architect (CB-548), identified by the slot it occupies and the pane bound to it.
|
||||
*
|
||||
* <p>Carries {@link Role#ARCHITECT}. {@code slotName} is reporting only — it lets
|
||||
* {@code bridge_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* {@code fleet_whoami} say <em>which</em> architect slot is asking, and it is the key the
|
||||
* (future) spawn lifecycle reads a profile back from. Identity is the {@code terminal}: like a
|
||||
* worker's it comes from the connection and the live terminal→slot binding, so
|
||||
* {@code ownsSession} works exactly as it does for a worker — an architect acts as its own
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.auth;
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
/**
|
||||
* What a caller is allowed to be on the bus (CB-501).
|
||||
@@ -0,0 +1,618 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link FleetConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into four classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
||||
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||
* ({@code () -> config.get().fleet()}). <strong>But {@code fleet:} as a whole is NOT in this
|
||||
* class</strong>: {@code fleet.leaders} inside the same key is frozen, which is exactly what
|
||||
* makes {@code fleet:} split rather than hot — see below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code idleSleepGuard:} ({@code Fleetd.java} reads it once, at startup, to decide whether
|
||||
* to construct an {@code IdleSleepGuard} and wire {@code SessionManager}'s
|
||||
* {@code onAcquire}/{@code onRelease} hooks to it — neither is rebuilt on reload, so a
|
||||
* running daemon keeps whatever this was at startup regardless of a later edit),
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, {@code worktreeGroup:} and {@code memberSkills:}
|
||||
* (all three of the latter baked once into the {@code GitWorktrees} built at
|
||||
* {@code Fleetd.java:251} and never rebuilt — fleetd #323 instance 2 found
|
||||
* {@code worktreeGroup} missing from this list and from {@link #changedDeferredKeys};
|
||||
* {@code memberSkills} (fleetd #362) followed the same shape), {@code primary:} (fleetd #326 — {@code Fleetd.java:506, 519,
|
||||
* 520} read {@code cfg.primary()} only off the startup snapshot to build {@code
|
||||
* PrimaryRegistry} and size {@code ReplyPushLoop}'s reminder cap/backoff, and neither is
|
||||
* rebuilt on reload. Say the consequence exactly: {@code primary.terminal} is DEPRECATED
|
||||
* (CB-532, and {@code Fleetd.java:511} warns about it at startup) — a lead's identity comes
|
||||
* from {@code leaders:}/{@code leadScan:}, so changing this pin does not demote or promote a
|
||||
* lead that uses those. What a changed pin still does not take effect on until a restart is
|
||||
* the fallback nudge destination the pin remains, the deprecated identity path for an operator
|
||||
* who still relies on it, and {@code pushReminders}/{@code pushBackoffMs}), {@code configReload:} (fleetd #326 — {@code
|
||||
* Fleetd.java:679-680} read it only at startup to decide whether to build a {@code
|
||||
* ConfigWatcher} at all and with what interval; the watcher that would apply a later change is
|
||||
* itself built once, so a running watcher keeps polling on its original enabled flag and
|
||||
* interval regardless of what a reload changes it to, the same shape as {@code lifecycle} —
|
||||
* not cold, because no already-open resource goes inconsistent with the new value, the watcher
|
||||
* (if any) simply keeps its old settings), adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), {@code errorPattern} (fleetd #201 Unit 5 — compiled once into
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup, the same way),
|
||||
* {@code ideProjectDir} / {@code ideOpenCommand} / {@code autoCompactWindow} (fleetd #323
|
||||
* instance 1 — all three are read at spawn off the same frozen profile map and were missing
|
||||
* from {@link #sameLaunchSettings}), and the rest of {@link #sameLaunchSettings}.
|
||||
* {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Split</strong> (fleetd #330; extended to a third key by fleetd #333) — read
|
||||
* <em>both</em> ways at different sites, so the key does not fit any class above as a whole:
|
||||
* {@code health:}, {@code coordinator:} and {@code fleet:}. Each is read off the startup
|
||||
* snapshot to build a long-lived object, and read live off {@link #get()} at a different,
|
||||
* unrelated site — so half of a reload's effect already applies while the other half waits
|
||||
* for a restart, and a bare "config reloaded" would under-claim by exactly that half.
|
||||
* <ul>
|
||||
* <li>{@code health:} — the monitor itself ({@code enabled}, {@code intervalSeconds},
|
||||
* {@code workingSuspectAfterSeconds}) is built once at {@code Fleetd.java:556-563}
|
||||
* and never rebuilt, so a changed value needs a restart to actually start, stop, or
|
||||
* retime it. The coverage string {@code fleet_profiles} reports
|
||||
* ({@code Fleetd.java:648-650}) is read live off {@link #get()} on every call, so it
|
||||
* already reflects the new value.</li>
|
||||
* <li>{@code coordinator:} — the {@code LeadMailbox} connection ({@code uri},
|
||||
* {@code uriEnv}, {@code selfId}, {@code prefetch}) is opened once at
|
||||
* {@code Fleetd.java:502} and never reopened, so a changed value needs a restart —
|
||||
* {@code selfId} in particular names this daemon's own AMQP inbox queue, and a peer
|
||||
* lead that learned the old name would not discover a new one on its own. The broker
|
||||
* URI env-var <em>name</em> that {@code MemberEnvAllowList} keeps out of a member's
|
||||
* environment is read live off {@link #get()} on every spawn
|
||||
* ({@code HerdrPeerLauncher.java:1530}), so it already applies.</li>
|
||||
* <li>{@code fleet:} (fleetd #333) — {@code fleet.leaders} is the frozen half:
|
||||
* {@code Fleetd.java:281} reads {@code cfg.fleet().leaders()} off the startup snapshot
|
||||
* for two long-lived objects built right after it and never rebuilt — the
|
||||
* {@code LeadTabScanner}'s {@code tab label → lead name} map ({@code Fleetd.java:301},
|
||||
* wired into {@code CallerResolver.withLeadsAndMembers} at {@code Fleetd.java:620/624},
|
||||
* which is how a caller's pane is recognised as a lead at all) and, when herdr answered
|
||||
* at startup, {@code LeadLauncher(...).ensureLeads()} ({@code Fleetd.java:315}), which
|
||||
* auto-launches each declared lead up to its {@code instances} count. So a lead added,
|
||||
* removed, or given a new {@code tab:} label under {@code fleet.leaders} needs a
|
||||
* restart — until then it is invisible to identity resolution, and this is exactly the
|
||||
* scenario fleetd #333 named: an operator edits a lead's {@code tab:} to match a
|
||||
* renamed pane, sees "config reloaded", and the pane keeps resolving as a worker,
|
||||
* because {@code CallerResolver} is still matching against the old label. The live
|
||||
* half is the rest of {@code fleet:} — {@code architects}/{@code developers}/
|
||||
* {@code reviewers}, {@code charters}, {@code tabLabel} — read live through the same
|
||||
* {@code CompositePeerLauncher} supplier the Hot bullet above names, so a reload that
|
||||
* only touches those already applies with nothing to report. Because the hot and frozen
|
||||
* halves of {@code fleet:} are disjoint sub-fields rather than the same fields read two
|
||||
* ways (contrast {@code coordinator.uriEnv} above), {@link #changedSplitKeys} compares
|
||||
* {@code fleet.leaders} alone, not the whole {@code Fleet} record — comparing the whole
|
||||
* record would report "split" for a {@code tabLabel}-only change that is actually fully
|
||||
* hot, over-claiming in exactly the direction this class exists to avoid under-claiming
|
||||
* in.</li>
|
||||
* </ul>
|
||||
* A split change is still accepted — {@link Outcome#applied()} stays {@code true}, the same
|
||||
* as a deferred change — because the live half genuinely took effect; refusing the whole
|
||||
* reload would leave the operator worse off than today. {@link Outcome#split()} names the
|
||||
* key and says which half is which each time, rather than trying to score "how changed" a
|
||||
* mixed key is or handle "both halves changed in one reload" as a special case.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon. All five of
|
||||
* {@link #COLD_KEYS}: {@code bind:}, {@code herdrSocket:}, {@code memberHerdrSocket:},
|
||||
* {@code broker:} and {@code auth:}. The sockets are already connected, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening. This bullet omitted {@code memberHerdrSocket:} until fleetd #333 — say "all
|
||||
* five of COLD_KEYS" rather than re-listing them, so prose and set cannot drift again.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The denominator, measured on 2026-09-04 (fleetd #330; recounted for fleetd #333);
|
||||
* recounted again for fleetd #362, and again after {@code idleSleepGuard:} was added.</strong>
|
||||
* {@code FleetConfig} has 24 top-level record components: 5 cold, 13 deferred, 3 split, 3
|
||||
* hot-excluded. Three of them are named nowhere in this file, and the reason is the same for all
|
||||
* three: {@code placement}, {@code memberCredentials} and {@code memberLoginShell} are
|
||||
* <strong>hot</strong> and correctly absent — all three are read live off {@code config.get()}
|
||||
* (placement through the {@code CompositePeerLauncher} supplier the Hot bullet names;
|
||||
* {@code memberCredentials}/{@code memberLoginShell} at spawn time, {@code Fleetd.java:198, 205, 729}
|
||||
* and {@code HerdrPeerLauncher#configuredMemberLoginShell}), so a reload takes effect on the next
|
||||
* spawn with no entry needed here.
|
||||
* {@code health} and {@code coordinator} used to be a third kind — <strong>undecided</strong>, not
|
||||
* hot — until fleetd #330 added the <strong>split</strong> class above and gave them a home. A
|
||||
* reload touching either used to report a bare "config reloaded", which under-claimed; now it names
|
||||
* the key and says which half is which. {@code fleet} was the same story in reverse: fleetd #330's
|
||||
* own fact-find named {@code health}/{@code coordinator} as "the complete set of split-shaped keys"
|
||||
* and filed {@code fleet.leaders}'s restart requirement as a documented caveat sitting in the
|
||||
* <strong>hot-excluded</strong> escape hatch instead — correctly documented, but in the one bucket
|
||||
* this file's own coverage test cannot check the truth of (see that test's javadoc). fleetd #333
|
||||
* moved it into <strong>split</strong>, where {@link #changedSplitKeys} actually reports it.
|
||||
* <p>The point of writing the count down: "not mentioned in this file" looks identical for a key
|
||||
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
||||
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
||||
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
||||
* {@link ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
||||
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
||||
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
||||
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
||||
* any reporting code behind it at all — {@code ConfigRefTopLevelReportingCoverageTest} is what
|
||||
* fleetd #333 added for that, after measuring that a {@code SPLIT_KEYS} entry with its reporting
|
||||
* branch deleted passes both this file's own "kept in step" assert and
|
||||
* {@code ConfigRefTopLevelCoverageTest} unchanged. fleetd #337 extended it to {@code DEFERRED_KEYS}
|
||||
* after measuring the same one-way gap there directly: dropping {@code guard}'s branch out of
|
||||
* {@link #changedDeferredKeys} while {@code "guard"} stayed in the set left the whole suite green.
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart. A split change does <em>not</em> refuse, for a different reason than a
|
||||
* deferred change does not: its live half genuinely took effect, so refusing would throw that away
|
||||
* and leave the operator worse off than the partial-but-honest report {@link Outcome#split()} gives.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/**
|
||||
* Keys that cannot change under a running daemon — see the class doc.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} can fold it
|
||||
* into the top-level triage it checks, the same way it reads {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
/**
|
||||
* Keys read BOTH off the startup snapshot and live off {@link #get()} at different sites, so
|
||||
* neither the hot, deferred nor cold class fits them as a whole — see the class doc's Split
|
||||
* bullet (fleetd #330). A changed split key is accepted ({@link Outcome#applied()} stays
|
||||
* {@code true}) and reported by name, with a message naming which half is live and which needs
|
||||
* a restart.
|
||||
*/
|
||||
static final Set<String> SPLIT_KEYS = Set.of("health", "coordinator", "fleet");
|
||||
|
||||
/**
|
||||
* Top-level keys {@link #changedDeferredKeys} compares — see the class doc's Deferred bullet.
|
||||
* Promoted here from a test-side copy in {@code ConfigRefTopLevelCoverageTest} by fleetd #337,
|
||||
* the same reason {@link #COLD_KEYS} and {@link #SPLIT_KEYS} live here rather than in a test: a
|
||||
* second, hand-maintained copy of this set is exactly the kind of thing that silently drifts
|
||||
* from the method it is supposed to describe. {@code spawnReadyTimeoutMs} and
|
||||
* {@code spawnReadyPollMs} are compared together in one branch and reported under the combined
|
||||
* label {@code "spawnReady*"}; {@code profiles} is compared twice over (added/removed names,
|
||||
* then an existing profile's launch settings) — see {@link #changedDeferredKeys}.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} and
|
||||
* {@code ConfigRefTopLevelReportingCoverageTest} can both read it, the same way they already
|
||||
* read {@link #COLD_KEYS} and {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> DEFERRED_KEYS = Set.of(
|
||||
"guard", "worktreeRoot", "worktreeGroup", "memberSkills", "primary", "configReload",
|
||||
"leadHeartbeat", "lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs",
|
||||
"quarantineCooldownSeconds", "profiles", "idleSleepGuard");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(FleetConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public FleetConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
/** The file this ref reloads from, or {@code null} for a {@link #fixed} ref. */
|
||||
public Path path() {
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* <p>{@code split} is a separate field from {@code deferred} rather than a differently-worded
|
||||
* entry inside it, because the two carry different guarantees for any caller that branches on
|
||||
* them rather than just printing {@link #summary()}: every {@code deferred} entry means "this
|
||||
* key's whole change waits for a restart", while every {@code split} entry means "part of this
|
||||
* key's change already applied, and the message says which part" — collapsing them would force
|
||||
* a caller to re-parse the message to tell those apart. See the class doc's Split bullet
|
||||
* (fleetd #330) for why the key needs this at all.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits entirely on a
|
||||
* restart
|
||||
* @param split split keys that changed and were accepted, each named with which half of it
|
||||
* is already live and which half waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
List<String> split, String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
split = List.copyOf(split);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
public String summary() {
|
||||
if (error != null) {
|
||||
return "config reload refused — " + error;
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (deferred.isEmpty() && split.isEmpty()) {
|
||||
return "config reloaded";
|
||||
}
|
||||
StringBuilder out = new StringBuilder("config reloaded");
|
||||
if (!deferred.isEmpty()) {
|
||||
out.append("; these changes need a restart to take effect: ")
|
||||
.append(String.join(", ", deferred));
|
||||
}
|
||||
if (!split.isEmpty()) {
|
||||
out.append("; partially live — ").append(String.join(" | ", split));
|
||||
}
|
||||
return out.toString();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the file, validate it, and swap it in when nothing cold changed.
|
||||
*
|
||||
* <p>Never throws: a reload is a best-effort operation on a daemon that is already serving, and
|
||||
* a bad edit must not take it down. Every failure path leaves the previous config live and is
|
||||
* reported through the returned {@link Outcome}.
|
||||
*/
|
||||
public Outcome reload() {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
FleetConfig old = current.get();
|
||||
FleetConfig fresh;
|
||||
try {
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
return Outcome.failed(msg);
|
||||
}
|
||||
|
||||
List<String> cold = changedColdKeys(old, fresh);
|
||||
if (!cold.isEmpty()) {
|
||||
Outcome out = Outcome.refusedCold(cold);
|
||||
log.warn(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
List<String> split = changedSplitKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, split, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cold keys whose value differs between the running config and the candidate.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333).
|
||||
*/
|
||||
static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
}
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.memberHerdrSocket(), fresh.memberHerdrSocket())) {
|
||||
changed.add("memberHerdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
if (!Objects.equals(old.auth(), fresh.auth())) {
|
||||
changed.add("auth");
|
||||
}
|
||||
// Kept in step with COLD_KEYS so the doc and the code cannot drift apart silently.
|
||||
assert COLD_KEYS.containsAll(changed) : "a cold key was reported that COLD_KEYS omits";
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Changed keys that were accepted but whose effect waits for a restart.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #changedColdKeys} and {@link #changedSplitKeys} already are (fleetd #333, extended to
|
||||
* this method by fleetd #337 — membership in {@link #DEFERRED_KEYS} proved nothing about this
|
||||
* method on its own until then; see that test's class doc).
|
||||
*/
|
||||
static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
}
|
||||
if (!Objects.equals(old.leadHeartbeat(), fresh.leadHeartbeat())) {
|
||||
changed.add("leadHeartbeat");
|
||||
}
|
||||
if (!Objects.equals(old.guard(), fresh.guard())) {
|
||||
changed.add("guard");
|
||||
}
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
// Baked into the same GitWorktrees as worktreeRoot (Fleetd.java:251) and never rebuilt
|
||||
// either — see the class doc. Missing this check was fleetd #323 instance 2: a reload
|
||||
// that changed only worktreeGroup reported "config reloaded" with nothing deferred, and
|
||||
// newly provisioned worktrees kept the old sharing behaviour.
|
||||
if (!Objects.equals(old.worktreeGroup(), fresh.worktreeGroup())) {
|
||||
changed.add("worktreeGroup");
|
||||
}
|
||||
// fleetd #362: baked into the same GitWorktrees as worktreeRoot/worktreeGroup
|
||||
// (Fleetd.java:251) and never rebuilt either — a reload that changes only memberSkills
|
||||
// must be reported the same way, or a newly provisioned worktree keeps seeding from (or
|
||||
// skipping) the old source directory with nothing telling the operator why.
|
||||
if (!Objects.equals(old.memberSkills(), fresh.memberSkills())) {
|
||||
changed.add("memberSkills");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:506, 519, 520 read cfg.primary() only off the startup snapshot
|
||||
// (PrimaryRegistry's pinned terminal, ReplyPushLoop's reminder cap and backoff) — neither is
|
||||
// rebuilt on reload, so a changed value needs a restart. Note what it does NOT mean:
|
||||
// primary.terminal is deprecated (CB-532), identity comes from leaders:/leadScan:, so a lead
|
||||
// using those is unaffected by this pin either way. See the class doc for the exact scope.
|
||||
if (!Objects.equals(old.primary(), fresh.primary())) {
|
||||
changed.add("primary");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:679-680 read cfg.configReload() only at startup to decide whether
|
||||
// to build a ConfigWatcher at all and with what interval — the watcher that would apply a
|
||||
// later change is itself built once, so a running watcher keeps its original enabled flag and
|
||||
// interval regardless of what a reload changes it to. Not cold: no already-open resource goes
|
||||
// inconsistent with the new value, a watcher (if any) simply keeps polling on the old settings.
|
||||
if (!Objects.equals(old.configReload(), fresh.configReload())) {
|
||||
changed.add("configReload");
|
||||
}
|
||||
// Fleetd.java reads cfg.idleSleepGuard() once, at startup, to decide whether to construct
|
||||
// an IdleSleepGuard at all and wire SessionManager's onAcquire/onRelease hooks to it —
|
||||
// neither is rebuilt on reload, so a running daemon keeps whatever this was at startup
|
||||
// (armed or not) regardless of a later edit here. Not cold: nothing already-open goes
|
||||
// inconsistent with the new value, an armed-or-not guard just keeps its original answer.
|
||||
if (!Objects.equals(old.idleSleepGuard(), fresh.idleSleepGuard())) {
|
||||
changed.add("idleSleepGuard");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, FleetConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, FleetConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
if (!before.keySet().equals(after.keySet())) {
|
||||
Set<String> diff = new LinkedHashSet<>(before.keySet());
|
||||
diff.addAll(after.keySet());
|
||||
diff.removeIf(p -> before.containsKey(p) && after.containsKey(p));
|
||||
changed.add("profiles (added/removed: " + String.join(", ", diff) + ")");
|
||||
}
|
||||
// An EXISTING profile's launch settings are deferred too, and this is easy to get wrong:
|
||||
// `HerdrPeerLauncher` takes `Map.copyOf(profiles)` at construction and `spawn` resolves the
|
||||
// profile out of that snapshot, so a reloaded model/baseUrl/argv/env never reaches a launch.
|
||||
// Only weight and maxLoad are genuinely hot, because placement reads them through the
|
||||
// supplier on the composite rather than from the adapter's copy. Without this check a
|
||||
// changed model would report "config reloaded" and silently do nothing — the worst outcome
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
FleetConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
});
|
||||
if (!relaunch.isEmpty()) {
|
||||
changed.add("profiles." + String.join("/", relaunch) + " launch settings "
|
||||
+ "(model, baseUrl, argv, env, …) — the launcher holds a startup snapshot");
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Split keys whose value differs between the running config and the candidate — see the class
|
||||
* doc's Split bullet (fleetd #330, extended for {@code fleet:} by fleetd #333). Unlike
|
||||
* {@link #changedDeferredKeys}, this does not try to tell which sub-field moved for {@code
|
||||
* health:} or {@code coordinator:}: any change to either gets the same fixed message, because
|
||||
* the message already names both halves every time, so there is no "which half changed"
|
||||
* question left for the caller to answer. {@code fleet:} is different on purpose — see below.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333). That
|
||||
* test exists because membership in {@link #SPLIT_KEYS} proves nothing about this method on its
|
||||
* own: fleetd #333 measured that dropping the {@code coordinator} branch out of this method
|
||||
* while leaving {@code "coordinator"} in {@code SPLIT_KEYS} left the whole suite green except a
|
||||
* hand-written {@code ConfigRefTest} case — neither {@code ConfigRefTopLevelCoverageTest} (it
|
||||
* only reads the set) nor the "kept in step" assert below (it only checks the reported keys are
|
||||
* a SUBSET of {@code SPLIT_KEYS}, never that every {@code SPLIT_KEYS} member has a branch here)
|
||||
* would have caught it.
|
||||
*/
|
||||
static List<String> changedSplitKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.health(), fresh.health())) {
|
||||
changed.add("health: the monitor itself (enabled, interval, workingSuspectAfter) is "
|
||||
+ "frozen at startup and needs a restart; the coverage status fleet_profiles "
|
||||
+ "reports is read live and already applied");
|
||||
}
|
||||
if (!Objects.equals(old.coordinator(), fresh.coordinator())) {
|
||||
changed.add("coordinator: the LeadMailbox connection (uri, uriEnv, selfId, prefetch) is "
|
||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||
+ "member's environment is read live on every spawn and already applied");
|
||||
}
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (architects, developers,
|
||||
// reviewers, charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||
// with no restart note. Only fleet.leaders is frozen (Fleetd.java:281 reads
|
||||
// cfg.fleet().leaders() off the startup snapshot to build both the LeadTabScanner's
|
||||
// tab-label-to-name map, wired into CallerResolver.withLeadsAndMembers at Fleetd.java:620/624,
|
||||
// and — when herdr answered — LeadLauncher(...).ensureLeads() at Fleetd.java:315, which
|
||||
// auto-launches each lead up to its `instances` count; neither is rebuilt on reload). So this
|
||||
// compares fleet.leaders alone, not the whole Fleet record: comparing the whole record would
|
||||
// report "split" for a tabLabel-only or charters-only change that is actually fully hot,
|
||||
// which is the over-claim mirror of the under-claim bug this class exists to prevent.
|
||||
if (!Objects.equals(leadersOf(old), leadersOf(fresh))) {
|
||||
changed.add("fleet: fleet.leaders (each lead's tab, workspace, cwd, profile and "
|
||||
+ "instances count) is read once at startup to build the LeadTabScanner's "
|
||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||
+ "lead; the rest of fleet: (architects, developers, reviewers, charters, "
|
||||
+ "tabLabel) is read live through the supplier on CompositePeerLauncher and "
|
||||
+ "already applied");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
||||
// member with no branch above at all, only a branch whose message is mis-worded relative to
|
||||
// the set. ConfigRefTopLevelReportingCoverageTest is what proves the former.
|
||||
assert changed.stream().allMatch(m -> SPLIT_KEYS.stream().anyMatch(k -> m.startsWith(k + ":")))
|
||||
: "a split entry was reported that does not start with a SPLIT_KEYS name: " + changed;
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** {@code cfg.fleet().leaders()}, defensively, in case a caller hands in a non-defaulted config. */
|
||||
private static Map<String, FleetConfig.Leader> leadersOf(FleetConfig cfg) {
|
||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link FleetConfig.Profile} record components deliberately left out of
|
||||
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
||||
* the class doc's <em>Hot</em> bullet. {@code weight} and {@code maxLoad} are read live by the
|
||||
* placement policy on every spawn; {@code credentialId} is read live by
|
||||
* {@code CompositePeerLauncher} and the CB-578 stage B exhaustion sink. Nothing else is
|
||||
* excluded — see {@code sameLaunchSettingsComparesEveryProfileComponentOrExcludesIt} in
|
||||
* {@code ConfigRefProfileCoverageTest}, which enumerates every {@code Profile} record component
|
||||
* by reflection and fails the build if one is neither compared below nor named here.
|
||||
*/
|
||||
static final Set<String> LAUNCH_SETTINGS_EXCLUDED = Set.of("weight", "maxLoad", "credentialId");
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically.
|
||||
*
|
||||
* <p>This must compare every {@link FleetConfig.Profile} record component except the three in
|
||||
* {@link #LAUNCH_SETTINGS_EXCLUDED}. That is not a claim this javadoc can make good on by
|
||||
* itself — a javadoc saying "compares every component" is exactly what fleetd #323 found to be
|
||||
* false for three fields (and a sibling method's field list, for a fourth). The actual
|
||||
* guarantee comes from {@code ConfigRefProfileCoverageTest}: it enumerates every record
|
||||
* component of {@code FleetConfig.Profile} by reflection, mutates each one not in
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} on a base profile, and asserts this method reports a
|
||||
* difference — so a new component that is neither compared here nor added to
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} (with a reason) fails that test by name, rather than
|
||||
* silently reporting "config reloaded" for a value the daemon never picked up.
|
||||
*/
|
||||
static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.profile(), b.profile())
|
||||
&& Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
&& Objects.equals(a.argv(), b.argv())
|
||||
&& Objects.equals(a.placement(), b.placement())
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
// CB-634: the IDE MCP mount is a launch flag, fixed at spawn like mcpUrl — a
|
||||
// reload changes it only for members spawned after, so a changed value is deferred.
|
||||
&& Objects.equals(a.ideMcpUrl(), b.ideMcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern())
|
||||
// fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error
|
||||
// pattern map at startup (see BackendErrorPatternLookup wiring), the same way
|
||||
// exhaustedPattern is — a reload never re-reads it either.
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern())
|
||||
// fleetd #323 instance 1: ideProjectDir and ideOpenCommand are read at spawn off the
|
||||
// same frozen profile map as ideMcpUrl above (ClaudeCodeLauncher.java:267/269,
|
||||
// OpenCodeLauncher.java:474/480/486) and were missing from this comparison.
|
||||
&& Objects.equals(a.ideProjectDir(), b.ideProjectDir())
|
||||
&& Objects.equals(a.ideOpenCommand(), b.ideOpenCommand())
|
||||
// fleetd #323 instance 1: autoCompactWindow is read at spawn the same way
|
||||
// (ClaudeCodeLauncher.java:926, OpenCodeLauncher.java:650). Comparing it here only
|
||||
// makes the reload REPORT that a restart is needed — it deliberately does not make
|
||||
// autoCompactWindow take effect live, which is a separate, larger change.
|
||||
&& Objects.equals(a.autoCompactWindow(), b.autoCompactWindow());
|
||||
}
|
||||
}
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.config;
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -11,7 +11,7 @@ import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Polls {@code bridged.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* Polls {@code fleetd.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* (CB-559). Opt-in through {@code configReload.enabled}.
|
||||
*
|
||||
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
||||
+749
-98
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.guard;
|
||||
package dev.ltms.fleet.guard;
|
||||
|
||||
/** Thrown when the subscription boundary would be violated. Never swallow this. */
|
||||
public class GuardException extends RuntimeException {
|
||||
+2
-2
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.guard;
|
||||
package dev.ltms.fleet.guard;
|
||||
|
||||
import java.net.URI;
|
||||
import java.net.URISyntaxException;
|
||||
@@ -16,7 +16,7 @@ import java.util.Set;
|
||||
* means its traffic would leave the subscription. That is a hard stop.</li>
|
||||
* </ul>
|
||||
*
|
||||
* Both checks throw {@link GuardException} on violation. {@code bridged} calls
|
||||
* Both checks throw {@link GuardException} on violation. {@code fleetd} calls
|
||||
* {@link #assertWorker} before spawning a worker and {@link #assertPrimaryClean}
|
||||
* against its own environment at startup.
|
||||
*/
|
||||
+3
-3
@@ -1,7 +1,7 @@
|
||||
package dev.ltms.bridged.health;
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.bridged.herdr.AgentStatus;
|
||||
import dev.ltms.bridged.session.MemberSession;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
|
||||
/**
|
||||
* Pure classifier. Collection and repair are deliberately outside this package.
|
||||
@@ -0,0 +1,380 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
/**
|
||||
* fleetd #280: how long after a terminal transition to wait before the one bounded re-check
|
||||
* fires. Must exceed the worst-case reverse-rendezvous {@code fleet_ask} window (55-115s, see
|
||||
* {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS}) so that, if the
|
||||
* target was genuinely {@code ASKING} when {@code state} was first observed, its own ask has had
|
||||
* time to lapse (clearing {@code Task#question} back to {@code null}) before this fires.
|
||||
*/
|
||||
static final long ASK_LAPSE_RECHECK_DELAY_SECONDS = 120;
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code System.nanoTime()} (or whatever {@link #clock} is) does not advance while
|
||||
* macOS sleeps, so a raw {@code nowNanos - lastActivityAtNanos} comparison freezes with the
|
||||
* host and can never cross {@link #workingSuspectAfterNanos}. This is a second, wall-clock
|
||||
* source used ONLY inside the stall check ({@link #stallElapsedNanos}) to detect and correct
|
||||
* for that freeze. Nothing else in this class reads it — every other decision (readiness grace,
|
||||
* the fault classification itself) stays exactly on {@link #clock}, as the ticket requires.
|
||||
*/
|
||||
private static final LongSupplier DEFAULT_REALTIME_CLOCK =
|
||||
() -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis());
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final LongSupplier realtimeClock;
|
||||
private final long intervalSeconds;
|
||||
private final long tickIntervalNanos;
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
/**
|
||||
* fleetd #386 clock-drift bookkeeping. {@code haveClockBaseline}/{@code lastTickMonoNanos}/
|
||||
* {@code lastTickRealNanos} track the previous tick's pair of readings so each new tick can
|
||||
* measure how far the two clocks moved apart since then. {@code accumulatedDriftNanos} is the
|
||||
* running total of every such divergence observed since this monitor started (never decreases —
|
||||
* the monotonic clock can only lag real time, never lead it). {@code busyDriftBaselineNanos}/
|
||||
* {@code busyBaselineActivityNanos} record, per target, the value of {@code accumulatedDriftNanos}
|
||||
* at the moment this monitor first saw that target's CURRENT {@code lastActivityAtNanos} while
|
||||
* BUSY — so {@link #stallElapsedNanos} adds back only the drift observed DURING this BUSY span,
|
||||
* never drift from a sleep that happened before the member went busy. All five fields are touched
|
||||
* only from {@code tick()}, like {@link #priors}.
|
||||
*/
|
||||
private boolean haveClockBaseline = false;
|
||||
private long lastTickMonoNanos;
|
||||
private long lastTickRealNanos;
|
||||
private long accumulatedDriftNanos = 0;
|
||||
private final Map<String, Long> busyDriftBaselineNanos = new HashMap<>();
|
||||
private final Map<String, Long> busyBaselineActivityNanos = new HashMap<>();
|
||||
/**
|
||||
* The live classification per member, and the only one of this class's three maps that more
|
||||
* than one scheduler task touches. {@code tick} writes it (and prunes it to the roster);
|
||||
* fleetd #280's delayed {@link #recheckTerminalTarget} reads it from its own separate scheduled
|
||||
* task. Both run on the single-threaded scheduler {@code Fleetd} passes in today, so they are
|
||||
* serialised — but nothing in this class enforces that, and an unsynchronised {@link HashMap}
|
||||
* read racing a resize can spin a CPU forever rather than fail visibly. {@code priors} and
|
||||
* {@code orphanStreaks} stay plain maps because {@code tick} is still their only toucher.
|
||||
*/
|
||||
private final Map<String, HealthState> states = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
* for one tick during an ordinary race — an async ticket exists before its virtual thread has
|
||||
* reached {@code rendezvous.open()}, so for that instant nothing is accepted or queued behind
|
||||
* it. {@code decide} maps the field straight to {@code DELEGATION_ORPHANED} with no cross-tick
|
||||
* smoothing of its own, so a single racy read would log a fault that clears on the next tick.
|
||||
* Requiring two consecutive observations costs one interval of latency on a real orphan and
|
||||
* removes that false positive entirely.
|
||||
*/
|
||||
private final Map<String, Integer> orphanStreaks = new HashMap<>();
|
||||
|
||||
/** How many consecutive ticks a target must look orphaned before health reports it (CB-643). */
|
||||
static final int ORPHAN_CONFIRM_TICKS = 2;
|
||||
|
||||
// CB-643: every HealthSnapshot field now carries real evidence. The NOT_YET_OBSERVED placeholder
|
||||
// that stood in for 7 of the 12 is gone, and with it the reason 8 of the 9 fault states were
|
||||
// unreachable — GONE and NEVER_READY included, which is what kept CB-580's failTarget from ever
|
||||
// firing. Do not reintroduce a constant here: a field with no publisher is a dead state, and the
|
||||
// tests pass either way, so nothing else will tell you.
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code FleetMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
long workingSuspectAfterSeconds, BiConsumer<String, String> failTarget) {
|
||||
this(agents, roster, messages, scheduler, clock, DEFAULT_REALTIME_CLOCK, intervalSeconds,
|
||||
workingSuspectAfterSeconds, failTarget);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param realtimeClock fleetd #386: a wall-clock nanosecond source (e.g.
|
||||
* {@code System.currentTimeMillis()} converted to nanos) that keeps
|
||||
* advancing while {@code clock} is frozen by a host sleep. Used only to
|
||||
* correct the stall check — see the class-level javadoc on the
|
||||
* clock-drift fields.
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, LongSupplier realtimeClock,
|
||||
long intervalSeconds, long workingSuspectAfterSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.realtimeClock = Objects.requireNonNull(realtimeClock, "realtimeClock");
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.tickIntervalNanos = TimeUnit.SECONDS.toNanos(intervalSeconds);
|
||||
this.workingSuspectAfterNanos = TimeUnit.SECONDS.toNanos(workingSuspectAfterSeconds);
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
List<Agent> agentsNow;
|
||||
boolean controlLinkDown = false;
|
||||
try {
|
||||
agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
} catch (HerdrException error) {
|
||||
agentsNow = List.of();
|
||||
controlLinkDown = true;
|
||||
log.warn("fleet health control link unavailable; classifying roster", error);
|
||||
}
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
long nowNanos = clock.getAsLong();
|
||||
long driftBeforeThisTick = observeClockDrift(nowNanos);
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean present = agent != null;
|
||||
boolean targetNotFound = !controlLinkDown && !present
|
||||
&& session.state() != MemberSession.State.SPAWNING;
|
||||
boolean readinessGraceElapsed = nowNanos - session.spawnedAtNanos() >= READINESS_GRACE_NANOS;
|
||||
boolean stalled = session.state() == MemberSession.State.BUSY
|
||||
&& stallElapsedNanos(session, nowNanos, driftBeforeThisTick) >= workingSuspectAfterNanos;
|
||||
// CB-643: the three message-layer facts CB-640 published. Read them here rather than
|
||||
// leaving them false — that constant is what made 8 of the 9 fault states dead.
|
||||
boolean queuedDelivery = messages.hasQueuedDelivery(session.terminalId());
|
||||
boolean replyStranded = messages.hasStrandedReply(session.terminalId());
|
||||
boolean orphanedDelegation = confirmOrphan(session.terminalId(),
|
||||
messages.hasOrphanedDelegation(session.terminalId()));
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, queuedDelivery,
|
||||
messages.hasInboxMessage(session.terminalId()), present, targetNotFound, controlLinkDown,
|
||||
readinessGraceElapsed, orphanedDelegation, replyStranded, stalled);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
nowNanos);
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
orphanStreaks.keySet().retainAll(current);
|
||||
busyDriftBaselineNanos.keySet().retainAll(current);
|
||||
busyBaselineActivityNanos.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// Any unclassified collection failure must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Debounce {@link MessageService#hasOrphanedDelegation} across ticks (CB-643). Returns true only
|
||||
* once {@code observed} has held for {@link #ORPHAN_CONFIRM_TICKS} consecutive ticks; a single
|
||||
* false reading resets the streak, so a transient race never reaches the classifier.
|
||||
*/
|
||||
private boolean confirmOrphan(String target, boolean observed) {
|
||||
if (!observed) {
|
||||
orphanStreaks.remove(target);
|
||||
return false;
|
||||
}
|
||||
int streak = orphanStreaks.merge(target, 1, Integer::sum);
|
||||
return streak >= ORPHAN_CONFIRM_TICKS;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: compare this tick's monotonic and real-time readings against the previous
|
||||
* tick's, and fold any positive divergence into {@link #accumulatedDriftNanos} (a ratchet — it
|
||||
* never decreases, since the monotonic clock can only fall behind real time, never ahead of
|
||||
* it). Logs once, at WARN, when that single tick's divergence exceeds one full tick interval —
|
||||
* the signature of a host that slept between the two ticks (a tick literally cannot run while
|
||||
* the process itself is suspended, so the whole sleep duration lands inside one tick's gap).
|
||||
*
|
||||
* @return {@link #accumulatedDriftNanos} as it stood BEFORE this tick's divergence was folded
|
||||
* in — the baseline {@link #stallElapsedNanos} needs when a target is observed BUSY
|
||||
* for the first time this tick, so a sleep that happened before this member went busy
|
||||
* is not attributed to it.
|
||||
*/
|
||||
private long observeClockDrift(long nowNanos) {
|
||||
long nowRealNanos = realtimeClock.getAsLong();
|
||||
long driftBeforeThisTick = accumulatedDriftNanos;
|
||||
if (haveClockBaseline) {
|
||||
long monoDelta = nowNanos - lastTickMonoNanos;
|
||||
long realDelta = nowRealNanos - lastTickRealNanos;
|
||||
long tickDrift = realDelta - monoDelta;
|
||||
if (tickDrift > tickIntervalNanos) {
|
||||
log.warn("fleet health: the monotonic clock did not advance for about {}s that the "
|
||||
+ "real clock did since the last tick (host likely slept); the stall "
|
||||
+ "detector could not see that time", TimeUnit.NANOSECONDS.toSeconds(tickDrift));
|
||||
}
|
||||
if (tickDrift > 0) {
|
||||
accumulatedDriftNanos = driftBeforeThisTick + tickDrift;
|
||||
}
|
||||
}
|
||||
lastTickMonoNanos = nowNanos;
|
||||
lastTickRealNanos = nowRealNanos;
|
||||
haveClockBaseline = true;
|
||||
return driftBeforeThisTick;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code nowNanos - lastActivityAtNanos} alone freezes across a host sleep, since
|
||||
* both come from the monotonic {@link #clock}. This adds back the real-time drift observed
|
||||
* since this BUSY span started — not the monitor's whole lifetime, so a sleep that happened
|
||||
* before this member went busy never leaks into its stall reading (see the class-level javadoc
|
||||
* on the drift fields). The baseline resets whenever {@code lastActivityAtNanos} changes (a new
|
||||
* turn) or the member is not currently BUSY.
|
||||
*/
|
||||
private long stallElapsedNanos(MemberSession session, long nowNanos, long driftBeforeThisTick) {
|
||||
String target = session.terminalId();
|
||||
if (session.state() != MemberSession.State.BUSY) {
|
||||
busyDriftBaselineNanos.remove(target);
|
||||
busyBaselineActivityNanos.remove(target);
|
||||
return nowNanos - session.lastActivityAtNanos();
|
||||
}
|
||||
Long baselineActivity = busyBaselineActivityNanos.get(target);
|
||||
if (baselineActivity == null || baselineActivity != session.lastActivityAtNanos()) {
|
||||
busyBaselineActivityNanos.put(target, session.lastActivityAtNanos());
|
||||
busyDriftBaselineNanos.put(target, driftBeforeThisTick);
|
||||
}
|
||||
long driftSinceBusyStart = accumulatedDriftNanos - busyDriftBaselineNanos.get(target);
|
||||
return (nowNanos - session.lastActivityAtNanos()) + driftSinceBusyStart;
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
scheduleTerminalRecheck(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: schedule the one bounded, delayed follow-up for a terminal transition — never a
|
||||
* per-tick retry (CB-580 rejected that shape; {@link #reportTransition} still fires
|
||||
* {@link #failTerminalTarget} exactly once per transition, unconditionally on the tick loop).
|
||||
* This is a single one-shot task, scheduled once per transition into GONE/NEVER_READY, so a
|
||||
* member stuck terminal for the rest of its life gets exactly one extra attempt, not one per
|
||||
* tick. See {@link #recheckTerminalTarget} for why the extra attempt is safe.
|
||||
*/
|
||||
private void scheduleTerminalRecheck(String target, HealthState state) {
|
||||
if (scheduler.isShutdown()) return;
|
||||
try {
|
||||
scheduler.schedule(() -> recheckTerminalTarget(target, state),
|
||||
ASK_LAPSE_RECHECK_DELAY_SECONDS, TimeUnit.SECONDS);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("fleet health: could not schedule terminal re-check for member={} state={}",
|
||||
target, state, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: the delayed re-check {@link #scheduleTerminalRecheck} scheduled for one terminal
|
||||
* transition. By now, a {@code fleet_ask} that was still open when {@code state} was first
|
||||
* observed has had time to lapse on its own (see {@link #ASK_LAPSE_RECHECK_DELAY_SECONDS}),
|
||||
* clearing {@code Task#question} back to {@code null} — which is exactly what
|
||||
* {@link MessageService#abandon(String, String, boolean)}'s {@code sweepAsking=false} filter
|
||||
* needs to finally match it. Calling {@link #failTerminalTarget} again is safe only because
|
||||
* {@code sweepAsking} stays {@code false}: a task genuinely still {@code ASKING} is skipped
|
||||
* exactly as it was on the very first attempt — this never fails a ticket whose ask has not yet
|
||||
* lapsed.
|
||||
*
|
||||
* <p><strong>Guarded on "target is still classified {@code state}."</strong> Without this guard,
|
||||
* a member that recovered (or was released and dropped from the roster) between the transition
|
||||
* and this re-check would still take a blind {@code failTarget} call — reaching into whatever
|
||||
* brand-new, unrelated turn it has since picked up and failing it too. {@link #states} already
|
||||
* carries the live classification (updated every tick, pruned to the current roster on release),
|
||||
* so a stale or recovered target simply reads as a mismatch here and this is a no-op.
|
||||
*/
|
||||
void recheckTerminalTarget(String target, HealthState state) {
|
||||
if (states.get(target) != state) return;
|
||||
failTerminalTarget(target, state);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -1,4 +1,4 @@
|
||||
package dev.ltms.bridged.health;
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
/** Classification plus the private fact that the next pure decision needs. */
|
||||
public record HealthDecision(HealthState state, HealthPrior prior) { }
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user