Compare commits
499 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 36870836aa | |||
| 136312fb11 | |||
| 4f28da62a3 | |||
| 708f1795ad | |||
| 599419f9e6 | |||
| ac351ee1de | |||
| 274afafde6 | |||
| 8f59019305 | |||
| e966cbadf9 | |||
| b17f37a683 | |||
| c87cc25aa6 | |||
| 3fb331145a | |||
| dcd505286f | |||
| 60fa958da2 | |||
| 9950361bc9 | |||
| 9f74b6619a | |||
| f687046450 | |||
| c6058652be | |||
| 2a95da2ff4 | |||
| 7f9137fcb6 | |||
| 3bf3968bc7 | |||
| 261aa056f9 | |||
| 042b8c99dd | |||
| 4bfab6b718 | |||
| 008a457557 | |||
| 9494a6b99a | |||
| eb0557621e | |||
| a0eed6f01b | |||
| e2a91e883e | |||
| 62646957ea | |||
| 802c0ab701 | |||
| 4bab23e241 | |||
| a94262271b | |||
| 5c12865c25 | |||
| 3f8c956325 | |||
| 2051ceeb26 | |||
| 325d0771a4 | |||
| 7b97aae85b | |||
| 49a404ddf3 | |||
| 72d6a6878b | |||
| 4466ee0ef2 | |||
| 25ba7f16bb | |||
| 5ba69c9cf2 | |||
| 17052bb515 | |||
| e95ed99bf7 | |||
| 1477e4358a | |||
| 9e4e423ad6 | |||
| 6c2d6e93cb | |||
| 789b6a8716 | |||
| 01462c9695 | |||
| 8b4ff78546 | |||
| d4a2cd720c | |||
| 5a467e1f8b | |||
| 1fb6176783 | |||
| 49df79203c | |||
| 235644c0f0 | |||
| 7772b41993 | |||
| eccd0548ce | |||
| c4e23eebad | |||
| 7df503dfc2 | |||
| f5e02fedd6 | |||
| 29e7a06c49 | |||
| 92a96fcbd8 | |||
| 13482872bb | |||
| c1ca6273fc | |||
| 5f1b260c81 | |||
| 30d6872779 | |||
| 9d1306d442 | |||
| e54e3d87ea | |||
| bf027f10b9 | |||
| 3c5873dfe2 | |||
| e29227d5f4 | |||
| 45aca9eb3e | |||
| 2af13ab1ff | |||
| 0788d84be8 | |||
| 20c1094cbf | |||
| 9011c59b9f | |||
| c11ad71ed0 | |||
| bdcf285265 | |||
| d4f93a7b13 | |||
| cfebc575ea | |||
| 822327eed5 | |||
| 5289eb509f | |||
| e4c703a51a | |||
| 703a05db41 | |||
| 3f036b2a62 | |||
| 82fae94c55 | |||
| b1f34c2e6b | |||
| 3f807d9f1b | |||
| e70263062c | |||
| 1a1e586b62 | |||
| c16d118f09 | |||
| 4b10d02207 | |||
| c5fbfdbf4a | |||
| 12cff28abb | |||
| 77a6a7142e | |||
| 9f3671b801 | |||
| 84034b34d1 | |||
| 1e9b2c9b7e | |||
| 5d422f85fa | |||
| ed2027b202 | |||
| 6b0a99b2b7 | |||
| 6b7caba248 | |||
| f8b0d42a5c | |||
| d1e7d71eee | |||
| 7fd914df1a | |||
| b066eb1903 | |||
| a196d34455 | |||
| 051d320ea0 | |||
| 766772763f | |||
| eab8185d7b | |||
| 7f672f0fb8 | |||
| e2801b9bbc | |||
| ea02c7b248 | |||
| ce74e164c6 | |||
| e60f892efd | |||
| ce05886831 | |||
| be123d0ac7 | |||
| 2d09c8b027 | |||
| ed54f0224e | |||
| 8d5bc3ee89 | |||
| ab0cc71aa4 | |||
| 4cd9046353 | |||
| 4e98a74047 | |||
| a502ba53e0 | |||
| 446cc11d1d | |||
| ddd81fe174 | |||
| ae74cc081f | |||
| cf8da1d5fa | |||
| b8aedeafcb | |||
| 3d61af6f6f | |||
| ed99c209ac | |||
| af4c88d54b | |||
| 44c735f6f5 | |||
| b540a1744b | |||
| 0f51d53098 | |||
| c670792ffe | |||
| 3982ace544 | |||
| 7180b1aad0 | |||
| c26f695402 | |||
| 48877315ca | |||
| 5c08054533 | |||
| b6db9c31f5 | |||
| e7b33fe3a0 | |||
| e3e403e5c8 | |||
| 799014e99d | |||
| 69e09b10fa | |||
| b9d09e044e | |||
| fd8650cda4 | |||
| 11050e24ed | |||
| 769f282408 | |||
| 6f71f40047 | |||
| fb36c5238f | |||
| cc9cdc938b | |||
| 5fede82468 | |||
| 1515025804 | |||
| 7754f53662 | |||
| a507f7b31b | |||
| 71c322f104 | |||
| fde2c15627 | |||
| 2830735644 | |||
| 4e3ac91a22 | |||
| 29d3f0b41f | |||
| cbc732444f | |||
| 6f828b8c38 | |||
| 145a8c8862 | |||
| 7057291739 | |||
| 2302b3bc11 | |||
| 3a004dc1b3 | |||
| a9a3c12232 | |||
| 94ec77a1bc | |||
| 49f285cfda | |||
| 5a12ae7930 | |||
| 127e6832a9 | |||
| 6442a583ae | |||
| 24b96d29ae | |||
| 4721771052 | |||
| 2f71a30bd7 | |||
| 22cdebbdbe | |||
| 154971c2b8 | |||
| fef287c346 | |||
| a37acd5ee3 | |||
| d6ef0c8013 | |||
| dfb70871b4 | |||
| 4ee7b16929 | |||
| 8557289dc0 | |||
| dd2efd8541 | |||
| 5af786d135 | |||
| 380eb63277 | |||
| 6c61355f8f | |||
| 96c406b968 | |||
| a5d81c3f70 | |||
| 92c0f164f1 | |||
| 9e813ec179 | |||
| 395b3b5c46 | |||
| 6e9e464d62 | |||
| 105c065615 | |||
| 01492059d4 | |||
| f84824ee29 | |||
| c4d40fbc2b | |||
| 7c684e40d3 | |||
| 6938f52155 | |||
| 0df34f3220 | |||
| 457458437f | |||
| 3759c41f99 | |||
| 09159f2857 | |||
| 29cd1194c2 | |||
| 815e8f8b23 | |||
| 1e60ac0745 | |||
| 650a4c146b | |||
| 23f299e105 | |||
| dbf6fef0e9 | |||
| d292522d00 | |||
| 0241e0d3a8 | |||
| e4973eb8a4 | |||
| b6b88c5f1c | |||
| 86dddfe240 | |||
| 0d5944af63 | |||
| 4a5030a5c6 | |||
| c1c8794c48 | |||
| 65a78932c1 | |||
| f429ca1a50 | |||
| 73aab3f83e | |||
| 887aca0183 | |||
| 3fd23ecafa | |||
| f379847942 | |||
| ea12107497 | |||
| 591df91de1 | |||
| 6a814176f0 | |||
| d11d1d157c | |||
| d057d56156 | |||
| d703ce1313 | |||
| b32a30fd47 | |||
| 464dbc0930 | |||
| a8cadd9150 | |||
| 57b8c0b56d | |||
| 147f50c19e | |||
| eee4d576a2 | |||
| b4f9d7f53a | |||
| 3aca53b967 | |||
| 4aa1fae296 | |||
| 0c865032f9 | |||
| ea41bbf6b9 | |||
| 7b918c51ff | |||
| 554395b104 | |||
| 823976c1b5 | |||
| c8388a7f92 | |||
| 02e6aef98c | |||
| 6d493bc7bb | |||
| e5cb51a90e | |||
| e545c08082 | |||
| efa0deb9b2 | |||
| ca47e90c01 | |||
| fa1f49675b | |||
| 8426c3528f | |||
| 65f98ba910 | |||
| d05205d1eb | |||
| 667254df47 | |||
| 2926cd1784 | |||
| c801851c66 | |||
| de70aa38f1 | |||
| 53a533afb4 | |||
| 77ad88631b | |||
| d88017807b | |||
| 2159a5a94a | |||
| b9c2cf69f4 | |||
| 8beae50fe7 | |||
| b2a58cb966 | |||
| b8b25cf74c | |||
| f159ca7d27 | |||
| 83f2aea60f | |||
| 002329adb5 | |||
| a49671ceb9 | |||
| 76672ff016 | |||
| 9379f92c23 | |||
| 21ff63b11d | |||
| 21844b54d7 | |||
| 4769481515 | |||
| bbbb4c1eb3 | |||
| 38c248e617 | |||
| 9debc0de27 | |||
| f34361b263 | |||
| be5ba22c75 | |||
| 85c90d440a | |||
| e97502d550 | |||
| aef14ff46e | |||
| cba516bda4 | |||
| ba51e0c6cc | |||
| 086c59848e | |||
| 0c10079755 | |||
| ece2091b53 | |||
| b3f917e6f5 | |||
| fa39a5f55e | |||
| d5128a1d35 | |||
| 61097e5cf0 | |||
| ef507bcd12 | |||
| 94f50e507a | |||
| 75b15086b0 | |||
| dab9645906 | |||
| e93b5f6512 | |||
| bfabe13e8f | |||
| 7e49c6eca2 | |||
| 11cbfa79b4 | |||
| c2c2746922 | |||
| 6ed70700a0 | |||
| 0087645da4 | |||
| 19cacf5b62 | |||
| 4dd12083ab | |||
| 30e21adec7 | |||
| 719b79f892 | |||
| 66e5247b6d | |||
| 1e41bd63b4 | |||
| 4887d03d88 | |||
| 7d5434455d | |||
| f71ee4926e | |||
| 282a2fc2b8 | |||
| f04e934b94 | |||
| 3fae35c357 | |||
| 18aecbfe67 | |||
| 5d75f72473 | |||
| e028a0ae54 | |||
| 2fa673d4c0 | |||
| 9020d01b40 | |||
| 1006805027 | |||
| 27aefbf9a0 | |||
| d42c2bc204 | |||
| 3916adc372 | |||
| fa97f598dd | |||
| ea9aa4fd77 | |||
| a2b8caf6b5 | |||
| b5ddbe5757 | |||
| d223a93039 | |||
| 96d8191149 | |||
| f0e7ac73d6 | |||
| e2fe861b4d | |||
| 279d6f5fbd | |||
| 34480cebef | |||
| 9d0bf14c46 | |||
| 5a3ab5764c | |||
| 3bad9f5785 | |||
| 8308c0b68f | |||
| 3b3063eb2b | |||
| df9086263d | |||
| eb568ff451 | |||
| ac790e4cce | |||
| 6b5f3f472f | |||
| 38dec72152 | |||
| 0e8bfb74fc | |||
| 51f7b0a3ca | |||
| 21c539f22e | |||
| bd2774b5f1 | |||
| c50f5b2d61 | |||
| 01a840cc14 | |||
| c4deef08be | |||
| c796eac09c | |||
| e897e5257b | |||
| 2afa3652bb | |||
| 80092ff359 | |||
| 9d37f3aa29 | |||
| eaf89abaf6 | |||
| d895f02bc1 | |||
| 43206cac2f | |||
| 8bba3a8184 | |||
| 5cf3ca9a89 | |||
| ac474981e4 | |||
| ba04b2359b | |||
| 2e5b63f6f6 | |||
| 3437d6313d | |||
| 743377d6cd | |||
| 5952d559c7 | |||
| 205ad823b0 | |||
| d654ccb818 | |||
| a89dcc9b7e | |||
| 0d7b4fb026 | |||
| 321d8dcbb5 | |||
| ef8c97871e | |||
| 26bafe824b | |||
| 838a701109 | |||
| e5eb3534c7 | |||
| 959c83534f | |||
| 31b028e860 | |||
| 7840e9adf6 | |||
| c3672f5472 | |||
| cf54aed451 | |||
| bbf68f3e3c | |||
| 776743cbe2 | |||
| 826e0aeb2a | |||
| c935b181dd | |||
| ee932fd85b | |||
| fe2e5ede34 | |||
| c325054242 | |||
| 7662e2d0c8 | |||
| 4877992a70 | |||
| 1c051c4e47 | |||
| adb7a67880 | |||
| 723fe494e9 | |||
| 0f08b93659 | |||
| 432c1d92d1 | |||
| 60b7e67b42 | |||
| 3fbd43fe3f | |||
| 32ebf065ac | |||
| 1178b3f684 | |||
| 2a434ced2f | |||
| cc919aa2b6 | |||
| 6417b0edd9 | |||
| 39c7ce76f3 | |||
| b40f477210 | |||
| 5c56cb347f | |||
| e694deace3 | |||
| 748367b7d6 | |||
| dcf5fb3be3 | |||
| f9d2ee2a2b | |||
| 866c7f2e9a | |||
| d9168de43e | |||
| c3fa1136d4 | |||
| cabcd87b66 | |||
| bf0e09b1a2 | |||
| e1eb50ce65 | |||
| 049e7d9d54 | |||
| ff3b49cd1e | |||
| 0373b6c41b | |||
| 445a45f6e1 | |||
| 966c58a3b8 | |||
| de026b8f8a | |||
| 97f6c33a45 | |||
| 735c837604 | |||
| 457dc0330d | |||
| a1a9015217 | |||
| 5a811a3695 | |||
| 1fdaa74eb3 | |||
| 7d4a4339c2 | |||
| 847e8bd3fa | |||
| f9fb387427 | |||
| a55079afbd | |||
| e18ad4723b | |||
| 6de8ac8972 | |||
| a052975420 | |||
| 6d82ca95a4 | |||
| 8067ee4ec4 | |||
| 3743789e8d | |||
| 388aba7632 | |||
| ad587eafa3 | |||
| 08ce9aef11 | |||
| ee5f8b932b | |||
| 4ac688b6d9 | |||
| a814d1ef00 | |||
| ea98856130 | |||
| a49e96835a | |||
| 63c19dcba7 | |||
| 2823349c8e | |||
| 6fc301d62c | |||
| bc99d64786 | |||
| 615af4ed0a | |||
| 045d229728 | |||
| d1fd5700f5 | |||
| 2d55b0b9a5 | |||
| d89ae94a2e | |||
| e6193c4098 | |||
| b66f0677ed | |||
| 23ada1981e | |||
| a22480c117 | |||
| 24f404f989 | |||
| a237fbff9d | |||
| 31d5516991 | |||
| 17af61e8dd | |||
| 6af87b6ad6 | |||
| 5ba05d0bdb | |||
| fc655e78c2 | |||
| 25726a5ae7 | |||
| 11c3ff67b6 | |||
| d867c87100 | |||
| b0c4cedfab | |||
| 85417d5215 | |||
| 4accc746bd | |||
| 21c4c8cbef | |||
| 430f5b0dae | |||
| 42731833d0 | |||
| 65acf066ad | |||
| ee8f570fd7 | |||
| fa3f910d44 | |||
| 46ac6e4e38 | |||
| 3bfa82839b | |||
| 65ccf2e4ad | |||
| 7a3b27f76f | |||
| 82e7be564c | |||
| c5e24197bf | |||
| bcb402b688 | |||
| 7c4170ff6d | |||
| a6095743f0 | |||
| 66e776d178 | |||
| 26f64cba45 | |||
| 51047848f1 | |||
| d8c0b657e8 | |||
| 312c0584ce | |||
| da2625acfb | |||
| 97d9cebc59 | |||
| 3ce76a5d69 | |||
| 2e138a199b | |||
| 450a5ed9c5 |
@@ -1,15 +1,15 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the bridged MCP gateway.",
|
||||
"name": "fleetd",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleet",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the bridged MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"description": "Mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.2.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
|
||||
@@ -3,7 +3,7 @@ name: architect
|
||||
description: Refine work into clear, independent units before implementation.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
|
||||
@@ -3,7 +3,7 @@ name: dev
|
||||
description: Implement one assigned unit, test it, and open a pull request.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
|
||||
@@ -3,7 +3,7 @@ name: reviewer
|
||||
description: Review one assigned scope and report the most important real issue.
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
@@ -0,0 +1,274 @@
|
||||
---
|
||||
name: fleets-status
|
||||
description: Report the status of every fleet that shares one LavinMQ instance. Use for local daemon health, broker-wide fleet presence, and cross-host lead coordination checks.
|
||||
---
|
||||
|
||||
# Status of every fleet on the shared LavinMQ instance
|
||||
|
||||
**The headline: always report what is missing.** This skill starts with the local fleet, then adds
|
||||
broker-wide facts when its read-only credential exists. A missing fleet must appear as `unknown` or
|
||||
`not reachable`, with the reason and the fix. Never leave it out.
|
||||
|
||||
The known topology has one LavinMQ instance on `10.10.20.13` (`fleet01`). AMQP uses port `5672`,
|
||||
and the management API uses port `15672`. The Mac fleet owns vhost `/mac`. The fleet01 fleet owns
|
||||
vhost `/fleet01`.
|
||||
|
||||
## 1. Protect credentials before any probe
|
||||
|
||||
**Hard rule — never print `LAVINMQ_URI`.** It is an AMQP URI with its password inline. It only
|
||||
resolves in a login shell because `${SHARED_ENV}/tools/secrets.sh` supplies it. A non-login shell
|
||||
can make every broker probe look empty.
|
||||
|
||||
- Never run `echo "$LAVINMQ_URI"`.
|
||||
- Never put `${LAVINMQ_URI:-something}` in output. That form expands to the secret value when set.
|
||||
- Parse the user, host, and password into shell or Python variables. Use them without printing them.
|
||||
- Prefer `resolves` or `does not resolve` over any part of the value.
|
||||
- Every command that can read `LAVINMQ_URI` must send all output through this redaction before it
|
||||
reaches the report:
|
||||
|
||||
```bash
|
||||
sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
```
|
||||
|
||||
**The `g` flag is not optional.** Without it `sed` replaces only the first match on each line, so a
|
||||
line carrying two URIs leaks the second one. `scripts/redeploy-fleetd.sh --check` prints lines like
|
||||
that. Checked on 2026-08-27: without `g`, `amqp://u1:p1@h1/mac and http://u2:p2@h2:15672/api`
|
||||
redacts the first pair and prints `u2:p2` in the clear.
|
||||
|
||||
Keep `pipefail` on when applying that filter. Otherwise the filter can hide a failed probe. Apply
|
||||
the same no-print rule to the management password below, even though it is not in an AMQP URI.
|
||||
|
||||
## 2. Tier 1 — this fleet (always run)
|
||||
|
||||
Start here even when the broker tier is blocked. Work from the local fleetd checkout.
|
||||
|
||||
First run the read-only deployment check. It already checks the daemon process, deployed jar versus
|
||||
the checkout `HEAD`, launchd state, and whether each configured token resolves in a login shell.
|
||||
Do not copy those checks into new shell code. The script reads `LAVINMQ_URI`, so redact all output:
|
||||
|
||||
```bash
|
||||
set -o pipefail
|
||||
scripts/redeploy-fleetd.sh --check 2>&1 \
|
||||
| sed -E 's#://[^@]*@#://<redacted>@#g'
|
||||
git rev-parse HEAD
|
||||
```
|
||||
|
||||
Treat jar drift as a top-level warning. A merge is not a deployment. State the running jar result
|
||||
as `matches HEAD`, `drift`, or `unknown`; do not turn an unclear timestamp into a match.
|
||||
|
||||
Report the process identifier (PID) and uptime too:
|
||||
|
||||
```bash
|
||||
PIDS="$(pgrep -f 'target/fleetd.jar' || true)"
|
||||
if [ -z "$PIDS" ]; then
|
||||
printf '%s\n' 'fleetd: not running'
|
||||
else
|
||||
for PID in $PIDS; do
|
||||
ps -p "$PID" -o pid=,etime=,lstart=,command=
|
||||
done
|
||||
fi
|
||||
```
|
||||
|
||||
Read the full health response. Keep the HTTP status because `503` means fleetd is running but herdr
|
||||
is not reachable. Report both `herdr.version` and `herdr.protocol` when present:
|
||||
|
||||
```bash
|
||||
curl -sS --max-time 5 -w '\nHTTP %{http_code}\n' http://127.0.0.1:8765/healthz
|
||||
```
|
||||
|
||||
Call `fleet_whoami`, then call `fleet_list`. Preserve its sections in the report:
|
||||
|
||||
- `leads`, including which row is this lead;
|
||||
- `members`, including state, role, profile, branch, and worktree when present;
|
||||
- every per-profile `capacity` row, including `maxLoad`, `live`, `free`, and quarantine facts;
|
||||
- the exact `healthCoverage` value.
|
||||
|
||||
Do not describe an empty `members` list as an empty fleet. It says only that no members are spawned.
|
||||
Also do not hide a profile with `free: 0`; say whether load or credential quarantine caused it.
|
||||
|
||||
Show only WARN, ERROR, and SEVERE lines after the last `fleetd listening` line. This anchor stops an
|
||||
old incident from looking current:
|
||||
|
||||
```bash
|
||||
python3 - <<'PY'
|
||||
from pathlib import Path
|
||||
import re
|
||||
|
||||
path = Path("fleetd/fleetd.out")
|
||||
if not path.exists():
|
||||
print("cannot check current WARN/ERROR: fleetd/fleetd.out does not exist")
|
||||
else:
|
||||
lines = path.read_text(errors="replace").splitlines()
|
||||
starts = [i for i, line in enumerate(lines) if "fleetd listening" in line]
|
||||
if not starts:
|
||||
print("cannot anchor WARN/ERROR: no 'fleetd listening' line exists")
|
||||
else:
|
||||
current = lines[starts[-1]:]
|
||||
alerts = [line for line in current if re.search(r"\b(?:WARN|ERROR|SEVERE)\b", line)]
|
||||
print(f"current WARN/ERROR/SEVERE count: {len(alerts)}")
|
||||
for line in alerts[-50:]:
|
||||
print(line)
|
||||
PY
|
||||
```
|
||||
|
||||
**What this tier cannot see:** it proves facts only about the Mac daemon at `127.0.0.1:8765`.
|
||||
It cannot show the fleet01 daemon, broker queue depth, or broker consumers. The fleet01 REST service
|
||||
at `10.10.20.13:8765` is not reachable from the Mac. Say this in the report rather than omitting
|
||||
fleet01.
|
||||
|
||||
**But fleet01 IS reachable over SSH — checked 2026-08-28.** An older version of this line said SSH
|
||||
was denied. That is true only for the user `dai.ha`. The host alias `fleet01` maps to user `ltms`,
|
||||
and `ssh fleet01` works with key auth:
|
||||
|
||||
```bash
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=6 fleet01 'echo $(id -un)@$(hostname)'
|
||||
```
|
||||
|
||||
So fleet01's daemon PID, uptime, jar and `/healthz` **can** be reported — over SSH, not over REST.
|
||||
Do that rather than writing `not reachable`. `ltms` also has passwordless sudo there.
|
||||
|
||||
## 3. Tier 2 — the shared broker (run when management access exists)
|
||||
|
||||
**This tier is blocked today.** The AMQP user in `LAVINMQ_URI` can connect on port `5672`, but gets
|
||||
HTTP `401` from the management API on port `15672`. An AMQP connection does not grant monitoring
|
||||
access.
|
||||
|
||||
The operator must create a separate, read-only LavinMQ management user with the `monitoring` tag.
|
||||
It needs access to inspect both `/mac` and `/fleet01`. Store its values as
|
||||
`LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD` in
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Do not reuse or print the AMQP URI. Full multi-fleet status stays
|
||||
blocked until this user exists.
|
||||
|
||||
When both variables resolve, run this from a login shell. It calls `GET /api/overview`,
|
||||
`GET /api/vhosts`, `GET /api/queues`, and `GET /api/connections`. It prints selected status fields,
|
||||
but never the user, password, Authorization header, or AMQP URI:
|
||||
|
||||
```bash
|
||||
zsh -lc 'python3 - "$@"' -- <<'PY'
|
||||
import base64
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
base = "http://10.10.20.13:15672"
|
||||
user = os.environ.get("LAVINMQ_MANAGEMENT_USER", "")
|
||||
password = os.environ.get("LAVINMQ_MANAGEMENT_PASSWORD", "")
|
||||
if not user or not password:
|
||||
print("broker tier: BLOCKED — management credential does not resolve in a login shell")
|
||||
sys.exit(0)
|
||||
|
||||
token = base64.b64encode(f"{user}:{password}".encode()).decode()
|
||||
|
||||
def get(path):
|
||||
request = urllib.request.Request(
|
||||
base + path,
|
||||
headers={"Authorization": "Basic " + token, "Accept": "application/json"},
|
||||
)
|
||||
with urllib.request.urlopen(request, timeout=5) as response:
|
||||
return json.load(response)
|
||||
|
||||
try:
|
||||
overview = get("/api/overview")
|
||||
vhosts = get("/api/vhosts")
|
||||
queues = get("/api/queues")
|
||||
connections = get("/api/connections")
|
||||
except urllib.error.HTTPError as error:
|
||||
print(f"broker tier: BLOCKED — management API returned HTTP {error.code}")
|
||||
sys.exit(0)
|
||||
except Exception as error:
|
||||
print(f"broker tier: BLOCKED — management API is not reachable: {type(error).__name__}")
|
||||
sys.exit(0)
|
||||
|
||||
fleet_names = {"/mac": "Mac fleet", "/fleet01": "fleet01 fleet"}
|
||||
print(json.dumps({
|
||||
"overview": {
|
||||
"lavinmq_version": overview.get("lavinmq_version"),
|
||||
"rabbitmq_version": overview.get("rabbitmq_version"),
|
||||
"queue_totals": overview.get("queue_totals", {}),
|
||||
"object_totals": overview.get("object_totals", {}),
|
||||
},
|
||||
"fleets": [
|
||||
{
|
||||
"fleet": fleet_names.get(vhost.get("name"), "UNKNOWN FLEET"),
|
||||
"vhost": vhost.get("name"),
|
||||
"queues": [
|
||||
{
|
||||
"name": queue.get("name"),
|
||||
"messages": queue.get("messages", 0),
|
||||
"messages_ready": queue.get("messages_ready", 0),
|
||||
"messages_unacknowledged": queue.get("messages_unacknowledged", 0),
|
||||
"consumers": queue.get("consumers", 0),
|
||||
}
|
||||
for queue in queues if queue.get("vhost") == vhost.get("name")
|
||||
],
|
||||
"connections": [
|
||||
{
|
||||
"name": connection.get("name"),
|
||||
"peer_host": connection.get("peer_host"),
|
||||
"state": connection.get("state"),
|
||||
}
|
||||
for connection in connections if connection.get("vhost") == vhost.get("name")
|
||||
],
|
||||
}
|
||||
for vhost in vhosts
|
||||
],
|
||||
}, indent=2, sort_keys=True))
|
||||
PY
|
||||
```
|
||||
|
||||
Map `/mac` to the Mac fleet and `/fleet01` to the fleet01 fleet. Keep any other vhost in the
|
||||
report as `UNKNOWN FLEET`; do not drop it. For each vhost, total the ready, unacknowledged, and all
|
||||
messages. Report every queue's consumer count and each live connection.
|
||||
|
||||
**A vhost with queues but zero consumers means that fleet's daemon is down while its durable state
|
||||
survives. Call this out as a top-level warning.** This is the main reason to use the management API
|
||||
instead of calling each remote daemon.
|
||||
|
||||
**What this tier cannot see:** without the new `monitoring` credential it cannot enumerate any
|
||||
vhost, queue, depth, consumer, or connection. With the credential it still cannot report fleet01's
|
||||
daemon PID, uptime, jar revision, `/healthz`, herdr version, or member capacity. Those need reachable
|
||||
fleet01 REST or SSH access, which the Mac does not have today.
|
||||
|
||||
## 4. Tier 3 — cross-fleet lead coordination
|
||||
|
||||
Use the queue data from Tier 2. Select queues whose names match `lead.<coordId>.inbox`. Report each
|
||||
queue's vhost, depth, consumer count, and the `coordId` between the prefix and suffix.
|
||||
|
||||
- A lead inbox with a consumer shows that a lead mailbox is live on that vhost.
|
||||
- A durable lead inbox with zero consumers shows saved coordination state, but no live receiver.
|
||||
- No lead inbox is not proof that coordination is disabled. The daemon may be down before declaring
|
||||
its queue, or this account may not be allowed to see the vhost.
|
||||
|
||||
This Mac fleet currently sets both `broker.uriEnv` and `coordinator.uriEnv` to the same variable,
|
||||
`LAVINMQ_URI`. Therefore its coordinator connects to `/mac`. Cross-host `fleet_send{coordId}` routes
|
||||
only when both leads share the same coordinator vhost. If the fleet01 lead uses `/fleet01` for its
|
||||
coordinator, the leads cannot see each other and the send will not route.
|
||||
|
||||
**Open question:** the fleet01 coordinator vhost has not been checked. Surface this question in
|
||||
every report until a live `lead.<coordId>.inbox` consumer or fleet01's config proves the answer. Do
|
||||
not claim that fleet01 uses `/fleet01` just because its member queues do.
|
||||
|
||||
Also compare these broker facts with the `leads` rows from local `fleet_list`. A missing remote lead
|
||||
is `not visible from this coordinator`, not `down`, unless the broker consumer facts prove it.
|
||||
|
||||
**What this tier cannot see:** without Tier 2 management access it cannot list lead inboxes or their
|
||||
consumers. Even with that access, a stopped fleet01 daemon leaves only durable queue history. That
|
||||
history cannot prove which coordinator URI its current config would use after restart.
|
||||
|
||||
## 5. Report all fleets
|
||||
|
||||
Use one row per known or discovered fleet. Include blocked rows.
|
||||
|
||||
| Fleet | Daemon | Deployment | Herdr | Members/capacity | Queues/consumers | Lead coordination | Cannot check |
|
||||
|---|---|---|---|---|---|---|---|
|
||||
| Mac (`/mac`) | PID + uptime | jar vs `HEAD` | health + version + protocol | `fleet_list` + `healthCoverage` | facts or blocked reason | inbox facts or open question | exact missing facts |
|
||||
| fleet01 (`/fleet01`) | reachable/down/unknown | value or `not reachable` | value or `not reachable` | value or `not reachable` | facts or blocked reason | inbox facts plus coordinator-vhost question | exact missing facts and fix |
|
||||
|
||||
Add rows for unknown vhosts. End with three short sections: `Current warnings`, `Checks that were
|
||||
blocked`, and `Operator action`. Until the management user exists, `Operator action` must say:
|
||||
|
||||
> Create a read-only LavinMQ management user with the `monitoring` tag and access to `/mac` and
|
||||
> `/fleet01`. Put its user and password in `${SHARED_ENV}/tools/secrets.sh` as
|
||||
> `LAVINMQ_MANAGEMENT_USER` and `LAVINMQ_MANAGEMENT_PASSWORD`.
|
||||
@@ -0,0 +1,226 @@
|
||||
---
|
||||
name: handover
|
||||
description: Procedure for an outgoing lead to write the handover file that a fresh lead session inherits. Load this when your context is filling up and you are about to be replaced, whether you hand off by hand or fleetd does it for you. The file is the new lead's only inheritance — follow it exactly.
|
||||
---
|
||||
|
||||
# Handover — write the file the next lead depends on
|
||||
|
||||
A lead session fills up its context and has to be replaced by a fresh one. The outgoing lead
|
||||
writes a handover file, and the new session reads that file and carries on.
|
||||
|
||||
**There are two ways to hand off, and the file is the same either way.**
|
||||
|
||||
- **By hand.** You write the file, then tell the operator where it is. The operator starts the new
|
||||
session and points it at the file. This always works.
|
||||
- **With `fleet_handover`** (fleetd #480, merged 2026-09-11). You ask fleetd to do the swap: it
|
||||
checks the file, clears your pane, and tells the fresh session to read it. This needs
|
||||
`leadRollover:` in `fleetd.yaml`; without it every action answers a clean refusal naming
|
||||
`NOT_CONFIGURED`, and you fall back to the manual path. Section 11 below is the procedure.
|
||||
|
||||
Nothing else in this skill changes between the two. Only who performs the swap changes.
|
||||
|
||||
**The new lead's only inheritance is that file.** It does not see your conversation, your plan,
|
||||
or your screen. If the file is thin or wrong, the new lead re-derives what you already knew, and
|
||||
that wastes hours. Writing a good handover file is real work. It is not paperwork you rush
|
||||
through at the end of a session.
|
||||
|
||||
This skill is the procedure for writing it. Every rule below earned its place because a past
|
||||
handover got it wrong.
|
||||
|
||||
## 1. Confirm you are the right session to write this
|
||||
|
||||
Run `fleet_whoami` first. It must answer `primary`. Only a primary (lead) session writes a
|
||||
handover file. A worker's job ends with its own pull request, not a fleet-wide handoff.
|
||||
|
||||
## 2. Every number needs a command, run in this turn
|
||||
|
||||
A number is a claim: a count, a commit hash, a process id, a percentage, a queue depth. Before
|
||||
you write one, run the command that produces it — now, in this turn, against the live state.
|
||||
|
||||
Never take a number from:
|
||||
|
||||
- earlier in your own conversation — the state has moved since then,
|
||||
- a peer lead's report — that is their measurement, not yours,
|
||||
- your own memory of an earlier session.
|
||||
|
||||
Put the command, or its real output, next to the number. That lets the next lead re-run it and
|
||||
check it still matches. If you cannot measure something yourself, say so instead of guessing:
|
||||
"the fleet01 lead reports 91 commits behind; I have not checked this myself."
|
||||
|
||||
## 3. Say what you measured and what you did not
|
||||
|
||||
Mark every claim as one of two things:
|
||||
|
||||
- **"I checked this myself, in the code or on this host, at `<time>`."**
|
||||
- **"I did not check this myself; `<who>` reported it."**
|
||||
|
||||
Never present someone else's measurement as your own. This matters most for cross-host claims —
|
||||
a peer lead's daemon, a worker's report, or something the operator said earlier that you cannot
|
||||
re-verify from here.
|
||||
|
||||
## 4. Record open decisions, and who owns them
|
||||
|
||||
List three things:
|
||||
|
||||
- what the operator actually asked for, in their own words where you have them,
|
||||
- what is still unanswered,
|
||||
- any question you decided yourself instead of asking, with your reason.
|
||||
|
||||
Write the decision so it cannot be mistaken for the operator's instruction. Say plainly: "the
|
||||
operator never answered X; I decided Y, because Z." Without this, the next lead either silently
|
||||
reopens a closed question or assumes the operator chose something they never did.
|
||||
|
||||
## 5. Record live hazards
|
||||
|
||||
List anything that will break if the next lead does the obvious thing next. This includes:
|
||||
|
||||
- unpushed commits or unmerged branches,
|
||||
- a build, a spawn, or a redeploy still running,
|
||||
- code merged to `main` but not yet redeployed to the live daemon,
|
||||
- any trap that looks safe and is not — say what goes wrong and why, not only that something is
|
||||
"tricky."
|
||||
|
||||
## 6. Record what is explicitly not owed
|
||||
|
||||
List work that is finished, and work that another party has said they do not want touched. Name
|
||||
who said so and when. Without this line, the next lead re-does closed work or reopens a question
|
||||
a peer already declined to revisit.
|
||||
|
||||
## 7. Open the file with three re-measurement commands
|
||||
|
||||
The file's own first section must give the next lead three concrete commands to run before
|
||||
acting on anything else in the file:
|
||||
|
||||
1. confirm role — for example `fleet_whoami`,
|
||||
2. confirm the state of the working tree — for example `git status` and
|
||||
`git rev-list origin/main..HEAD`,
|
||||
3. read the live fleet — for example `fleet_list`.
|
||||
|
||||
Record what each command answered when you wrote the file, and tell the reader to run it again
|
||||
rather than trust your answer. The point of this section is that the reader checks live state
|
||||
before acting on any claim in the rest of the file, including yours.
|
||||
|
||||
## 8. Stamp the file with time and commit
|
||||
|
||||
Near the top of the file, write:
|
||||
|
||||
- the date and time you wrote it,
|
||||
- the commit the tree was on (`git rev-parse HEAD`),
|
||||
- whether the tree was clean (`git status`).
|
||||
|
||||
Without this, nobody can tell how old the file is, or which code it describes.
|
||||
|
||||
## 9. State plainly that the file goes stale fast
|
||||
|
||||
Say near the top: **re-measure anything you act on.** The file goes stale the moment anyone
|
||||
merges a branch, spawns a member, or restarts the daemon. Everything in the file is a snapshot
|
||||
of one moment, not a live fact.
|
||||
|
||||
## 10. What to leave out
|
||||
|
||||
Do not include:
|
||||
|
||||
- narration of how the session felt, or how hard something was,
|
||||
- anything the repo already records — code structure, git history, or a rule already written in
|
||||
`CLAUDE.md`. Point at it instead of repeating it,
|
||||
- advice that is only true for the session that is ending — a half-open terminal, a local
|
||||
variable, a train of thought with no state behind it.
|
||||
|
||||
A handover file is a record of state and decisions. It is not a diary.
|
||||
|
||||
## 11. Using `fleet_handover` (only if `leadRollover:` is configured)
|
||||
|
||||
**Run the three steps in this order. The order is not a style choice — the wrong order is
|
||||
refused.**
|
||||
|
||||
1. **`fleet_handover{action: "open", reason: "<why now>"}`.** It returns a `token` and the
|
||||
`handoverPath` you must write to. Nothing has happened to your pane yet.
|
||||
|
||||
**Write to exactly that path, and do not resolve it yourself.** It is always absolute, even when
|
||||
the operator configured a relative `handoverPath`: fleetd resolves a relative one against your
|
||||
own workspace before it hands it to you. The daemon and your pane can run in different
|
||||
directories, so a path you resolve yourself can point at a different file from the one the daemon
|
||||
will check.
|
||||
|
||||
**Check that the path is ignored by git before you write to it (#491).** A relative
|
||||
`handoverPath` resolves inside YOUR workspace, which is usually a repository — and usually not
|
||||
the `fleetd` one, so an ignore rule added to `fleetd` does not protect it. Run
|
||||
`grep -n handover <your workspace>/.gitignore`. No output means the file you are about to write
|
||||
will show up as untracked content in that repo. The file is a snapshot of live state and must
|
||||
never be committed, so tell the operator rather than committing it or silently editing their
|
||||
`.gitignore`.
|
||||
2. **Write the handover file at that path**, following sections 1–10 above.
|
||||
3. **Ask the operator, then `fleet_handover{action: "confirm", token, operatorConfirmed: true}`.**
|
||||
|
||||
Why that order: `confirm` refuses with `HANDOVER_STALE` unless the file was modified **after** the
|
||||
`open` request. That check stops a leftover file from an earlier session being accepted as this
|
||||
one's handover. So writing the file first and then calling `open` — the obvious order — always
|
||||
fails.
|
||||
|
||||
`{action: "cancel", token}` drops a pending request without rolling.
|
||||
|
||||
**Things that will surprise you:**
|
||||
|
||||
- **`accepted` does not mean your pane has been cleared.** It means every gate passed and the roll
|
||||
is scheduled to run once your current turn ends. Say your goodbye in the same turn — you will not
|
||||
get another one.
|
||||
- **There is no terminal or session parameter, on purpose.** The pane is always your own, resolved
|
||||
from your connection, so you can only ever roll yourself.
|
||||
- **`operatorConfirmed` is your report of what a human told you.** Do not pass `true` because you
|
||||
are confident. Ask, wait for the answer, then pass what they said. `requireOperatorConfirm`
|
||||
defaults to `true` and this is the only thing standing between a judgement call and a wiped
|
||||
session.
|
||||
- **The roll can still refuse after `confirm` returns**, and by then there is no caller to tell.
|
||||
Those outcomes are logged only, as `lead-rollover:` lines in the daemon log.
|
||||
- **The bootstrap prompt has never yet landed, and the fix is unproven (fleetd #489).** The first
|
||||
real rollover, on 2026-09-12, joined `/clear` and the bootstrap text into one line and Claude Code
|
||||
refused it as `Unknown command: /clearFresh`. The pane was never cleared and no context was lost,
|
||||
so the failure was safe — the roll simply did nothing. PR #490 fixed the cause and is deployed,
|
||||
but no roll has bootstrapped a fresh session end to end yet. **Assume it may still fail, and tell
|
||||
the operator so before you confirm.** The recovery is the same either way: the file is already
|
||||
written, so the operator starts a session and points it at the file. That is why you write the
|
||||
file before you confirm, and never the other way round.
|
||||
|
||||
## Writing style
|
||||
|
||||
Write in plain English. Use everyday words, one idea per sentence, and active voice. Keep every
|
||||
class, method, file, flag, and config key exactly as it appears in the code — replacing a
|
||||
precise term with a vague one makes the sentence wrong, not simpler. Explain an abbreviation the
|
||||
first time you use it.
|
||||
|
||||
If a diagram genuinely helps, put it in the `.md` file as a fenced ` ```mermaid ` block with no
|
||||
hardcoded colors, so it stays readable on light and dark backgrounds. Quote any label that has
|
||||
brackets, colons, or slashes.
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
# Handover — <fleet name> lead session, <date and time>
|
||||
|
||||
Written at commit `<output of git rev-parse HEAD>`. Tree was <clean, or dirty: `<git status
|
||||
summary>`>. Re-measure anything you act on — this file goes stale the moment anyone merges,
|
||||
spawns, or restarts.
|
||||
|
||||
## 0. Do these three things first
|
||||
|
||||
1. Confirm your role: `fleet_whoami` — must answer `primary`. (Answered `<result>` at `<time>`.)
|
||||
2. Confirm tree state: `git status`, `git rev-list origin/main..HEAD`. (`<result>` at `<time>`.)
|
||||
3. Read the live fleet: `fleet_list`. (`<result>` at `<time>`.)
|
||||
|
||||
## 1. What the operator asked for
|
||||
|
||||
<the live instructions, in their words where you have them; what is still open; any decision
|
||||
you made yourself, and why>
|
||||
|
||||
## 2. Open decisions, and who owns them
|
||||
|
||||
<one line per decision: who owns it, what is unanswered>
|
||||
|
||||
## 3. Live hazards
|
||||
|
||||
<one entry per hazard: what breaks, and why, if the next lead does the obvious thing>
|
||||
|
||||
## 4. What is not owed
|
||||
|
||||
<finished work, and work another party has declined; name who said so and when>
|
||||
```
|
||||
@@ -0,0 +1,102 @@
|
||||
---
|
||||
name: hunter
|
||||
description: Defect-hunt procedure for a fleetd worker — sweep an assigned package for real bugs and report several ranked findings without fixing anything. Load this when the lead asks you to hunt or audit a scope rather than review one diff. Do NOT load `reviewer` for this; the two want different output.
|
||||
---
|
||||
|
||||
# Hunter worker — procedure
|
||||
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies.
|
||||
|
||||
**This skill is not `reviewer`.** `reviewer` judges one diff and reports the *single* most
|
||||
important issue in about 90 words. A hunt sweeps a whole package and reports *several* findings
|
||||
in a long structured form. Loading both gives you two contradictory output contracts, and the
|
||||
usual result is a worker that writes a good report into its terminal and ends the turn without
|
||||
sending it. Load exactly one.
|
||||
|
||||
## 0. Read this before you read code: how the report gets home
|
||||
|
||||
Your terminal reaches nobody. The lead sees **only** the text inside your `fleet_reply` call.
|
||||
|
||||
A long report is exactly the case where this goes wrong, so plan for it:
|
||||
|
||||
- **Write the report into the `fleet_reply` argument itself.** Do not compose it in your terminal
|
||||
and then summarise it into the call.
|
||||
- If the report is long, **send it anyway** — one `fleet_reply` with everything.
|
||||
- If you end the turn without replying, the bridge scrapes your pane instead. That scrape carries
|
||||
at most the last 4000 characters, and on a hunt it usually captures the tail of the lead's own
|
||||
brief rather than your findings. The lead then has nothing and has to ask you again.
|
||||
|
||||
## 1. Change nothing
|
||||
|
||||
A hunt is read-only. Do not edit a production file, do not "quickly fix" what you find, and do
|
||||
not run a formatter. You may run the build and tests to *check* a claim, and you should say so
|
||||
when you did.
|
||||
|
||||
## 2. Read the whole scope first
|
||||
|
||||
Read every file in the assigned package before you judge any of it. A defect that a caller
|
||||
elsewhere in the same package makes unreachable is not a defect, and you cannot know that from
|
||||
one file.
|
||||
|
||||
Stay inside the scope. If a defect there depends on a class outside it, read that class to
|
||||
confirm — but the defect itself must live in the scope you were given.
|
||||
|
||||
## 3. The bar — this matters more than the count
|
||||
|
||||
**Name the path into the bad state.** Say which caller, in which state, reaches it. A defect on
|
||||
paper is not a reachable defect. If you cannot name that path, keep the finding but mark it
|
||||
`unproven` and say exactly what you could not check. Do not drop it, and do not dress it up.
|
||||
|
||||
**Say which direction the harm goes.** Data loss, privilege escalation and silent wrong answers
|
||||
are worth reporting even when the window is narrow. A finding whose worst outcome is a worse log
|
||||
line is not worth a block.
|
||||
|
||||
Two workers once ran the same scope: the one that applied the direction-of-harm filter found ten
|
||||
real defects, the one that did not found none. Fewer findings the lead can act on beat many the
|
||||
lead has to triage.
|
||||
|
||||
## 4. Shapes that have produced real merged fixes here
|
||||
|
||||
Read for these first:
|
||||
|
||||
1. **A one-way gate.** A guard added after an incident closes only the direction that incident
|
||||
came from. Do not only ask what closes the gate — ask **which states still open it**.
|
||||
2. **A value read once, then used later to authorise something destructive**, after something
|
||||
else has had a chance to change it.
|
||||
3. **A failure downgraded to a value that looks like a legitimate result** — `-1`, `null`, an
|
||||
empty list, `false` — which a caller then trusts.
|
||||
4. **A lock held for one half of a read-modify-write and not the other**, or two collections
|
||||
updated under different locks.
|
||||
5. **A comment or javadoc stating an invariant the code no longer keeps.** Comments are
|
||||
load-bearing in this repo; a stale one has already caused a bug.
|
||||
|
||||
## 5. What you cannot check, and must not claim you did
|
||||
|
||||
- `fleetd/fleetd.yaml` is gitignored and **absent from your worktree**. You cannot read it. If a
|
||||
finding depends on live configuration, name the key and say you could not check it.
|
||||
- `.mcp.json`, `opencode.json` and `.autoenv` in your worktree are neutralised stubs, not the
|
||||
repo's real files.
|
||||
- The `wiki/` submodule pointer is months old. Do not cite it.
|
||||
|
||||
Reporting a fact you took from the lead's brief as something you measured yourself is a false
|
||||
report, even when the fact is correct. Say where each fact came from.
|
||||
|
||||
## 6. The report — what goes in `fleet_reply`
|
||||
|
||||
One block per finding, most severe first:
|
||||
|
||||
```
|
||||
FINDING N — <one line>
|
||||
file:line
|
||||
Path in: <which caller, in which state, reaches this>
|
||||
Direction: <data loss | escalation | silent wrong answer | outage | ...>
|
||||
Window/trigger: <when it actually happens>
|
||||
Confidence: <confirmed by reading | unproven — say what you could not check>
|
||||
Why nothing else catches it: <the guard or test you checked, and why it misses>
|
||||
```
|
||||
|
||||
End with one line naming every file you read, so the lead knows the denominator.
|
||||
|
||||
**Nothing clears the bar?** Reply `NO FINDINGS`, name the files you read, and say what you ruled
|
||||
out. A clean sweep is a valid result; an invented defect is worse than none.
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
name: implementer
|
||||
description: Implementer-role procedure for a bridged worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over bridged.
|
||||
description: Implementer-role procedure for a fleetd worker — verify your worktree, implement the scope, commit, push, open your own PR, and hand off the PR URL. Load this when the lead delegates you an implementation task over fleetd.
|
||||
---
|
||||
|
||||
# Implementer worker — procedure
|
||||
@@ -39,6 +39,19 @@ a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
**Never run `git stash` (or `git stash pop`/`apply`/`drop`).** Your worktree is isolated, but the
|
||||
stash is **not**: `refs/stash` is one stack shared by the primary's checkout and every other
|
||||
worker's worktree of this repo. Measured on 2026-09-04 — `git stash list` from a worker's worktree
|
||||
and from the primary's tree returned byte-identical output. So a `git stash` you run can be popped
|
||||
into someone else's tree, and a `git stash pop` you run can drop **another worker's** uncommitted
|
||||
edits on top of yours. This has already happened here: two workers were running in parallel and one
|
||||
of them had its in-progress edit silently overwritten by the other's stash.
|
||||
|
||||
The branch is your isolation, so use it instead. To set work aside, commit it on your own branch
|
||||
(`git commit -m "wip: ..."`) and carry on; to try something and back out, use
|
||||
`git diff > /tmp/<your-branch>.patch` then `git checkout -- <file>`. Both stay inside your worktree.
|
||||
If you find a stash entry you did not create, leave it alone and say so in your report.
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
@@ -49,7 +62,7 @@ test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-t
|
||||
your worktree*:
|
||||
|
||||
```bash
|
||||
cd "$(git rev-parse --show-toplevel)/bridged" && mvn clean install
|
||||
cd "$(git rev-parse --show-toplevel)/fleetd" && mvn clean install
|
||||
echo "exit=$?"
|
||||
```
|
||||
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
---
|
||||
name: redeploy-fleetd
|
||||
description: Rebuild and restart the live fleetd daemon after a merge (lead / primary only). Load this before redeploying — it holds the script, the drain step, the permission grant, and the five checks that have each gone wrong here before. Workers must never do this.
|
||||
---
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk
|
||||
```
|
||||
|
||||
`--no-build` skips the build and restarts whatever jar is at `fleetd/target/fleetd.jar`. Use it only
|
||||
when you just built and nothing changed since. It gives up the protection in the next paragraph: no
|
||||
build runs, so a stale or missing jar is not caught early. The script still checks the file is there
|
||||
and dies with `no jar at … — run without --no-build` if it is not, but it cannot tell you the jar is
|
||||
old. A `mvn clean` in the tree deletes that jar while the daemon keeps running on it, and nothing
|
||||
degrades until the next restart. Run `--check` first: it prints the jar's hash and its modification
|
||||
time, so you can see for yourself whether the jar is missing or older than the code you mean to ship.
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
---
|
||||
name: reviewer
|
||||
description: Reviewer-role procedure for a bridged worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over bridged.
|
||||
description: Reviewer-role procedure for a fleetd worker — how to work a review scope and the exact shape of the finding to report. Load this when the lead delegates you a code review over fleetd.
|
||||
---
|
||||
|
||||
# Reviewer worker — procedure
|
||||
@@ -10,6 +10,11 @@ never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and alrea
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
**Wrong skill for a sweep.** This one reviews *one* diff or scope and reports the *single* most
|
||||
important issue. If the lead asked you to hunt or audit a whole package for several defects, load
|
||||
`hunter` instead and ignore this file — the two want different output, and following both is how a
|
||||
worker ends its turn with a good report that never gets sent.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
|
||||
+16
-10
@@ -14,7 +14,7 @@ jobs:
|
||||
# does not depend on the wiki repo being reachable.
|
||||
- uses: actions/checkout@v4
|
||||
|
||||
# The runner image ships an older default-jdk; bridged sets maven.compiler.release=25, so
|
||||
# The runner image ships an older default-jdk; fleetd sets maven.compiler.release=25, so
|
||||
# provision the JDK explicitly rather than apt-installing whatever "default" means today.
|
||||
- name: Set up JDK 25
|
||||
uses: actions/setup-java@v4
|
||||
@@ -32,7 +32,7 @@ jobs:
|
||||
mvn -version
|
||||
|
||||
- name: Build and test
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
# This IS the mock-socket surface CB-503 asks for: the pom's `default-excludes` profile
|
||||
# already sets excludedGroups=contract, so the @Tag("contract") tests — which need a live
|
||||
# herdr socket and a RabbitMQ container — are excluded without any flag here. Everything
|
||||
@@ -46,7 +46,7 @@ jobs:
|
||||
# the log instead, where they are actually readable.
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
# AmqpReplyInboxContractTest reads AMQP_URI (set below to the service's network alias) and binds
|
||||
# straight to it — no Docker, no skipped tests. This separation (build job hermetic and
|
||||
# Docker-free; contract job broker-provided) is deliberate — see the default-excludes/contract
|
||||
# profiles in bridged/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# profiles in fleetd/pom.xml. `setup-java` provides the JDK only; Maven is installed separately,
|
||||
# exactly as in the build job above.
|
||||
contract:
|
||||
runs-on: ubuntu-latest
|
||||
@@ -87,16 +87,22 @@ jobs:
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
# The `contract` profile clears the default-excludes group, so the @Tag("contract") AMQP test
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
# The `contract` profile clears the default-excludes group, so `-Dgroups=contract` runs every
|
||||
# @Tag("contract") test and nothing from the unit suite the `build` job already covered — a
|
||||
# tag selects the whole group, so a test added to it later runs here automatically. A prior
|
||||
# version of this step pinned `-Dtest=AmqpReplyInboxContractTest` by class name instead: that
|
||||
# silently excluded every other contract test (including the herdr ones) from CI, and nobody
|
||||
# noticed until the herdr protocol drifted out from under a test that never ran here
|
||||
# (fleetd #449). If this runner has no herdr socket, the herdr-backed tests in the group
|
||||
# skip on their own `assumeTrue` and only the broker-backed ones actually run — check the
|
||||
# step output rather than assuming which.
|
||||
- name: Contract tests
|
||||
working-directory: bridged
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
working-directory: fleetd
|
||||
run: mvn -B -Pcontract test -Dgroups=contract
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
working-directory: bridged
|
||||
working-directory: fleetd
|
||||
run: |
|
||||
for f in target/surefire-reports/*.txt; do
|
||||
[ -f "$f" ] || continue
|
||||
|
||||
+10
-4
@@ -14,8 +14,14 @@
|
||||
.env
|
||||
.envrc
|
||||
|
||||
# Daemon runtime artefacts. bridged appends its log wherever it is launched from, so both the
|
||||
# repo root and bridged/ collect one; neither belongs in git.
|
||||
bridged.out
|
||||
bridged/bridged.out
|
||||
# Daemon runtime artefacts. fleetd appends its log wherever it is launched from, so both the
|
||||
# repo root and fleetd/ collect one; neither belongs in git.
|
||||
fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
# fleetd #480: the lead rollover handover file. `leadRollover.handoverPath` points here, and the
|
||||
# outgoing lead rewrites it on every rollover. It is a snapshot of one moment's live state —
|
||||
# unpushed branches, running builds, open questions — so it is stale the moment it is written and
|
||||
# has no business in git history.
|
||||
.handover/
|
||||
|
||||
@@ -3,7 +3,7 @@ description: Refine work into clear, independent units before implementation.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You are an architect in this fleet. You refine work before anyone builds it: scope,
|
||||
acceptance criteria, risks, and a unit split. You read the repo and write analysis.
|
||||
|
||||
@@ -3,7 +3,7 @@ description: Implement one assigned unit, test it, and open a pull request.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You implement the one unit you were given and nothing else. Work in your assigned
|
||||
git worktree and branch. Never check out, rebase onto, or push to `main`. Confirm
|
||||
|
||||
@@ -3,7 +3,7 @@ description: Review one assigned scope and report the most important real issue.
|
||||
mode: primary
|
||||
---
|
||||
|
||||
<!-- CB-617: The model comes from bridged.yaml because the launch flag overrides model here on both backends. -->
|
||||
<!-- CB-617: The model comes from fleetd.yaml because the launch flag overrides model here on both backends. -->
|
||||
|
||||
You review the diff you were given. Report bugs, risks, and missing tests. You do
|
||||
not change code.
|
||||
|
||||
@@ -7,10 +7,18 @@
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
>
|
||||
> **Anything you measure in an addendum is perishable.** Date it, give the command that
|
||||
> re-measures it and what each outcome means, and tell the reader to delete the section once
|
||||
> it stops reproducing. The four parts work together: deciding what would falsify a claim is
|
||||
> the expensive step, and a reader in the middle of another task will not pay it, so a bare
|
||||
> "verify before relying on this" costs the same space and does nothing. The case this is for
|
||||
> is a note that goes stale as a live restriction — it will tell a future session it cannot do
|
||||
> the thing at the moment doing it becomes the job.
|
||||
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
`bridged` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
`fleetd` is the **sole communication gateway** between agents here. The orchestrating session (the
|
||||
**primary**) and every delegated peer (a **member**) mount the *same* MCP server and talk only
|
||||
through its `fleet_*` tools. No session addresses a peer, a broker, or the network directly.
|
||||
|
||||
@@ -50,8 +58,9 @@ and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
5. **Never move a fleet session, pane or peer except through the bridge.** The bridge owns policy;
|
||||
the multiplexer owns PTYs. Any route that changes fleet state without the bridge's checks
|
||||
bypasses every rule above — the `herdr` CLI and its socket are the usual example.
|
||||
|
||||
### Primary (lead) — run this on every task, in order
|
||||
|
||||
@@ -71,8 +80,11 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model, cost and
|
||||
LIVENESS, not in tier, so the default is rarely what you want. The default is whatever the
|
||||
daemon reports, and on a host where it sits on an exhausted or withdrawn credential every
|
||||
unqualified spawn fails — sometimes loudly, sometimes as a member that spawns fine and then
|
||||
produces nothing. `fleet_profiles` reports the default; check it once per session.
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
@@ -94,6 +106,18 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
**If the forge refuses you the merge** — a protected branch, a token without the grant — the
|
||||
adjudication is still yours. Read the diff, decide, and hand the operator a merge-ready queue
|
||||
with the refusal quoted. Never report a PR as merged, and never call one "ready to merge"
|
||||
without having read the diff yourself. A refusal is exactly when that shortcut is tempting,
|
||||
because no action is left that forces you to look, and taking it turns this step into
|
||||
forwarding a reviewer's verdict — which is delegating the merge by proxy, two lines above.
|
||||
**Test a refusal; do not read it off a permissions field.** A protected branch holds its merge
|
||||
rights separately from the repository permissions, so that field can say yes while the merge is
|
||||
refused, and still say no after a grant makes it work. Probe instead, with a request that cannot
|
||||
succeed on its merits, so a rejection can only mean the refusal. Treat a transport failure as a
|
||||
third answer that proves nothing: a timeout, a DNS error or a bad URL is not a refusal, and
|
||||
counting it as one makes you sure of something you never measured.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
@@ -114,9 +138,11 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_reply{content}` — the one case a lead replies |
|
||||
| Answer a peer lead that messaged you | `fleet_send{coordId}` — or `{sessionId}` if they are on this host. **Not** `fleet_reply`: it has no peer route and the publish is refused |
|
||||
| Read your own held lead-to-lead mail (no ack) | `fleet_poll{coordId: <your own coord-id, from fleet_list's coordinator.selfId>}` — primary-only; never acks, so `fleet_list`'s `held[]` still shows it after. `fleet_list`'s `held[]` gives only a truncated preview — this is the only way to read the full body |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
| Replace your OWN lead session when its context is full | `fleet_handover{action:"open", reason?}` → write the handover file it names → `fleet_handover{action:"confirm", token, operatorConfirmed}`. Primary-only. **In that order**: the file must be modified *after* `open`, or `confirm` refuses it as stale. There is no terminal parameter — the pane is always your own, so you can never roll another lead. `{action:"cancel", token}` drops a pending request |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
@@ -140,10 +166,21 @@ The traffic between leads is coordination and nothing else:
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
**N observations are N data points only if they differ in the axis you are trusting.** This cuts
|
||||
both ways. N *failures* blamed on one cause are one data point when the cases share what you are
|
||||
not varying. N *agreeing measurements* are also one data point when they share an instrument —
|
||||
two hosts, two operators and the same formula is one formula, not two confirmations.
|
||||
4. **Ask a peer to read your project addendum.** Your addendum is instruction surface: every future
|
||||
session on your host obeys it, and a wrong one is obeyed just as faithfully as a right one. The
|
||||
author is the worst reader of their own qualifier placement — measured here, one addendum carried
|
||||
two defects and a non-author found both. If you have no peer, at least re-read it asking "which
|
||||
sentence goes false first, and would a reader reach the caveat before acting?"
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
Being messaged by a peer does not make you its worker: answer the way you would open —
|
||||
`fleet_send{coordId}` for another daemon, `fleet_send{sessionId}` on this host — and push back on
|
||||
the substance if it is wrong. `fleet_reply` resolves a member's blocked `fleet_send`; a peer's
|
||||
coord-id message is durable and non-blocking, so there is nothing for it to resolve. A peer that
|
||||
simply complies has thrown away the reason there are two of you.
|
||||
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
@@ -185,78 +222,88 @@ must obey belongs in the charter, not here.
|
||||
|
||||
## Project addendum — claude-bridge (not part of the canonical block)
|
||||
|
||||
- **This repo is the bridge.** The daemon is `bridged`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Herdr socket tests (measured 2026-09-10).** In this repo, herdr is a subject under test. A
|
||||
worker assigned to herdr code, and the lead, may let a test open the herdr socket directly in a
|
||||
throwaway workspace that the test tears down. This only covers
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/AgentControlContractTest.java`,
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/HerdrContractTest.java`,
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/PaneLocatorContractTest.java`, and
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/WorkspacePlacementContractTest.java`. It is not a
|
||||
general licence. Using the herdr CLI or socket to move a real fleet session, pane, or peer stays
|
||||
banned. That is the control plane that invariant 5 protects. Re-measure with
|
||||
`grep -rl 'UnixSocketHerdrClient.connect()' fleetd/src/test/java --include='*.java'`. A non-empty
|
||||
result means tests still open the socket and this note still applies. An empty result means nobody
|
||||
does this any more; delete this section. Canonical invariant 5 restatement is tracked in #458 and
|
||||
is not part of this change.
|
||||
- **`fleet_profiles`/`fleet_list` report two separate outage states, and they are not the same
|
||||
thing.** *Quarantined* (CB-578) means the backend told us it is out of capacity — a long,
|
||||
1800s-default cooldown. *Cooling off* (fleetd #201/#227) means a profile's credential threw two
|
||||
distinct backend errors (a non-exhaustion failure such as an HTTP 5xx) within 60 seconds — a
|
||||
short, fixed 60s cooldown, not configurable per profile. Each check runs independently, so a
|
||||
profile can show both at once. In the JSON: a cooling profile carries `credentialId` and
|
||||
`coolingOffForSeconds`; a quarantined profile carries `quarantinedForSeconds`; a profile hit by
|
||||
both carries all three fields, and either state alone already sets that profile's `free` to `0`.
|
||||
A `fleet_spawn` naming a cooling-off profile is refused before it ever reaches the backend
|
||||
adapter, with a message naming the credential and the remaining seconds ("cooling off after
|
||||
repeated backend errors") — distinct wording from a quarantine refusal, so don't conflate the
|
||||
two when reading a spawn failure.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR),
|
||||
`reviewer` (one diff → one structured finding) and `hunter` (sweep a package → several ranked
|
||||
findings, change nothing). Name exactly one in every delegation. **`reviewer` and `hunter` are
|
||||
not interchangeable** — `reviewer` caps the answer at one finding in about 90 words, so naming
|
||||
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
||||
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
||||
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace).
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace),
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance),
|
||||
`redeploy-fleetd` (rebuild and restart the live daemon after a merge) and
|
||||
`handover` (write the file a fresh lead session inherits when the outgoing one hands off,
|
||||
fleetd #480).
|
||||
- **This repo is also a Claude Code marketplace, and ships a plugin.** `.claude-plugin/marketplace.json`
|
||||
points at `plugin/`, which carries the MCP mount and the `setup` skill
|
||||
(`/claude-bridge:setup` — make any project bridge-ready). It was added in CB-527 and then went
|
||||
unmentioned by every instruction file, so it drifted and a later session planned it from scratch
|
||||
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
||||
structural, not bugs: a plugin cannot carry the role agent files, because
|
||||
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
||||
worktree; and a plugin cannot deliver anything to members at all, because
|
||||
`ClaudeCodeLauncher.java:285` exports `CLAUDE_CONFIG_DIR` and every Claude profile here sets it,
|
||||
so a member never reads the operator's plugin store. **The plugin is the lead-side surface;
|
||||
member-facing assets travel in the worktree.**
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
committed copies would otherwise mount the primary's IDE and forge servers (fleetd #134). The
|
||||
worktree's copy of each is a stub, **not** the repo's real file, so a worker that reads one and
|
||||
reports what it found is reporting on the stub. The daemon logs a per-spawn summary, but the
|
||||
worker cannot see that log. From inside its own worktree a worker — or a lead debugging one —
|
||||
reads the list with `git config --worktree --get-all fleet.neutralizedConfig`, and the
|
||||
consequence with `git config --worktree --get fleet.neutralizedConfigNote`. Never brief a worker
|
||||
to edit one of these files: the edit cannot be committed, and it will not tell you so.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, the turn-done
|
||||
fallback and status gating — are diagrammed in `docs/MCP-Contract.md`. That page is now flows
|
||||
only: its pre-build tool catalogue, parameter tables and REST paths were deleted rather than
|
||||
corrected, because a hand-maintained second copy of the tool surface is what drifted for a month
|
||||
while this line pointed every session at it (CB-609 / #114). **The live MCP schema is the tool
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `bridged` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. The lead **may and
|
||||
should** redeploy rather than hand the job back to the operator. **Workers must never do this** — a
|
||||
worker has no business restarting the daemon it is talking through, and stopping it kills the
|
||||
worker's own channel mid-turn.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-bridged.sh --check # report state, change nothing
|
||||
scripts/redeploy-bridged.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-bridged.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `bridged listening` line at the end of
|
||||
`bridged/bridged.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-bridged.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
**Load the `redeploy-fleetd` skill before you redeploy.** It holds `scripts/redeploy-fleetd.sh`
|
||||
and its flags, the drain step, the operator's permission grant, and the five checks that have each
|
||||
gone wrong here before. Do not hand-roll the steps from memory.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
@@ -277,7 +324,7 @@ Before you call any work done, check the row that matches what you touched:
|
||||
| worktree provisioning or the parity overlay | the "both roles read this file" premise — it rests on the worker's worktree being a checkout of this repo |
|
||||
| `.claude/skills/**` | the addendum's skill list, and the "name the playbook" rule |
|
||||
| a new peer kind (non-Claude adapter) | what that peer can read — anything it must obey belongs in its charter, not in the block |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `bridged.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
| **anything an operator can use, configure, or observe** — an MCP tool, a `fleetd.yaml` knob, an endpoint, a visible behaviour | **[Features](wiki/11-Features.md)** — one entry: what it does · the knob that turns it on · **why it exists** · the gotcha |
|
||||
|
||||
That last row is not bookkeeping. Chapters 1–10 answer *how is this built* and *why this way*;
|
||||
none of them has a home for *what can it do and how do I turn it on*, so for twenty tickets a
|
||||
@@ -304,6 +351,16 @@ print("in sync:", w[i:w.index("\n```\n", i) + 1] == block)
|
||||
PY
|
||||
```
|
||||
|
||||
**Only the lead can run that check (measured 2026-09-10).** A member's provisioned worktree has
|
||||
`wiki/` uninitialized, so the script dies with `FileNotFoundError: wiki/7-Use-Cases.md`. Measured
|
||||
in three worker worktrees: `git submodule status` printed a leading `-` and `wiki/` held 0
|
||||
entries; the primary's own clone printed a leading `+` and the file was there. So never make this
|
||||
check a member's acceptance criterion — it is unsatisfiable for them, and a brief that asks for it
|
||||
is asking a worker to invent a pass. A member told to check it must say it could not run it, and
|
||||
must never report it as passed. The lead runs it in the main clone before merging. Re-measure with
|
||||
`git submodule status` in a member's worktree: a leading `-` means this still applies; once it
|
||||
prints a commit with no `-`, delete this paragraph.
|
||||
|
||||
## IDE MCP tools & validation workflow (enforced)
|
||||
|
||||
> **Primary only.** Workers have no IDE MCP mount — if you are a worker, skip this section and
|
||||
@@ -311,10 +368,10 @@ PY
|
||||
|
||||
Two IDE MCP servers are connected: **intellij-index** (semantic code intelligence) and
|
||||
**jetbrains** (file problems, reformat, debugger). IntelliJ has multiple projects open; our
|
||||
module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
module is **`fleetd`**. Always pass these to IDE MCP tools:
|
||||
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/bridged`
|
||||
- IDE paths are relative to `bridged/` (e.g. `src/main/java/dev/ltms/bridged/...`)
|
||||
- `project_path` = `/Users/dai.ha/LTMS/claude-bridge/fleetd`
|
||||
- IDE paths are relative to `fleetd/` (e.g. `src/main/java/dev/ltms/fleet/...`)
|
||||
|
||||
### After editing any file — mandatory
|
||||
|
||||
@@ -328,7 +385,7 @@ module is **`bridged`**. Always pass these to IDE MCP tools:
|
||||
whole-project gate before declaring work done or committing.
|
||||
|
||||
**Whenever dependencies change (or a `pom.xml` edit), validate CVEs with
|
||||
`jetbrains get_file_problems{filePath: "bridged/pom.xml"}`** — its Mend.io check reflects the
|
||||
`jetbrains get_file_problems{filePath: "fleetd/pom.xml"}`** — its Mend.io check reflects the
|
||||
dependencies on disk. (Note: `ide_diagnostics` / intellij-index does NOT re-resolve dependencies
|
||||
after a pom edit without a full Maven reimport, so it reports stale CVE results — don't trust it
|
||||
for this.) Treat a CVE warning like any other: bump to a patched version and confirm
|
||||
|
||||
@@ -56,9 +56,8 @@ flowchart LR
|
||||
ever addresses a broker, a peer, or the network
|
||||
directly**; any queue is `fleetd`-internal. MCP tool I/O never sets `ANTHROPIC_BASE_URL`, so
|
||||
mounting the bridge is subscription-safe by construction.
|
||||
**Tool naming:** the tools were renamed from `bridge_*` to `fleet_*` (CB-622). The daemon
|
||||
still answers the old `bridge_*` names for one release, but they are deprecated — use the
|
||||
`fleet_*` names.
|
||||
**Tool naming:** the tools are `fleet_*` (renamed from `bridge_*` in CB-622). The old
|
||||
`bridge_*` names were removed in CB-634 — only `fleet_*` answers now.
|
||||
- **How the primary consumes a reply:** a single **blocking MCP call** (`fleet_send`);
|
||||
`fleetd` holds it open until the worker calls `fleet_reply` or its turn hits
|
||||
`agent_status=done`, then returns the reply as the tool result. No cross-turn busy-poll, so
|
||||
|
||||
@@ -1,897 +0,0 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.ConfigWatcher;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.LeadTabScanner;
|
||||
import dev.ltms.fleet.lead.LeadLauncher;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.UnixSocketHerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.CompletionResolver;
|
||||
import dev.ltms.fleet.inject.ExhaustedPatternLookup;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.inject.StatusPoller;
|
||||
import dev.ltms.fleet.inject.TurnListener;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.mcp.ConnectionIdentity;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.health.FleetHealthMonitor;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.mcp.LsofPeerPidLookup;
|
||||
import dev.ltms.fleet.mcp.LsofProcessCwdLookup;
|
||||
import dev.ltms.fleet.msg.AmqpReplyInbox;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadChannel;
|
||||
import dev.ltms.fleet.msg.LeadCoordLoop;
|
||||
import dev.ltms.fleet.msg.LeadMailbox;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.ReplyInbox;
|
||||
import dev.ltms.fleet.msg.LeadHeartbeatLoop;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.rest.FleetApp;
|
||||
import dev.ltms.fleet.session.GitWorktrees;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.session.SessionReaper;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.member.HerdrPeerLauncher;
|
||||
import dev.ltms.fleet.member.OpenCodeLauncher;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import io.javalin.Javalin;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* {@code bridged} entry point. Wires the real herdr socket client to the REST app and
|
||||
* starts listening. Before anything else it asserts its own environment is clean —
|
||||
* {@code bridged} is not a Claude process and must never carry a base_url.
|
||||
*/
|
||||
public final class Fleetd {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(Fleetd.class);
|
||||
|
||||
/** CB-504: how long to wait at startup for herdr's socket before serving degraded. */
|
||||
private static final long HERDR_WAIT_SECONDS = 30;
|
||||
/**
|
||||
* CB-637: how often the lead coordination loop looks for peer messages. A few seconds — slow
|
||||
* enough that an idle fleet is not polling a broker in a tight loop, fast enough that a peer
|
||||
* lead's message is not left sitting once the local lead reaches a turn boundary. The mailbox
|
||||
* pushes into the loop's held set on its own consumer thread, so this interval bounds only the
|
||||
* pane delivery, never the receive.
|
||||
*/
|
||||
private static final long LEAD_COORD_INTERVAL_MS = 3_000L;
|
||||
private static final long HERDR_WAIT_POLL_MILLIS = 500;
|
||||
|
||||
/**
|
||||
* CB-632: prefer {@code fleetd.yaml} in {@code dir}; fall back to {@code bridged.yaml} when
|
||||
* the new name is not there. The operator's live file is still named {@code bridged.yaml},
|
||||
* so the old name keeps working until that file moves.
|
||||
*/
|
||||
static Path chooseDefaultConfigFile(Path dir) {
|
||||
Path fleetd = dir.resolve("fleetd.yaml");
|
||||
if (Files.exists(fleetd)) {
|
||||
return fleetd;
|
||||
}
|
||||
return dir.resolve("bridged.yaml");
|
||||
}
|
||||
|
||||
static void main(String[] args) {
|
||||
Path configPath = args.length > 0 ? Path.of(args[0]) : chooseDefaultConfigFile(Path.of(""));
|
||||
// CB-632: the config file is being renamed bridged.yaml -> fleetd.yaml. Name the file we
|
||||
// actually loaded, whichever of the two names it carries.
|
||||
log.info("Using configuration file {}", configPath);
|
||||
FleetConfig cfg = FleetConfig.load(configPath);
|
||||
// CB-594: report which secret env vars the config actually needs, by name, before anything
|
||||
// else can fail on a silently-empty one. A daemon started without a login shell (launchd)
|
||||
// boots fine either way — this is the only thing that says so out loud.
|
||||
reportRequiredSecrets(cfg);
|
||||
// CB-596: an absent (or empty) memberCredentials: block blocks NOTHING — no credential
|
||||
// name is hardcoded any more to fall back on. Say so loudly, the same way a missing
|
||||
// secret is reported above, so upgrading past this commit never silently drops CB-592's
|
||||
// protection.
|
||||
reportMemberCredentialsGap(cfg);
|
||||
// CB-559: `cfg` stays the startup snapshot — every validation and every piece of one-time
|
||||
// wiring below reads it, and must, because those decisions cannot be unmade. `config` is the
|
||||
// live reference the hot paths read per use. Which keys can actually move is ConfigRef's
|
||||
// contract; adding a reader here does not make a key reloadable by itself.
|
||||
ConfigRef config = new ConfigRef(configPath, cfg);
|
||||
|
||||
// The primary/host env that launched bridged must not be tainted.
|
||||
SubscriptionGuard guard = new SubscriptionGuard(cfg.guard().hostSet());
|
||||
guard.assertPrimaryClean(System.getenv());
|
||||
|
||||
// CB-501: refuse to start if the bind is wider than the auth mode can defend. Under
|
||||
// loopback-trust, "not a known worker" means "the primary" — sound only because the OS
|
||||
// refuses remote connections to a loopback socket. This throws rather than warns so the
|
||||
// dangerous configuration cannot be reached by ignoring a log line.
|
||||
cfg.validateAuthExposure();
|
||||
cfg.validateLeadTabPrefixes();
|
||||
// CB-542: a subscription:true profile whose env: reseats ANTHROPIC_BASE_URL/AUTH_TOKEN would
|
||||
// reach an unguarded endpoint (the launcher skips SubscriptionGuard for it). Refuse at load.
|
||||
cfg.validateSubscriptionProfiles();
|
||||
cfg.validateCharters();
|
||||
// CB-548: every architect slot must name a configured workers: profile — the strong-model
|
||||
// backend the future spawn lifecycle would read. A stale reference dies here, not later.
|
||||
cfg.validateMembers();
|
||||
|
||||
Path socket = cfg.herdrSocket() != null && !cfg.herdrSocket().isBlank()
|
||||
? Path.of(cfg.herdrSocket())
|
||||
: UnixSocketHerdrClient.defaultSocketPath();
|
||||
|
||||
UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect(socket, new com.fasterxml.jackson.databind.ObjectMapper());
|
||||
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
// CB-402: one adapter per configured peer kind, fronted by a composite router. A profile's
|
||||
// `kind:` selects its adapter — claude-code (the default) and opencode partition the profile
|
||||
// set — and the composite dispatches each SPI call to the adapter that owns the profile/pane.
|
||||
Map<String, FleetConfig.Profile> claudeProfiles = new LinkedHashMap<>();
|
||||
Map<String, FleetConfig.Profile> opencodeProfiles = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, w) -> {
|
||||
if (w.isOpenCode()) {
|
||||
opencodeProfiles.put(name, w);
|
||||
} else {
|
||||
claudeProfiles.put(name, w);
|
||||
}
|
||||
});
|
||||
List<HerdrPeerLauncher> adapters = new ArrayList<>();
|
||||
// The claude-code adapter is the always-present default; keep it even with no profiles (so a
|
||||
// bridge configured with no workers, or opencode-only, still has a well-defined base adapter)
|
||||
// unless opencode is the only kind configured.
|
||||
if (!claudeProfiles.isEmpty() || opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new ClaudeCodeLauncher(agents, spaces, guard,
|
||||
claudeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
if (!opencodeProfiles.isEmpty()) {
|
||||
adapters.add(new OpenCodeLauncher(agents, spaces,
|
||||
opencodeProfiles, cfg.effectiveDefaultProfile(), System::getenv,
|
||||
cfg.spawnReadyTimeoutMs(), cfg.spawnReadyPollMs(),
|
||||
() -> config.get().fleet(),
|
||||
() -> config.get().memberCredentials()));
|
||||
}
|
||||
AtomicReference<Function<String, Integer>> liveCountRef = new AtomicReference<>(_ -> 0);
|
||||
// CB-578 stage B: one quarantine tracker for the whole daemon, shared between the launcher
|
||||
// (checked at spawn) and the exhaustion sink wired in below (written on BACKEND_EXHAUSTED).
|
||||
// The cooldown is deferred (see FleetConfig#quarantineCooldownSeconds): it is read once
|
||||
// here, at startup, and a config reload only changes it for a daemon restart.
|
||||
BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));
|
||||
PeerLauncher workers = new CompositePeerLauncher(
|
||||
adapters,
|
||||
cfg.effectiveDefaultProfile(),
|
||||
config,
|
||||
profileName -> liveCountRef.get().apply(profileName),
|
||||
quarantine);
|
||||
// CB-504: under supervision (launchd/systemd) bridged can start before herdr's socket
|
||||
// exists. The client itself is lazy — it connects per call — but the orphan reap below is
|
||||
// the first thing that actually talks to herdr, so without this wait a boot-order race
|
||||
// would crash the daemon into a restart loop. Wait, then degrade rather than die: serving
|
||||
// with /healthz reporting "degraded" is strictly more useful than exiting.
|
||||
boolean herdrUp = awaitHerdr(herdr);
|
||||
if (herdrUp) {
|
||||
// CB-117: herdr keeps worker panes alive across a daemon restart, and their ids died
|
||||
// with the previous process — reap those leaked orphans now, before we start serving.
|
||||
workers.reapOrphanWorkers();
|
||||
} else {
|
||||
log.warn("herdr did not answer within {}s — starting anyway; /healthz will report "
|
||||
+ "degraded until it comes up. Orphaned worker panes (if any) were NOT reaped.",
|
||||
HERDR_WAIT_SECONDS);
|
||||
}
|
||||
|
||||
// CB-301: authoritative session registry + lifecycle FSM on top of ClaudeCodeLauncher.
|
||||
// CB-301-ext: worktree provisioning seam, optionally rooted at a configured directory.
|
||||
// CB-303 part 2: context cap is opt-in and disabled (0) when absent/null.
|
||||
int contextCap = 0;
|
||||
if (cfg.lifecycle() != null && cfg.lifecycle().contextCap() != null
|
||||
&& cfg.lifecycle().contextCap() > 0) {
|
||||
contextCap = cfg.lifecycle().contextCap();
|
||||
}
|
||||
boolean clearAfterTurn = cfg.lifecycle() != null && cfg.lifecycle().clearAfterTurn();
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(cfg.worktreeRoot()),
|
||||
System::nanoTime, contextCap, clearAfterTurn);
|
||||
liveCountRef.set(profileName -> (int) sessions.roster().stream()
|
||||
.filter(s -> profileName.equals(s.profile()))
|
||||
.count());
|
||||
|
||||
// CB-303 part 1: idle-ttl reaper — only when configured, defaults to disabled.
|
||||
final SessionReaper reaper;
|
||||
if (cfg.lifecycle() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() != null
|
||||
&& cfg.lifecycle().idleTtlSeconds() > 0) {
|
||||
reaper = new SessionReaper(sessions, cfg.lifecycle().idleTtlSeconds());
|
||||
reaper.start();
|
||||
} else {
|
||||
reaper = null;
|
||||
}
|
||||
|
||||
// CB-530: every pane the config names as a lead, merged from `leaders:` and the legacy
|
||||
// singular pin. PrimaryRegistry below still tracks ONE terminal — it addresses the push
|
||||
// loop's nudges, which need a single destination — so it keeps the legacy pin.
|
||||
Map<String, String> leadTerminals = cfg.leaderTerminals();
|
||||
if (leadTerminals.size() > 1) {
|
||||
log.info("leads: {} panes recognised {}", leadTerminals.size(), leadTerminals.values());
|
||||
}
|
||||
// CB-531: on top of the legacy primary.terminal pin, discover leads by the tab labels the
|
||||
// operator writes. CB-557 moved the settings onto the lead they describe, so scanning is on
|
||||
// whenever a `fleet.leaders:` entry exists — with no leads configured the supplier is a
|
||||
// constant and never touches herdr, exactly as a missing `leadScan:` block used to behave.
|
||||
// CB-579: each lead now names its own exact `tab:` label, so one scanner discovers every
|
||||
// configured lead regardless of how differently their tabs are labelled — the old
|
||||
// single-shared-tabPrefix limitation (and its warning) is gone.
|
||||
final Supplier<Map<String, String>> leads;
|
||||
var leaders = cfg.fleet().leaders();
|
||||
if (!leaders.isEmpty()) {
|
||||
Set<String> memberSpaces = cfg.profiles().values().stream()
|
||||
.map(FleetConfig.Profile::workspace)
|
||||
.filter(Objects::nonNull)
|
||||
.collect(Collectors.toSet());
|
||||
Map<String, String> tabToName = new LinkedHashMap<>();
|
||||
leaders.forEach((name, leader) -> {
|
||||
if (leader != null && leader.tab() != null && !leader.tab().isBlank()) {
|
||||
tabToName.put(leader.tab(), name);
|
||||
}
|
||||
});
|
||||
// One shared rescan cadence: still taken from the first entry, as before — it is an
|
||||
// operational cadence, not identity, so there is no correctness reason to give every
|
||||
// lead its own scanner.
|
||||
int scanIntervalSeconds = leaders.values().iterator().next().scanIntervalSeconds();
|
||||
leads = new LeadTabScanner(herdr, tabToName, memberSpaces,
|
||||
TimeUnit.SECONDS.toNanos(scanIntervalSeconds), System::nanoTime);
|
||||
log.info("lead scan: tabs {} host a lead (rescan every {}s, member spaces {} excluded)",
|
||||
tabToName.keySet(), scanIntervalSeconds, memberSpaces);
|
||||
} else {
|
||||
leads = () -> leadTerminals;
|
||||
}
|
||||
|
||||
// CB-558: start any declared lead that is not already running. After the scanner is built,
|
||||
// because both read the same tab labels and the ordering makes that dependency visible; and
|
||||
// only when herdr answered, because the launcher's whole safety property is that it can
|
||||
// count live leads first — it must never guess and risk a second orchestrator.
|
||||
if (herdrUp && !leaders.isEmpty()) {
|
||||
int launched = new LeadLauncher(agents, spaces, cfg).ensureLeads();
|
||||
if (launched > 0) {
|
||||
log.info("lead auto-launch: {} lead(s) started", launched);
|
||||
}
|
||||
}
|
||||
|
||||
// CB-548: config-declared architect slots. Config supplies only the stable name → profile
|
||||
// map; the terminal → slot binding is owned by the registry and is empty at startup, so no
|
||||
// pane resolves to an architect until the later spawn lifecycle binds one. The registry is
|
||||
// what CallerResolver resolves against and what that lifecycle will read profiles from;
|
||||
// nothing here spawns a slot.
|
||||
MemberRegistry members = new MemberRegistry(cfg.fleet());
|
||||
sessions.setMemberLifecycle(members);
|
||||
if (!members.slots().isEmpty()) {
|
||||
log.info("member slots: {} configured {} — none bound yet (a slot is idle until the "
|
||||
+ "spawn lifecycle binds a live terminal to it)",
|
||||
members.slots().size(), members.slots().keySet());
|
||||
}
|
||||
|
||||
// Status-gated injector (CB-103): the single writer into workers, fed by a poller.
|
||||
// The blocking message endpoint (CB-104) is the producer; the poller is inert until then.
|
||||
// CB-106: a confirmed turn completion resolves a blocked send whose worker never replied.
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
// CB-578 stage A: classify a completion-fallback scrape that matches a profile's configured
|
||||
// usage-limit refusal as BACKEND_EXHAUSTED rather than handing it back as a real answer.
|
||||
// Compiled once at startup, keyed by profile name; a profile with no exhaustedPattern is
|
||||
// simply absent here, so its workers keep today's completion-fallback behaviour unchanged.
|
||||
Map<String, Pattern> exhaustedPatternsByProfile = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (profile.hasExhaustedPattern()) {
|
||||
exhaustedPatternsByProfile.put(name, Pattern.compile(profile.exhaustedPattern()));
|
||||
}
|
||||
});
|
||||
ExhaustedPatternLookup exhaustedPatterns = target -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(session -> exhaustedPatternsByProfile.get(session.profile()))
|
||||
.orElse(null);
|
||||
log.info("backend-exhausted classification (CB-578 stage A): {}",
|
||||
CompletionResolver.coverage(cfg.profiles().keySet(), exhaustedPatternsByProfile.keySet()));
|
||||
// CB-578 stage B: on a classification that actually wins, quarantine the exhausted profile's
|
||||
// CREDENTIAL — not the profile name — so a profile sharing that credential (e.g. two models
|
||||
// on one OpenAI account) is refused too, not just the one that happened to report it. Reads
|
||||
// the profile config live off `config`, so a credentialId edit is hot: no restart needed.
|
||||
ExhaustionSink exhaustionSink = (target, reason) -> sessions.roster().stream()
|
||||
.filter(session -> target.equals(session.terminalId()))
|
||||
.findFirst()
|
||||
.map(MemberSession::profile)
|
||||
.map(profileName -> config.get().profiles().get(profileName))
|
||||
.ifPresent(profile -> {
|
||||
String credentialId = profile.effectiveCredentialId();
|
||||
quarantine.quarantine(credentialId);
|
||||
log.warn("credential '{}' quarantined for {}s (profile '{}' classified "
|
||||
+ "BACKEND_EXHAUSTED): {}", credentialId,
|
||||
cfg.quarantineCooldownSeconds(), profile.profile(), reason);
|
||||
});
|
||||
CompletionResolver completion = new CompletionResolver(agents, rendezvous, exhaustedPatterns, exhaustionSink);
|
||||
// CB-113: deliver only to an available worker (its MCP is connected), never its boot window.
|
||||
// CB-301: the manager's presence bridge records availability and drives SPAWNING → READY.
|
||||
MemberPresence presence = sessions.asPresence();
|
||||
TurnListener turnListener = new TurnListener() {
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
completion.onTurnComplete(target);
|
||||
sessions.onTurnComplete(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
return sessions.hasPostTurnAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean onTurnCompleteWithPostAction(String target) {
|
||||
completion.resolveBeforePostAction(target);
|
||||
return sessions.onTurnCompleteWithPostAction(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, dev.ltms.fleet.msg.TurnToken token) {
|
||||
completion.onDelivered(target, token);
|
||||
sessions.onDelivered(target, token);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
completion.onTurnFailed(target);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
completion.onTurnFailed(target, reason);
|
||||
sessions.onTurnFailed(target);
|
||||
}
|
||||
};
|
||||
Injector injector = new Injector(agents, turnListener, deliverableTo(presence, leads),
|
||||
presence::forget);
|
||||
StatusPoller poller = new StatusPoller(agents, injector, Injector.POLL_INTERVAL_MILLIS);
|
||||
poller.start();
|
||||
|
||||
// CB-307: reply inbox. A broker: block selects the AMQP-backed durable adapter; absent (or
|
||||
// unusable), bridged stays soft-state on the in-memory inbox. The AMQP inbox owns a broker
|
||||
// connection, so keep the reference to close it in the ordered shutdown hook.
|
||||
final ReplyInbox replyInbox = selectReplyInbox(cfg.broker(), System.getenv(), AmqpReplyInbox::open);
|
||||
// CB-637: this daemon's lead-to-lead mailbox on the SHARED coordination vhost — a separate
|
||||
// broker from the reply inbox by design (see FleetConfig.Coordinator). Absent a coordinator:
|
||||
// block this is null and every lead path below is simply not wired, which is exactly the
|
||||
// behaviour before this ticket. It owns a broker connection, so keep the reference for the
|
||||
// ordered shutdown hook.
|
||||
final LeadMailbox leadMailbox = openLeadMailbox(cfg.coordinator(), System.getenv(), LeadMailbox::open);
|
||||
// CB-307: learn the primary's terminal from orchestration tool calls (or pin from config).
|
||||
// The pin also feeds CallerResolver below: a primary running inside a herdr pane would
|
||||
// otherwise resolve as a worker and be refused every orchestration tool.
|
||||
String pinnedPrimaryTerminal = cfg.primary() != null ? cfg.primary().terminal() : null;
|
||||
PrimaryRegistry primaryRegistry = new PrimaryRegistry(pinnedPrimaryTerminal);
|
||||
// CB-532: `primary.terminal` is superseded and no longer needed for either of its jobs —
|
||||
// identity comes from `leaders:`/`leadScan:`, and reply nudges now follow the delegating
|
||||
// lead. Say so once at startup rather than leaving a redundant pin to look load-bearing.
|
||||
if (pinnedPrimaryTerminal != null && !pinnedPrimaryTerminal.isBlank()) {
|
||||
log.warn("primary.terminal is DEPRECATED (CB-532) and can be deleted: identity now comes "
|
||||
+ "from leaders:/leadScan:, and reply nudges follow the lead that delegated. "
|
||||
+ "It still works, and is still the fallback nudge destination when a restart "
|
||||
+ "has lost the delegation map. Its pushReminders/pushBackoffMs stay valid.");
|
||||
}
|
||||
// CB-307: active push-to-primary loop — nudge the primary when replies land without an
|
||||
// open fleet_send. Uses its own lightweight scheduled executor, separate from the injector.
|
||||
int maxReminders = cfg.primary() != null ? cfg.primary().remindersOrDefault() : 5;
|
||||
long backoffMs = cfg.primary() != null ? cfg.primary().backoffMsOrDefault() : 15_000L;
|
||||
var pushScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-push-").unstarted(r));
|
||||
// CB-502: the registry is built before the service and the push loop so send/reply outcomes
|
||||
// are counted at their single funnel rather than at each of the two caller-facing surfaces.
|
||||
// CB-512: the push loop takes it too, so nudge outcomes (delivered|exhausted) are counted.
|
||||
Metrics metrics = FleetMetrics.create(sessions, replyInbox);
|
||||
var pushLoop = new ReplyPushLoop(primaryRegistry, agents, replyInbox,
|
||||
pushScheduler, maxReminders, backoffMs, metrics);
|
||||
// CB-551: idle-lead heartbeat. Opt-in; absent `leadHeartbeat:` this is never constructed, so
|
||||
// an upgraded daemon cannot silently start spending subscription on nudging an idle lead.
|
||||
// It has its own single-thread scheduler and holds its own scheduler shutdown via close().
|
||||
final LeadHeartbeatLoop heartbeat;
|
||||
var heartbeatScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-heartbeat-").unstarted(r));
|
||||
if (cfg.leadHeartbeat() != null) {
|
||||
var hb = cfg.leadHeartbeat();
|
||||
heartbeat = new LeadHeartbeatLoop(primaryRegistry, agents, replyInbox, sessions::roster,
|
||||
pushLoop, heartbeatScheduler, System::nanoTime,
|
||||
TimeUnit.SECONDS.toNanos(hb.idleAfterSeconds()), hb.backoffMs(), hb.quietNudgeCap(),
|
||||
metrics);
|
||||
heartbeat.start();
|
||||
} else {
|
||||
heartbeat = null;
|
||||
heartbeatScheduler.shutdownNow();
|
||||
}
|
||||
MessageService messages = new MessageService(agents, injector, rendezvous, replyInbox,
|
||||
pushLoop, metrics);
|
||||
|
||||
// Health is a slow whole-fleet observer. Keep it separate from the 250ms delivery poller.
|
||||
final FleetHealthMonitor healthMonitor;
|
||||
var healthScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-health-").unstarted(r));
|
||||
if (cfg.health() != null && cfg.health().isEnabled()) {
|
||||
// CB-580: a member found GONE/NEVER_READY must fail whatever ticket is waiting on it,
|
||||
// through the same idempotent target-wide operation CB-516 already uses on release.
|
||||
healthMonitor = new FleetHealthMonitor(agents, sessions::roster, messages, healthScheduler,
|
||||
System::nanoTime, cfg.health().intervalOrDefault(), messages::abandon);
|
||||
String coverage = FleetHealthMonitor.coverage(true,
|
||||
cfg.health().notifications() != null && cfg.health().notifications().configured());
|
||||
if ("detection-only".equals(coverage)) {
|
||||
log.warn("fleet health: {} (no notification sink configured)", coverage);
|
||||
} else {
|
||||
log.info("fleet health: {}", coverage);
|
||||
}
|
||||
healthMonitor.start();
|
||||
} else {
|
||||
healthMonitor = null;
|
||||
healthScheduler.shutdownNow();
|
||||
}
|
||||
|
||||
// CB-520: the reply inbox only consumes for agents this gateway owns. own on acquire,
|
||||
// release on teardown. Do this before CB-516 so the inbox is owned before any reply can land.
|
||||
sessions.onAcquire(replyInbox::own);
|
||||
// CB-516: releasing a worker must fail whatever send was waiting on it. Without this a
|
||||
// torn-down delegation kept reporting PENDING until the 30-minute async timeout, and never
|
||||
// reached /metrics — the delegation was unresolvable and nothing said so.
|
||||
sessions.onRelease(detail -> {
|
||||
// CB-578 stage C, acceptance criterion 10: a failed ticket's detail should tell a lead
|
||||
// where to re-dispatch onto the same tree, not just that the worker vanished.
|
||||
String reason = "the worker session was released before it replied";
|
||||
if (detail.worktreePath() != null) {
|
||||
reason += "; worktree=" + detail.worktreePath() + " branch=" + detail.branch()
|
||||
+ " snapshot=" + (detail.snapshotRef() != null ? detail.snapshotRef() : "none");
|
||||
}
|
||||
// CB-584 (issue #65 criterion 5): also name the agent session, so a lead can resume the
|
||||
// member's conversation instead of only re-dispatching a fresh one onto the same files.
|
||||
if (detail.agentSessionId() != null) {
|
||||
reason += " agentSessionId=" + detail.agentSessionId();
|
||||
}
|
||||
messages.abandon(detail.terminalId(), reason);
|
||||
replyInbox.release(detail.terminalId());
|
||||
primaryRegistry.forgetDelegation(detail.terminalId()); // CB-532: don't leak the lead binding
|
||||
});
|
||||
|
||||
// MCP server face (CB-105): fleet_send/fleet_reply/fleet_status, mounted at /mcp.
|
||||
// Caller identity is resolved from the connection (peer PID → herdr pane), not arguments.
|
||||
ConnectionIdentity identity = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), new LsofPeerPidLookup(), new LsofProcessCwdLookup());
|
||||
|
||||
// CB-501: one resolver behind both entry paths. Worker identity still comes from the
|
||||
// connection and is never token-gated, so enabling token mode cannot lock the fleet out.
|
||||
final CallerResolver callers;
|
||||
if (cfg.auth().tokenMode()) {
|
||||
String token = System.getenv(cfg.auth().tokenEnv());
|
||||
if (token == null || token.isBlank()) {
|
||||
throw new IllegalStateException("auth.mode=token but env var " + cfg.auth().tokenEnv()
|
||||
+ " is unset or empty — export it before starting bridged");
|
||||
}
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, true, token, leads, members);
|
||||
log.info("auth: token mode (bearer required for non-worker callers, env {})",
|
||||
cfg.auth().tokenEnv());
|
||||
} else {
|
||||
callers = CallerResolver.withLeadsAndMembers(identity, false, null, leads, members);
|
||||
log.info("auth: loopback-trust (any loopback non-worker caller is the primary)");
|
||||
}
|
||||
|
||||
FleetMcp mcp = new FleetMcp(messages, workers, sessions, identity, presence,
|
||||
primaryRegistry, callers, metrics, new FleetMcp.CapacitySource(profile -> liveCountRef.get().apply(profile),
|
||||
profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.maxLoad();
|
||||
}, () -> config.get().profiles().keySet(), System::nanoTime),
|
||||
new FleetMcp.HealthCoverageSource(() -> {
|
||||
var health = config.get().health();
|
||||
return FleetHealthMonitor.coverage(health != null && health.isEnabled(),
|
||||
health != null && health.notifications() != null && health.notifications().configured());
|
||||
}),
|
||||
new FleetMcp.QuarantineSource(profile -> {
|
||||
var configured = config.get().profiles().get(profile);
|
||||
return configured == null ? null : configured.effectiveCredentialId();
|
||||
}, quarantine),
|
||||
leadMailbox);
|
||||
|
||||
// CB-637: the receive half. Only constructed when a lead mailbox actually opened — with no
|
||||
// coordinator (or an unreachable one) there is nothing to deliver, so no scheduler is
|
||||
// created and no thread runs. It reads the SAME live lead supplier the injector's
|
||||
// deliverability gate does, so a lead found by the tab scan after startup is reachable
|
||||
// without a restart.
|
||||
final LeadCoordLoop leadCoordLoop;
|
||||
final ScheduledExecutorService leadCoordSchedulerRef;
|
||||
if (leadMailbox != null) {
|
||||
var leadCoordScheduler = Executors.newSingleThreadScheduledExecutor(r ->
|
||||
Thread.ofVirtual().name("bridge-leadcoord-").unstarted(r));
|
||||
leadCoordLoop = new LeadCoordLoop(leadMailbox, agents, leads, leadCoordScheduler,
|
||||
LEAD_COORD_INTERVAL_MS);
|
||||
leadCoordLoop.start();
|
||||
leadCoordSchedulerRef = leadCoordScheduler;
|
||||
} else {
|
||||
leadCoordLoop = null;
|
||||
leadCoordSchedulerRef = null;
|
||||
}
|
||||
|
||||
// CB-559: opt-in config reload. With no `configReload:` block nothing is constructed, so an
|
||||
// upgraded daemon behaves exactly as before — the file is read once at boot and never again.
|
||||
final ConfigWatcher configWatcher;
|
||||
if (cfg.configReload() != null && cfg.configReload().isEnabled()) {
|
||||
configWatcher = new ConfigWatcher(config, cfg.configReload().intervalSeconds());
|
||||
configWatcher.start();
|
||||
} else {
|
||||
configWatcher = null;
|
||||
}
|
||||
|
||||
// CB-303 part 3: single ordered shutdown hook. Drain sessions first while herdr is still
|
||||
// open (so releases reach the daemon), then stop poller/message/mcp/reaper, and close herdr
|
||||
// last. This replaces the earlier independent hooks that could race and close herdr early.
|
||||
Runtime.getRuntime().addShutdownHook(new Thread(() -> {
|
||||
sessions.close(cfg.lifecycle() != null ? cfg.lifecycle().drainTimeoutSeconds() : null);
|
||||
poller.stop();
|
||||
messages.close();
|
||||
pushLoop.close();
|
||||
if (heartbeat != null) heartbeat.close(); // CB-551: stop the idle-lead heartbeat scheduler
|
||||
if (leadCoordLoop != null) leadCoordLoop.close(); // CB-637: stop delivering peer-lead messages
|
||||
if (leadCoordSchedulerRef != null) leadCoordSchedulerRef.shutdownNow();
|
||||
if (healthMonitor != null) healthMonitor.stop();
|
||||
if (configWatcher != null) configWatcher.stop(); // CB-559: stop polling the config file
|
||||
mcp.close();
|
||||
if (reaper != null) reaper.stop();
|
||||
// Release the broker connection last among message resources (no-op for the in-memory inbox).
|
||||
if (replyInbox instanceof AutoCloseable closeable) {
|
||||
try {
|
||||
closeable.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("reply inbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
// CB-637: the coordination connection goes with it — after the loop that reads it has
|
||||
// stopped, so no tick can be mid-ack against a closed channel.
|
||||
if (leadMailbox != null) {
|
||||
try {
|
||||
leadMailbox.close();
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox close: {}", e.toString());
|
||||
}
|
||||
}
|
||||
herdr.close();
|
||||
}));
|
||||
|
||||
Javalin app = new FleetApp(herdr, workers, sessions, messages, presence, mcp.servlet(),
|
||||
callers, metrics).build();
|
||||
app.start(cfg.bind().host(), cfg.bind().port());
|
||||
log.info("bridged listening on {}:{}, herdr socket {}",
|
||||
cfg.bind().host(), cfg.bind().port(), socket);
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@link Injector}'s readiness gate (CB-534): a target is deliverable if it is a spawned
|
||||
* member whose agent has connected the bridge MCP, <em>or</em> a lead.
|
||||
*
|
||||
* <p>The gate exists for one reason — to hold a delivery out of a <em>spawned</em> member's boot
|
||||
* window, where herdr already reports {@code idle} but the TUI would drop an injected paste. That
|
||||
* hazard is a property of spawning. A lead is never spawned: the operator started it and named it
|
||||
* (or labelled its tab) only once it was up, so there is no boot window to guard.
|
||||
*
|
||||
* <p>A lead is also never enrolled in {@link MemberPresence} — {@code FleetMcp} marks presence
|
||||
* for every spawned member (worker and architect), deliberately, since that map doubles as the
|
||||
* member roster's availability signal and a lead counted there would show up as an available
|
||||
* member. So without the second disjunct a lead is permanently un-deliverable: every
|
||||
* lead→lead send sat on the gate for {@code READINESS_GRACE_POLLS} (~60s) and then failed
|
||||
* having never been typed into the pane.
|
||||
*
|
||||
* <p>The lead set is read through the supplier on each call rather than snapshotted, so a lead
|
||||
* discovered by {@code leadScan} after startup becomes deliverable without a restart.
|
||||
*/
|
||||
static Predicate<String> deliverableTo(MemberPresence presence, Supplier<Map<String, String>> leads) {
|
||||
return target -> presence.isPresent(target) || leads.get().containsKey(target);
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #selectReplyInbox}: production binds {@link AmqpReplyInbox#open}. */
|
||||
@FunctionalInterface
|
||||
interface AmqpOpener {
|
||||
ReplyInbox open(String uri, int prefetch);
|
||||
}
|
||||
|
||||
/** Injection seam for {@link #openLeadMailbox}: production binds {@link LeadMailbox#open}. */
|
||||
@FunctionalInterface
|
||||
interface LeadMailboxOpener {
|
||||
LeadMailbox open(String uri, String selfCoordId, int prefetch);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-637: open this daemon's lead-to-lead mailbox, or return {@code null} to leave the feature
|
||||
* off. Package-private and env-injected for the same reason as {@link #selectReplyInbox}: the
|
||||
* selection is then testable without a broker or a mutable process environment.
|
||||
*
|
||||
* <p>Every "off" path returns {@code null}, and each says why at the level it deserves:
|
||||
*
|
||||
* <ul>
|
||||
* <li>no {@code coordinator:} block — silent. Lead coordination is opt-in; an operator who
|
||||
* never configured it does not need to be told it is off on every boot.</li>
|
||||
* <li>a block whose {@code uriEnv} does not resolve — INFO, the same "you moved to the secret
|
||||
* store and the variable is not there" case {@code selectReplyInbox} warns about.</li>
|
||||
* <li>a configured broker but no {@code selfId} — WARN. This one is a half-finished config: a
|
||||
* mailbox is named after the coord-id that owns it, so with no id there is no queue to own
|
||||
* and no {@code from} to send as. Loud, because the operator plainly intended the feature.</li>
|
||||
* <li>the broker refuses at boot — WARN, and carry on. Mirrors {@code openAmqpOrFallback}: a
|
||||
* coordination broker that is down must never take a whole fleet's daemon with it, and the
|
||||
* fleet still works exactly as it did before this feature existed.</li>
|
||||
* </ul>
|
||||
*/
|
||||
static LeadMailbox openLeadMailbox(FleetConfig.Coordinator coordinator, Map<String, String> env,
|
||||
LeadMailboxOpener opener) {
|
||||
if (coordinator == null) {
|
||||
return null; // opt-in: nothing configured, nothing to say
|
||||
}
|
||||
String uri = coordinator.effectiveUri(env);
|
||||
if (uri == null) {
|
||||
log.info("lead coordination: OFF — coordinator{} has no usable broker uri",
|
||||
coordinator.uriEnv() == null ? "" : ".uriEnv=" + coordinator.uriEnv());
|
||||
return null;
|
||||
}
|
||||
if (coordinator.selfId() == null || coordinator.selfId().isBlank()) {
|
||||
log.warn("coordinator.selfId is unset — lead coordination is OFF. A lead mailbox is the "
|
||||
+ "queue named after the coord-id that owns it, so with no id there is nothing to "
|
||||
+ "own and no sender identity to publish as. Set coordinator.selfId to a name that "
|
||||
+ "is unique across every daemon sharing {} and restart bridged.",
|
||||
stripCredentials(uri));
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
LeadMailbox mailbox = opener.open(uri, coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
log.info("lead coordination: ON as coord-id {} (prefetch={})",
|
||||
coordinator.selfId(), coordinator.prefetchOrDefault());
|
||||
return mailbox;
|
||||
} catch (IllegalStateException e) {
|
||||
log.warn("cannot reach the AMQP coordination broker ({}) — lead-to-lead messaging is OFF "
|
||||
+ "for this process lifetime. fleet_send{{coordId}} will report it as "
|
||||
+ "not configured, and peer messages already queued stay on the broker until "
|
||||
+ "a restart picks them up. Reason: {}",
|
||||
stripCredentials(uri), reasonOf(e));
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-151/152: pick the reply inbox. A usable broker — a literal {@code uri}, or a {@code
|
||||
* uriEnv} whose variable resolves (both read from {@code env}) — selects the durable AMQP inbox.
|
||||
* Everything else falls back to the in-memory inbox: no broker block, a blank {@code uri}, a
|
||||
* {@code uriEnv} whose variable is unset or blank, or a broker unreachable at boot. The two
|
||||
* lossy paths warn <em>loudly</em> — never silently — because what is lost is durable,
|
||||
* cross-restart reply delivery. Package-private and env-injected so the selection is testable
|
||||
* without a real broker or a mutable process environment.
|
||||
*/
|
||||
static ReplyInbox selectReplyInbox(FleetConfig.Broker broker, Map<String, String> env, AmqpOpener amqp) {
|
||||
if (broker == null) {
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
if (broker.hasUriEnv()) {
|
||||
// uriEnv is authoritative whenever set (CB-151): the operator moved off clear text, so
|
||||
// it must not quietly fall back onto a stale literal uri.
|
||||
if (broker.uri() != null && !broker.uri().isBlank()) {
|
||||
log.info("broker.uri is ignored because broker.uriEnv={} is set", broker.uriEnv());
|
||||
}
|
||||
String effectiveUri = broker.effectiveUri(env);
|
||||
if (effectiveUri == null) {
|
||||
log.warn("broker.uriEnv={} is unset or blank — durable AMQP reply inbox DISABLED. "
|
||||
+ "Replies are soft-state and will not survive a restart. Set {} in the "
|
||||
+ "daemon's environment (see scripts/redeploy-bridged.sh) and restart to "
|
||||
+ "use the durable broker inbox.",
|
||||
broker.uriEnv(), broker.uriEnv());
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
log.info("reply inbox: AMQP broker (durable) via env var {} (prefetch={})",
|
||||
broker.uriEnv(), broker.prefetchOrDefault());
|
||||
return openAmqpOrFallback(effectiveUri, broker.prefetchOrDefault(), "uriEnv " + broker.uriEnv(), amqp);
|
||||
}
|
||||
// No uriEnv: the literal uri path (existing behaviour).
|
||||
if (broker.effectiveUri(env) == null) {
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
log.info("reply inbox: AMQP broker (durable) (prefetch={})", broker.prefetchOrDefault());
|
||||
return openAmqpOrFallback(broker.uri(), broker.prefetchOrDefault(), "uri", amqp);
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the AMQP inbox, falling back to the in-memory inbox for this process lifetime if the
|
||||
* broker cannot be reached at boot (CB-152). Not silent: the warning says durable delivery is
|
||||
* off, replies are soft-state and will not survive a restart, plus the source that failed and
|
||||
* the URI <em>with credentials stripped</em>. Never retries in the background — a broker that
|
||||
* drops <em>after</em> startup already self-heals via the connection factory's automatic
|
||||
* recovery; only the boot path is changed here.
|
||||
*/
|
||||
private static ReplyInbox openAmqpOrFallback(String effectiveUri, int prefetch, String source,
|
||||
AmqpOpener amqp) {
|
||||
try {
|
||||
return amqp.open(effectiveUri, prefetch);
|
||||
} catch (IllegalStateException e) {
|
||||
log.warn("cannot reach AMQP broker ({}, {}) — falling back to the in-memory reply inbox "
|
||||
+ "for this process lifetime. Durable, cross-restart reply delivery is OFF; "
|
||||
+ "replies are soft-state and will not survive a restart. Reason: {}",
|
||||
source, stripCredentials(effectiveUri), reasonOf(e));
|
||||
log.info("reply inbox: in-memory (soft-state)");
|
||||
return new InMemoryReplyInbox();
|
||||
}
|
||||
}
|
||||
|
||||
/** An AMQP URI carries {@code user:pass@} inline — show the host/port, never the credentials. */
|
||||
static String stripCredentials(String uri) {
|
||||
return uri == null ? null : uri.replaceAll("://[^@/]*@", "://");
|
||||
}
|
||||
|
||||
/** The deepest cause's class and message — the outermost {@code IllegalStateException} echoes the URI (with password). */
|
||||
private static String reasonOf(Throwable e) {
|
||||
Throwable t = e;
|
||||
while (t.getCause() != null && t.getCause() != t) {
|
||||
t = t.getCause();
|
||||
}
|
||||
String msg = t.getMessage();
|
||||
return t.getClass().getSimpleName() + (msg == null || msg.isBlank() ? "" : ": " + msg);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: which env vars the loaded config actually needs, and why — every non-{@code
|
||||
* subscription} profile's {@code tokenEnv} (a subscription profile never reads one, see
|
||||
* {@link FleetConfig.Profile#isSubscription()}), plus every profile's {@code gitTokenEnv}
|
||||
* where set (opt-in), plus a configured {@code broker.uriEnv} (CB-151). Derived from the
|
||||
* config, not hard-coded, so a new profile is covered for free. A var required by more than one
|
||||
* profile is one entry naming every profile that needs it. Deliberately excludes {@code
|
||||
* auth.tokenEnv}: that one is already enforced loudly, by a startup throw in {@code main()} —
|
||||
* about 370 lines <em>below</em> this method's call site
|
||||
* ({@link #reportRequiredSecrets(FleetConfig)}), not a few lines above it. That throw only
|
||||
* fires when {@code auth.mode: token} is configured; under the default loopback-trust mode it
|
||||
* never runs, and {@code auth.tokenEnv} is simply not required.
|
||||
*
|
||||
* <p>Package-private and pure (no I/O, no logging) so the derivation is unit-testable without
|
||||
* capturing log output; {@link #reportRequiredSecrets(FleetConfig)} is the logging caller.
|
||||
*/
|
||||
static Map<String, List<String>> requiredSecretEnvVars(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = new LinkedHashMap<>();
|
||||
cfg.profiles().forEach((name, profile) -> {
|
||||
if (!profile.isSubscription()) {
|
||||
requiredBy.computeIfAbsent(profile.tokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' tokenEnv");
|
||||
}
|
||||
if (profile.hasGitToken()) {
|
||||
requiredBy.computeIfAbsent(profile.gitTokenEnv(), _ -> new ArrayList<>())
|
||||
.add("profile '" + name + "' gitTokenEnv");
|
||||
}
|
||||
});
|
||||
FleetConfig.Broker broker = cfg.broker();
|
||||
if (broker != null && broker.hasUriEnv()) {
|
||||
requiredBy.computeIfAbsent(broker.uriEnv(), _ -> new ArrayList<>())
|
||||
.add("broker uriEnv");
|
||||
}
|
||||
return requiredBy;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-594: log, by name only, which required env vars (see {@link #requiredSecretEnvVars}) are
|
||||
* set in the daemon's own process environment — the environment every profile's {@code
|
||||
* tokenEnv}/{@code gitTokenEnv} is read from at spawn time (see
|
||||
* {@code HerdrPeerLauncher.resolveEnv}). Never logs a value, a prefix, or a length.
|
||||
*
|
||||
* <p>A missing entry only warns — it must never refuse to start. A daemon that boots and says
|
||||
* what is wrong is strictly more useful than one that will not boot at all.
|
||||
*/
|
||||
private static void reportRequiredSecrets(FleetConfig cfg) {
|
||||
Map<String, List<String>> requiredBy = requiredSecretEnvVars(cfg);
|
||||
if (requiredBy.isEmpty()) {
|
||||
log.info("startup secrets: no profile references a token env var — nothing to check");
|
||||
return;
|
||||
}
|
||||
Map<String, String> env = System.getenv();
|
||||
requiredBy.forEach((varName, sources) -> {
|
||||
String value = env.get(varName);
|
||||
if (value != null && !value.isBlank()) {
|
||||
log.info("startup secret {}: set ({})", varName, String.join(", ", sources));
|
||||
} else {
|
||||
log.warn("startup secret {}: MISSING ({}) — the daemon will start anyway, and this "
|
||||
+ "failure stays invisible until a worker actually needs it. Fix "
|
||||
+ "${SHARED_ENV}/tools/secrets.sh and restart bridged from a LOGIN "
|
||||
+ "shell (see scripts/redeploy-bridged.sh).",
|
||||
varName, String.join(", ", sources));
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: {@code known:} empty (block absent entirely, or present but empty) means {@link
|
||||
* FleetConfig.MemberCredentials#blockedSet()} is empty too — every member pane inherits the
|
||||
* operator's whole secret store, unblocked, exactly the defect this ticket fixes. Unlike a
|
||||
* missing token ({@link #reportRequiredSecrets}), there is no name to point at: the point is
|
||||
* that the block itself is missing. Warn once at startup and say what to add; never refuse to
|
||||
* start over it — see {@link #reportRequiredSecrets} for why a daemon that boots and says
|
||||
* what is wrong beats one that will not boot at all.
|
||||
*
|
||||
* <p>Package-private so the test can capture the log directly, the same way {@link
|
||||
* #requiredSecretEnvVars} is exposed for {@link #reportRequiredSecrets}'s own test.
|
||||
*/
|
||||
static void reportMemberCredentialsGap(FleetConfig cfg) {
|
||||
FleetConfig.MemberCredentials creds = cfg.memberCredentials();
|
||||
if (creds != null && !creds.known().isEmpty()) {
|
||||
log.info("memberCredentials: policy={}, {} known name(s), {} allowed — blocking {} on "
|
||||
+ "every spawn{}",
|
||||
creds.policy(), creds.known().size(), creds.allow().size(), creds.blockedSet().size(),
|
||||
creds.isAllowList()
|
||||
? " (allow-list: known/allow are reporting only — the control is the derived ZDOTDIR scrub)"
|
||||
: "");
|
||||
return;
|
||||
}
|
||||
log.warn("memberCredentials: absent or empty — the daemon will start anyway, and every "
|
||||
+ "member pane inherits the operator's WHOLE secret store, unblocked (CB-592's "
|
||||
+ "protection is lost). Add a memberCredentials: block (policy/allow/known) to "
|
||||
+ "bridged.yaml — see fleetd.example.yaml — and restart.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll herdr's {@code ping} until it answers or {@link #HERDR_WAIT_SECONDS} elapses (CB-504).
|
||||
*
|
||||
* @return true if herdr answered, false if it never did
|
||||
*/
|
||||
private static boolean awaitHerdr(HerdrClient herdr) {
|
||||
long deadline = System.nanoTime() + HERDR_WAIT_SECONDS * 1_000_000_000L;
|
||||
boolean waited = false;
|
||||
while (true) {
|
||||
try {
|
||||
herdr.call("ping");
|
||||
if (waited) {
|
||||
log.info("herdr is up");
|
||||
}
|
||||
return true;
|
||||
} catch (HerdrException e) {
|
||||
if (System.nanoTime() >= deadline) {
|
||||
return false;
|
||||
}
|
||||
if (!waited) {
|
||||
log.info("waiting up to {}s for the herdr socket…", HERDR_WAIT_SECONDS);
|
||||
waited = true;
|
||||
}
|
||||
try {
|
||||
Thread.sleep(HERDR_WAIT_POLL_MILLIS);
|
||||
} catch (InterruptedException ie) {
|
||||
Thread.currentThread().interrupt();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
private Fleetd() {
|
||||
}
|
||||
}
|
||||
@@ -1,21 +0,0 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
}
|
||||
@@ -1,232 +0,0 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
* profile it points at, plus the <em>live</em> bindings from a live architect's herdr terminal to
|
||||
* its slot.
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
*/
|
||||
public final class MemberRegistry implements MemberLifecycle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MemberRegistry.class);
|
||||
|
||||
/**
|
||||
* One flattened {@code fleet:} entry.
|
||||
*
|
||||
* <p>Flattened because a slot name is unique only <em>within</em> its pool — {@code sonnet} may
|
||||
* legitimately be both a developer and a reviewer — while a terminal binds to exactly one thing.
|
||||
* The qualified {@link #key()} is what that binding uses.
|
||||
*
|
||||
* @param name the slot's key inside its pool
|
||||
* @param role the pool it came from
|
||||
* @param profile the backend it runs on
|
||||
*/
|
||||
public record Entry(String name, MemberRole role, String profile) {
|
||||
/** {@code "architect:opus"} — unique across pools, unlike {@link #name()}. */
|
||||
public String key() {
|
||||
return role.wireName() + ":" + name;
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
fleet.pool(role).forEach((name, slot) -> {
|
||||
if (slot != null) {
|
||||
Entry e = new Entry(name, role, slot.profile());
|
||||
flat.put(e.key(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code fleet_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
synchronized (terminalToSlot) {
|
||||
return Map.copyOf(terminalToSlot);
|
||||
}
|
||||
}
|
||||
|
||||
/** The slot a live terminal is bound to, or {@code null} if it is not an architect slot. */
|
||||
public String slotForTerminal(String terminal) {
|
||||
if (terminal == null) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
return terminalToSlot.get(terminal);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under — what the spawn lifecycle reads.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind {@code terminal} to {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it stands a slot up. The bind is atomic and preserves
|
||||
* the two cardinality invariants: a terminal may occupy at most one slot, and a slot may host at
|
||||
* most one terminal. Binding the same terminal to the same slot again is a harmless no-op.
|
||||
*
|
||||
* @param slot a configured slot name, or the bind is refused
|
||||
* @param terminal the pane that will act as this architect
|
||||
* @return {@code true} if the binding is now {@code terminal → slot}; {@code false} if it was
|
||||
* refused — an unknown slot, a terminal already bound to a different slot, or a slot
|
||||
* already hosting a different terminal
|
||||
*/
|
||||
public boolean bind(String slot, String terminal) {
|
||||
if (slot == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!isSlot(slot)) {
|
||||
return false; // unknown slot — nothing to bind to
|
||||
}
|
||||
String existingSlot = terminalToSlot.get(terminal);
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare-safe unbind of {@code expectedTerminal} from {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it tears a slot down. Only the exact binding
|
||||
* {@code expectedTerminal → slot} is removed; if that terminal was since rebound to a different
|
||||
* slot (or the slot to a different terminal), the call is a no-op returning {@code false} — a
|
||||
* stale unbind must never remove a replacement.
|
||||
*
|
||||
* @param slot the slot the caller believes the terminal is bound to
|
||||
* @param expectedTerminal the terminal it expects to be bound there
|
||||
* @return {@code true} if {@code expectedTerminal → slot} was removed; {@code false} if nothing
|
||||
* was (no such binding, or the binding had already moved)
|
||||
*/
|
||||
public boolean unbind(String slot, String expectedTerminal) {
|
||||
if (slot == null || expectedTerminal == null) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
String current = terminalToSlot.get(expectedTerminal);
|
||||
if (current == null || !slot.equals(current)) {
|
||||
return false; // absent, or a replacement/moved binding — leave it in place
|
||||
}
|
||||
terminalToSlot.remove(expectedTerminal);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind only architect sessions to a free slot with the resolved profile.
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
String slot = slotForTerminal(terminal);
|
||||
if (slot != null) {
|
||||
unbind(slot, terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,294 +0,0 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link FleetConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Fleetd.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(FleetConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public FleetConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
/** The file this ref reloads from, or {@code null} for a {@link #fixed} ref. */
|
||||
public Path path() {
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
public String summary() {
|
||||
if (error != null) {
|
||||
return "config reload refused — " + error;
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart bridged to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
}
|
||||
return "config reloaded";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the file, validate it, and swap it in when nothing cold changed.
|
||||
*
|
||||
* <p>Never throws: a reload is a best-effort operation on a daemon that is already serving, and
|
||||
* a bad edit must not take it down. Every failure path leaves the previous config live and is
|
||||
* reported through the returned {@link Outcome}.
|
||||
*/
|
||||
public Outcome reload() {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
FleetConfig old = current.get();
|
||||
FleetConfig fresh;
|
||||
try {
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
return Outcome.failed(msg);
|
||||
}
|
||||
|
||||
List<String> cold = changedColdKeys(old, fresh);
|
||||
if (!cold.isEmpty()) {
|
||||
Outcome out = Outcome.refusedCold(cold);
|
||||
log.warn(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
}
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
if (!Objects.equals(old.auth(), fresh.auth())) {
|
||||
changed.add("auth");
|
||||
}
|
||||
// Kept in step with COLD_KEYS so the doc and the code cannot drift apart silently.
|
||||
assert COLD_KEYS.containsAll(changed) : "a cold key was reported that COLD_KEYS omits";
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
}
|
||||
if (!Objects.equals(old.leadHeartbeat(), fresh.leadHeartbeat())) {
|
||||
changed.add("leadHeartbeat");
|
||||
}
|
||||
if (!Objects.equals(old.guard(), fresh.guard())) {
|
||||
changed.add("guard");
|
||||
}
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, FleetConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, FleetConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
if (!before.keySet().equals(after.keySet())) {
|
||||
Set<String> diff = new LinkedHashSet<>(before.keySet());
|
||||
diff.addAll(after.keySet());
|
||||
diff.removeIf(p -> before.containsKey(p) && after.containsKey(p));
|
||||
changed.add("profiles (added/removed: " + String.join(", ", diff) + ")");
|
||||
}
|
||||
// An EXISTING profile's launch settings are deferred too, and this is easy to get wrong:
|
||||
// `HerdrPeerLauncher` takes `Map.copyOf(profiles)` at construction and `spawn` resolves the
|
||||
// profile out of that snapshot, so a reloaded model/baseUrl/argv/env never reaches a launch.
|
||||
// Only weight and maxLoad are genuinely hot, because placement reads them through the
|
||||
// supplier on the composite rather than from the adapter's copy. Without this check a
|
||||
// changed model would report "config reloaded" and silently do nothing — the worst outcome
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
FleetConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
});
|
||||
if (!relaunch.isEmpty()) {
|
||||
changed.add("profiles." + String.join("/", relaunch) + " launch settings "
|
||||
+ "(model, baseUrl, argv, env, …) — the launcher holds a startup snapshot");
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
&& Objects.equals(a.argv(), b.argv())
|
||||
&& Objects.equals(a.placement(), b.placement())
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
// CB-634: the IDE MCP mount is a launch flag, fixed at spawn like mcpUrl — a
|
||||
// reload changes it only for members spawned after, so a changed value is deferred.
|
||||
&& Objects.equals(a.ideMcpUrl(), b.ideMcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
}
|
||||
}
|
||||
@@ -1,151 +0,0 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final long intervalSeconds;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
|
||||
// These facts need the evidence publishers introduced by later M4 units. They are not negatives.
|
||||
private static final boolean NOT_YET_OBSERVED = false;
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code FleetMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<Agent> agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, NOT_YET_OBSERVED,
|
||||
messages.hasInboxMessage(session.terminalId()), agent != null, NOT_YET_OBSERVED,
|
||||
NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED, NOT_YET_OBSERVED);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
clock.getAsLong());
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// A list failure is health evidence, and must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
@@ -1,59 +0,0 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.Map;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
* identity (CB-105). Given the PID that opened an MCP connection, {@link #terminalForPid} finds
|
||||
* the agent pane whose process tree contains it, so {@code bridged} can tell <em>which worker</em>
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane owns it (e.g. the caller is the primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(paneId, pid)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private boolean paneOwnsPid(String paneId, long pid) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -1,372 +0,0 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* The CB-106 completion fallback: bridges the {@link Injector}'s turn-completion signal to the
|
||||
* {@link Rendezvous} so a blocking {@code fleet_send} resolves even when the worker finishes its
|
||||
* task without ever calling {@code fleet_reply} — the common case for a real delegated coding task.
|
||||
*
|
||||
* <p>On a confirmed {@code working → idle} boundary it scrapes the worker's recent transcript and
|
||||
* resolves the awaiting send with that tail (a {@link Rendezvous.Kind#COMPLETION} resolution, so the
|
||||
* caller can tell a scrape from a structured reply). It scrapes only when a send is actually waiting
|
||||
* — a fleet worker's own turns, or a send that already timed out, cost no herdr traffic. An explicit
|
||||
* {@code fleet_reply} that raced in first wins; {@link Rendezvous#resolveCompletion} is then a no-op.
|
||||
*
|
||||
* <p>It also handles the CB-109 stall signal ({@link #onTurnFailed}): a worker that ran a turn then
|
||||
* wedged in an {@code unknown} state resolves the send as a failure (with the error screen as
|
||||
* context) rather than leaving it to time out.
|
||||
*
|
||||
* <p>The scrape is cleaned to the last {@code ⏺} assistant block (stripping TUI chrome) and guarded
|
||||
* against misattribution (CB-115): the pane content is baselined on delivery ({@link #onDelivered}),
|
||||
* and a completion whose scrape is unchanged from that baseline — the previous turn's wind-down
|
||||
* sampled as this turn's boundary on a rapid back-to-back send — is suppressed rather than resolving
|
||||
* the send with a stale answer.
|
||||
*
|
||||
* <p><strong>Waiter-specific resolution (CB-116).</strong> On delivery we also capture the exact
|
||||
* {@link Rendezvous} waiter this turn belongs to, and the completion/failure fallbacks resolve
|
||||
* <em>that</em> waiter — never "whatever send is waiting now". A completion fallback runs on a virtual
|
||||
* thread and can land after the worker's {@code fleet_reply} already resolved the turn and the
|
||||
* <em>next</em> send opened its own waiter on the same session; resolving the current waiter would
|
||||
* then deliver turn N's stale scrape as turn N+1's answer. Targeting the captured waiter makes a late
|
||||
* completion a harmless no-op (its waiter is already done) instead of a cross-turn stale reply.
|
||||
*
|
||||
* <p>Wired as the {@link Injector}'s {@link TurnListener}; the handlers hand off to a virtual thread
|
||||
* so the scrape's herdr round-trip never stalls the status poller. The captured waiter is read on the
|
||||
* poller thread (before any next-turn delivery can overwrite it) and passed into the virtual thread.
|
||||
*/
|
||||
public final class CompletionResolver implements TurnListener {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompletionResolver.class);
|
||||
|
||||
/**
|
||||
* herdr {@code agent.read} source for the completion scrape. {@code recent} returns the tail of
|
||||
* the transcript (the worker's last output), which is what a delegator wants when the worker
|
||||
* didn't structure a reply.
|
||||
*/
|
||||
static final String SCRAPE_SOURCE = "recent";
|
||||
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call fleet_reply.]";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
* delivering send opened, plus the assistant block present when it was delivered.
|
||||
*
|
||||
* <p>The {@code waiter} is what makes a late fallback safe (CB-116): we resolve it, not "whoever
|
||||
* is waiting now", so a completion that fires after the next send has opened its own waiter is a
|
||||
* no-op rather than a cross-turn stale reply. The {@code baseline} is the CB-115 staleness
|
||||
* reference: a completion scrape equal to it means the worker produced no new output (the previous
|
||||
* turn's wind-down sampled as this boundary), so it is suppressed. Overwritten on each delivery;
|
||||
* cleared when the turn resolves. Package-private so tests can capture and replay a specific turn.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* @param exhaustedPatterns CB-578 stage A: per-target lookup for a profile's configured
|
||||
* usage-limit refusal pattern. Required — there is deliberately no
|
||||
* defaulting overload; a caller that does not want the classification
|
||||
* must pass an explicit inert value ({@link ExhaustedPatternLookup#none()}).
|
||||
* @param exhaustionSink CB-578 stage B: notified when a {@code BACKEND_EXHAUSTED}
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onDelivered(String target, TurnToken token) {
|
||||
// Capture the exact waiter this turn belongs to (CB-116) and snapshot the pane's pre-turn
|
||||
// content — what it shows *before* the just-delivered turn produces output — as the staleness
|
||||
// reference (CB-115). Done synchronously (like the delivering send itself) so both are in
|
||||
// place before this turn's completion can fire.
|
||||
captureBaseline(target, token);
|
||||
}
|
||||
|
||||
/** Capture the in-flight turn: its waiter and pre-turn baseline (the testable core of {@link #onDelivered}). */
|
||||
void captureBaseline(String target, TurnToken token) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = token.waiter();
|
||||
if (waiter == null) {
|
||||
inFlight.remove(target); // no send is waiting on this delivery — nothing to resolve later
|
||||
return;
|
||||
}
|
||||
String baseline;
|
||||
try {
|
||||
// Clip to the same cap resolve() applies to the tail (line ~134): the CB-115 misattribution
|
||||
// guard compares baseline.equals(tail), so both sides must be the same capped representation.
|
||||
// An unclipped baseline vs a clipped tail would never match for a >MAX_SCRAPE_CHARS block,
|
||||
// defeating the guard and letting a stale completion resolve the send.
|
||||
baseline = clip(lastAssistantBlock(agents.read(target, SCRAPE_SOURCE)));
|
||||
} catch (RuntimeException e) {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
InFlight inFlight(String target) {
|
||||
return inFlight.get(target);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnComplete(String target) {
|
||||
// Read the in-flight turn on the poller thread — before any next-turn delivery can overwrite
|
||||
// it — then off-load the scrape (a herdr round-trip we must not block polling on) to a vthread.
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("completion-" + target).start(() -> resolve(target, turn));
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the completed turn before adapter housekeeping can erase its rendered output. This is
|
||||
* intentionally synchronous and used only when a post-turn context reset is enabled; the normal
|
||||
* path remains off-loaded so polling is not blocked by a scrape.
|
||||
*/
|
||||
public void resolveBeforePostAction(String target) {
|
||||
resolve(target, inFlight.get(target));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, null));
|
||||
}
|
||||
|
||||
@Override
|
||||
public void onTurnFailed(String target, String reason) {
|
||||
InFlight turn = inFlight.get(target);
|
||||
Thread.ofVirtual().name("turn-failed-" + target).start(() -> fail(target, turn, reason));
|
||||
}
|
||||
|
||||
/** Synchronous resolve (the unit-testable core of {@link #onTurnComplete}). */
|
||||
void resolve(String target, InFlight turn) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = turn == null ? null : turn.waiter();
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
// Nobody is blocked on THIS turn (it had no send, or its fleet_reply already won). Skip
|
||||
// the scrape; resolving the current waiter here would be the CB-116 cross-turn stale reply.
|
||||
inFlight.remove(target, turn);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
} catch (RuntimeException e) {
|
||||
// The worker finished but we couldn't read its screen — still resolve the send so the
|
||||
// caller unblocks; an empty tail beats hanging until the caller's timeout.
|
||||
log.warn("completion scrape for {} failed; resolving with an empty tail: {}",
|
||||
target, e.getMessage());
|
||||
tail = "";
|
||||
scrapeFailed = true;
|
||||
}
|
||||
// Misattribution guard (CB-115): if the scrape is byte-identical to the pane content at
|
||||
// delivery, this turn produced no new output — the boundary belongs to the previous turn's
|
||||
// wind-down (common on rapid back-to-back sends). Suppress rather than resolve the send with
|
||||
// a stale answer; the real fleet_reply (or a later genuine completion) resolves it instead.
|
||||
// A scrape that failed to read is exempt — an empty tail there is "couldn't see", not "no change".
|
||||
String baseline = turn.baseline();
|
||||
if (!scrapeFailed && baseline != null && baseline.equals(tail)) {
|
||||
log.debug("suppressing misattributed completion for {} (no output change since delivery)",
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
}
|
||||
// CB-578 stage A: a turn that ended with no fleet_reply AND whose scrape matches the
|
||||
// backend's configured usage-limit pattern is a refusal, not an answer. Classify it as
|
||||
// BACKEND_EXHAUSTED rather than handing the caller a scrape that reads like a real reply.
|
||||
if (!scrapeFailed) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(assistantBlock, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED (no fleet_reply; scrape "
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
log.warn("completion scrape for {} clipped from {} chars to the {} char cap; "
|
||||
+ "member did not call fleet_reply, so the pane tail is partial",
|
||||
target, originalLength, MAX_SCRAPE_CHARS);
|
||||
}
|
||||
log.debug("resolved send to {} via turn-completion fallback ({} chars scraped)",
|
||||
target, tail.length());
|
||||
}
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
}
|
||||
|
||||
/** Synchronous fail with an optional reason supplied by a dropped worker queue. */
|
||||
void fail(String target, InFlight turn, String explicitReason) {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fall back to the currently-registered waiter (unambiguous — that send never completed, so
|
||||
// no next turn exists to confuse it with).
|
||||
CompletableFuture<Rendezvous.Resolution> waiter =
|
||||
turn != null ? turn.waiter() : rendezvous.currentWaiter(target);
|
||||
if (waiter == null || waiter.isDone()) {
|
||||
inFlight.remove(target, turn); // nobody blocked on this worker — nothing to fail
|
||||
return;
|
||||
}
|
||||
String reason = explicitReason;
|
||||
if (reason == null || reason.isBlank()) {
|
||||
try {
|
||||
reason = clip(agents.read(target, SCRAPE_SOURCE));
|
||||
} catch (RuntimeException e) {
|
||||
reason = "";
|
||||
}
|
||||
if (reason.isBlank()) {
|
||||
// No screen to scrape — either the worker is stuck (CB-109) or gone (CB-110).
|
||||
reason = "worker did not reply; its turn ended in an unrecoverable state "
|
||||
+ "(worker unreachable or stuck)";
|
||||
}
|
||||
}
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The first line of {@code text} matching {@code pattern}, stripped — the CB-578 stage A
|
||||
* evidence carried in a {@code BACKEND_EXHAUSTED} reason so the operator sees the real refusal
|
||||
* text, never a generic label. {@code null} if no line matches.
|
||||
*/
|
||||
static String firstMatchingLine(String text, Pattern pattern) {
|
||||
if (text == null || text.isEmpty()) return null;
|
||||
for (String line : text.split("\n", -1)) {
|
||||
if (pattern.matcher(line).find()) {
|
||||
return line.strip();
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* the way {@link dev.ltms.fleet.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
return unconfigured.isEmpty()
|
||||
? "full (all profiles configured: " + sorted(allProfiles) + ")"
|
||||
: "partial (configured: " + sorted(configuredProfiles) + "; not configured: " + sorted(unconfigured) + ")";
|
||||
}
|
||||
|
||||
private static List<String> sorted(Set<String> names) {
|
||||
return names.stream().sorted().toList();
|
||||
}
|
||||
|
||||
private static String clip(String s) {
|
||||
if (s == null) return "";
|
||||
String trimmed = s.strip();
|
||||
return trimmed.length() <= MAX_SCRAPE_CHARS
|
||||
? trimmed
|
||||
: trimmed.substring(trimmed.length() - MAX_SCRAPE_CHARS);
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract the last assistant message from a raw Claude Code pane scrape (CB-115). Claude Code
|
||||
* prefixes each assistant turn with {@code ⏺}; the delegator wants that answer, not the TUI
|
||||
* chrome around it. Take everything from the final {@code ⏺} onward and stop at the <em>first</em>
|
||||
* hard interface boundary below it — the spinner/status line, input box, {@code ❯} prompt (which
|
||||
* may echo the <em>next</em> turn's text), footer, or tips/warnings. Stopping at the first
|
||||
* boundary (rather than trimming only trailing chrome) is what keeps a following turn's echoed
|
||||
* prompt out of this reply. Blank lines are not boundaries, so a multi-paragraph answer survives;
|
||||
* trailing blanks are trimmed at the end. With no {@code ⏺} marker (an unusual render) the whole
|
||||
* text is scanned the same way, so we never lose the reply.
|
||||
*
|
||||
* <p>Package-private and pure so it is unit-testable without herdr.
|
||||
*/
|
||||
static String lastAssistantBlock(String raw) {
|
||||
if (raw == null || raw.isBlank()) return "";
|
||||
int marker = raw.lastIndexOf('⏺');
|
||||
String block = marker >= 0 ? raw.substring(marker + 1) : raw;
|
||||
StringBuilder out = new StringBuilder();
|
||||
int kept = 0;
|
||||
for (String line : block.split("\n", -1)) {
|
||||
if (isBoundary(line)) break; // first TUI boundary ends the assistant message
|
||||
if (kept++ > 0) out.append('\n');
|
||||
out.append(line);
|
||||
}
|
||||
return out.toString().strip();
|
||||
}
|
||||
|
||||
/**
|
||||
* A hard TUI boundary line that marks the end of an assistant message and the start of interface
|
||||
* chrome (input box, prompt, spinner, footer, tips/warnings). Blank lines are <em>not</em>
|
||||
* boundaries — an answer may contain them — so they are kept and trimmed only if trailing.
|
||||
*/
|
||||
private static boolean isBoundary(String line) {
|
||||
String t = line.strip();
|
||||
if (t.isEmpty()) return false;
|
||||
// A horizontal rule / all box-drawing separators (e.g. "──────").
|
||||
if (t.chars().allMatch(c -> c == '─' || c == '—' || c == '━' || c == '═' || c == '-')) {
|
||||
return true;
|
||||
}
|
||||
String lower = t.toLowerCase();
|
||||
return t.startsWith("╭") || t.startsWith("│") || t.startsWith("╰") || t.startsWith("┌")
|
||||
|| t.startsWith("└") || t.startsWith("❯") || t.startsWith("⏵")
|
||||
|| t.startsWith("⎿") || t.startsWith("⚠")
|
||||
// Status/spinner lines Claude Code renders below a settled or in-flight turn,
|
||||
// e.g. "✻ Baked for 21s", "✶ Forming…".
|
||||
|| t.startsWith("✻") || t.startsWith("✳") || t.startsWith("✽") || t.startsWith("·")
|
||||
|| t.startsWith("●") || t.startsWith("◐") || t.startsWith("✢") || t.startsWith("✶")
|
||||
|| lower.contains("auto mode") || lower.contains("for shortcuts")
|
||||
|| lower.contains("esc to interrupt") || lower.contains("bypass permissions");
|
||||
}
|
||||
}
|
||||
@@ -1,31 +0,0 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a {@code BACKEND_EXHAUSTED}
|
||||
* classification to a waiting send (CB-578 stage B) — never on a race that lost (see
|
||||
* {@link CompletionResolver#resolve}, which only calls this after
|
||||
* {@code Rendezvous.resolveExhausted} returns {@code true}).
|
||||
*
|
||||
* <p>{@link CompletionResolver} knows only {@code target} (a herdr terminal id); it has no notion of
|
||||
* profiles or credentials, so mapping {@code target} to whatever should be quarantined is entirely
|
||||
* the sink's job — see {@code Fleetd.main}'s wiring, which resolves target → session → profile →
|
||||
* {@code effectiveCredentialId()} and calls {@code BackendQuarantine.quarantine} on it.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustionSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
*/
|
||||
void onExhausted(String target, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on exhaustion. The explicit stand-in a caller (or a test not
|
||||
* exercising this feature) passes instead of a defaulting overload, exactly like
|
||||
* {@link ExhaustedPatternLookup#none()}.
|
||||
*/
|
||||
static ExhaustionSink none() {
|
||||
return (target, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -1,66 +0,0 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
|
||||
/**
|
||||
* Resolves <em>who is calling</em> an MCP tool from the connection alone — the anti-spoofing
|
||||
* identity model of the MCP contract. It ties the connection's loopback peer PID (from the OS)
|
||||
* to a herdr agent pane (from herdr), yielding the caller's worker {@code terminal_id}. A caller
|
||||
* that maps to no worker pane — the primary, or an off-host client — resolves to {@code null}.
|
||||
*
|
||||
* <p>Both sources are authoritative and unforgeable: the OS reports the real connecting PID, and
|
||||
* herdr owns the PID→pane mapping. A worker cannot claim to be another worker, nor the primary.
|
||||
* Single-host only (the herd shares the {@code bridged} host); the token path is the split-host
|
||||
* fallback.
|
||||
*/
|
||||
public final class ConnectionIdentity {
|
||||
|
||||
private final PaneLocator panes;
|
||||
private final PeerPidLookup pids;
|
||||
private final ProcessCwdLookup cwds;
|
||||
|
||||
/** Identity only (no cwd resolution — {@link #cwdForPid} returns {@code null}). */
|
||||
public ConnectionIdentity(PaneLocator panes, PeerPidLookup pids) {
|
||||
this(panes, pids, _ -> null);
|
||||
}
|
||||
|
||||
/** Identity plus cwd resolution (CB-112 — inherit the primary's directory on spawn). */
|
||||
public ConnectionIdentity(PaneLocator panes, PeerPidLookup pids, ProcessCwdLookup cwds) {
|
||||
this.panes = panes;
|
||||
this.pids = pids;
|
||||
this.cwds = cwds;
|
||||
}
|
||||
|
||||
/**
|
||||
* The caller resolved from the connection: its worker {@code terminal} (or {@code null} for the
|
||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
||||
*/
|
||||
public record Caller(String terminal, long pid) {
|
||||
}
|
||||
|
||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||
public Caller resolve(String remoteAddr, int remotePort) {
|
||||
if (!isLoopback(remoteAddr)) {
|
||||
return new Caller(null, -1); // only same-host callers can be workers
|
||||
}
|
||||
long pid = pids.pidForLocalPort(remotePort);
|
||||
return new Caller(panes.terminalForPid(pid), pid);
|
||||
}
|
||||
|
||||
/**
|
||||
* The calling worker's {@code terminal_id}, or {@code null} if the caller is not a known
|
||||
* on-host worker (treat as the primary).
|
||||
*/
|
||||
public String callerTerminal(String remoteAddr, int remotePort) {
|
||||
return resolve(remoteAddr, remotePort).terminal();
|
||||
}
|
||||
|
||||
/** The working directory of {@code pid} (the primary's cwd on an MCP spawn), or {@code null}. */
|
||||
public String cwdForPid(long pid) {
|
||||
return pid > 0 ? cwds.cwdForPid(pid) : null;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String addr) {
|
||||
return "127.0.0.1".equals(addr) || "::1".equals(addr) || "0:0:0:0:0:0:0:1".equals(addr);
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,546 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.UUID;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>Claude Code</strong> — the safe path from a
|
||||
* delegation request to a running off-subscription Claude.
|
||||
*
|
||||
* <p>Everything transport-related (tab/pane placement, the CB-306 spawn-readiness gate, unique
|
||||
* naming, CB-117 orphan reap, teardown, listing, cwd resolution) lives in the base. This class
|
||||
* supplies only the two Claude-specific seams:
|
||||
* <ul>
|
||||
* <li>the {@code claude} name prefix (so reap matches {@code claude-*} panes, never another
|
||||
* adapter's), and</li>
|
||||
* <li>{@link #buildLaunch}, which encodes the subscription boundary: build the worker env with
|
||||
* {@code ANTHROPIC_BASE_URL}, assert that host is on the allowlist <em>before</em> touching
|
||||
* herdr, and mount the bridge MCP + reply charter as inline launch flags. A worker's base_url
|
||||
* lives in the env map handed to herdr and nowhere else; {@code bridged}'s own environment is
|
||||
* never mutated, and nothing is written to the worker's profile.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "claude";
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so
|
||||
* existing deployments and tests keep the legacy non-blocking spawn semantics.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300));
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with spawn-ready gate enabled. The gate polls {@code agents.status()}
|
||||
* until the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet, memberCredentials);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply
|
||||
* fakes for the clock ({@code nowMillis}) and poll-loop wait ({@code sleeper}). The
|
||||
* {@code sleeper} is never called when the gate is disabled ({@code spawnReadyTimeoutMs == 0}).
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param guard subscription-boundary guard (checked before spawning)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (e.g. {@code () -> Thread.sleep(pollMs)}); it
|
||||
* already encodes the poll interval, so the 8th positional argument
|
||||
* (poll ms) is accepted for API symmetry but otherwise unused here
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<FleetConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus an injectable host-env-names source for the CB-596
|
||||
* criterion-4 gap detector. Test seam only — every production call site leaves this at the
|
||||
* default (the real {@code System.getenv()} key set) via the constructor above.
|
||||
*/
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>The spawn sequence encodes the subscription boundary: assert the profile's base_url is on
|
||||
* the allowlist <em>before</em> any herdr call, then build the worker env with
|
||||
* {@code ANTHROPIC_*}, the parity-neutral git-forge grant, and the bridge MCP + reply charter
|
||||
* mounted as inline launch flags. When the request carries session identity (CB-547a) it is
|
||||
* applied here — see {@link #applySessionIdentity}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
// CB-539: a profile may deliberately opt into the subscription (subscription: true) when no
|
||||
// off-subscription endpoint exists for it — e.g. `sonnet` on `ccs`. That profile gets no
|
||||
// ANTHROPIC_BASE_URL/AUTH_TOKEN (there is nothing to point them at) and the guard's base_url
|
||||
// requirement is skipped FOR IT ONLY. Every other profile keeps the hard boundary below.
|
||||
boolean onSubscription = cfg.isSubscription();
|
||||
String baseUrl = cfg.baseUrl();
|
||||
|
||||
if (onSubscription) {
|
||||
// NO SILENT CONTRADICTION: subscription:true + a baseUrl state opposite intents; refuse
|
||||
// loudly rather than pick a winner.
|
||||
if (baseUrl != null && !baseUrl.isBlank()) {
|
||||
throw new IllegalStateException("profile '" + cfg.profile()
|
||||
+ "' sets both subscription: true and a baseUrl ('" + baseUrl + "') — the two "
|
||||
+ "are contradictory: a subscription profile must not point at an endpoint. "
|
||||
+ "Drop baseUrl, or drop subscription: true.");
|
||||
}
|
||||
// Visible without anyone going looking for it: this worker bills the subscription.
|
||||
log.warn("spawning profile '{}' on the Claude subscription (subscription: true) — this "
|
||||
+ "worker WILL bill the operator's subscription", cfg.profile());
|
||||
} else {
|
||||
guard.assertWorker(baseUrl); // hard stop before we spawn anything
|
||||
}
|
||||
|
||||
// CB-634: the IDE guidance is delivered as an on-disk CLAUDE.local.md overlay, NOT through
|
||||
// the charter — the charter returns to role -> reply only. Best-effort: a failed overlay
|
||||
// must never fail the spawn, and `writeIdeOverlay` no-ops unless the cwd is a provisioned
|
||||
// worktree (see its .git-file safety gate). The overlay pins, and the auto-open opens, the
|
||||
// module dir (this repo's pom is in `bridged/`, not at the worktree root) — see ideProjectPath.
|
||||
if (cfg.hasIdeMcp()) {
|
||||
String projectPath = PeerLauncher.ideProjectPath(spec.cwd(), cfg.ideProjectDir());
|
||||
writeIdeOverlay(spec.cwd(), projectPath);
|
||||
PeerLauncher.openInIde(projectPath, cfg.ideOpenCommand(), log);
|
||||
}
|
||||
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
if (onSubscription) {
|
||||
// CB-542 belt-and-braces: on the subscription path no guard vets these two keys, and the
|
||||
// profile's env: is layered in by baseEnv — so strip any that rode in there. Config load
|
||||
// already rejects this (loudly, naming the profile); this makes the boundary hold even
|
||||
// for a profile built in code that never passed through that validation.
|
||||
workerEnv.remove("ANTHROPIC_BASE_URL");
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so bridged chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
// handle is known BEFORE the agent has written anything; a resume spawn adopts its prior
|
||||
// id via -r and passes no --session-id (the two conflict). Both are injected before the
|
||||
// model flag so --model keeps outranking the operator's own argv.
|
||||
// mutableArgv: argvWithFleet may hand back the profile's own (immutable) List.of when it
|
||||
// has neither MCP nor a charter — session flags must be added into a list we own.
|
||||
List<String> argv = mutableArgv(argvWithFleet(cfg, spec));
|
||||
String agentSessionId = applySessionIdentity(argv, spec.sessionName(), spec.resumeSessionId());
|
||||
return new Launch(workerEnv, argvWithAutoCompact(argvWithModel(argv, cfg), cfg), agentSessionId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
}
|
||||
if (resuming) {
|
||||
argv.add("-r");
|
||||
argv.add(resumeSessionId);
|
||||
return resumeSessionId;
|
||||
}
|
||||
String minted = UUID.randomUUID().toString();
|
||||
argv.add("--session-id");
|
||||
argv.add(minted);
|
||||
return minted;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv, plus an inline {@code --mcp-config} when {@code worker.mcpUrl} is set, the
|
||||
* CB-617 charter flags, and {@code --agent <role>} when the role has an agent-definition file
|
||||
* under the worker's cwd. Neither touches the profile's config; all are pure command-line flags.
|
||||
* This inline-flag mount is Claude Code specific — other adapters mount MCP and instructions
|
||||
* their own way.
|
||||
*
|
||||
* <p>CB-617: the role charter is operator-authored and often multi-line, so it can never be a
|
||||
* single inline argv element — herdr refuses to shell-encode a multi-line argument
|
||||
* ({@code invalid_agent_argument}). It is written to a temp file instead and mounted with
|
||||
* {@code --append-system-prompt-file}, which this host confirms Claude Code accepts for a
|
||||
* multi-line file.
|
||||
*
|
||||
* <p>CB-618: Claude Code refuses to start when BOTH {@code --append-system-prompt} and
|
||||
* {@code --append-system-prompt-file} are on the command line ("Cannot use both ... Please use
|
||||
* only one"), so the two charters can never travel on separate flags. When both are present they
|
||||
* are concatenated into the one file, role charter first and reply charter last — last is where
|
||||
* the reply rule must sit, because it is the rule that must survive. When only the reply charter
|
||||
* is present it keeps its proven inline {@code --append-system-prompt} delivery, which is also
|
||||
* the only form that reaches a member with no repo checkout.
|
||||
*/
|
||||
private List<String> argvWithFleet(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
String roleCharter = nonBlank(spec.roleCharter());
|
||||
// CB-634: the IDE guidance is delivered as an on-disk overlay (writeIdeOverlay), not through
|
||||
// the charter. The charter file is role -> reply only.
|
||||
String replyCharter = nonBlank(spec.replyCharter());
|
||||
Path agentFile = agentDefinitionFile(spec.cwd(), spec.role(), ".claude", "agents");
|
||||
if (!cfg.mountsAnyMcp() && roleCharter == null
|
||||
&& replyCharter == null && agentFile == null) {
|
||||
return cfg.argv();
|
||||
}
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
if (cfg.mountsAnyMcp()) {
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpConfigJson(cfg));
|
||||
}
|
||||
// Combine the charters in order role -> reply, dropping any that are absent. When two
|
||||
// or more survive they must ride one --append-system-prompt-file (CB-618 forbids the inline
|
||||
// flag and the file flag together). A lone reply charter keeps its proven inline delivery.
|
||||
List<String> charters = new java.util.ArrayList<>(2);
|
||||
if (roleCharter != null) charters.add(roleCharter);
|
||||
if (replyCharter != null) charters.add(replyCharter);
|
||||
if (charters.size() == 1 && replyCharter != null && roleCharter == null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(replyCharter);
|
||||
} else if (!charters.isEmpty()) {
|
||||
argv.add("--append-system-prompt-file");
|
||||
argv.add(writeCharterFile(String.join("\n\n", charters)).toString());
|
||||
}
|
||||
if (agentFile != null) {
|
||||
argv.add("--agent");
|
||||
argv.add(spec.role().wireName());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code --mcp-config} JSON for this member: always the bridge mount when {@link
|
||||
* FleetConfig.Profile#hasMcp()}, plus the IDE Index MCP as a second server named {@code
|
||||
* intellij} when {@link FleetConfig.Profile#hasIdeMcp()} (CB-634). At least one is present —
|
||||
* the caller only reaches here when {@link FleetConfig.Profile#mountsAnyMcp()} is true.
|
||||
*/
|
||||
private static String mcpConfigJson(FleetConfig.Profile cfg) {
|
||||
StringBuilder servers = new StringBuilder();
|
||||
if (cfg.hasMcp()) {
|
||||
servers.append('"').append(PeerLauncher.MCP_MOUNT_NAME)
|
||||
.append("\":{\"type\":\"http\",\"url\":\"").append(cfg.mcpUrl()).append("\"}");
|
||||
}
|
||||
if (cfg.hasIdeMcp()) {
|
||||
if (servers.length() > 0) servers.append(',');
|
||||
servers.append("\"intellij\":{\"type\":\"http\",\"url\":\"")
|
||||
.append(cfg.ideMcpUrl()).append("\"}");
|
||||
}
|
||||
return "{\"mcpServers\":{" + servers + "}}";
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver the shared IDE guidance ({@link PeerLauncher#ideOverlayText}) as an on-disk
|
||||
* {@code CLAUDE.local.md} overlay beside the project's own {@code CLAUDE.md} (CB-634), and
|
||||
* register the overlay in the repository's common {@code info/exclude} so it never shows as
|
||||
* untracked (git reads a worktree's excludes from the common dir, not the per-worktree gitdir).
|
||||
*
|
||||
* <p><strong>Safety gate:</strong> the overlay is written ONLY when {@code cwd/.git} is a
|
||||
* <em>regular file</em> — a provisioned worktree keeps a {@code .git} FILE holding a
|
||||
* {@code gitdir: <path>} pointer, while the primary's real checkout has a {@code .git}
|
||||
* DIRECTORY. Returning without writing when {@code .git} is a directory is the whole safety of
|
||||
* the feature: it must never write into a non-worktree cwd, i.e. never clobber a project that
|
||||
* does not want the overlay.
|
||||
*
|
||||
* <p>Best-effort: a failure is logged at debug and swallowed — a failed overlay must never fail
|
||||
* the spawn.
|
||||
*
|
||||
* @param cwd the member's worktree root, where the {@code CLAUDE.local.md} file is written
|
||||
* @param projectPath the module dir the overlay pins {@code project_path} to (see
|
||||
* {@link PeerLauncher#ideProjectPath}); equals {@code cwd} when no module subdir
|
||||
*/
|
||||
private static void writeIdeOverlay(String cwd, String projectPath) {
|
||||
try {
|
||||
Path dotGit = Path.of(cwd, ".git");
|
||||
if (!Files.isRegularFile(dotGit)) {
|
||||
// Not a provisioned worktree (primary's real checkout has a .git directory, or the
|
||||
// cwd is not a repo at all). Never write into it.
|
||||
return;
|
||||
}
|
||||
// The overlay FILE lives at the worktree root (claude-code's cwd), but its CONTENT pins
|
||||
// project_path to the module dir the IDE opened (projectPath), not the worktree root.
|
||||
Files.writeString(Path.of(cwd, "CLAUDE.local.md"), PeerLauncher.ideOverlayText(projectPath));
|
||||
String gitdirLine = Files.readString(dotGit).trim();
|
||||
Path gitDir = Path.of(gitdirLine.replaceFirst("^gitdir:\\s*", ""));
|
||||
if (!gitDir.isAbsolute()) {
|
||||
gitDir = Path.of(cwd).resolve(gitDir).normalize();
|
||||
}
|
||||
// git reads info/exclude from the COMMON dir, never the per-worktree gitdir (only
|
||||
// info/sparse-checkout is per-worktree). A provisioned worktree's gitdir is
|
||||
// <common>/worktrees/<name>, so the common dir is two levels up; writing the entry into
|
||||
// the per-worktree gitdir leaves it un-honoured and the overlay shows as untracked.
|
||||
Path commonDir = gitDir;
|
||||
if (gitDir.getParent() != null && gitDir.getParent().getFileName() != null
|
||||
&& "worktrees".equals(gitDir.getParent().getFileName().toString())) {
|
||||
commonDir = gitDir.getParent().getParent();
|
||||
}
|
||||
Path exclude = commonDir.resolve("info").resolve("exclude");
|
||||
Files.createDirectories(exclude.getParent());
|
||||
String overlayLine = "CLAUDE.local.md";
|
||||
if (!Files.exists(exclude) || Files.readAllLines(exclude).stream().noneMatch(overlayLine::equals)) {
|
||||
Files.writeString(exclude, (Files.exists(exclude) ? System.lineSeparator() : "")
|
||||
+ overlayLine + System.lineSeparator());
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("cannot write IDE overlay into worktree '{}'", cwd, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** {@code s}, or {@code null} when {@code s} is null/blank — the charter-presence test used above. */
|
||||
private static String nonBlank(String s) {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write the role charter to a fresh temp file so it can be mounted with
|
||||
* {@code --append-system-prompt-file} instead of riding inline in argv (CB-617). Best-effort
|
||||
* cleaned via {@code deleteOnExit} — the same disposable-worker-config cleanup
|
||||
* {@link OpenCodeLauncher#writeConfig} already uses for its charter file, since the process that
|
||||
* reads this file (the spawned peer) outlives this JVM call and there is no spawn-scoped teardown
|
||||
* hook to delete it synchronously.
|
||||
*/
|
||||
private static Path writeCharterFile(String charterText) {
|
||||
try {
|
||||
Path file = Files.createTempFile("bridged-role-charter-", ".md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
return file;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot write role charter temp file", e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin the model on the command line as well as in {@code ANTHROPIC_MODEL} (CB-533).
|
||||
*
|
||||
* <p>The env var alone is not a reliable pin for this adapter, because the argv is usually a
|
||||
* launcher rather than {@code claude} itself — {@code ["ccs", "<profile>"]} — and {@code ccs}
|
||||
* exports its profile's own model family ({@code ANTHROPIC_MODEL}, {@code DEFAULT_OPUS/SONNET/
|
||||
* HAIKU}, {@code CLAUDE_CODE_SUBAGENT_MODEL}) over whatever it inherited. A worker profile that
|
||||
* set {@code model:} therefore got silently overruled by its own launcher. Claude Code's
|
||||
* {@code --model} flag outranks the environment, and {@code ccs <profile> [claude-args...]}
|
||||
* passes trailing arguments through, so the flag survives the wrapper.
|
||||
*
|
||||
* <p>Appended last so it also outranks anything in the operator's own {@code argv}. Profiles
|
||||
* that deliberately leave {@code model:} unset (letting {@code ccs} own model selection, as
|
||||
* {@code gx10} does) are untouched — this adds nothing when there is nothing to add. This is
|
||||
* the {@code kind: claude} counterpart of the opencode adapter's {@code -m provider/model}.
|
||||
*/
|
||||
private static List<String> argvWithModel(List<String> argv, FleetConfig.Profile cfg) {
|
||||
if (cfg.model() == null || cfg.model().isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withModel = mutableArgv(argv);
|
||||
withModel.add("--model");
|
||||
withModel.add(cfg.model());
|
||||
return withModel;
|
||||
}
|
||||
|
||||
/**
|
||||
* Pin a bounded auto-compaction window on the command line via {@code --autocompact <tokens>},
|
||||
* opt-in per profile (CB-634's sibling ticket: a member that runs out of context dies mid-turn
|
||||
* and its {@code fleet_reply} — the whole point of the turn — is lost with it; opencode already
|
||||
* forces {@code compaction.auto: true} unconditionally, CB-523, but Claude Code has no equivalent
|
||||
* and runs at the backend's own default window).
|
||||
*
|
||||
* <p>Mirrors {@link #argvWithModel}: appended after it, so it survives the {@code ccs <profile>}
|
||||
* wrapper the same way {@code --model} does, and outranks env/settings and the operator's own
|
||||
* {@code argv}. Verified: {@code claude 2.1.241 --help} lists {@code --autocompact <auto|tokens>}
|
||||
* (either the literal {@code auto}, or an integer 100k–1M) — {@link FleetConfig#load} rejects a
|
||||
* configured value outside that band before this ever runs, so the flag Claude Code receives here
|
||||
* is always in range.
|
||||
*/
|
||||
private static List<String> argvWithAutoCompact(List<String> argv, FleetConfig.Profile cfg) {
|
||||
if (cfg.autoCompactWindow() == null) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withAutoCompact = mutableArgv(argv);
|
||||
withAutoCompact.add("--autocompact");
|
||||
withAutoCompact.add(String.valueOf(cfg.autoCompactWindow()));
|
||||
return withAutoCompact;
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.CONTEXT_RESET, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_NAME, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
String target = agentTarget(id);
|
||||
if (target == null) {
|
||||
return false;
|
||||
}
|
||||
// This deliberately bypasses Injector: /clear is housekeeping, not a delegated turn.
|
||||
agents().send(target, "/clear");
|
||||
return true;
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(FleetConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (Claude prefix), kept for direct unit testing -------------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is a Claude Code bridge worker started by a <em>different</em> process
|
||||
* than {@code currentNonce}. A thin {@code claude}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -1,510 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementCandidate;
|
||||
import dev.ltms.fleet.placement.PlacementContext;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.placement.PlacementPolicy;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* The {@link PeerLauncher} the core actually holds when more than one adapter is configured — a thin
|
||||
* router in front of one {@link HerdrPeerLauncher} per peer {@code kind} (Claude Code, opencode, …).
|
||||
* It owns no transport of its own; it dispatches each SPI call to the delegate that owns the profile
|
||||
* involved, and fans the fleet-wide queries (list/reap/caps/profiles) across all delegates.
|
||||
*
|
||||
* <p>Routing rules:
|
||||
* <ul>
|
||||
* <li><strong>By profile</strong> — {@link #spawn}, {@link #effectiveCwd}, {@link #parityOverlay}
|
||||
* resolve the profile (a null/blank name → the global {@link #defaultProfile}) and delegate to
|
||||
* the single adapter that declares it. Profiles partition cleanly across adapters: the
|
||||
* constructor rejects a name claimed by two.</li>
|
||||
* <li><strong>By pane id</strong> — {@link #stop} routes to the adapter that spawned that pane
|
||||
* (recorded at spawn time). A pane the composite never spawned (only real for a caller that
|
||||
* hand-rolls an id) falls back to the first delegate; teardown is pane-id addressed and
|
||||
* tab cleanup is single-occupant guarded, so it is safe either way.</li>
|
||||
* <li><strong>Fleet-wide</strong> — {@link #reapOrphanWorkers} and {@link #capabilities} fan out
|
||||
* and combine. {@link #list} is deduplicated by pane id because every herdr-backed delegate
|
||||
* shares one herdr connection and so reports the same global agent set.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>CB-518: an unqualified spawn is routed through a {@link PlacementPolicy}. The default
|
||||
* {@code fixed} policy reproduces the historical default-profile behaviour; {@code weighted} uses
|
||||
* smooth weighted round-robin with {@code maxLoad} gating. If a chosen profile fails with
|
||||
* {@link PeerUnreachableException}, the composite advances to the next available candidate and
|
||||
* retries, bounded by the number of candidates.
|
||||
*/
|
||||
public final class CompositePeerLauncher implements PeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CompositePeerLauncher.class);
|
||||
|
||||
private final List<HerdrPeerLauncher> delegates;
|
||||
private final Map<String, HerdrPeerLauncher> byProfile;
|
||||
private final String defaultProfile;
|
||||
|
||||
/** paneId → the delegate that spawned it, so {@link #stop} tears down through the right adapter. */
|
||||
private final Map<String, HerdrPeerLauncher> spawnedBy = new ConcurrentHashMap<>();
|
||||
|
||||
private final Function<String, Integer> liveCount;
|
||||
|
||||
/**
|
||||
* CB-559: the placement inputs are read <em>per spawn</em>, not captured at construction, so a
|
||||
* config reload changes where the next member lands without a restart. These are the hot keys —
|
||||
* role pools, an existing profile's weight/maxLoad, and the placement policy. What cannot change
|
||||
* this way is the set of adapters ({@link #byProfile}), because a new backend needs a launcher
|
||||
* and launchers are built once; {@code ConfigRef} classifies that as deferred and says so.
|
||||
*/
|
||||
private final Supplier<Map<String, FleetConfig.Profile>> profileConfigs;
|
||||
private final Supplier<PlacementPolicy> placementPolicy;
|
||||
|
||||
/** CB-578 stage B: credential cooldown, checked before an explicit spawn and filtered into placement. */
|
||||
private final BackendQuarantine quarantine;
|
||||
|
||||
/**
|
||||
* CB-557: the role pools an unqualified spawn draws its candidates from. A supplier that yields
|
||||
* {@code null}, and an empty pool for a role, both fall back to every configured profile — the
|
||||
* pre-CB-557 behaviour.
|
||||
*/
|
||||
private final Supplier<FleetConfig.Fleet> fleet;
|
||||
|
||||
/**
|
||||
* Backward-compatible constructor: fixed placement, no live-counting. Use this for tests and
|
||||
* simple wiring; it preserves the pre-CB-518 behaviour exactly.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to (may be null)
|
||||
* @throws IllegalArgumentException if {@code delegates} is empty or two adapters claim one profile
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates, String defaultProfile) {
|
||||
this(delegates, defaultProfile, Map.of(), PlacementPolicies.fixed(), _ -> 0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with a placement policy and live-worker counter. Quarantine (CB-578
|
||||
* stage B) is off for this constructor — {@link BackendQuarantine#none()} — since it predates
|
||||
* the feature and existing callers of this exact overload never exercised it; use the 7-arg
|
||||
* overload below to wire a real {@link BackendQuarantine}.
|
||||
*
|
||||
* @param delegates one adapter per configured peer kind; must be non-empty and declare
|
||||
* disjoint profile-name sets
|
||||
* @param defaultProfile the profile a no-argument spawn resolves to under {@code fixed} policy
|
||||
* @param profileConfigs all configured worker profiles (used for candidate weights/caps)
|
||||
* @param placementPolicy which policy governs unqualified spawns
|
||||
* @param liveCount live worker count per profile (must never return {@code null})
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null, BackendQuarantine.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools (CB-557). An unqualified spawn draws its candidates from
|
||||
* {@code fleet.<role>} instead of from every configured profile, so a reviewer is placed on a
|
||||
* reviewer backend and never on, say, the architect-only one. Quarantine is off for this
|
||||
* constructor too, for the same reason as the 5-arg overload above.
|
||||
*
|
||||
* @param fleet the configured role pools; {@code null} ⇒ every profile is a candidate for every
|
||||
* role, which is the pre-CB-557 behaviour
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, BackendQuarantine.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B). The full-featured
|
||||
* non-reloading form; {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine)}
|
||||
* is what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), so a config
|
||||
* reload retargets the next member without a restart.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<FleetConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile,
|
||||
() -> config.get().profiles(),
|
||||
() -> PlacementPolicies.fromName(config.get().placement()),
|
||||
liveCount,
|
||||
() -> config.get().fleet(),
|
||||
quarantine);
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
private CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<Map<String, FleetConfig.Profile>> profileConfigs,
|
||||
Supplier<PlacementPolicy> placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
this.fleet = fleet;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
if (delegates.isEmpty()) {
|
||||
throw new IllegalArgumentException("at least one peer adapter must be configured");
|
||||
}
|
||||
this.delegates = List.copyOf(delegates);
|
||||
this.defaultProfile = defaultProfile;
|
||||
this.profileConfigs = profileConfigs;
|
||||
this.placementPolicy = placementPolicy;
|
||||
this.liveCount = liveCount;
|
||||
Map<String, HerdrPeerLauncher> index = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : this.delegates) {
|
||||
for (String profile : d.profiles()) {
|
||||
HerdrPeerLauncher prev = index.putIfAbsent(profile, d);
|
||||
if (prev != null) {
|
||||
throw new IllegalArgumentException(
|
||||
"worker profile '" + profile + "' is claimed by two peer adapters");
|
||||
}
|
||||
}
|
||||
}
|
||||
// Order-preserving for the same reason, and because profiles() is user-visible (fleet_profiles).
|
||||
this.byProfile = Collections.unmodifiableMap(index);
|
||||
}
|
||||
|
||||
/** A supplier of a value fixed at construction — how the non-reloading constructors funnel in. */
|
||||
private static <T> Supplier<T> constant(T value) {
|
||||
return () -> value;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-configured profiles, never null.
|
||||
*
|
||||
* <p>Read fresh on every call so a reload is visible; a caller that needs two consistent reads
|
||||
* takes one local, as {@link #poolFor} does.
|
||||
*/
|
||||
private Map<String, FleetConfig.Profile> profiles0() {
|
||||
Map<String, FleetConfig.Profile> m = profileConfigs.get();
|
||||
return m == null ? Map.of() : m;
|
||||
}
|
||||
|
||||
/** The adapter owning {@code profileName} (null/blank → the default). Throws on an unknown profile. */
|
||||
private HerdrPeerLauncher route(String profileName) {
|
||||
String resolved = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
if (resolved == null) {
|
||||
// No profile and no default configured — hand to the first delegate so it raises the
|
||||
// same "no default" error it would on its own; keeps the SPI contract single-sourced.
|
||||
return delegates.getFirst();
|
||||
}
|
||||
HerdrPeerLauncher d = byProfile.get(resolved);
|
||||
if (d == null) {
|
||||
throw new IllegalArgumentException("unknown worker profile: " + resolved);
|
||||
}
|
||||
return d;
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String requestedProfile = req.profileName();
|
||||
if (requestedProfile != null && !requestedProfile.isBlank()) {
|
||||
// An explicit profile bypasses the placement policy, but not the capacity cap: maxLoad
|
||||
// is documented as an unconditional limit on this profile (FleetConfig.Profile), and
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
enforceNotQuarantined(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
}
|
||||
|
||||
// CB-557: an unqualified spawn is placed inside the pool of the role it asked for, not across
|
||||
// the whole profile list. An EXPLICIT profile (above) is left alone on purpose — it is the
|
||||
// operator overriding, and refusing it would break `fleet_spawn{profile:"opus"}`, which
|
||||
// carries no role and so would be judged against the dev pool it was never meant for.
|
||||
List<PlacementCandidate> candidates = candidates(req.role());
|
||||
String roleDefault = defaultProfileFor(req.role());
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
// Deliberately uncaught: when no candidate is left (all at cap, or all unreachable) the
|
||||
// policy already throws a clear message. Catching it to rethrow a generic
|
||||
// PeerUnreachableException would replace a precise diagnosis with a vague one.
|
||||
PlacementCandidate chosen = placementPolicy.get().select(ctx);
|
||||
|
||||
HerdrPeerLauncher d = byProfile.get(chosen.profile());
|
||||
if (d == null) {
|
||||
// A configured profile with no adapter is a wiring bug; fail fast.
|
||||
unreachable.add(chosen.profile());
|
||||
continue;
|
||||
}
|
||||
|
||||
// CB-547a: route the chosen profile but keep the caller's session identity — dropping it
|
||||
// here would silently sever the resume handle on every policy-routed spawn. CB-557: the
|
||||
// role rides along for the same reason, or a routed spawn would be labelled as a dev.
|
||||
SpawnRequest routedReq = new SpawnRequest(chosen.profile(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
try {
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
} catch (PeerUnreachableException e) {
|
||||
log.warn("spawn on profile {} unreachable, will retry next candidate if any: {}",
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
}
|
||||
}
|
||||
|
||||
throw new PeerUnreachableException(
|
||||
"no reachable worker profile available after trying " + unreachable.size()
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn when the profile is at its {@code maxLoad} cap.
|
||||
*
|
||||
* <p>maxLoad is a documented, unconditional capacity limit (see {@code FleetConfig.Profile#maxLoad}),
|
||||
* and the charter makes explicit-profile spawns the normal path — so enforcing it only in placement
|
||||
* ({@code PlacementPolicyUtil}, package-private, hence not linked) would leave the cap dead config
|
||||
* on every call that names a profile. Same rule as placement: {@code live >= cap} is at capacity.
|
||||
*
|
||||
* <p>Deliberately no fallback to another profile: the caller named {@code profile} for a cost/model
|
||||
* reason, and silently re-routing a paid-tier (subscription) request elsewhere is worse than
|
||||
* refusing it. A caller that wants placement should omit the profile and let the policy pick.
|
||||
*
|
||||
* <p>Known TOCTOU limitation — documented, not fixed. {@link #liveCount} is read outside any lock and
|
||||
* {@code SessionManager} registers a session only after {@code launcher.spawn} returns, so two
|
||||
* genuinely concurrent spawns can both pass this check. The race already exists on the placement
|
||||
* path. Closing it needs slot reservation in the registry; serializing spawn here would block on
|
||||
* the readiness gate and is a far worse trade.
|
||||
*
|
||||
* @param profile the profile the caller explicitly named
|
||||
* @throws PlacementException when the profile is at capacity
|
||||
*/
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose credential is quarantined (CB-578 stage B): a prior
|
||||
* {@code BACKEND_EXHAUSTED} classification on this profile, or on another profile sharing its
|
||||
* {@code credentialId}, is still on cooldown.
|
||||
*
|
||||
* <p>Checked before {@link #enforceMaxLoad}, and for the same reason that check exists: an
|
||||
* explicit profile bypasses the placement policy's filtering entirely, so without this the cap
|
||||
* (there, quarantine here) would be dead config on the very path the charter calls normal.
|
||||
* Deliberately no fallback to another profile, matching {@link #enforceMaxLoad}'s own reasoning —
|
||||
* the caller named this profile for a cost/model reason.
|
||||
*
|
||||
* @throws PlacementException naming the profile, its credential, and the remaining cooldown
|
||||
*/
|
||||
private void enforceNotQuarantined(String profile) {
|
||||
String credentialId = credentialIdFor(profile);
|
||||
quarantine.remainingSeconds(credentialId).ifPresent(remaining -> {
|
||||
throw new PlacementException("worker profile '" + profile + "' is quarantined "
|
||||
+ "(credential '" + credentialId + "' exhausted; ~" + remaining
|
||||
+ "s remaining) — refusing spawn");
|
||||
});
|
||||
}
|
||||
|
||||
/** {@code profile}'s credential group (CB-578 stage B), or the profile's own name if unconfigured. */
|
||||
private String credentialIdFor(String profile) {
|
||||
FleetConfig.Profile cfg = profiles0().get(profile);
|
||||
return cfg == null ? profile : cfg.effectiveCredentialId();
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose credential is currently quarantined (CB-578 stage B). */
|
||||
private Set<String> quarantinedProfiles(List<PlacementCandidate> candidates) {
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> quarantine.isQuarantined(credentialIdFor(p)))
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (ABSENT ⇒ unlimited at load),
|
||||
// means no cap — never cap what wasn't configured. Note "non-positive ⇒ unlimited" was true
|
||||
// until CB-585: an explicit `maxLoad: 0` now survives as 0 and is a real cap of zero, so the
|
||||
// check below refuses every spawn on that profile, and a negative value is refused at config
|
||||
// load rather than normalized away.
|
||||
FleetConfig.Profile cfg = profiles0().get(profile);
|
||||
Integer cap = (cfg == null) ? null : cfg.maxLoad();
|
||||
if (cap == null) {
|
||||
return;
|
||||
}
|
||||
int live = liveCount.apply(profile);
|
||||
if (live >= cap) {
|
||||
throw new PlacementException("worker profile '" + profile + "' is at maxLoad: " + live
|
||||
+ " live >= " + cap + " cap; refusing spawn — no fallback to another profile");
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile names {@code role} may be placed on, in definition order.
|
||||
*
|
||||
* <p>An empty or absent pool means "unconstrained", not "nothing allowed": a config that declares
|
||||
* no pool for a role must keep spawning, so it falls back to every configured profile. Names in a
|
||||
* pool that no adapter declares are dropped here rather than thrown — config load already rejects
|
||||
* a pool entry with no profile, so a survivor is a profile this particular composite does not own.
|
||||
*/
|
||||
private List<String> poolFor(MemberRole role) {
|
||||
Map<String, FleetConfig.Profile> configured = profiles0();
|
||||
FleetConfig.Fleet f = fleet.get();
|
||||
List<String> pool = (f == null) ? List.of() : f.profilesFor(role);
|
||||
List<String> known = pool.stream().filter(configured::containsKey).toList();
|
||||
return known.isEmpty() ? List.copyOf(configured.keySet()) : known;
|
||||
}
|
||||
|
||||
/** The profile an unqualified spawn for {@code role} falls back to under {@code fixed} placement. */
|
||||
private String defaultProfileFor(MemberRole role) {
|
||||
List<String> pool = poolFor(role);
|
||||
return pool.isEmpty() ? defaultProfile : pool.getFirst();
|
||||
}
|
||||
|
||||
/** Build the candidate list from {@code role}'s pool, in definition order. */
|
||||
private List<PlacementCandidate> candidates(MemberRole role) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (String name : poolFor(role)) {
|
||||
FleetConfig.Profile w = profiles0().get(name);
|
||||
if (w != null) {
|
||||
out.add(new PlacementCandidate(name, null, w.weight(), w.maxLoad()));
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return route(req.profileName()).effectiveCwd(req);
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return route(profileName).parityOverlay(profileName);
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.remove(id);
|
||||
if (d == null) {
|
||||
log.debug("stop({}) — no recorded owner, routing to the first adapter (pane-addressed)", id);
|
||||
d = delegates.getFirst();
|
||||
}
|
||||
d.stop(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
HerdrPeerLauncher delegate = spawnedBy.get(id);
|
||||
if (delegate == null) {
|
||||
log.debug("clearContext({}) ignored — no recorded owning adapter", id);
|
||||
return false;
|
||||
}
|
||||
return delegate.clearContext(id);
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return byProfile.keySet();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Routes to the specific delegate {@code profileName} resolves to, not the fleet-wide
|
||||
* union {@link #capabilities()} returns — the whole reason this method exists (CB-584): in a
|
||||
* mixed fleet, one adapter's capability must never be read as every profile's.
|
||||
*/
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
return route(profileName).capabilities();
|
||||
}
|
||||
|
||||
/** Every herdr agent, deduplicated by pane id (all delegates share one herdr and list globally). */
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
Map<String, Agent> byPane = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
for (Agent a : d.list()) {
|
||||
if (a.paneId() != null) {
|
||||
byPane.putIfAbsent(a.paneId(), a);
|
||||
}
|
||||
}
|
||||
}
|
||||
return List.copyOf(byPane.values());
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
int reaped = 0;
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
reaped += d.reapOrphanWorkers();
|
||||
}
|
||||
return reaped;
|
||||
}
|
||||
|
||||
/** The union of every adapter's capabilities — a capability any adapter offers, the fleet offers. */
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
EnumSet<Capability> caps = EnumSet.noneOf(Capability.class);
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
caps.addAll(d.capabilities());
|
||||
}
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
}
|
||||
@@ -1,276 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.time.Duration;
|
||||
import java.time.Instant;
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* CB-633: generates the per-spawn {@code ZDOTDIR} directory whose startup files enforce
|
||||
* {@code memberCredentials.policy: allow-list}.
|
||||
*
|
||||
* <p>The seam: zsh reads its startup files from {@code $ZDOTDIR}, and the daemon puts that variable
|
||||
* in the pane-creation env map. The operator's whole chain ({@code ~/.zshrc} → secret store) runs
|
||||
* inside those files, so a scrub appended to the LAST one runs after everything the operator
|
||||
* sourced, and nothing later can re-export over it. This is the property CB-596's env-overlay
|
||||
* control lacked: herdr applies that overlay BEFORE the shell starts, so any sourced file can undo
|
||||
* it — and did.
|
||||
*
|
||||
* <p><b>Which file is last depends on the platform, so the scrub runs from two of them.</b> zsh
|
||||
* reads {@code .zshenv} always, {@code .zprofile} and {@code .zlogin} only for a LOGIN shell, and
|
||||
* {@code .zshrc} only for an INTERACTIVE one. herdr does not open the same kind of shell
|
||||
* everywhere — measured on herdr 0.8.0: macOS panes run {@code -zsh} (login, so {@code .zlogin}
|
||||
* runs), Linux panes run a plain {@code /usr/bin/zsh} (interactive but NOT login, so
|
||||
* {@code .zlogin} never runs at all). A scrub in {@code .zlogin} alone is therefore a control that
|
||||
* silently does nothing on Linux — the exact failure this class exists to remove, one platform
|
||||
* over.
|
||||
*
|
||||
* <p>So both {@code .zshrc} and {@code .zlogin} source the same generated {@code scrub.zsh} after
|
||||
* sourcing their {@code $HOME} counterpart. On Linux only the first fires; on macOS both do, and
|
||||
* the second pass is deliberate rather than merely harmless — it re-scrubs anything the operator's
|
||||
* own {@code ~/.zlogin} exported after {@code .zshrc} had finished. Re-running is idempotent: a
|
||||
* name already blank is blanked again, and the report is rewritten with the same counts.
|
||||
*
|
||||
* <p>Each generated file sources its {@code $HOME} counterpart FIRST, so {@code PATH} and every
|
||||
* toolchain binary still resolve exactly as the operator configured them; only afterwards does
|
||||
* {@code .zlogin} run the scrub: every EXPORTED variable not on the derived allow-list is re-exported
|
||||
* blank. Blank, not credential-shaped-pattern-filtered: a pattern list ({@code *TOKEN*}, …) is an
|
||||
* enumeration and misses what it did not think of — a username is the other half of a credential and
|
||||
* is shaped like none. Credential-SHAPED names among the blanked set go to the WARN log only,
|
||||
* never to the control.
|
||||
*
|
||||
* <p>The scrub also writes {@code scrub-report.txt} into its own directory: one {@code allowed N of
|
||||
* M} line (N = exports left untouched, M = exports present when the scrub ran), then the blanked
|
||||
* NAMES — never values. The launcher reads this back at teardown and logs it, because a blocked
|
||||
* count next to an unknown denominator is not a finding.
|
||||
*/
|
||||
public final class EnvAllowListScrub {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(EnvAllowListScrub.class);
|
||||
|
||||
/** Name of the report file written into the generated directory by the scrub itself. */
|
||||
static final String REPORT_FILE = "scrub-report.txt";
|
||||
|
||||
/** The scrub body, generated once and sourced from both {@code .zshrc} and {@code .zlogin}. */
|
||||
static final String SCRUB_FILE = "scrub.zsh";
|
||||
|
||||
/** Prefix of every generated directory — also what {@link #reapOrphans} matches on. */
|
||||
static final String DIR_PREFIX = "bridged-zdotdir-";
|
||||
|
||||
/**
|
||||
* How old an orphan must be before {@link #reapOrphans} removes it. Comfortably longer than any
|
||||
* spawn takes, so a directory belonging to a pane that is still starting is never removed.
|
||||
*/
|
||||
private static final Duration ORPHAN_AGE = Duration.ofHours(24);
|
||||
|
||||
/** Appended to the two startup files that must run the scrub, after their {@code $HOME} source. */
|
||||
private static final String SOURCE_SCRUB =
|
||||
"source \"$ZDOTDIR/" + SCRUB_FILE + "\"\n";
|
||||
|
||||
private EnvAllowListScrub() {
|
||||
}
|
||||
|
||||
/**
|
||||
* A parsed {@code scrub-report.txt}: how many exported variables existed when the scrub ran,
|
||||
* how many were left untouched (allowed), and the NAMES that were blanked. Values never appear.
|
||||
*/
|
||||
record ScrubReport(int allowed, int total, List<String> blanked) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a fresh ZDOTDIR directory under {@code parentDir} holding the four zsh startup files.
|
||||
* Every file (and the directory) registers {@code deleteOnExit}, next to the existing per-spawn
|
||||
* charter/config temp cleanup; the launcher additionally deletes eagerly at pane release.
|
||||
*
|
||||
* @param allowedNames the DERIVED allow-list — exact variable names that must survive the scrub
|
||||
* @return the directory path (to be passed as the pane's {@code ZDOTDIR})
|
||||
* @throws UncheckedIOException when the directory or any file cannot be written — a spawn whose
|
||||
* protection cannot even be materialized must fail loudly rather
|
||||
* than start unprotected
|
||||
*/
|
||||
public static Path generate(Path parentDir, Set<String> allowedNames) {
|
||||
try {
|
||||
reapOrphans(parentDir);
|
||||
Path dir = Files.createTempDirectory(parentDir, DIR_PREFIX);
|
||||
dir.toFile().deleteOnExit();
|
||||
// The report is written by zsh, after these hooks are registered, so register its path
|
||||
// too — otherwise the directory is non-empty at JVM exit and cannot be removed at all.
|
||||
dir.resolve(REPORT_FILE).toFile().deleteOnExit();
|
||||
write(dir, SCRUB_FILE, scrubScript(allowedNames));
|
||||
write(dir, ".zshenv", homeSourcingFile(".zshenv"));
|
||||
write(dir, ".zprofile", homeSourcingFile(".zprofile"));
|
||||
write(dir, ".zshrc", homeSourcingFile(".zshrc") + SOURCE_SCRUB);
|
||||
write(dir, ".zlogin", homeSourcingFile(".zlogin") + SOURCE_SCRUB);
|
||||
return dir;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot generate ZDOTDIR scrub files under " + parentDir, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** One operator-sourcing startup file: source the {@code $HOME} counterpart, change nothing else. */
|
||||
private static String homeSourcingFile(String name) {
|
||||
return """
|
||||
# generated by fleetd (CB-633 memberCredentials policy=allow-list) — do not edit.
|
||||
# Source the operator's own %s first, so PATH and the agent binaries resolve as usual.
|
||||
[ -r "$HOME/%s" ] && source "$HOME/%s"
|
||||
""".formatted(name, name, name);
|
||||
}
|
||||
|
||||
/**
|
||||
* The generated {@code scrub.zsh} — the scrub body on its own, so the two startup files that
|
||||
* must run it ({@code .zshrc} and {@code .zlogin}) hold one copy between them rather than two
|
||||
* that can drift. Package-private so tests can assert on the exact script handed to zsh — the
|
||||
* artefact here IS a shell file, and a test that checks only the Java string assembly proves
|
||||
* nothing about whether zsh accepts it.
|
||||
*/
|
||||
static String scrubScript(Set<String> allowedNames) {
|
||||
StringBuilder names = new StringBuilder();
|
||||
for (String n : allowedNames.stream().sorted().toList()) {
|
||||
if (names.length() > 0) {
|
||||
names.append(' ');
|
||||
}
|
||||
// Names are validated against [A-Za-z_][A-Za-z0-9_]* before they get here; single quotes
|
||||
// keep even a non-conforming name inert rather than executable.
|
||||
names.append('\'').append(n.replace("'", "")).append('\'');
|
||||
}
|
||||
return """
|
||||
# generated by fleetd (CB-633 memberCredentials policy=allow-list) — do not edit.
|
||||
# Sourced from .zshrc and again from .zlogin, each time AFTER that file has sourced
|
||||
# its $HOME counterpart — so this runs after everything the operator sourced, on a
|
||||
# login shell (macOS panes) and on a plain interactive one (Linux panes) alike.
|
||||
# Running twice is idempotent and deliberate: the second pass catches anything
|
||||
# ~/.zlogin exported after ~/.zshrc had finished.
|
||||
|
||||
typeset -A _cb633_allowed
|
||||
for _cb633_n in %s; do _cb633_allowed[$_cb633_n]=1; done
|
||||
|
||||
# Enumerate EXPORTED variable NAMES from `env` itself. Deliberately NOT the special
|
||||
# `parameters` assoc: its subscript is evaluated arithmetically on this host's zsh
|
||||
# and blows up on some names ("bad math expression"). Names not matching the
|
||||
# identifier pattern (junk from multi-line values) are skipped, never scrubbed.
|
||||
typeset -a _cb633_names
|
||||
_cb633_names=("${(@f)$(command env | command cut -d= -f1)}")
|
||||
typeset -a _cb633_blank
|
||||
_cb633_blank=()
|
||||
integer _cb633_total=0
|
||||
for _cb633_n in "${_cb633_names[@]}"; do
|
||||
[[ "$_cb633_n" =~ ^[A-Za-z_][A-Za-z0-9_]*$ ]] || continue
|
||||
(( _cb633_total += 1 ))
|
||||
[[ -n "${_cb633_allowed[$_cb633_n]-}" ]] && continue
|
||||
case "$_cb633_n" in %s) continue ;; esac
|
||||
_cb633_blank+=("$_cb633_n")
|
||||
done
|
||||
|
||||
{ for _cb633_n in "${_cb633_blank[@]}"; do export "$_cb633_n="; done; } 2>/dev/null
|
||||
|
||||
integer _cb633_kept=$(( _cb633_total - ${#_cb633_blank} ))
|
||||
{
|
||||
print -r -- "allowed $_cb633_kept of $_cb633_total"
|
||||
for _cb633_n in "${_cb633_blank[@]}"; do print -r -- "$_cb633_n"; done
|
||||
} > "$ZDOTDIR/%s" 2>/dev/null
|
||||
|
||||
unset _cb633_allowed _cb633_names _cb633_blank _cb633_n _cb633_total _cb633_kept
|
||||
""".formatted(names, MemberEnvAllowList.zshCasePattern(), REPORT_FILE);
|
||||
}
|
||||
|
||||
private static void write(Path dir, String fileName, String content) throws IOException {
|
||||
Path file = dir.resolve(fileName);
|
||||
Files.writeString(file, content);
|
||||
file.toFile().deleteOnExit();
|
||||
}
|
||||
|
||||
/**
|
||||
* Read and parse {@link #REPORT_FILE} out of a generated ZDOTDIR directory. Returns {@code null}
|
||||
* when absent or unreadable (the pane may have been torn down before its login shell ever got to
|
||||
* the scrub) — callers treat that as "no measurement available", never as success.
|
||||
*/
|
||||
static ScrubReport readReport(Path zdotdir) {
|
||||
Path report = zdotdir.resolve(REPORT_FILE);
|
||||
if (!Files.isRegularFile(report)) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
List<String> lines = Files.readAllLines(report);
|
||||
if (lines.isEmpty() || !lines.getFirst().startsWith("allowed ")) {
|
||||
return null;
|
||||
}
|
||||
String[] parts = lines.getFirst().substring("allowed ".length()).trim().split("\\s+");
|
||||
if (parts.length != 3 || !"of".equals(parts[1])) {
|
||||
return null;
|
||||
}
|
||||
List<String> blanked = new ArrayList<>();
|
||||
for (int i = 1; i < lines.size(); i++) {
|
||||
if (!lines.get(i).isBlank()) {
|
||||
blanked.add(lines.get(i));
|
||||
}
|
||||
}
|
||||
return new ScrubReport(Integer.parseInt(parts[0]), Integer.parseInt(parts[2]),
|
||||
List.copyOf(blanked));
|
||||
} catch (IOException | NumberFormatException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Best-effort recursive delete; failures are swallowed — JVM-exit cleanup is the backstop. */
|
||||
/**
|
||||
* Remove generated directories left behind by an earlier daemon process.
|
||||
*
|
||||
* <p>{@link #generate} registers each directory for deletion at JVM exit, which covers a clean
|
||||
* shutdown and covers nothing else. A {@code kill -9}, a crash, or a host reboot leaves the
|
||||
* directory in the temp dir for good, and the daemon is restarted often enough that these
|
||||
* accumulate. They hold no secrets — the generated files contain variable NAMES and a report of
|
||||
* names, never a value — but an unbounded pile of them in {@code /tmp} is still our mess to
|
||||
* clear.
|
||||
*
|
||||
* <p>Called from {@link #generate}, so it runs on the path that creates them and needs no
|
||||
* separate wiring or scheduler. Only directories older than {@link #ORPHAN_AGE} are touched,
|
||||
* which keeps it clear of any pane that is merely still starting, including one belonging to a
|
||||
* different daemon instance running right now. Best-effort: every failure is ignored, because
|
||||
* tidying temp files must never be the reason a spawn fails.
|
||||
*/
|
||||
static void reapOrphans(Path parentDir) {
|
||||
Instant cutoff = Instant.now().minus(ORPHAN_AGE);
|
||||
try (Stream<Path> entries = Files.list(parentDir)) {
|
||||
entries.filter(d -> d.getFileName().toString().startsWith(DIR_PREFIX))
|
||||
.filter(Files::isDirectory)
|
||||
.filter(d -> olderThan(d, cutoff))
|
||||
.forEach(EnvAllowListScrub::deleteRecursively);
|
||||
} catch (IOException | RuntimeException e) {
|
||||
log.debug("could not scan {} for orphaned ZDOTDIRs: {}", parentDir, e.toString());
|
||||
}
|
||||
}
|
||||
|
||||
private static boolean olderThan(Path dir, Instant cutoff) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(dir).toInstant().isBefore(cutoff);
|
||||
} catch (IOException e) {
|
||||
return false; // unreadable timestamp ⇒ leave it alone
|
||||
}
|
||||
}
|
||||
|
||||
static void deleteRecursively(Path dir) {
|
||||
if (dir == null || !Files.exists(dir)) {
|
||||
return;
|
||||
}
|
||||
try (Stream<Path> walk = Files.walk(dir)) {
|
||||
walk.sorted(java.util.Comparator.reverseOrder()).forEach(p -> {
|
||||
try {
|
||||
Files.deleteIfExists(p);
|
||||
} catch (IOException ignored) {
|
||||
// best effort — deleteOnExit retries at JVM shutdown
|
||||
}
|
||||
});
|
||||
} catch (IOException ignored) {
|
||||
// same
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,606 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
/**
|
||||
* The {@link HerdrPeerLauncher} adapter for <strong>opencode</strong> — an open-source,
|
||||
* provider-agnostic terminal coding agent. Its whole reason for existing is to prove the
|
||||
* {@code PeerLauncher} SPI is genuinely provider-neutral: opencode shares none of Claude Code's
|
||||
* private launch seams, yet reuses every line of shared transport in the base (tab/pane placement,
|
||||
* the CB-306 readiness gate, unique naming + CB-117 reap, teardown, listing, cwd).
|
||||
*
|
||||
* <p>The divergences from {@link ClaudeCodeLauncher}, all confined to {@link #buildLaunch}:
|
||||
* <ul>
|
||||
* <li><strong>No subscription boundary.</strong> opencode carries no {@code ANTHROPIC_BASE_URL}
|
||||
* and there is no {@link dev.ltms.fleet.guard.SubscriptionGuard} — the guard is a
|
||||
* Claude-private concern, not part of the SPI. opencode reads the operator's own provider
|
||||
* credentials from its global {@code auth.json}; the bridge injects none.</li>
|
||||
* <li><strong>File-based MCP mount + instructions.</strong> opencode has no inline
|
||||
* {@code --mcp-config}/{@code --append-system-prompt}. Instead the bridge writes an ephemeral
|
||||
* {@code opencode.json} that declares the bridge as a {@code remote} MCP server and lists a
|
||||
* member-charter file under {@code instructions}, then points the worker at it with
|
||||
* {@code OPENCODE_CONFIG}. This is the one place the launcher touches disk — Claude never did.</li>
|
||||
* <li><strong>Model as a flag.</strong> the {@code provider/model} selector is passed as
|
||||
* {@code -m}, not an env var.</li>
|
||||
* <li><strong>{@code opencode} name prefix</strong> so reap matches {@code opencode-*} panes and
|
||||
* never another adapter's.</li>
|
||||
* </ul>
|
||||
*/
|
||||
public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(OpenCodeLauncher.class);
|
||||
|
||||
/** Label prefix for this adapter's herdr agent names (drives naming + orphan reap). */
|
||||
private static final String NAME_PREFIX = "opencode";
|
||||
|
||||
/** Writer for the generated {@code opencode.json}. */
|
||||
private static final ObjectMapper JSON = new ObjectMapper();
|
||||
|
||||
/** Root under which per-spawn opencode config dirs are created (injectable for tests). */
|
||||
private final Path configRoot;
|
||||
|
||||
/**
|
||||
* Session discovery against opencode's on-disk storage ({@link OpenCodeSessionDiscovery}) —
|
||||
* the one seam that knows opencode's private session-file layout. Its root is injectable for
|
||||
* tests so they never touch the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, 0,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(300),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with the spawn-ready gate enabled. Polls {@code agents.status()} until
|
||||
* the pane reports an injectable state or {@code spawnReadyTimeoutMs} elapses.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs) {
|
||||
this(agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, spawnReadyPollMs, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor carrying the fleet-wide tab-label template (CB-557). The template comes
|
||||
* from {@code fleet.tabLabel}, which a profile cannot know because it names the member's
|
||||
* <em>role</em>; a profile may still override it with its own {@code tabLabel}.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor. Every injectable collaborator is explicit so unit tests supply a
|
||||
* fake clock ({@code nowMillis}), poll-loop wait ({@code sleeper}), and a temp {@code configRoot}
|
||||
* they can inspect the generated {@code opencode.json}/charter under.
|
||||
*
|
||||
* @param agents herdr agent control (start, status, close)
|
||||
* @param spaces workspace / tab control (ensure, create, close)
|
||||
* @param profiles configured worker profiles
|
||||
* @param defaultProfile profile a no-argument spawn uses (nullable)
|
||||
* @param env host env lookup (injectable for tests)
|
||||
* @param spawnReadyTimeoutMs max ms to wait for injectable state (0 disables the gate)
|
||||
* @param nowMillis monotonic clock source (e.g. {@code System::currentTimeMillis})
|
||||
* @param sleeper sleep/wait hook (encodes the poll interval; never called when the
|
||||
* gate is disabled)
|
||||
* @param configRoot existing directory under which per-spawn config dirs are created
|
||||
* @param discoveryRoot opencode's on-disk storage root to scan for session records
|
||||
* (injectable for tests; opencode's layout is matched at
|
||||
* {@link OpenCodeSessionDiscovery})
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
nowMillis, sleeper, configRoot, discoveryRoot, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the fleet-wide tab-label template (CB-557).
|
||||
*
|
||||
* @param fleet live fleet config, read once for each spawn
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the CB-596 {@code memberCredentials} policy supplier.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Builds the opencode launch: no {@code ANTHROPIC_*} and no guard (opencode reads its own
|
||||
* provider credentials); when the profile mounts the bridge MCP or has a member charter,
|
||||
* generate an ephemeral {@code opencode.json} (remote MCP server + member-charter instructions)
|
||||
* and point the worker at it via {@code OPENCODE_CONFIG}; carry the parity-neutral git-forge
|
||||
* grant; and select the model with {@code -m}.
|
||||
*/
|
||||
@Override
|
||||
protected Launch buildLaunch(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
Map<String, String> workerEnv = baseEnv(cfg);
|
||||
// autoCompactWindow's opencode lever (limit.context) only targets a specific provider/model
|
||||
// entry, so it needs model: in "provider/model" form. A profile that opts in without that
|
||||
// shape gets no silent no-op — log it, once, here, whether or not writeConfig ends up running.
|
||||
boolean wantsContextLimit = cfg.autoCompactWindow() != null && splitProviderModel(cfg.model()) != null;
|
||||
if (cfg.autoCompactWindow() != null && !wantsContextLimit) {
|
||||
log.warn("profile '{}' sets autoCompactWindow but model '{}' is not \"<provider>/<model>\" "
|
||||
+ "form — opencode's per-model context limit could not be applied for this profile",
|
||||
cfg.profile(), cfg.model());
|
||||
}
|
||||
// A config file is needed for the bridge MCP mount, a member charter, the IDE MCP (+ its
|
||||
// guidance overlay, CB-634), a pinned endpoint (CB-508), or a resolvable autoCompactWindow.
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp() || spec.charter() != null || hasCustomProvider(cfg)
|
||||
|| wantsContextLimit) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter(), spec.cwd()).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
List<String> argv = argvWithResume(argvWithModel(argvWithAuto(cfg), cfg), spec.resumeSessionId());
|
||||
return new Launch(workerEnv, argvWithAgent(argv, spec));
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus, when the role has an agent-definition file under the worker's cwd,
|
||||
* opencode's {@code --agent <role>} flag (CB-617). A role with no such file gets nothing added —
|
||||
* the member must still spawn.
|
||||
*/
|
||||
private List<String> argvWithAgent(List<String> argv, LaunchSpec spec) {
|
||||
Path agentFile = agentDefinitionFile(spec.cwd(), spec.role(), ".opencode", "agent");
|
||||
if (agentFile == null) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withAgent = mutableArgv(argv);
|
||||
withAgent.add("--agent");
|
||||
withAgent.add(spec.role().wireName());
|
||||
return withAgent;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
*
|
||||
* <p>Note this reuses {@code baseUrl}, the same field the Claude adapter injects as
|
||||
* {@code ANTHROPIC_BASE_URL} — but it does <em>not</em> go through {@code SubscriptionGuard}.
|
||||
* That asymmetry is deliberate and safe: the guard exists to stop a worker borrowing the
|
||||
* primary's Anthropic subscription, and an opencode process has no Anthropic credential path
|
||||
* at all. Pointing it at a local vLLM cannot leak the subscription.
|
||||
*/
|
||||
private static boolean hasCustomProvider(FleetConfig.Profile cfg) {
|
||||
return cfg.baseUrl() != null && !cfg.baseUrl().isBlank();
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus the unconditional {@code --auto} flag, which auto-approves the
|
||||
* permissions opencode does not explicitly deny. It is unconditional, not a preference: a
|
||||
* spawned peer has no human at its pane — the bridge spawned it — so one that stops at an
|
||||
* approval prompt is a wedged agent, indistinguishable from a legitimate mid-turn wait and
|
||||
* unable to end its turn with {@code fleet_reply}. opencode's own help calls this
|
||||
* "dangerous!", but the blast radius here is already bounded by design: a worker runs in its
|
||||
* own git worktree on its own branch, is off-subscription, and cannot merge — the lead is the
|
||||
* gate.
|
||||
*/
|
||||
private List<String> argvWithAuto(FleetConfig.Profile cfg) {
|
||||
List<String> argv = mutableArgv(cfg.argv());
|
||||
argv.add("--auto");
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* The launch argv plus, on a resumed spawn, opencode's {@code -s <id>} flag to continue a prior
|
||||
* conversation by its session id. {@code -s, --session <id>} resumes an existing session; on a
|
||||
* fresh spawn (no resume target) no flag is added, letting opencode start a brand-new session.
|
||||
* The id comes from the base launch spec.
|
||||
*/
|
||||
private List<String> argvWithResume(List<String> argv, String id) {
|
||||
if (id == null || id.isBlank()) {
|
||||
return argv;
|
||||
}
|
||||
List<String> withResume = mutableArgv(argv);
|
||||
withResume.add("-s");
|
||||
withResume.add(id);
|
||||
return withResume;
|
||||
}
|
||||
|
||||
/** The launch argv plus, when a model is configured, the opencode {@code -m provider/model} flag. */
|
||||
private List<String> argvWithModel(List<String> argv, FleetConfig.Profile cfg) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()) {
|
||||
argv.add("-m");
|
||||
argv.add(cfg.model());
|
||||
}
|
||||
return argv;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write an ephemeral {@code opencode.json} (and the member-charter file it references) into a
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*/
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "bridged-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
root.put("$schema", "https://opencode.ai/config.json");
|
||||
// CB-523, opencode side: a worker that runs out of context dies mid-turn, and its reply
|
||||
// — the entire point of the turn — is lost with it. Auto-compaction is therefore not an
|
||||
// operator preference for a bridged worker, it is a condition of the turn contract.
|
||||
//
|
||||
// Stated deliberately even though it is redundant today: OPENCODE_CONFIG is MERGED over
|
||||
// ~/.config/opencode/config.json rather than replacing it, so a worker already inherits
|
||||
// an `auto: true` set at home. We do not want that inheritance to be what the guarantee
|
||||
// rests on — the home file is outside this repo, differs per machine, and is not ours.
|
||||
//
|
||||
// Know the cost before removing it: this key WINS over the home config (verified — an
|
||||
// OPENCODE_CONFIG value overrides the home value, it does not defer to it), so an
|
||||
// operator who sets `compaction.auto: false` at home cannot turn it off for bridged
|
||||
// workers. That is the intended trade for peers we spawn and whose turns we must land;
|
||||
// if per-profile control is ever wanted, add a profile knob rather than dropping this.
|
||||
root.putObject("compaction").put("auto", true);
|
||||
|
||||
if (charterText != null) {
|
||||
Path charter = dir.resolve("member-charter.md");
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp()) {
|
||||
// One shared mcp node for both servers — putObject would replace the node (and thus
|
||||
// the other server) on the second call, so build into a single get-or-create node.
|
||||
ObjectNode mcp = root.withObject("mcp");
|
||||
if (cfg.hasMcp()) {
|
||||
ObjectNode mount = mcp.putObject(PeerLauncher.MCP_MOUNT_NAME);
|
||||
mount.put("type", "remote");
|
||||
mount.put("url", cfg.mcpUrl());
|
||||
mount.put("enabled", true);
|
||||
}
|
||||
// CB-634: mount the IDE Index MCP in the same shape as the bridge remote server, and
|
||||
// deliver its guidance via the instructions array (opencode does not read
|
||||
// CLAUDE.local.md) rather than any system-prompt string.
|
||||
if (cfg.hasIdeMcp()) {
|
||||
ObjectNode ide = mcp.putObject("intellij");
|
||||
ide.put("type", "remote");
|
||||
ide.put("url", cfg.ideMcpUrl());
|
||||
ide.put("enabled", true);
|
||||
|
||||
// CB-634: pin the overlay and open the IDE at the module dir (this repo's pom is
|
||||
// in `bridged/`, not at the worktree root) — see PeerLauncher.ideProjectPath.
|
||||
String projectPath = PeerLauncher.ideProjectPath(cwd, cfg.ideProjectDir());
|
||||
Path rules = dir.resolve("ide-rules.md");
|
||||
Files.writeString(rules, PeerLauncher.ideOverlayText(projectPath));
|
||||
rules.toFile().deleteOnExit();
|
||||
// The array may already hold the member-charter path; withArray gets-or-creates.
|
||||
root.withArray("instructions").add(rules.toAbsolutePath().toString());
|
||||
PeerLauncher.openInIde(projectPath, cfg.ideOpenCommand(), log);
|
||||
}
|
||||
}
|
||||
if (hasCustomProvider(cfg)) {
|
||||
addCustomProvider(root, cfg);
|
||||
}
|
||||
if (cfg.autoCompactWindow() != null) {
|
||||
applyContextLimit(root, cfg);
|
||||
}
|
||||
|
||||
Path cfgFile = dir.resolve("opencode.json");
|
||||
// Built with Jackson rather than string concatenation: the provider block is nested and
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
"cannot write opencode config for profile " + cfg.profile(), e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
*
|
||||
* <p>The provider id comes from the {@code provider/model} selector in {@code model:}, so one
|
||||
* field drives both the declaration and the {@code -m} flag and they cannot drift apart.
|
||||
*/
|
||||
private void addCustomProvider(ObjectNode root, FleetConfig.Profile cfg) {
|
||||
String[] parts = splitModelSelector(cfg);
|
||||
String providerId = parts[0];
|
||||
String modelId = parts[1];
|
||||
|
||||
ObjectNode provider = root.putObject("provider").putObject(providerId);
|
||||
provider.put("npm", "@ai-sdk/openai-compatible");
|
||||
provider.put("name", providerId + " (bridged)");
|
||||
|
||||
ObjectNode options = provider.putObject("options");
|
||||
options.put("baseURL", openAiBaseUrl(cfg.baseUrl()));
|
||||
// vLLM and friends usually ignore the key, but the AI SDK still requires a non-empty one.
|
||||
String token = resolveEnv(cfg.tokenEnv());
|
||||
options.put("apiKey", (token == null || token.isBlank()) ? "fleetd-local-noauth" : token);
|
||||
|
||||
provider.putObject("models").putObject(modelId).put("name", modelId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Split {@code model:} into its {@code provider} and {@code model} halves. A pinned endpoint
|
||||
* needs both, so a bare model name is rejected loudly rather than silently falling back to the
|
||||
* default gateway — a worker quietly talking to the wrong endpoint is the failure this avoids.
|
||||
*/
|
||||
private static String[] splitModelSelector(FleetConfig.Profile cfg) {
|
||||
String[] parts = splitProviderModel(cfg.model());
|
||||
if (parts == null) {
|
||||
throw new IllegalArgumentException(
|
||||
"profile " + cfg.profile() + " sets baseUrl (a pinned opencode endpoint) so"
|
||||
+ " model: must be \"<provider>/<model>\", e.g."
|
||||
+ " \"local-vllm/deepseek-v4-flash\"; got "
|
||||
+ (cfg.model() == null ? "null" : '"' + cfg.model() + '"'));
|
||||
}
|
||||
return parts;
|
||||
}
|
||||
|
||||
/**
|
||||
* Split {@code model} into its {@code provider} and {@code model} halves, or {@code null} when
|
||||
* it is not in that shape (unset/blank, or no non-trailing {@code /}). Unlike
|
||||
* {@link #splitModelSelector}, non-throwing — callers that only *optionally* need the split
|
||||
* (autoCompactWindow's context-limit application) use this to fall back to a WARN rather than an
|
||||
* exception, since a profile without {@code baseUrl} is not required to name a provider/model.
|
||||
*/
|
||||
private static String[] splitProviderModel(String model) {
|
||||
int slash = model == null ? -1 : model.indexOf('/');
|
||||
if (model == null || model.isBlank() || slash <= 0 || slash == model.length() - 1) {
|
||||
return null;
|
||||
}
|
||||
return new String[]{model.substring(0, slash), model.substring(slash + 1)};
|
||||
}
|
||||
|
||||
/**
|
||||
* Apply the per-profile {@code autoCompactWindow} as opencode's per-model context limit.
|
||||
*
|
||||
* <p>opencode has no absolute "compact at N tokens" knob — its {@code compaction} block only
|
||||
* exposes {@code auto}/{@code prune}/{@code reserved}/{@code tail_turns}/
|
||||
* {@code preserve_recent_tokens} — so the real lever is the model's own
|
||||
* {@code provider.<p>.models.<m>.limit.context}, which bounds the window opencode compacts
|
||||
* <em>within</em> rather than compacting exactly AT it the way Claude Code's {@code --autocompact}
|
||||
* does.
|
||||
*
|
||||
* <p>Uses get-or-create nodes ({@code withObject}) at every level so this MERGES with any provider
|
||||
* block {@link #addCustomProvider} already wrote for a custom-provider (pinned-endpoint) profile —
|
||||
* it must never overwrite that block's {@code npm}/{@code name}/{@code options}. For a gateway
|
||||
* profile (no {@code baseUrl}, so no prior provider block) this writes a partial
|
||||
* {@code provider.<p>.models.<m>.limit} override, which opencode merges over its own built-in
|
||||
* provider definition.
|
||||
*
|
||||
* <p>opencode's {@code limit} schema requires both {@code context} and {@code output}; there is no
|
||||
* independent signal for the latter here, so 16384 is written as a safe default (documented in
|
||||
* {@code fleetd.example.yaml}).
|
||||
*
|
||||
* <p>Silently does nothing when {@code model:} is not in {@code provider/model} form — a warning
|
||||
* for that case is already logged once in {@code buildLaunch}, so this stays quiet rather than
|
||||
* duplicating it.
|
||||
*/
|
||||
private static void applyContextLimit(ObjectNode root, FleetConfig.Profile cfg) {
|
||||
String[] parts = splitProviderModel(cfg.model());
|
||||
if (parts == null) {
|
||||
return;
|
||||
}
|
||||
ObjectNode limit = root.withObject("provider").withObject(parts[0])
|
||||
.withObject("models").withObject(parts[1]).withObject("limit");
|
||||
limit.put("context", cfg.autoCompactWindow());
|
||||
limit.put("output", 16384);
|
||||
}
|
||||
|
||||
/**
|
||||
* The OpenAI-compatible base URL for {@code baseUrl}. A bare {@code host:port} gets {@code /v1}
|
||||
* appended (where these servers put the API); a URL that already carries a path is taken as-is,
|
||||
* so an endpoint mounted somewhere unusual is still reachable.
|
||||
*/
|
||||
private static String openAiBaseUrl(String baseUrl) {
|
||||
String trimmed = baseUrl.trim();
|
||||
while (trimmed.endsWith("/")) {
|
||||
trimmed = trimmed.substring(0, trimmed.length() - 1);
|
||||
}
|
||||
int schemeEnd = trimmed.indexOf("://");
|
||||
String afterScheme = schemeEnd < 0 ? trimmed : trimmed.substring(schemeEnd + 3);
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
}
|
||||
|
||||
/**
|
||||
* A {@link PeerHandle} that delegates everything to the base's worker handle but resolves
|
||||
* {@link #agentSessionId()} lazily through opencode session discovery. Delegate-only, so the
|
||||
* base's id/terminalId/profile semantics (CB-519's host-unique routing key, herdr coordinates)
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String id() {
|
||||
return delegate.id();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String terminalId() {
|
||||
return delegate.terminalId();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String profile() {
|
||||
return delegate.profile();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String sessionName() {
|
||||
return delegate.sessionName();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
}
|
||||
|
||||
@Override
|
||||
public CharterReceipt charterReceipt() {
|
||||
return delegate.charterReceipt();
|
||||
}
|
||||
}
|
||||
|
||||
// --- Agent-returning convenience spawns (used by callers/tests that want the herdr Agent) ---
|
||||
|
||||
/** Spawn a worker for the default profile in the resolved default cwd. */
|
||||
public Agent spawn() {
|
||||
return spawnInternal(null, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile (null → default) in the resolved default cwd. */
|
||||
public Agent spawn(String profileName) {
|
||||
return spawnInternal(profileName, null, null);
|
||||
}
|
||||
|
||||
/** Spawn a worker for a named profile with an explicit requested/caller cwd (CB-112). */
|
||||
public Agent spawn(String profileName, String requestedCwd, String callerCwd) {
|
||||
return spawnInternal(profileName, requestedCwd, callerCwd);
|
||||
}
|
||||
|
||||
// --- capabilities --------------------------------------------------------------------------
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
Set<Capability> caps = EnumSet.of(Capability.MID_TURN_ASK, Capability.WORKTREE,
|
||||
Capability.ORPHAN_REAP, Capability.SESSION_RESUME);
|
||||
if (hasGitTokenProfile()) {
|
||||
caps.add(Capability.SELF_PR);
|
||||
}
|
||||
// Deliberately NOT SESSION_NAME: opencode has no display-name flag, so the bridge's logical
|
||||
// name can't surface in the peer's own UI — declaring the capability would hide that
|
||||
// asymmetry rather than make it honest. For opencode the name lives only in the bridge's
|
||||
// roster (see PeerHandle.sessionName() returning null).
|
||||
return Set.copyOf(caps);
|
||||
}
|
||||
|
||||
/** Whether any configured profile opts into a git-forge token (required for {@link Capability#SELF_PR}). */
|
||||
private boolean hasGitTokenProfile() {
|
||||
return profileConfigs().stream().anyMatch(FleetConfig.Profile::hasGitToken);
|
||||
}
|
||||
|
||||
// --- CB-117 reap predicate (opencode prefix), kept for direct unit testing -----------------
|
||||
|
||||
/**
|
||||
* Whether {@code name} is an opencode bridge worker started by a <em>different</em> process than
|
||||
* {@code currentNonce}. A thin {@code opencode}-prefix binding of
|
||||
* {@link HerdrPeerLauncher#isForeignWorker(String, String, String)}.
|
||||
*/
|
||||
static boolean isForeignWorker(String name, String currentNonce) {
|
||||
return HerdrPeerLauncher.isForeignWorker(NAME_PREFIX, name, currentNonce);
|
||||
}
|
||||
}
|
||||
@@ -1,122 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a bridged worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every bridged worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A bridged worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
*/
|
||||
String sessionIdForDirectory(String directory) {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
} catch (IOException e) {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,40 +0,0 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* The lead-to-lead message channel this daemon speaks, as its callers need it — one lead's own
|
||||
* mailbox: publish to a peer's coord-id, look at what has arrived for me, and ack what I have
|
||||
* delivered.
|
||||
*
|
||||
* <p>Extracted from {@link LeadMailbox} purely as a seam. {@code LeadMailbox} is the one production
|
||||
* implementation and owns a live AMQP connection, so a test that wanted to exercise the routing in
|
||||
* {@code FleetMcp} or the delivery in {@link LeadCoordLoop} would have had to stand up a broker —
|
||||
* which is exactly the kind of test that gets tagged {@code contract} and then does not run. With
|
||||
* this interface both of those are hermetic: they inject a fake channel and assert on what was
|
||||
* published, peeked and acked.
|
||||
*
|
||||
* <p>Note what is <em>not</em> here: {@code drain()} and {@code close()}. Draining is a convenience
|
||||
* over peek+ack that no caller on this seam uses, and closing is the owner's job — {@code Fleetd}
|
||||
* holds the concrete {@link LeadMailbox} for its shutdown hook and hands only this narrower view to
|
||||
* everyone else.
|
||||
*/
|
||||
public interface LeadChannel {
|
||||
|
||||
/**
|
||||
* Send {@code m} to {@code toCoordId}'s mailbox, blocking until the broker confirms it is
|
||||
* durably queued. Throws {@link IllegalStateException} when it is not — unroutable (nobody owns
|
||||
* that coord-id), nacked, or unconfirmed within the implementation's timeout. A caller must
|
||||
* report that as a failed send, never as a delivered one.
|
||||
*/
|
||||
void publish(String toCoordId, LeadMessage m);
|
||||
|
||||
/** Non-destructive FIFO snapshot of the messages held for this daemon's own coord-id. */
|
||||
List<LeadMessage> peek();
|
||||
|
||||
/** Drop {@code msgId} from the held set and ack it on the broker. A no-op if it is not held. */
|
||||
void ack(String msgId);
|
||||
|
||||
/** This daemon's own lead coordination id — the mailbox it owns, and the {@code from} it sends as. */
|
||||
String selfCoordId();
|
||||
}
|
||||
@@ -1,831 +0,0 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.CompletionException;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.locks.ReentrantLock;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* The blocking delegation feature (CB-104): deliver {@code content} into a worker and block until
|
||||
* the worker returns a <em>structured reply</em> via {@code fleet_reply} (the {@link Rendezvous}),
|
||||
* then hand that reply back. Delivery is the {@link Injector}'s job (the background poller sends it
|
||||
* when the worker is injectable); this service never drives the injector or scrapes the terminal —
|
||||
* completion is the worker's explicit reply, not a guess about {@code agent_status}.
|
||||
*
|
||||
* <p>Sends are serialized per session so exactly one reply can be outstanding per worker, which is
|
||||
* what lets a reply map unambiguously to its send (no cross-talk between concurrent callers).
|
||||
*
|
||||
* <p>If the worker never replies within the timeout, the caller gets a typed "still working" /
|
||||
* "queued" outcome — the message may still be mid-flight. A finished-but-unreplied turn is caught
|
||||
* by the CB-106 completion fallback (see {@link Rendezvous#resolveCompletion}).
|
||||
*
|
||||
* <p><strong>Async fire-and-poll (CB-107).</strong> A caller's MCP client caps a blocking call at
|
||||
* ~60s, but a real delegated task runs for minutes. {@link #sendAsync} therefore runs the same
|
||||
* blocking {@link #send} on a background virtual thread and hands back a <em>ticket</em> the caller
|
||||
* polls with {@link #poll}. The blocking and async paths share one code path (and the same per-target
|
||||
* serialization), so async inherits the reply + completion resolution behaviour for free.
|
||||
*/
|
||||
public final class MessageService {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MessageService.class);
|
||||
|
||||
/**
|
||||
* The window a fire-and-poll send waits for resolution — generous, since no caller is blocked on
|
||||
* it; a real delegated task resolves (reply or completion) well within this, and only a genuinely
|
||||
* hung worker rides it out.
|
||||
*/
|
||||
private static final long ASYNC_TIMEOUT_MS = 30 * 60 * 1_000L;
|
||||
|
||||
/**
|
||||
* How long a finished (terminal) ticket is retained for polling before it is pruned. Package-
|
||||
* private (not {@code private}) so a test can advance an injected clock past it deterministically
|
||||
* instead of duplicating the magic number or sleeping for real.
|
||||
*/
|
||||
static final long TICKET_TTL_NANOS = 10 * 60 * 1_000_000_000L;
|
||||
|
||||
/** Outcome of a blocking send. */
|
||||
public enum Outcome {
|
||||
/** The worker called {@code fleet_reply}; {@code text} holds the structured answer. */
|
||||
REPLIED,
|
||||
/**
|
||||
* The worker's delegated turn finished without a {@code fleet_reply} (CB-106 fallback);
|
||||
* {@code text} is the scraped transcript tail rather than a structured answer.
|
||||
*/
|
||||
COMPLETED_UNREPLIED,
|
||||
/**
|
||||
* The worker ran the turn then wedged in an unrecoverable state (CB-109); {@code text} is the
|
||||
* failure context (e.g. the error screen). Terminal, but not a successful completion.
|
||||
*/
|
||||
WORKER_FAILED,
|
||||
/**
|
||||
* The turn finished without a {@code fleet_reply} and the scrape matched the backend's
|
||||
* configured usage-limit refusal pattern (CB-578 stage A); {@code text} is the reason,
|
||||
* carrying the matched line. The worker's pane is healthy — only its account is refusing —
|
||||
* so this is never reported as a completed reply, and is kept distinct from
|
||||
* {@link #WORKER_FAILED} (a wedged worker) and a session simply going {@code GONE}.
|
||||
*/
|
||||
BACKEND_EXHAUSTED,
|
||||
/**
|
||||
* The worker paused mid-turn to ask the primary a question (CB-205); {@code text} is the
|
||||
* question and {@code turnId} correlates the answer. Not terminal — the primary answers with
|
||||
* {@link #answer(String, String, long)} and the turn resumes.
|
||||
*/
|
||||
QUESTION,
|
||||
/** Timed out after the message was delivered — the worker is still working. */
|
||||
TIMED_OUT_WORKING,
|
||||
/** Timed out before delivery — the message is still queued for the worker. */
|
||||
TIMED_OUT_QUEUED,
|
||||
/** Another send to this session was in flight for the whole window. */
|
||||
BUSY,
|
||||
/**
|
||||
* An answer ({@link #answer(String, String, long)}) referenced a {@code turnId} that is no
|
||||
* longer open — the worker's {@code fleet_ask} already timed out or was answered.
|
||||
*/
|
||||
STALE_TURN
|
||||
}
|
||||
|
||||
/**
|
||||
* @param outcome how the send ended (or paused)
|
||||
* @param text the worker's answer when {@link #completed()} (a structured {@code fleet_reply}
|
||||
* for {@link Outcome#REPLIED}, a scraped transcript tail for
|
||||
* {@link Outcome#COMPLETED_UNREPLIED}), or the question for {@link Outcome#QUESTION},
|
||||
* else {@code null}
|
||||
* @param turnId correlation id for a {@link Outcome#QUESTION} (answered via
|
||||
* {@link #answer(String, String, long)}), else {@code null}
|
||||
*/
|
||||
public record Reply(Outcome outcome, String text, String turnId) {
|
||||
/** A reply with no correlation id (the common terminal outcomes). */
|
||||
public Reply(Outcome outcome, String text) {
|
||||
this(outcome, text, null);
|
||||
}
|
||||
|
||||
/** Whether the worker's turn actually finished with an answer (replied or scraped). */
|
||||
public boolean completed() {
|
||||
return outcome == Outcome.REPLIED || outcome == Outcome.COMPLETED_UNREPLIED;
|
||||
}
|
||||
}
|
||||
|
||||
/** How a worker's {@code fleet_ask} (CB-205) resolved. */
|
||||
public enum AskOutcome {
|
||||
/** The primary answered; {@link AskResult#answer} carries it. */
|
||||
ANSWERED,
|
||||
/** No delegation was open to surface the question to — the worker has no one to ask. */
|
||||
NO_WAITER,
|
||||
/** The primary did not answer within the window. */
|
||||
TIMED_OUT
|
||||
}
|
||||
|
||||
/** The outcome of a worker's {@code fleet_ask}: how it resolved and (if answered) the answer. */
|
||||
public record AskResult(AskOutcome outcome, String answer) {
|
||||
}
|
||||
|
||||
/** Lifecycle phase of an async delegation ticket. */
|
||||
public enum Phase {
|
||||
/** Delegated and in flight — queued for the worker or being worked. */
|
||||
PENDING,
|
||||
/** The worker is paused in {@code fleet_ask}; {@link TaskView#reply} and {@link TaskView#turnId} identify it. */
|
||||
ASKING,
|
||||
/** The worker's turn finished; {@link TaskView#reply} holds the answer. */
|
||||
DONE,
|
||||
/** The delegation could not complete (timed out, worker gone, or busy). */
|
||||
FAILED
|
||||
}
|
||||
|
||||
/**
|
||||
* A poll snapshot of an async delegation.
|
||||
*
|
||||
* @param reply the answer when {@link #phase} is {@link Phase#DONE}, or the question when
|
||||
* {@link #phase} is {@link Phase#ASKING}; otherwise {@code null}
|
||||
* @param replySource {@code "reply"} (structured {@code fleet_reply}) or {@code "transcript"}
|
||||
* (completion scrape) when {@link Phase#DONE}, else {@code null}
|
||||
* @param detail a human note (live worker status while pending, ask state, or failure reason)
|
||||
* @param turnId correlation id for an {@link Phase#ASKING} ticket, else {@code null}
|
||||
*/
|
||||
public record TaskView(String ticket, Phase phase, String reply, String replySource, String detail,
|
||||
String turnId) {
|
||||
}
|
||||
|
||||
/** An in-flight or finished async delegation, keyed by its ticket. */
|
||||
private static final class Task {
|
||||
private final String ticket;
|
||||
private final String target;
|
||||
private final CompletableFuture<Reply> future = new CompletableFuture<>();
|
||||
private final long createdNanos;
|
||||
private volatile Reply question;
|
||||
private volatile String turnId;
|
||||
|
||||
private Task(String ticket, String target, long createdNanos) {
|
||||
this.ticket = ticket;
|
||||
this.target = target;
|
||||
this.createdNanos = createdNanos;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker session's currently-open {@code fleet_ask} question, surfaced so {@code fleet_status}
|
||||
* can show it without the caller needing the ticket first (CB-582). Only covers async
|
||||
* (fire-and-poll) delegations, which track the question on their {@link Task}; a blocking
|
||||
* ({@code wait:true}) send already hands the question straight back to its own caller, so there is
|
||||
* nothing hidden left for {@code fleet_status} to surface in that case.
|
||||
*/
|
||||
public record PendingAsk(String ticket, String question, String turnId) {
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Injector injector;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ReplyInbox inbox;
|
||||
private final ReplyPushLoop pushLoop;
|
||||
private final Metrics metrics; // CB-502: nullable — no registry in unit tests
|
||||
// CB-588: injectable so pruneTerminalTickets' 10-minute TICKET_TTL_NANOS can be exercised in a
|
||||
// test without a real wait — same seam SessionManager already uses for its idle reaper (nowNanos).
|
||||
private final LongSupplier nowNanos;
|
||||
private final ConcurrentHashMap<String, ReentrantLock> sessionLocks = new ConcurrentHashMap<>();
|
||||
private final ConcurrentHashMap<String, Task> tasks = new ConcurrentHashMap<>();
|
||||
/** Async task that owns each exact forward rendezvous waiter. */
|
||||
private final ConcurrentHashMap<CompletableFuture<Rendezvous.Resolution>, Task> asyncTasksByWaiter =
|
||||
new ConcurrentHashMap<>();
|
||||
/** Async tickets paused on a specific {@code fleet_ask} turn. */
|
||||
private final ConcurrentHashMap<String, Task> asyncTasksByTurn = new ConcurrentHashMap<>();
|
||||
private final AtomicLong ticketSeq = new AtomicLong();
|
||||
private final ExecutorService asyncExecutor = Executors.newThreadPerTaskExecutor(
|
||||
Thread.ofVirtual().name("bridge-async-", 0).factory());
|
||||
|
||||
/**
|
||||
* Create with an explicit {@link ReplyInbox} and optional {@link ReplyPushLoop}.
|
||||
*
|
||||
* @param pushLoop nullable — when non-null, the push loop is notified on the no-waiter reply
|
||||
* branch ({@link #reply}) so it can nudge the primary to drain the inbox,
|
||||
* (CB-588) whenever an async ticket started by {@link #sendAsync} reaches a
|
||||
* terminal phase, whenever {@link #poll} hands a terminal ticket to its caller,
|
||||
* and (CB-582) whenever an async ticket's worker pauses mid-turn in
|
||||
* {@code fleet_ask} or that pause ends (answered or lapsed)
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, with a metric registry (CB-502). Instrumenting here rather than at the REST and MCP
|
||||
* edges means both surfaces are counted by one piece of code and cannot drift.
|
||||
*
|
||||
* @param metrics nullable — when null, nothing is recorded
|
||||
*/
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous,
|
||||
ReplyInbox inbox, ReplyPushLoop pushLoop, Metrics metrics) {
|
||||
this(agents, injector, rendezvous, inbox, pushLoop, metrics, System::nanoTime);
|
||||
}
|
||||
|
||||
/** Test constructor with an injectable clock (CB-588: exercise the ticket-prune TTL without a real wait). */
|
||||
MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox,
|
||||
ReplyPushLoop pushLoop, Metrics metrics, LongSupplier nowNanos) {
|
||||
this.agents = agents;
|
||||
this.injector = injector;
|
||||
this.rendezvous = rendezvous;
|
||||
this.inbox = inbox;
|
||||
this.pushLoop = pushLoop;
|
||||
this.metrics = metrics;
|
||||
this.nowNanos = nowNanos;
|
||||
}
|
||||
|
||||
/** Create with an explicit {@link ReplyInbox} and no push loop. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous, ReplyInbox inbox) {
|
||||
this(agents, injector, rendezvous, inbox, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible constructor that uses a default {@link InMemoryReplyInbox}. */
|
||||
public MessageService(AgentControl agents, Injector injector, Rendezvous rendezvous) {
|
||||
this(agents, injector, rendezvous, new InMemoryReplyInbox());
|
||||
}
|
||||
|
||||
/** Current lifecycle status of a worker (the {@code GET /sessions/{id}/status} surface). */
|
||||
public AgentStatus status(String target) {
|
||||
return agents.status(target);
|
||||
}
|
||||
|
||||
/** Read-only delegation fact for fleet views. */
|
||||
public boolean hasAcceptedDelivery(String target) {
|
||||
return rendezvous.isWaiting(target);
|
||||
}
|
||||
|
||||
/** Read-only inbox fact for fleet views. */
|
||||
public boolean hasInboxMessage(String target) {
|
||||
return !inbox.peek(target).isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Route a worker's explicit {@code fleet_reply}: resolve an open send, or queue it in the
|
||||
* inbox if no send is currently open. Unlike the bare {@link Rendezvous#resolve}, a no-waiter
|
||||
* result is <em>not</em> a failure — the reply is held for later drain.
|
||||
*
|
||||
* <p><strong>Do NOT use this for mid-turn questions.</strong> {@code fleet_ask} /
|
||||
* {@link Rendezvous#resolveQuestion} must keep today's {@code NO_WAITER} behaviour — questions
|
||||
* are interactive and must never be queued.
|
||||
*
|
||||
* @return always {@code true} — the reply either resolved a live send or was queued
|
||||
*/
|
||||
public boolean reply(String session, String content) {
|
||||
if (rendezvous.resolve(session, content)) {
|
||||
count(FleetMetrics.REPLIES, "path", "rendezvous");
|
||||
return true; // a live send took it — unchanged fast path
|
||||
}
|
||||
inbox.publish(session, UUID.randomUUID().toString(), content);
|
||||
// A rising inbox share is the signal CB-307 exists to make visible: the worker finished but
|
||||
// nobody was waiting, so delivery now depends on the push loop and a drain.
|
||||
count(FleetMetrics.REPLIES, "path", "inbox");
|
||||
if (pushLoop != null) {
|
||||
pushLoop.onReplyQueued(session);
|
||||
}
|
||||
return true; // held, not lost
|
||||
}
|
||||
|
||||
/** Record a counter sample when a registry is wired; a no-op in unit tests. */
|
||||
private void count(String name, String... labels) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(name, labels);
|
||||
}
|
||||
}
|
||||
|
||||
/** Count a send's terminal outcome and pass the reply through unchanged. */
|
||||
private Reply recorded(Reply r) {
|
||||
String label = sendOutcomeLabel(r.outcome());
|
||||
if (label != null) {
|
||||
count(FleetMetrics.SENDS, "outcome", label);
|
||||
}
|
||||
return r;
|
||||
}
|
||||
|
||||
/** Map a terminal send outcome to its metric label, or {@code null} for non-terminal ones. */
|
||||
private static String sendOutcomeLabel(Outcome o) {
|
||||
return switch (o) {
|
||||
case REPLIED -> "replied";
|
||||
case COMPLETED_UNREPLIED -> "completion_fallback";
|
||||
case TIMED_OUT_WORKING, TIMED_OUT_QUEUED, BUSY -> "timeout";
|
||||
case WORKER_FAILED -> "failed";
|
||||
case BACKEND_EXHAUSTED -> "backend_exhausted";
|
||||
case STALE_TURN, QUESTION -> null; // not a completed delegation
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Abandon any send still waiting on {@code target} because its session has gone away (CB-516).
|
||||
*
|
||||
* <p>Without this, tearing a worker down left its rendezvous waiter open: a blocking
|
||||
* {@code fleet_send} kept blocking, and an async one kept reporting {@code PENDING} until
|
||||
* {@link #ASYNC_TIMEOUT_MS} — thirty minutes — even though the worker provably no longer
|
||||
* existed and the delegation could never complete. Worse, {@code poll} already had the evidence
|
||||
* (it calls {@code liveStatus} to build its detail string and gets back {@code "unknown"}) and
|
||||
* reported {@code PENDING} anyway.
|
||||
*
|
||||
* <p>Resolving the waiter as a failure — rather than letting it time out — also means the
|
||||
* outcome is counted, so a torn-down delegation stops being invisible to {@code /metrics}.
|
||||
*
|
||||
* @return true if a live waiter was failed
|
||||
*/
|
||||
public boolean abandon(String target, String reason) {
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(target);
|
||||
boolean failed = waiter != null && !waiter.isDone() && rendezvous.resolveFailure(waiter, reason);
|
||||
boolean asyncFailed = false;
|
||||
for (Task task : tasks.values()) {
|
||||
if (target.equals(task.target) && task.question == null
|
||||
&& task.future.complete(new Reply(Outcome.WORKER_FAILED, reason))) {
|
||||
asyncFailed = true;
|
||||
}
|
||||
}
|
||||
if (failed) {
|
||||
log.warn("abandoning the blocked send to {}: {}", target, reason);
|
||||
}
|
||||
return failed || asyncFailed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Acknowledge a specific reply by {@code msgId} for {@code target}. Removes it from the inbox
|
||||
* so that a subsequent drain or peek no longer returns it.
|
||||
*/
|
||||
public void ackReply(String target, String msgId) {
|
||||
inbox.ack(target, msgId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain (peek + ack) all pending inbox replies for {@code target}.
|
||||
*
|
||||
* <p><strong>The ack happens here, before the caller has the messages</strong> — before the MCP
|
||||
* or REST response carrying them has been written, and long before the client has processed
|
||||
* them. That ordering is what the two adapters disagree about, so do not read this method as
|
||||
* "at-least-once" without qualifying which inbox is behind it (CB-529):
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code InMemoryReplyInbox} — the ack only drops an entry from a local map. The messages
|
||||
* are already in the returned list, so nothing can be lost after this point.
|
||||
* <li>{@code AmqpReplyInbox} — the ack is a broker-side {@code basicAck}. Once it lands the
|
||||
* broker has forgotten the message. If the daemon dies while writing the response, the
|
||||
* reply is gone from the broker <em>and</em> the client never received it. Re-polling
|
||||
* cannot recover it, because there is nothing left to re-deliver.
|
||||
* </ul>
|
||||
*
|
||||
* <p>So the loss window is the response write, and it is a genuine loss rather than a
|
||||
* redelivery. This is accepted, not overlooked: the alternative — ack on the next poll — turns
|
||||
* every normal drain into a double delivery, which costs more than the window it closes. A
|
||||
* caller that needs certainty re-polls; that is idempotent for every case except this one.
|
||||
*
|
||||
* <p>Any change here must be checked against <em>both</em> adapters. The previous version of
|
||||
* this javadoc claimed "the ack is local", which was true when only the in-memory inbox existed
|
||||
* and silently became false when the AMQP adapter landed.
|
||||
*
|
||||
* @return the drained messages, newest last (FIFO); empty list if none
|
||||
*/
|
||||
public List<ReplyInbox.InboxMessage> drainReplies(String target) {
|
||||
var messages = inbox.peek(target);
|
||||
for (var msg : messages) {
|
||||
inbox.ack(target, msg.msgId());
|
||||
}
|
||||
return messages;
|
||||
}
|
||||
|
||||
/**
|
||||
* Deliver {@code content} to {@code target} (a herdr {@code terminal_id}) and block until the
|
||||
* worker replies via {@link Rendezvous} or {@code timeoutMillis} elapses.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis) {
|
||||
return send(target, content, timeoutMillis, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #send(String, String, long)}, but with an accepted-delivery hook.
|
||||
*
|
||||
* <p>{@code onAccepted} is invoked exactly once, once this send has won {@code target}'s send
|
||||
* lock and so become the <em>accepted target turn</em> — it runs <em>before</em> delivery is
|
||||
* queued, so a throwing hook fails the send cleanly (the waiter it already opened is closed and
|
||||
* nothing is left queued). It is <em>not</em> invoked when the send is {@link Outcome#BUSY}
|
||||
* (lock never taken). A caller uses this to record that <em>it</em> now owns the delegation's
|
||||
* reply routing (CB-548: {@code PrimaryRegistry} delegator ownership) — recording only on
|
||||
* acceptance means a concurrent sender that times out {@code BUSY} can never steal ownership it
|
||||
* never earned. {@code null} disables the hook.
|
||||
*/
|
||||
public Reply send(String target, String content, long timeoutMillis, Runnable onAccepted) {
|
||||
return send(target, content, timeoutMillis, onAccepted, null);
|
||||
}
|
||||
|
||||
/** Run a send, optionally stopping an async task that teardown already failed before acceptance. */
|
||||
private Reply send(String target, String content, long timeoutMillis, Runnable onAccepted, Task task) {
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(target, _ -> new ReentrantLock());
|
||||
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null); // another send held the session the whole window
|
||||
}
|
||||
try {
|
||||
if (task != null && task.future.isDone()) {
|
||||
return task.future.getNow(null);
|
||||
}
|
||||
if (hasAsyncQuestion(target)) {
|
||||
return new Reply(Outcome.BUSY, null); // the worker's current turn is paused for its lead
|
||||
}
|
||||
// Open the waiter BEFORE queueing delivery (CB-548). A fast reply — the worker already
|
||||
// injectable the instant we enqueue — otherwise arrives before the waiter is registered
|
||||
// and orphans into the inbox while this send blocks to the timeout (the enqueue-before-
|
||||
// open race). Opening first also means a throwing onAccepted (fired before enqueue) or an
|
||||
// enqueue failure is safely closed by the finally below: nothing is left queued, and the
|
||||
// failed send leaves no stale waiter behind.
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(target);
|
||||
try {
|
||||
if (task != null) {
|
||||
asyncTasksByWaiter.put(reply, task);
|
||||
}
|
||||
TurnToken token = new TurnToken(target, reply);
|
||||
// The send has won the lock; the accepted-delivery hook records delegator ownership
|
||||
// here (CB-548). It runs BEFORE enqueue so a throwing hook — onAccepted is now a
|
||||
// public callback — fails the send without queuing a message that would orphan.
|
||||
if (onAccepted != null) {
|
||||
onAccepted.run();
|
||||
}
|
||||
CompletableFuture<Void> delivered = injector.enqueue(target, content, token);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
return recorded(new Reply(outcomeOf(r.kind()), r.text(), r.turnId()));
|
||||
} catch (TimeoutException e) {
|
||||
boolean wasDelivered = delivered.isDone() && !delivered.isCompletedExceptionally();
|
||||
log.debug("send to {} timed out (delivered={})", target, wasDelivered);
|
||||
return recorded(new Reply(
|
||||
wasDelivered ? Outcome.TIMED_OUT_WORKING : Outcome.TIMED_OUT_QUEUED, null));
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + target, e);
|
||||
}
|
||||
} finally {
|
||||
asyncTasksByWaiter.remove(reply);
|
||||
rendezvous.close(target, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A worker's mid-turn question (CB-205 reverse rendezvous): surface {@code question} to the
|
||||
* primary by resolving its open blocking {@code fleet_send}, then block this (worker) call until
|
||||
* the primary answers via {@link #answer} or {@code timeoutMillis} elapses. Identity is the
|
||||
* worker's own session — it does not address the primary.
|
||||
*
|
||||
* <p>Returns {@link AskOutcome#NO_WAITER} when no delegation is open to surface the question to
|
||||
* (nothing to answer it), {@link AskOutcome#ANSWERED} with the primary's answer, or
|
||||
* {@link AskOutcome#TIMED_OUT} if the primary stayed silent. The worker resumes its turn either
|
||||
* way — an answered ask hands back the answer; an unanswered one leaves it to proceed alone.
|
||||
*/
|
||||
public AskResult ask(String workerSession, String question, long timeoutMillis) {
|
||||
Rendezvous.AskTicket ticket = rendezvous.openAsk(workerSession);
|
||||
// Only the freshly-opening caller surfaces the question; a coalesced duplicate simply blocks on
|
||||
// the shared answer future that the fresh owner is already responsible for.
|
||||
if (ticket.fresh()) {
|
||||
// Register the reverse waiter first, then surface the question — so the answer, which can
|
||||
// arrive the instant the primary reacts, always finds an open waiter to resolve.
|
||||
CompletableFuture<Rendezvous.Resolution> waiter = rendezvous.currentWaiter(workerSession);
|
||||
Task task = markAsyncQuestion(waiter, question, ticket.turnId());
|
||||
if (!rendezvous.resolveQuestion(workerSession, question, ticket.turnId())) {
|
||||
if (task != null) {
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
}
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
return new AskResult(AskOutcome.NO_WAITER, null); // no primary is blocked on this worker
|
||||
}
|
||||
// CB-582: the question just became visible via fleet_poll (Phase.ASKING) for an async
|
||||
// (wait:false) delegation — nudge the lead's own pane the same way a terminal ticket does
|
||||
// (CB-588), since the lead's normal poll cadence is minutes away and the reverse-rendezvous
|
||||
// window (~55s, see FleetMcp/FleetApp) is far shorter. A blocking (wait:true) send has
|
||||
// no Task and gets the question directly in its own reply, so task == null there — nothing
|
||||
// to nudge.
|
||||
if (task != null && pushLoop != null) {
|
||||
pushLoop.onQuestionOpened(task.ticket, workerSession, ticket.turnId(), question);
|
||||
}
|
||||
}
|
||||
try {
|
||||
String answer = ticket.answer().get(timeoutMillis, TimeUnit.MILLISECONDS);
|
||||
return new AskResult(AskOutcome.ANSWERED, answer);
|
||||
} catch (TimeoutException e) {
|
||||
log.debug("fleet_ask from {} went unanswered in {}ms", workerSession, timeoutMillis);
|
||||
clearAsyncQuestion(ticket.turnId(), true);
|
||||
return new AskResult(AskOutcome.TIMED_OUT, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the primary's answer for " + workerSession, e);
|
||||
} finally {
|
||||
// Only the fresh owner tears down the shared turn; a duplicate must leave it open.
|
||||
if (ticket.fresh()) {
|
||||
rendezvous.closeAsk(ticket.turnId());
|
||||
// CB-582: tear the push loop's copy down at the same point, not only on the three
|
||||
// paths that call clearAsyncQuestion. The answer future can complete exceptionally
|
||||
// (ExecutionException) or the thread be interrupted, and both leave this method by
|
||||
// throwing — the question would stay pending forever, keep being named in nudges
|
||||
// until its own cap, and never be removed from the map. Already-closed is a no-op.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.questionClosed(ticket.turnId());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The primary's answer to a worker's {@code fleet_ask} (CB-205): resolve the worker's blocked
|
||||
* question identified by {@code turnId}, then — like a fresh {@link #send} — block for the worker's
|
||||
* eventual {@code fleet_reply} as it finishes the resumed turn. The worker session is derived from
|
||||
* {@code turnId}, never a caller argument.
|
||||
*
|
||||
* <p>Unlike {@link #send} this does not re-inject through the {@link Injector}: the worker is
|
||||
* mid-turn (already picked up), so the answer flows back through its own open {@code fleet_ask}
|
||||
* call, not a new status-gated delivery. The forward waiter is opened <em>before</em> the worker
|
||||
* is unblocked so a reply that lands the instant it resumes is not lost.
|
||||
*/
|
||||
public Reply answer(String turnId, String content, long timeoutMillis) {
|
||||
String workerSession = rendezvous.askSession(turnId);
|
||||
if (workerSession == null) {
|
||||
return new Reply(Outcome.STALE_TURN, null); // the ask lapsed (timed out or already answered)
|
||||
}
|
||||
long deadlineNanos = System.nanoTime() + timeoutMillis * 1_000_000L;
|
||||
ReentrantLock lock = sessionLocks.computeIfAbsent(workerSession, _ -> new ReentrantLock());
|
||||
if (!tryLock(lock, remainingMillis(deadlineNanos))) {
|
||||
return new Reply(Outcome.BUSY, null);
|
||||
}
|
||||
try {
|
||||
CompletableFuture<Rendezvous.Resolution> reply = rendezvous.open(workerSession);
|
||||
if (!rendezvous.answerAsk(turnId, content)) {
|
||||
rendezvous.close(workerSession, reply);
|
||||
return new Reply(Outcome.STALE_TURN, null); // lapsed between the lookup and the unblock
|
||||
}
|
||||
clearAsyncQuestion(turnId, false);
|
||||
try {
|
||||
Rendezvous.Resolution r = reply.get(remainingMillis(deadlineNanos), TimeUnit.MILLISECONDS);
|
||||
Reply result = new Reply(outcomeOf(r.kind()), r.text(), r.turnId());
|
||||
finishAsyncTask(turnId, result);
|
||||
return result;
|
||||
} catch (TimeoutException e) {
|
||||
// The worker resumed but hasn't replied yet — no completion fallback arms an answered
|
||||
// turn (it never re-entered the injector), so a silent worker rides out the window.
|
||||
return new Reply(Outcome.TIMED_OUT_WORKING, null);
|
||||
} catch (ExecutionException e) {
|
||||
Throwable cause = e.getCause();
|
||||
throw cause instanceof RuntimeException re ? re : new IllegalStateException(cause);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting reply from " + workerSession, e);
|
||||
} finally {
|
||||
rendezvous.close(workerSession, reply);
|
||||
}
|
||||
} finally {
|
||||
lock.unlock();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Fire-and-poll variant of {@link #send}: deliver {@code content} to {@code target} on a
|
||||
* background virtual thread and return immediately with a ticket to {@link #poll}. This is how a
|
||||
* long task is delegated without tripping the caller's MCP client call timeout.
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content) {
|
||||
return sendAsync(target, content, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #sendAsync(String, String)}, with the accepted-delivery hook of
|
||||
* {@link #send(String, String, long, Runnable)} — the running {@code send} invokes {@code onAccepted}
|
||||
* the moment it becomes the accepted target turn, so async flooding records delegator ownership
|
||||
* exactly as the blocking path does (CB-548).
|
||||
*
|
||||
* @return the ticket to poll for the eventual result
|
||||
*/
|
||||
public String sendAsync(String target, String content, Runnable onAccepted) {
|
||||
String ticket = "task-" + ticketSeq.incrementAndGet();
|
||||
Task task = new Task(ticket, target, nowNanos.getAsLong());
|
||||
tasks.put(ticket, task);
|
||||
if (pushLoop != null) {
|
||||
// CB-588: task.future only ever completes on a terminal phase (DONE or a failure) — a
|
||||
// worker paused in fleet_ask leaves it running, per finishAsyncTask's own contract — so
|
||||
// this fires exactly once, from whichever path completes it: finishAsyncTask(task, result)
|
||||
// below on any non-QUESTION outcome of send() — a worker's fleet_reply, the CB-106
|
||||
// completion fallback, a CB-109 wedge, TIMED_OUT, BUSY, or BACKEND_EXHAUSTED — the same
|
||||
// finishAsyncTask reached via answer()'s finishAsyncTask(turnId, result) once a QUESTION
|
||||
// is resolved, completeExceptionally(t) just below when send() itself throws, or a CB-516
|
||||
// abandon() on teardown. Without this, MessageService.reply's rendezvous fast path (the
|
||||
// one an async ticket always takes) never told the push loop anything happened — see the
|
||||
// class javadoc on sendAsync/CB-107.
|
||||
task.future.whenComplete((reply, ex) -> {
|
||||
boolean failed = ex != null || reply == null || !reply.completed();
|
||||
pushLoop.onTicketTerminal(ticket, target, failed);
|
||||
});
|
||||
}
|
||||
asyncExecutor.submit(() -> {
|
||||
try {
|
||||
Reply result = send(target, content, ASYNC_TIMEOUT_MS, onAccepted, task);
|
||||
if (result.outcome() == Outcome.QUESTION) {
|
||||
// Keep the accepted owner until answer() finishes it. markAsyncQuestion may run
|
||||
// just after resolveQuestion wakes this thread.
|
||||
} else {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
} catch (Throwable t) {
|
||||
task.future.completeExceptionally(t);
|
||||
}
|
||||
});
|
||||
pruneTerminalTickets();
|
||||
log.debug("async send {} -> {}", ticket, target);
|
||||
return ticket;
|
||||
}
|
||||
|
||||
/**
|
||||
* Snapshot the state of an async delegation. Returns {@code null} for an unknown/expired ticket;
|
||||
* otherwise a {@link Phase#PENDING} view (with the live worker status as detail), a
|
||||
* {@link Phase#DONE} view carrying the reply, or a {@link Phase#FAILED} view with the reason.
|
||||
*/
|
||||
public TaskView poll(String ticket) {
|
||||
Task task = tasks.get(ticket);
|
||||
if (task == null) {
|
||||
return null;
|
||||
}
|
||||
CompletableFuture<Reply> f = task.future;
|
||||
if (!f.isDone()) {
|
||||
Reply question = task.question;
|
||||
if (question != null) {
|
||||
return new TaskView(ticket, Phase.ASKING, question.text(), null,
|
||||
"worker is waiting for your answer", question.turnId());
|
||||
}
|
||||
return new TaskView(ticket, Phase.PENDING, null, null, "worker " + liveStatus(task.target), null);
|
||||
}
|
||||
// CB-588: the ticket is terminal and being handed to the caller right here — tell the push
|
||||
// loop it is collected so a later tick's nudge never names a ticket the lead already has.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.ticketCollected(ticket);
|
||||
}
|
||||
Reply r;
|
||||
try {
|
||||
r = f.getNow(null);
|
||||
} catch (CompletionException | java.util.concurrent.CancellationException e) {
|
||||
Throwable cause = (e instanceof CompletionException ce && ce.getCause() != null) ? ce.getCause() : e;
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, cause.getMessage(), null);
|
||||
}
|
||||
if (r.completed()) {
|
||||
String source = r.outcome() == Outcome.REPLIED ? "reply" : "transcript";
|
||||
return new TaskView(ticket, Phase.DONE, r.text(), source, null, null);
|
||||
}
|
||||
// A wedged worker (CB-109) or a backend-exhausted classification (CB-578 stage A) carries
|
||||
// the real cause as its reason; the timeout/busy outcomes carry none, so fall back to the
|
||||
// outcome name.
|
||||
boolean carriesReason = r.outcome() == Outcome.WORKER_FAILED || r.outcome() == Outcome.BACKEND_EXHAUSTED;
|
||||
String detail = carriesReason && r.text() != null
|
||||
? r.text()
|
||||
: "no reply — " + r.outcome().name().toLowerCase();
|
||||
return new TaskView(ticket, Phase.FAILED, null, null, detail, null);
|
||||
}
|
||||
|
||||
/** Best-effort live worker status for a pending poll; never throws (a lookup error is just noise). */
|
||||
private String liveStatus(String target) {
|
||||
try {
|
||||
return agents.status(target).name().toLowerCase();
|
||||
} catch (RuntimeException e) {
|
||||
return "unknown";
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop finished tickets older than the TTL so {@link #tasks} cannot grow without bound.
|
||||
*
|
||||
* <p>{@code tasks} is the sole authority on whether a ticket still exists — {@link #poll} returns
|
||||
* {@code null} the instant a ticket is gone from here, before it ever reaches the terminal branch
|
||||
* that calls {@link ReplyPushLoop#ticketCollected}. Without telling the push loop about a prune
|
||||
* too, its own {@code pendingTickets} entry would outlive the ticket it names: an unpolled ticket
|
||||
* (or one the reminder cap already gave up on) is pruned here but never collected there, so it
|
||||
* lingers in {@code pendingTickets} forever and rides along on every later nudge to the same lead
|
||||
* — naming a ticket {@code fleet_poll} can no longer find (CB-588 follow-up).
|
||||
*/
|
||||
private void pruneTerminalTickets() {
|
||||
long cutoff = nowNanos.getAsLong() - TICKET_TTL_NANOS;
|
||||
tasks.entrySet().removeIf(e -> {
|
||||
Task t = e.getValue();
|
||||
boolean expired = t.future.isDone() && t.createdNanos < cutoff;
|
||||
if (expired && pushLoop != null) {
|
||||
pushLoop.ticketCollected(e.getKey());
|
||||
}
|
||||
return expired;
|
||||
});
|
||||
}
|
||||
|
||||
/** Record the active question for an async ticket; blocking sends have no entry and stay unchanged. */
|
||||
private Task markAsyncQuestion(CompletableFuture<Rendezvous.Resolution> waiter, String text, String turnId) {
|
||||
Task task = waiter == null ? null : asyncTasksByWaiter.get(waiter);
|
||||
if (task != null) {
|
||||
task.question = new Reply(Outcome.QUESTION, text, turnId);
|
||||
task.turnId = turnId;
|
||||
asyncTasksByTurn.put(turnId, task);
|
||||
}
|
||||
return task;
|
||||
}
|
||||
|
||||
/** Clear an answered or lapsed question, but only when it matches the ticket's current turn. */
|
||||
private void clearAsyncQuestion(String turnId, boolean forgetTurn) {
|
||||
// CB-582: tell the push loop first — like ticketCollected, a removal for a turnId it never
|
||||
// nudged about (or already dropped) is a harmless no-op, so this is safe to call unconditionally
|
||||
// rather than threading the guard below through it.
|
||||
if (pushLoop != null) {
|
||||
pushLoop.questionClosed(turnId);
|
||||
}
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null && turnId.equals(task.turnId)) {
|
||||
task.question = null;
|
||||
if (forgetTurn) {
|
||||
asyncTasksByTurn.remove(turnId, task);
|
||||
task.turnId = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete and detach an async ticket after its worker's actual terminal reply. */
|
||||
private void finishAsyncTask(Task task, Reply result) {
|
||||
task.future.complete(result);
|
||||
if (task.turnId != null) {
|
||||
asyncTasksByTurn.remove(task.turnId, task);
|
||||
}
|
||||
}
|
||||
|
||||
/** Complete the async ticket correlated to a specific answered turn. */
|
||||
private void finishAsyncTask(String turnId, Reply result) {
|
||||
Task task = asyncTasksByTurn.get(turnId);
|
||||
if (task != null) {
|
||||
finishAsyncTask(task, result);
|
||||
}
|
||||
}
|
||||
|
||||
/** A new send must not open a waiter while an async ticket owns this worker's paused turn. */
|
||||
private boolean hasAsyncQuestion(String target) {
|
||||
return asyncTasksByTurn.values().stream().anyMatch(task -> target.equals(task.target));
|
||||
}
|
||||
|
||||
/**
|
||||
* The question {@code workerSession} is currently paused on via {@code fleet_ask}, if any
|
||||
* (CB-582) — {@code fleet_status} uses this to show a pending question without the caller
|
||||
* needing the ticket. {@code null} when the session has no open async question (including a
|
||||
* session mid a <em>blocking</em> {@code fleet_ask}, which has no {@link Task} to look up — see
|
||||
* {@link PendingAsk}).
|
||||
*/
|
||||
public PendingAsk pendingAsk(String workerSession) {
|
||||
for (Task task : tasks.values()) {
|
||||
Reply q = task.question;
|
||||
if (q != null && workerSession.equals(task.target)) {
|
||||
return new PendingAsk(task.ticket, q.text(), q.turnId());
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/** Release the async executor. */
|
||||
public void close() {
|
||||
asyncExecutor.shutdown();
|
||||
}
|
||||
|
||||
/** Map a rendezvous {@link Rendezvous.Kind} onto its send {@link Outcome} (shared by send/answer). */
|
||||
private static Outcome outcomeOf(Rendezvous.Kind kind) {
|
||||
return switch (kind) {
|
||||
case REPLY -> Outcome.REPLIED;
|
||||
case COMPLETION -> Outcome.COMPLETED_UNREPLIED;
|
||||
case FAILED -> Outcome.WORKER_FAILED;
|
||||
case BACKEND_EXHAUSTED -> Outcome.BACKEND_EXHAUSTED;
|
||||
case QUESTION -> Outcome.QUESTION;
|
||||
};
|
||||
}
|
||||
|
||||
private static boolean tryLock(ReentrantLock lock, long millis) {
|
||||
try {
|
||||
return lock.tryLock(Math.max(0, millis), TimeUnit.MILLISECONDS);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
throw new IllegalStateException("interrupted awaiting the session send lock", e);
|
||||
}
|
||||
}
|
||||
|
||||
private static long remainingMillis(long deadlineNanos) {
|
||||
return (deadlineNanos - System.nanoTime()) / 1_000_000L;
|
||||
}
|
||||
}
|
||||
@@ -1,195 +0,0 @@
|
||||
package dev.ltms.fleet.peer;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
|
||||
/**
|
||||
* SPI for materializing a connected peer — the only way the bridge core creates or tears down
|
||||
* a peer process. Every launcher is a first-party, in-tree adapter selected by (future) profile
|
||||
* config; today's single adapter is the {@code ClaudeCodeLauncher} / Claude Code over herdr.
|
||||
*
|
||||
* <p>The core delegates spawn and teardown to this interface without knowing how the peer is set
|
||||
* up. Environment variables, CLI flags, subscription guards, transport (herdr tab/pane) layout,
|
||||
* and naming conventions are all adapter-private — the core sees only the returned
|
||||
* {@link PeerHandle} whose {@code id()} is the registry/routing key.
|
||||
*
|
||||
* <p>The interface is a superset of what {@code SessionManager} and {@code Fleetd.main} call
|
||||
* on the concrete launcher today.
|
||||
*/
|
||||
public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The name every launcher gives the bridge's MCP server in the config it writes for its peer.
|
||||
* The peer's tools are addressed as {@code mcp__<this>__fleet_*}, and {@code CLAUDE.md}'s
|
||||
* role-detection ladder names that prefix, so the two must agree.
|
||||
*
|
||||
* <p>It is a constant because three launchers write it — {@code ClaudeCodeLauncher} and
|
||||
* {@code LeadLauncher} into a {@code --mcp-config} literal, {@code OpenCodeLauncher} into an
|
||||
* {@code opencode.json} node. Three hand-written copies of one name is how a rename lands in
|
||||
* two of them (CB-632).
|
||||
*/
|
||||
String MCP_MOUNT_NAME = "fleet";
|
||||
|
||||
/**
|
||||
* Shared IDE-guidance text (CB-634), delivered per-backend as an on-disk overlay rather than
|
||||
* any one adapter's system-prompt charter, so a project's own {@code CLAUDE.md} is never
|
||||
* clobbered. It pins every {@code ide_*} call to the member's own worktree, which is the whole
|
||||
* point of the mechanism. Both launchers render their own overlay from this single source.
|
||||
*
|
||||
* @param projectPath the path the member must pin every {@code ide_*} call to — the module dir
|
||||
* IntelliJ opened as the project, which is {@link #ideProjectPath} of the
|
||||
* member's own worktree (the worktree root when no module subdir is set)
|
||||
*/
|
||||
static String ideOverlayText(String projectPath) {
|
||||
return "## IDE code intelligence — your worktree only\n"
|
||||
+ "An IntelliJ IDE Index MCP server is mounted as `mcp__intellij__ide_*`. Prefer it "
|
||||
+ "over `grep`/`find` for symbol lookups, references, call and type hierarchy, and "
|
||||
+ "diagnostics — it resolves the real AST, text search does not.\n\n"
|
||||
+ "Every `ide_*` call MUST pass `project_path: \"" + projectPath + "\"` — your own "
|
||||
+ "worktree — and never any other path. A call without it errors "
|
||||
+ "`multiple_projects_open`; a call with a different path reads another checkout, "
|
||||
+ "not your changes. This is not the primary's IDE: it is your worktree, pinned to "
|
||||
+ "you.";
|
||||
}
|
||||
|
||||
/**
|
||||
* The absolute path IntelliJ must open as the project, and the {@code project_path} the overlay
|
||||
* pins (CB-634). It is {@code cwd} resolved against {@code ideProjectDir}. The distinction
|
||||
* matters because this repo (like {@code fleet/fleetd}) keeps its Maven module in a subdir
|
||||
* ({@code bridged/}), not at the worktree root: opening the root imports no module and
|
||||
* {@code ide_*} resolves nothing, so the module dir is the correct pin and open target.
|
||||
*
|
||||
* @param cwd the member's worktree root
|
||||
* @param ideProjectDir repo-relative module subdir, or {@code null}/blank for the worktree root
|
||||
* @return the absolute, normalized module dir as a string
|
||||
*/
|
||||
static String ideProjectPath(String cwd, String ideProjectDir) {
|
||||
Path base = Path.of(cwd);
|
||||
if (ideProjectDir == null || ideProjectDir.isBlank()) {
|
||||
return base.toString();
|
||||
}
|
||||
return base.resolve(ideProjectDir).normalize().toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Best-effort: open {@code projectPath} in the host IDE by running {@code openCommand} with
|
||||
* every {@code {dir}} replaced by {@code projectPath} (CB-634 auto-open). The command runs
|
||||
* through {@code /bin/sh -c} so an operator can set env inline — e.g.
|
||||
* {@code "env DISPLAY=:10.0 idea {dir}"} — because the daemon's own env may lack {@code DISPLAY}.
|
||||
*
|
||||
* <p>A blank command is a no-op: the profile opted into IDE MCP but not auto-open, so the
|
||||
* operator opens the module by hand. The child process is detached and its exit is not awaited;
|
||||
* any failure is logged and swallowed, because a member must spawn whether or not an IDE is
|
||||
* running. There is no close half yet (CB-634 defers it): an opened module stays open until the
|
||||
* operator closes it, and opening the same module again just refocuses it.
|
||||
*
|
||||
* @param projectPath the module dir to open (typically {@link #ideProjectPath})
|
||||
* @param openCommand the host command template, with {@code {dir}} substituted; null/blank ⇒ no-op
|
||||
* @param log the calling launcher's logger, for the best-effort WARN
|
||||
*/
|
||||
static void openInIde(String projectPath, String openCommand, Logger log) {
|
||||
if (openCommand == null || openCommand.isBlank()) {
|
||||
return;
|
||||
}
|
||||
String cmd = openCommand.replace("{dir}", projectPath);
|
||||
try {
|
||||
new ProcessBuilder("/bin/sh", "-c", cmd)
|
||||
.redirectOutput(ProcessBuilder.Redirect.DISCARD)
|
||||
.redirectError(ProcessBuilder.Redirect.DISCARD)
|
||||
.start();
|
||||
log.info("CB-634 auto-open: launched IDE open for {}", projectPath);
|
||||
} catch (Exception e) {
|
||||
log.warn("CB-634 auto-open of '{}' failed (member still spawns): {}", projectPath, e.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The set of {@link Capability capabilities} this launcher declares. A peer whose profile
|
||||
* opts into a git-forge token should include {@link Capability#SELF_PR}; the base set for
|
||||
* the Claude Code herdr adapter is always {@code MID_TURN_ASK, WORKTREE, ORPHAN_REAP}.
|
||||
*/
|
||||
Set<Capability> capabilities();
|
||||
|
||||
/**
|
||||
* The capabilities of the adapter that {@code profileName} resolves to (null/blank → the
|
||||
* default profile, the same resolution {@link #spawn} uses). Distinct from {@link
|
||||
* #capabilities()}, which unions every configured adapter: a caller that must know whether
|
||||
* <em>this</em> profile's backend supports a capability — e.g. {@link Capability#SESSION_RESUME}
|
||||
* before honoring {@link SpawnRequest#resumeSessionId()} — needs the per-profile answer, not
|
||||
* the fleet-wide union, or a mixed fleet could OK a resume that lands on a non-supporting
|
||||
* adapter (CB-584).
|
||||
*
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
Set<Capability> capabilitiesFor(String profileName);
|
||||
|
||||
/**
|
||||
* {@code profileName}/requestedCwd null/blank → default resolution. Returns after the peer
|
||||
* process is live (env + argv + placement complete). Never returns {@code null}.
|
||||
*
|
||||
* @param req the spawn parameters (profile, requested cwd, caller cwd)
|
||||
* @return a handle whose {@link PeerHandle#id()} is the registry/routing key
|
||||
* @throws IllegalArgumentException if the profile is unknown and no default is configured
|
||||
*/
|
||||
PeerHandle spawn(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The configured worker profile names — the set of names {@code spawn(profileName)} accepts.
|
||||
*/
|
||||
Set<String> profiles();
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
*
|
||||
* @return the resolved absolute path, never null/blank
|
||||
*/
|
||||
String effectiveCwd(SpawnRequest req);
|
||||
|
||||
/**
|
||||
* The parity-overlay file list for {@code profileName} (default list when unset). Used by
|
||||
* worktree provisioning to copy config files into the isolated checkout before spawning.
|
||||
*/
|
||||
List<String> parityOverlay(String profileName);
|
||||
|
||||
/**
|
||||
* The set of all agents this launcher currently tracks, transport-specific. Each element
|
||||
* exposes at minimum a pane-like {@code id()} matching this launcher's {@link PeerHandle}
|
||||
* scheme, plus transport-level status. Callers merge this set with the session registry to
|
||||
* build a live roster view.
|
||||
*/
|
||||
List<?> list();
|
||||
|
||||
/**
|
||||
* Reap orphaned peers left behind by a prior daemon process. Only peers whose naming scheme
|
||||
* matches this launcher's and whose nonce differs from the current process are eligible.
|
||||
* Best-effort: a failure to list or to stop any one peer is logged and never aborts startup.
|
||||
*
|
||||
* @return the number of orphaned peers reaped
|
||||
*/
|
||||
int reapOrphanWorkers();
|
||||
|
||||
/**
|
||||
* Tear a peer down by its registry/routing key ({@link PeerHandle#id()}). Tolerates an
|
||||
* already-gone peer. Also cleans up launcher-private resources (e.g. empty dedicated tabs)
|
||||
* when safe to do so.
|
||||
*/
|
||||
void stop(String id);
|
||||
|
||||
/**
|
||||
* Discard the context of the peer identified by {@code id}. Implementations must bypass normal
|
||||
* bridge delivery/turn accounting. Unsupported peer kinds return {@code false} without sending
|
||||
* a guessed command.
|
||||
*
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
}
|
||||
@@ -1,115 +0,0 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Where a credential (not a profile — see {@code FleetConfig.Profile#effectiveCredentialId()})
|
||||
* sits out a cooldown after a {@code BACKEND_EXHAUSTED} classification (CB-578 stage B), so a fresh
|
||||
* spawn does not walk straight back onto the account that just refused on a usage limit.
|
||||
*
|
||||
* <p>Keyed by credential id, never by profile name: two profiles sharing one credential (e.g. two
|
||||
* models on the same OpenAI account) share one quarantine — {@link #quarantine} one credential id
|
||||
* and every profile whose {@code effectiveCredentialId()} equals it is quarantined too, without this
|
||||
* class knowing anything about profiles at all. That mapping is the caller's job (see
|
||||
* {@code CompositePeerLauncher} and {@code dev.ltms.fleet.inject.ExhaustionSink}).
|
||||
*
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@code FleetHealthMonitor}), never read inline, so a quarantine's expiry is testable without a
|
||||
* real sleep.
|
||||
*/
|
||||
public final class BackendQuarantine {
|
||||
|
||||
private final ConcurrentHashMap<String, Long> quarantinedUntilNanos = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final long cooldownNanos;
|
||||
/** True only for {@link #none()}. See {@link #quarantine} for why this exists. */
|
||||
private final boolean inert;
|
||||
|
||||
/**
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos how long a fresh {@link #quarantine} call blocks the credential for;
|
||||
* must be positive
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, false);
|
||||
}
|
||||
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, boolean inert) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
if (cooldownNanos <= 0) {
|
||||
throw new IllegalArgumentException("cooldownNanos must be positive: " + cooldownNanos);
|
||||
}
|
||||
this.cooldownNanos = cooldownNanos;
|
||||
this.inert = inert;
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert quarantine — {@link #quarantine} does nothing on this instance, so nothing is ever
|
||||
* quarantined. The explicit stand-in a caller (or a test not exercising this feature) passes
|
||||
* instead of a defaulting overload, exactly like {@code ExhaustedPatternLookup.none()}.
|
||||
*/
|
||||
public static BackendQuarantine none() {
|
||||
return new BackendQuarantine(() -> 0L, 1, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Quarantine {@code credentialId} for the configured cooldown, starting now. A repeat call while
|
||||
* already quarantined restarts the cooldown at full length — a fresh refusal is fresh evidence the
|
||||
* account is still exhausted, not a reason to let an earlier, shorter wait stand.
|
||||
*
|
||||
* <p>On {@link #none()} this is a no-op. It has to be: that instance holds a clock frozen at 0,
|
||||
* so recording a deadline would produce a quarantine that never expires — a credential locked out
|
||||
* for the life of the daemon. Two production {@code CompositePeerLauncher} constructors default to
|
||||
* {@code none()}, so the failure would be silent and permanent. An inert stand-in must omit the
|
||||
* fact, never invent one.
|
||||
*/
|
||||
public void quarantine(String credentialId) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
if (inert) {
|
||||
return;
|
||||
}
|
||||
quarantinedUntilNanos.put(credentialId, nowNanos.getAsLong() + cooldownNanos);
|
||||
}
|
||||
|
||||
/** Whether {@code credentialId} is quarantined right now. */
|
||||
public boolean isQuarantined(String credentialId) {
|
||||
return remainingNanos(credentialId) > 0;
|
||||
}
|
||||
|
||||
/** Seconds left on {@code credentialId}'s quarantine, or empty when it is not quarantined. */
|
||||
public OptionalLong remainingSeconds(String credentialId) {
|
||||
long remaining = remainingNanos(credentialId);
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Every currently-quarantined credential id and its remaining seconds (CB-578 stage B fleet
|
||||
* reporting) — expired entries are never included. Not pruned from the backing map here: it stays
|
||||
* small (bounded by the number of distinct credentials ever exhausted) and a lazily-stale entry is
|
||||
* harmless, since every read already checks the deadline.
|
||||
*/
|
||||
public Map<String, Long> activeRemainingSeconds() {
|
||||
Map<String, Long> out = new LinkedHashMap<>();
|
||||
quarantinedUntilNanos.forEach((credentialId, deadline) -> {
|
||||
long remaining = deadline - nowNanos.getAsLong();
|
||||
if (remaining > 0) {
|
||||
out.put(credentialId, toSecondsRoundedUp(remaining));
|
||||
}
|
||||
});
|
||||
return out;
|
||||
}
|
||||
|
||||
private long remainingNanos(String credentialId) {
|
||||
Long deadline = quarantinedUntilNanos.get(credentialId);
|
||||
return deadline == null ? 0L : deadline - nowNanos.getAsLong();
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
return (nanos + 999_999_999L) / 1_000_000_000L;
|
||||
}
|
||||
}
|
||||
@@ -1,66 +0,0 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
*
|
||||
* <p>Two exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
* <li>Weight 0 (CB-554): {@code fixed} is still automatic selection, so a profile the operator
|
||||
* marked "never auto-select me" ({@code weight <= 0}) must be skipped here exactly as
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code fleet_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined or weight-0 never exercises either path, so today's
|
||||
* behaviour is unchanged.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !weightExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !c.excluded()) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
if (d != null && !d.isBlank()) {
|
||||
boolean dQuarantined = ctx.quarantined().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined && dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and has weight 0 (excluded from automatic selection), and no "
|
||||
+ "available candidate remains");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' has weight 0 (excluded "
|
||||
+ "from automatic selection) and no available candidate remains");
|
||||
}
|
||||
if (dQuarantined) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and no un-quarantined candidate is available");
|
||||
}
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException(
|
||||
"all worker profiles are excluded from automatic selection (quarantined or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
/** Whether {@code profile} carries {@code weight <= 0} (CB-554) among {@code ctx}'s candidates. */
|
||||
private static boolean weightExcluded(PlacementContext ctx, String profile) {
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.profile().equals(profile)) {
|
||||
return c.excluded();
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
}
|
||||
@@ -1,23 +0,0 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.function.Function;
|
||||
|
||||
/**
|
||||
* Everything a {@link PlacementPolicy} needs to make one selection.
|
||||
*
|
||||
* @param defaultProfile profile a {@code fixed} policy should return (may be {@code null})
|
||||
* @param candidates every configured candidate; the policy filters out those at cap or unreachable
|
||||
* @param liveCount current live worker count per profile (from the session registry)
|
||||
* @param unreachable profiles already known to have failed in this spawn attempt
|
||||
* @param quarantined profiles whose credential is currently quarantined (CB-578 stage B) — a
|
||||
* {@code BACKEND_EXHAUSTED} classification put it, or a profile it shares a
|
||||
* credential with, on cooldown. Filtered the same way as {@code unreachable}.
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable,
|
||||
Set<String> quarantined) {
|
||||
}
|
||||
@@ -1,85 +0,0 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Shared filtering and empty-set reporting used by the built-in placement policies.
|
||||
*/
|
||||
final class PlacementPolicyUtil {
|
||||
|
||||
private PlacementPolicyUtil() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Candidates that are not weight-excluded (CB-554: explicit {@code weight <= 0}, checked
|
||||
* first because it is a static config choice rather than transient state), not
|
||||
* known-unreachable, not quarantined (CB-578 stage B), and have not reached their maxLoad.
|
||||
* A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.excluded() || ctx.unreachable().contains(c.profile())
|
||||
|| ctx.quarantined().contains(c.profile())) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
if (cap != null) {
|
||||
int live = ctx.liveCount().apply(c.profile());
|
||||
if (live >= cap) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.add(c);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all weight-0, all
|
||||
* quarantined, all at capacity, all unreachable, or a mix. Each candidate is counted into
|
||||
* exactly one bucket (weight-excluded takes priority) so a candidate excluded for more than
|
||||
* one reason is never double-counted.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int weightExcluded = 0;
|
||||
int atCap = 0;
|
||||
int unreachable = 0;
|
||||
int quarantined = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (c.excluded()) {
|
||||
weightExcluded++;
|
||||
} else if (ctx.quarantined().contains(c.profile())) {
|
||||
quarantined++;
|
||||
} else if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
atCap++;
|
||||
}
|
||||
}
|
||||
|
||||
int total = ctx.candidates().size();
|
||||
if (total == 0) {
|
||||
return new PlacementException("no worker profiles configured");
|
||||
}
|
||||
if (weightExcluded == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles have weight 0 (excluded from automatic selection)");
|
||||
}
|
||||
if (quarantined == total) {
|
||||
return new PlacementException("all worker profiles are quarantined (backend exhausted)");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
if (unreachable == total) {
|
||||
return new PlacementException("all worker profiles are unreachable");
|
||||
}
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + quarantined + " quarantined, "
|
||||
+ weightExcluded + " weight-0, "
|
||||
+ (total - atCap - unreachable - quarantined - weightExcluded) + " remaining");
|
||||
}
|
||||
}
|
||||
@@ -1,510 +0,0 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.BufferedReader;
|
||||
import java.io.IOException;
|
||||
import java.io.InputStreamReader;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.security.SecureRandom;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
* Production {@link Worktrees} implementation that shells {@code git} via {@link ProcessBuilder}.
|
||||
* Non-zero exits become {@link WorktreeException}. Worktree directories live under a configurable
|
||||
* root (default: a sibling {@code .bridged-worktrees} of the repo root) so they are never nested
|
||||
* inside the primary working tree.
|
||||
*/
|
||||
public final class GitWorktrees implements Worktrees {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(GitWorktrees.class);
|
||||
|
||||
/** Project-level MCP config. Present in the repo, so every worktree would otherwise inherit the
|
||||
* primary's IDE server mounts (a CB-523 worker edited the primary checkout; see the isolation
|
||||
* javadoc). Neutralized unconditionally. */
|
||||
private static final String MCP_CONFIG = ".mcp.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .mcp.json}: a valid, explicitly empty server map. */
|
||||
private static final String NEUTRAL_MCP_CONFIG = "{\n \"mcpServers\": {}\n}\n";
|
||||
|
||||
/** OpenCode's repo-level config. Tracked here, so it lands in every worktree, and it mounts the
|
||||
* primary's gitea and context7 servers with the primary's credentials. Neutralized so the worker
|
||||
* gets only the config its launcher writes via {@code OPENCODE_CONFIG}.
|
||||
*
|
||||
* <p>The reason has changed shape and is now stronger. It used to be a crash: the file carried
|
||||
* {@code {file:.secrets/...}} references to gitignored files that never reached a worktree, and
|
||||
* opencode refuses to start on a dangling reference (CB-543). Those credentials now live in one
|
||||
* shell-level store and the file reads them as {@code {env:...}}, so in a worktree the reference
|
||||
* resolves instead of failing. That is worse, not better: a member would silently inherit the
|
||||
* primary's admin-scoped {@code GITEA_ACCESS_TOKEN}. A loud crash became a quiet privilege leak,
|
||||
* so this entry protects a boundary now rather than papering over a startup error. */
|
||||
private static final String OPENCODE_CONFIG = "opencode.json";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code opencode.json}: a valid, empty JSON object. */
|
||||
private static final String NEUTRAL_OPENCODE_CONFIG = "{}\n";
|
||||
|
||||
/** Autoenv's repo-level config. Not tracked today, but re-landing it must stay safe: autoenv
|
||||
* authorizes by path, so a fresh worktree path is always unauthorized and its interactive prompt
|
||||
* would block every spawn — neutralize it so it can never be committed. */
|
||||
private static final String AUTOENV_CONFIG = ".autoenv";
|
||||
|
||||
/** What {@link #isolateToolSurface} writes for {@code .autoenv}: a valid, empty env file. */
|
||||
private static final String NEUTRAL_AUTOENV_CONFIG = "";
|
||||
|
||||
/**
|
||||
* A tracked project config that is hostile in a provisioned worktree, and what to replace it
|
||||
* with. {@link #file} is the repo-relative path; {@link #stub} is a neutral but VALID payload for
|
||||
* that file's format — a malformed stub would only trade one crash for another;
|
||||
* {@link #createIfAbsent} keeps {@code .mcp.json}'s long-standing behaviour of writing its stub
|
||||
* even when the repo carries no such file, whereas the others are only touched when present.
|
||||
*/
|
||||
private record WorktreeHostileConfig(String file, String stub, boolean createIfAbsent) {}
|
||||
|
||||
/** The worktree-hostile configs neutralized in every provisioned worktree, in order. */
|
||||
private static final List<WorktreeHostileConfig> WORKTREE_HOSTILE_CONFIGS = List.of(
|
||||
new WorktreeHostileConfig(MCP_CONFIG, NEUTRAL_MCP_CONFIG, true),
|
||||
new WorktreeHostileConfig(OPENCODE_CONFIG, NEUTRAL_OPENCODE_CONFIG, false),
|
||||
new WorktreeHostileConfig(AUTOENV_CONFIG, NEUTRAL_AUTOENV_CONFIG, false)
|
||||
);
|
||||
|
||||
private final String configuredRoot;
|
||||
private final SecureRandom random = new SecureRandom();
|
||||
private final AtomicLong seq = new AtomicLong();
|
||||
|
||||
/** Default constructor: worktree root is derived per-repo as {@code <repoRoot>/../.bridged-worktrees}. */
|
||||
public GitWorktrees() {
|
||||
this(null);
|
||||
}
|
||||
|
||||
/** @param configuredRoot nullable absolute or relative path; null/blank derives a sibling of the repo root. */
|
||||
public GitWorktrees(String configuredRoot) {
|
||||
this.configuredRoot = configuredRoot;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
String base = (baseRef == null || baseRef.isBlank()) ? "HEAD" : baseRef;
|
||||
String nonce = nonce();
|
||||
Path root = resolveRoot(repoRoot);
|
||||
Path path = root.resolve(nonce);
|
||||
try {
|
||||
Files.createDirectories(root);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create worktree root " + root + ": " + e.getMessage(), e);
|
||||
}
|
||||
String wt = path.toAbsolutePath().toString();
|
||||
log.info("adding worktree branch={} path={} base={}", branch, wt, base);
|
||||
exec("git", "-C", repoRoot, "worktree", "add", wt, "-b", branch, base);
|
||||
isolateToolSurface(wt);
|
||||
return wt;
|
||||
}
|
||||
|
||||
/**
|
||||
* Neutralize the worktree's worktree-hostile project configs so a worker inherits only the tools
|
||||
* and environment its launcher mounts (the bridge via {@code --mcp-config}, the opencode config
|
||||
* via {@code OPENCODE_CONFIG}) — never the primary's.
|
||||
*
|
||||
* <p>This is unconditional, and it is not the same job as the parity overlay. The repo's own
|
||||
* committed {@code .mcp.json} declares the primary's IDE servers, so a fresh checkout mounts them
|
||||
* whether or not the overlay copies anything; a worker that inherits them navigates and edits
|
||||
* through tools bound to the <em>primary's</em> IntelliJ project, which silently hands it absolute
|
||||
* paths outside its own worktree. That is not hypothetical: a CB-523 worker made all 59 of its
|
||||
* edits in the primary checkout while compiling its worktree, so every build it ran was of code
|
||||
* that did not contain its changes. {@code opencode.json} is the same trap one tool over — tracked,
|
||||
* so it lands in every worktree, and it mounts gitea and context7 with the primary's own
|
||||
* credentials, which a member must never hold. {@code .autoenv} extends the principle to a
|
||||
* config that is not tracked today: autoenv authorizes by path, so a fresh worktree path is always
|
||||
* unauthorized and its interactive prompt would block every spawn, so re-landing one must be safe.
|
||||
*
|
||||
* <p>Where a config exists it is replaced by a valid neutral stub (an explicitly empty
|
||||
* map/object, or an empty env file — never a deletion, which would still let a later
|
||||
* {@code git checkout} restore the hostile copy). The {@code --skip-worktree} bit keeps the
|
||||
* neutralized copy from ever showing up as a local modification the worker might commit. A config
|
||||
* the repo does not carry is skipped silently — no stub is invented for a file the repo does not
|
||||
* have, and one missing file must never fail provisioning.
|
||||
*/
|
||||
private void isolateToolSurface(String worktreePath) {
|
||||
Path root = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (WorktreeHostileConfig cfg : WORKTREE_HOSTILE_CONFIGS) {
|
||||
neutralize(root, worktreePath, cfg);
|
||||
}
|
||||
}
|
||||
|
||||
private void neutralize(Path root, String worktreePath, WorktreeHostileConfig cfg) {
|
||||
Path target = root.resolve(cfg.file());
|
||||
if (!Files.exists(target) && !cfg.createIfAbsent()) {
|
||||
log.debug("{} absent in the worktree — skipping (repo does not carry it)", cfg.file());
|
||||
return;
|
||||
}
|
||||
try {
|
||||
Files.writeString(target, cfg.stub());
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot neutralize " + cfg.file() + " in the worktree: "
|
||||
+ e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(root, cfg.file())) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", cfg.file());
|
||||
}
|
||||
log.debug("neutralized {} — worker tool surface is launcher-mounted only", cfg.file());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing to remove", worktreePath);
|
||||
return;
|
||||
}
|
||||
log.info("removing worktree {}", worktreePath);
|
||||
exec("git", "-C", repoRoot, "worktree", "remove", "--force", worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
// A worktree that is already gone holds no work to lose, and it must not break teardown:
|
||||
// git -C <missing-dir> status exits non-zero and would throw where release() is mid-way
|
||||
// through stopping a pane. Mirror remove()'s already-gone tolerance by treating it as clean.
|
||||
Path p = Path.of(worktreePath);
|
||||
if (!Files.exists(p)) {
|
||||
log.debug("worktree {} already gone — nothing can be uncommitted", worktreePath);
|
||||
return false;
|
||||
}
|
||||
// No --untracked-files=no: the exact shape of the work lost in CB-576 was a new file
|
||||
// that was never added, so an untracked-only worktree is still dirty.
|
||||
String out = exec("git", "-C", worktreePath, "status", "--porcelain");
|
||||
return !out.isBlank();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
if (overlay == null || overlay.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Path srcRoot = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
Path dstRoot = Path.of(worktreePath).toAbsolutePath().normalize();
|
||||
for (String rel : overlay) {
|
||||
Path src = srcRoot.resolve(rel).normalize();
|
||||
if (!Files.exists(src)) {
|
||||
log.debug("parity overlay source missing — skipping {}", rel);
|
||||
continue;
|
||||
}
|
||||
Path dst = dstRoot.resolve(rel).normalize();
|
||||
try {
|
||||
Files.createDirectories(dst.getParent());
|
||||
Files.copy(src, dst, StandardCopyOption.REPLACE_EXISTING, StandardCopyOption.COPY_ATTRIBUTES);
|
||||
log.debug("copied parity overlay {}", rel);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy overlay " + rel + ": " + e.getMessage(), e);
|
||||
}
|
||||
if (isTracked(dstRoot, rel)) {
|
||||
exec("git", "-C", worktreePath, "update-index", "--skip-worktree", rel);
|
||||
log.debug("marked overlay --skip-worktree {}", rel);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
String out = exec("git", "-C", cwd, "rev-parse", "--show-toplevel");
|
||||
return Path.of(out.trim()).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, fixed by CB-587. Stages into a <em>temporary</em> index (never the worktree's
|
||||
* real one, which the worker may still be writing to) — but that temp index is first <em>seeded</em>
|
||||
* from the worktree's real one, rather than starting empty:
|
||||
*
|
||||
* <pre>
|
||||
* cp $(git -C worktree rev-parse --git-path index) <temp>
|
||||
* GIT_INDEX_FILE=<temp> git -C worktree add -A
|
||||
* tree=$(GIT_INDEX_FILE=<temp> git -C worktree write-tree)
|
||||
* commit=$(git -C worktree commit-tree $tree -p HEAD -m message)
|
||||
* git -C worktree update-ref refs/wip/branch $commit
|
||||
* </pre>
|
||||
*
|
||||
* A fresh empty index carries none of the real index's {@code --skip-worktree} /
|
||||
* {@code --assume-unchanged} bits, so {@code add -A} into it stages a skip-worktree file's local
|
||||
* on-disk content even though {@code git status --porcelain} correctly hides that file (CB-587).
|
||||
* Seeding from the real index preserves those bits, so {@code add -A} then skips exactly what
|
||||
* {@code git status} skips. {@code add -A} (never {@code -f}) also still respects
|
||||
* {@code .gitignore} exactly as it would in the real index — a gitignored file staying ignored is
|
||||
* what keeps secrets and local config out of the snapshot's tree. The temporary index file is
|
||||
* removed afterwards regardless of outcome; the worker's real index is never opened for writing.
|
||||
*/
|
||||
@Override
|
||||
public Optional<String> snapshot(String worktreePath, String branch, String message) {
|
||||
if (!Files.exists(Path.of(worktreePath))) {
|
||||
log.debug("worktree {} already gone — nothing to snapshot", worktreePath);
|
||||
return Optional.empty();
|
||||
}
|
||||
Path tempIndex;
|
||||
try {
|
||||
tempIndex = Files.createTempFile("bridged-wip-index-", ".tmp");
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot create a temporary index for snapshot: " + e.getMessage(), e);
|
||||
}
|
||||
Map<String, String> indexEnv = Map.of("GIT_INDEX_FILE", tempIndex.toAbsolutePath().toString());
|
||||
try {
|
||||
Path realIndex = resolveRealIndex(worktreePath);
|
||||
try {
|
||||
Files.copy(realIndex, tempIndex, StandardCopyOption.REPLACE_EXISTING);
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("cannot copy the worktree's real index (" + realIndex
|
||||
+ ") into the temporary snapshot index: " + e.getMessage(), e);
|
||||
}
|
||||
exec(indexEnv, "git", "-C", worktreePath, "add", "-A");
|
||||
String tree = exec(indexEnv, "git", "-C", worktreePath, "write-tree").trim();
|
||||
String commit = exec("git", "-C", worktreePath, "commit-tree", tree, "-p", "HEAD", "-m", message).trim();
|
||||
exec("git", "-C", worktreePath, "update-ref", "refs/wip/" + branch, commit);
|
||||
log.info("snapshotted worktree {} to refs/wip/{} commit={}", worktreePath, branch, commit);
|
||||
return Optional.of(commit);
|
||||
} finally {
|
||||
try {
|
||||
Files.deleteIfExists(tempIndex);
|
||||
} catch (IOException e) {
|
||||
log.debug("could not delete temporary snapshot index {}: {}", tempIndex, e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the path of {@code worktreePath}'s real index. Never assume {@code <worktree>/.git/index}:
|
||||
* in a linked worktree {@code .git} is a <em>file</em> pointing at the main repo's
|
||||
* {@code worktrees/<name>/} directory, and that is where the real per-worktree index lives.
|
||||
* {@code git rev-parse --git-path index} resolves this correctly for both a linked worktree and
|
||||
* the main checkout. Throws {@link WorktreeException} — same as every other failure in this
|
||||
* class — if the command fails or the resolved path does not exist, rather than silently
|
||||
* snapshotting from an empty index.
|
||||
*/
|
||||
private Path resolveRealIndex(String worktreePath) {
|
||||
String out = exec("git", "-C", worktreePath, "rev-parse", "--git-path", "index").trim();
|
||||
Path index = Path.of(out);
|
||||
if (!index.isAbsolute()) {
|
||||
index = Path.of(worktreePath).resolve(index).normalize();
|
||||
}
|
||||
if (!Files.exists(index)) {
|
||||
throw new WorktreeException("worktree's real index not found at resolved path " + index
|
||||
+ " (git rev-parse --git-path index reported '" + out + "')");
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
/**
|
||||
* One {@code refs/wip/<branch>} snapshot ref as read by {@link #listWipRefs}: its full ref name,
|
||||
* the snapshot commit's sha, and that commit's committer time in unix millis (the age of the
|
||||
* snapshot — a snapshot is written once and never rewritten, so the commit date is the ref's).
|
||||
*/
|
||||
private record WipRef(String refName, String sha, long committerMillis) {
|
||||
String branch() {
|
||||
return refName.substring("refs/wip/".length());
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public WipRefStats wipRefs(String repoRoot) {
|
||||
List<WipRef> refs = listWipRefs(repoRoot);
|
||||
long costBytes = 0;
|
||||
for (WipRef ref : refs) {
|
||||
costBytes += treeSize(repoRoot, ref.sha());
|
||||
}
|
||||
return new WipRefStats(refs.size(), costBytes);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
|
||||
// The rule is documented on Worktrees#pruneWipRefs: delete only a snapshot whose tree
|
||||
// content is already reachable from main AND that is older than minAgeMillis. Reachability
|
||||
// is the floor that keeps a worker's last copy; the age floor keeps a just-written snapshot
|
||||
// from being swept while a lead may still be looking at it.
|
||||
List<WipRef> refs = listWipRefs(repoRoot);
|
||||
if (refs.isEmpty()) {
|
||||
return 0;
|
||||
}
|
||||
long nowMillis = System.currentTimeMillis();
|
||||
// Resolve what main carries once per sweep, not once per ref.
|
||||
Set<String> mainObjects = reachableObjectsFromMain(repoRoot);
|
||||
int deleted = 0;
|
||||
for (WipRef ref : refs) {
|
||||
long ageMillis = nowMillis - ref.committerMillis();
|
||||
if (ageMillis <= minAgeMillis) {
|
||||
continue; // too recent — never swept, even if it looks recoverable (CB-586)
|
||||
}
|
||||
String tree = exec("git", "-C", repoRoot, "rev-parse", ref.sha() + "^{tree}").trim();
|
||||
if (!mainObjects.contains(tree)) {
|
||||
// Last copy of the snapshot's content — the worker's work exists nowhere else.
|
||||
// Never delete automatically (CB-586 criterion 2).
|
||||
continue;
|
||||
}
|
||||
exec("git", "-C", repoRoot, "update-ref", "-d", ref.refName());
|
||||
deleted++;
|
||||
log.info("pruned snapshot ref refs/wip/{} commit={} (age {}h): its tree is already "
|
||||
+ "reachable from main, so the work is preserved; recover from reflog via "
|
||||
+ "git update-ref refs/wip/{} {}",
|
||||
ref.branch(), ref.sha(), TimeUnit.MILLISECONDS.toHours(ageMillis),
|
||||
ref.branch(), ref.sha());
|
||||
}
|
||||
return deleted;
|
||||
}
|
||||
|
||||
/**
|
||||
* Every {@code refs/wip/*} ref (see {@link WipRef}). The committer date is read as a unix
|
||||
* count of seconds and converted to millis. {@code %00} (NUL) separates the fields because a
|
||||
* branch name may contain spaces.
|
||||
*/
|
||||
private List<WipRef> listWipRefs(String repoRoot) {
|
||||
String out = exec("git", "-C", repoRoot, "for-each-ref",
|
||||
"--format=%(refname)%00%(objectname)%00%(committerdate:unix)", "refs/wip/");
|
||||
List<WipRef> refs = new ArrayList<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
String[] parts = line.split("\u0000", -1);
|
||||
if (parts.length == 3 && !parts[1].isBlank()) {
|
||||
refs.add(new WipRef(parts[0], parts[1], Long.parseLong(parts[2]) * 1000L));
|
||||
}
|
||||
}
|
||||
return refs;
|
||||
}
|
||||
|
||||
/**
|
||||
* The set of object shas reachable from {@code main}, or an empty set when {@code main} cannot
|
||||
* be resolved. An empty set is the safe direction: the retention sweep then concludes nothing
|
||||
* is recoverable, so it deletes nothing — a repo with no {@code main} must never cause a
|
||||
* worker's last copy of a snapshot to be dropped on a reachability misreading.
|
||||
*/
|
||||
private Set<String> reachableObjectsFromMain(String repoRoot) {
|
||||
if (exitCode("git", "-C", repoRoot, "rev-parse", "--verify", "main") != 0) {
|
||||
log.debug("refs/wip retention: no 'main' ref in {} — treating nothing as reachable", repoRoot);
|
||||
return Set.of();
|
||||
}
|
||||
String out = exec("git", "-C", repoRoot, "rev-list", "--objects", "main");
|
||||
Set<String> objects = new HashSet<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
int sp = line.indexOf(' ');
|
||||
objects.add(sp < 0 ? line : line.substring(0, sp));
|
||||
}
|
||||
return objects;
|
||||
}
|
||||
|
||||
/** Approximate cost of a snapshot: the sum of every blob's size in its committed tree. */
|
||||
private long treeSize(String repoRoot, String sha) {
|
||||
String out = exec("git", "-C", repoRoot, "ls-tree", "-r", "-l", sha);
|
||||
long total = 0;
|
||||
for (String line : out.split("\\R")) {
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
// ls-tree -l row: "<mode> <type> <object> <size>\t<path>"; the size is only numeric for
|
||||
// blobs (trees read "-"), so gate on the type token and take the 4th whitespace field.
|
||||
String[] parts = line.split("\\s+");
|
||||
if (parts.length >= 4 && "blob".equals(parts[1])) {
|
||||
try {
|
||||
total += Long.parseLong(parts[3]);
|
||||
} catch (NumberFormatException ignored) {
|
||||
// a '-' size (or any anomaly) contributes nothing to the rough figure
|
||||
}
|
||||
}
|
||||
}
|
||||
return total;
|
||||
}
|
||||
|
||||
/** Resolve the directory that will hold per-session worktree checkouts. */
|
||||
private Path resolveRoot(String repoRoot) {
|
||||
if (configuredRoot != null && !configuredRoot.isBlank()) {
|
||||
return Path.of(configuredRoot).toAbsolutePath().normalize();
|
||||
}
|
||||
Path repo = Path.of(repoRoot).toAbsolutePath().normalize();
|
||||
return repo.resolveSibling(".bridged-worktrees");
|
||||
}
|
||||
|
||||
private String nonce() {
|
||||
return String.format("%06x", random.nextInt(1 << 24)) + "-" + seq.incrementAndGet();
|
||||
}
|
||||
|
||||
private boolean isTracked(Path worktreeRoot, String rel) {
|
||||
return exitCode("git", "-C", worktreeRoot.toString(), "ls-files", "--error-unmatch", rel) == 0;
|
||||
}
|
||||
|
||||
/**
|
||||
* Run a command and return its stdout. Non-zero exit → {@link WorktreeException} with both
|
||||
* stdout and stderr (merged by redirectErrorStream).
|
||||
*/
|
||||
private String exec(String... command) {
|
||||
return exec(Map.of(), command);
|
||||
}
|
||||
|
||||
/** Same as {@link #exec(String...)}, with extra environment variables set on the child process. */
|
||||
private String exec(Map<String, String> extraEnv, String... command) {
|
||||
String out;
|
||||
int code;
|
||||
Process p;
|
||||
try {
|
||||
ProcessBuilder pb = new ProcessBuilder(command).redirectErrorStream(true);
|
||||
if (extraEnv != null && !extraEnv.isEmpty()) {
|
||||
pb.environment().putAll(extraEnv);
|
||||
}
|
||||
p = pb.start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try (BufferedReader r = new BufferedReader(new InputStreamReader(p.getInputStream(), StandardCharsets.UTF_8))) {
|
||||
out = r.lines().collect(Collectors.joining("\n"));
|
||||
} catch (IOException e) {
|
||||
p.destroyForcibly();
|
||||
throw new UncheckedIOException(e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command) + "\n" + out);
|
||||
}
|
||||
code = p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
if (code != 0) {
|
||||
throw new WorktreeException("exit " + code + " for: " + String.join(" ", command)
|
||||
+ (out.isBlank() ? "" : "\n" + out));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
private int exitCode(String... command) {
|
||||
Process p;
|
||||
try {
|
||||
p = new ProcessBuilder(command).redirectErrorStream(true).start();
|
||||
} catch (IOException e) {
|
||||
throw new WorktreeException("failed to start " + command[0] + ": " + e.getMessage(), e);
|
||||
}
|
||||
try {
|
||||
if (!p.waitFor(30, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
throw new WorktreeException("command timed out: " + String.join(" ", command));
|
||||
}
|
||||
return p.exitValue();
|
||||
} catch (InterruptedException e) {
|
||||
p.destroyForcibly();
|
||||
Thread.currentThread().interrupt();
|
||||
throw new WorktreeException("interrupted waiting for command: " + String.join(" ", command), e);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,401 +0,0 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-559: re-reading {@code bridged.yaml} under a running daemon.
|
||||
*
|
||||
* <p>The tests that matter here are the refusals. A reload that applies a good file is the easy
|
||||
* half; the half that protects an operator is the one that keeps the running config when the new
|
||||
* file is bad, and the one that refuses a change the running daemon cannot honour.
|
||||
*/
|
||||
class ConfigRefTest {
|
||||
|
||||
/** A minimal file that loads and passes every startup validator. */
|
||||
private static String yaml(String extra) {
|
||||
return """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""" + extra;
|
||||
}
|
||||
|
||||
private static ConfigRef refFor(Path f) {
|
||||
return new ConfigRef(f, FleetConfig.load(f));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aHotChangeIsAppliedAndReadThroughGet(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
tabLabel: "{role}: {profile} #{n}"
|
||||
developers:
|
||||
a:
|
||||
profile: sonnet
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
assertEquals("{role}: {profile} #{n}", ref.get().fleet().tabLabel());
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
tabLabel: "[{profile}] {role}"
|
||||
developers:
|
||||
a:
|
||||
profile: sonnet
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty());
|
||||
assertEquals("config reloaded", out.summary());
|
||||
assertEquals("[{profile}] {role}", ref.get().fleet().tabLabel());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aCharterChangeIsHotAndReachesTheLiveConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: old charter
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
assertEquals("old charter", ref.get().fleet().charterFor(
|
||||
dev.ltms.fleet.peer.MemberRole.ARCHITECT));
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: new charter
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty());
|
||||
assertEquals("new charter", ref.get().fleet().charterFor(
|
||||
dev.ltms.fleet.peer.MemberRole.ARCHITECT));
|
||||
}
|
||||
|
||||
@Test
|
||||
void invalidChartersRefuseReloadAndKeepTheRunningConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: valid charter
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
FleetConfig before = ref.get();
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architect: " "
|
||||
"""));
|
||||
ConfigRef.Outcome blank = ref.reload();
|
||||
assertFalse(blank.applied());
|
||||
assertTrue(blank.error().contains("fleet.charters.architect is blank"));
|
||||
assertSame(before, ref.get());
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
charters:
|
||||
architetc: valid charter
|
||||
"""));
|
||||
ConfigRef.Outcome unknown = ref.reload();
|
||||
assertFalse(unknown.applied());
|
||||
assertTrue(unknown.error().contains("architetc"));
|
||||
assertTrue(unknown.error().contains("architect"));
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
/**
|
||||
* The point of the whole class: a consumer holding the ref sees the new value without being
|
||||
* rebuilt. A component that captured {@code get()} into a field would still show the old one.
|
||||
*/
|
||||
@Test
|
||||
void aConsumerHoldingTheRefSeesTheNewValue(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
java.util.function.Supplier<String> reader = () -> ref.get().placement();
|
||||
assertEquals("weighted", reader.get());
|
||||
|
||||
Files.writeString(f, yaml("placement: fixed\n"));
|
||||
assertTrue(ref.reload().applied());
|
||||
|
||||
assertEquals("fixed", reader.get());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aChangedColdKeyRefusesTheWholeReload(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
// Two changes in one file: a cold one (the port) and a hot one (placement).
|
||||
Files.writeString(f, yaml("placement: fixed\n").replace("port: 8765", "port: 9999"));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertEquals(java.util.List.of("bind"), out.coldKeys());
|
||||
assertTrue(out.summary().contains("Restart bridged"), out.summary());
|
||||
// The hot half must NOT have leaked in. A half-applied reload leaves the daemon matching no
|
||||
// file on disk, which is worse for an operator than no reload at all.
|
||||
assertEquals("weighted", ref.get().placement());
|
||||
assertEquals(8765, ref.get().bind().port());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFileThatNoLongerParsesKeepsTheRunningConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
FleetConfig before = ref.get();
|
||||
|
||||
Files.writeString(f, "profiles:\n sonnet:\n baseUrl: \"unclosed\n");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertTrue(out.summary().startsWith("config reload refused"), out.summary());
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
/** A file that would have refused to boot must not be able to slip in through a reload. */
|
||||
@Test
|
||||
void aFileThatFailsAValidatorKeepsTheRunningConfig(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("placement: weighted\n"));
|
||||
ConfigRef ref = refFor(f);
|
||||
FleetConfig before = ref.get();
|
||||
|
||||
// A member slot naming a profile that does not exist — validateMembers refuses this at
|
||||
// startup, so it must refuse it here too.
|
||||
Files.writeString(f, yaml("""
|
||||
fleet:
|
||||
developers:
|
||||
a:
|
||||
profile: no-such-profile
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aDeletedFileIsRefusedRatherThanCrashing(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
FleetConfig before = ref.get();
|
||||
|
||||
Files.delete(f);
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertSame(before, ref.get());
|
||||
}
|
||||
|
||||
/** A deferred change applies to the snapshot but the operator is told it needs a restart. */
|
||||
@Test
|
||||
void aDeferredChangeIsAppliedAndReported(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml("""
|
||||
lifecycle:
|
||||
drainTimeoutSeconds: 30
|
||||
"""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, yaml("""
|
||||
lifecycle:
|
||||
drainTimeoutSeconds: 60
|
||||
"""));
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(java.util.List.of("lifecycle"), out.deferred());
|
||||
assertTrue(out.summary().contains("needs a restart") || out.summary().contains("need a restart"),
|
||||
out.summary());
|
||||
assertEquals(60, ref.get().lifecycle().drainTimeoutSeconds());
|
||||
}
|
||||
|
||||
/**
|
||||
* Adding a profile is deferred, not hot: a new backend needs its own launcher, and launchers are
|
||||
* built once at startup. The snapshot carries it so a restart picks it up.
|
||||
*/
|
||||
@Test
|
||||
void addingAProfileIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
haiku:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: haiku
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size());
|
||||
assertTrue(out.deferred().getFirst().contains("haiku"), out.deferred().toString());
|
||||
}
|
||||
|
||||
/**
|
||||
* Changing an existing profile's weight or maxLoad IS hot: placement reads those live through
|
||||
* the supplier on the composite, so the next spawn already sees them.
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesWeightOrMaxLoadIsHot(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
maxLoad: 7
|
||||
weight: 3.0
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertTrue(out.deferred().isEmpty(), out.deferred().toString());
|
||||
assertEquals(7, ref.get().profiles().get("sonnet").maxLoad());
|
||||
}
|
||||
|
||||
/**
|
||||
* Changing an existing profile's MODEL is deferred, not hot — and saying so is the whole point.
|
||||
* HerdrPeerLauncher takes Map.copyOf(profiles) at construction and resolves every spawn out of
|
||||
* that copy, so a reloaded model never reaches a launch. Reporting it as applied would be the
|
||||
* worst outcome a reload can produce: the operator has no reason to doubt a clean "reloaded".
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesLaunchSettingsIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, yaml(""));
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: deepseek-v4-flash
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size(), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("sonnet"), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("launch settings"), out.deferred().toString());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("deepseek-v4-flash", ref.get().profiles().get("sonnet").model());
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map at startup
|
||||
* (see ExhaustedPatternLookup), so a reload never re-reads it — a changed pattern must be
|
||||
* reported deferred exactly like model/baseUrl, not silently claimed as applied.
|
||||
*/
|
||||
@Test
|
||||
void changingAProfilesExhaustedPatternIsReportedAsDeferred(@TempDir Path dir) throws Exception {
|
||||
Path f = dir.resolve("bridged.yaml");
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
exhaustedPattern: "usage limit has been reached"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef ref = refFor(f);
|
||||
|
||||
Files.writeString(f, """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
exhaustedPattern: "rate limit exceeded"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""");
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals(1, out.deferred().size(), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("sonnet"), out.deferred().toString());
|
||||
assertTrue(out.deferred().getFirst().contains("launch settings"), out.deferred().toString());
|
||||
// The snapshot still carries the new value — a restart is what makes it take effect.
|
||||
assertEquals("rate limit exceeded", ref.get().profiles().get("sonnet").exhaustedPattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFixedRefHasNoFileAndRefusesToReload() {
|
||||
FleetConfig cfg = new FleetConfig(null, null, null, null, null, null,
|
||||
null, null, null, null, null, null, null, null, null).withDefaults();
|
||||
ConfigRef ref = ConfigRef.fixed(cfg);
|
||||
|
||||
assertNull(ref.path());
|
||||
assertSame(cfg, ref.get());
|
||||
ConfigRef.Outcome out = ref.reload();
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
}
|
||||
}
|
||||
@@ -1,170 +0,0 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.function.BiConsumer;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
class FleetHealthMonitorTest {
|
||||
@Test void oneTickUsesOneFleetListForAnyRosterSize() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
FleetConfig.Profile profile = new FleetConfig.Profile("test", "http://test:1", null,
|
||||
null, null, null, null, null, null, null, null, null);
|
||||
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(agents, new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("test")), Map.of("test", profile), "test", _ -> "token");
|
||||
SessionManager sessions = new SessionManager(launcher);
|
||||
sessions.acquire("test", null, null, null);
|
||||
sessions.acquire("test", null, null, null);
|
||||
herdr.calls.clear();
|
||||
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox());
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, sessions::roster, messages, scheduler, () -> 1, 60,
|
||||
(_, _) -> { });
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
assertEquals(1, herdr.calls.stream().filter(call -> call.method().equals("agent.list")).count());
|
||||
}
|
||||
|
||||
@Test void failedTickDoesNotStopTheNextTick() {
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false);
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60, (_, _) -> { });
|
||||
monitor.tick();
|
||||
herdr.healthy(true);
|
||||
monitor.tick();
|
||||
monitor.stop();
|
||||
assertEquals(2, herdr.calls.stream().filter(call -> call.method().equals("agent.list")).count());
|
||||
}
|
||||
|
||||
@Test void faultTransitionLogsOnlyOnceUntilItChanges() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
FleetHealthMonitor monitor = new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60, (_, _) -> { });
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.stop();
|
||||
assertEquals(1, appender.list.stream().filter(event -> event.getFormattedMessage()
|
||||
.contains("member=term_a state=TURN_BOUNDARY_LOST")).count());
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- CB-580: a member that reaches GONE/NEVER_READY must fail its waiting tickets
|
||||
|
||||
private static FleetHealthMonitor monitorWith(BiConsumer<String, String> failTarget) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
AgentControl agents = new AgentControl(herdr);
|
||||
var scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
return new FleetHealthMonitor(agents, java.util.List::of,
|
||||
new MessageService(agents, new Injector(agents), new Rendezvous(), new InMemoryReplyInbox()),
|
||||
scheduler, () -> 1, 60, failTarget);
|
||||
}
|
||||
|
||||
@Test void terminalTransitionFailsTheTargetOnce() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
assertEquals("term_a", failTarget.calls.get(0).target());
|
||||
assertTrue(failTarget.calls.get(0).reason().contains("GONE"));
|
||||
}
|
||||
|
||||
@Test void neverReadyNamesItselfAsTheReason() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.NEVER_READY);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
assertTrue(failTarget.calls.get(0).reason().contains("NEVER_READY"));
|
||||
}
|
||||
|
||||
@Test void stayingInATerminalStateProducesOneFailureNotN() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(1, failTarget.calls.size());
|
||||
}
|
||||
|
||||
@Test void aNonTerminalFaultStateDoesNotFailTheTarget() {
|
||||
RecordingFailTarget failTarget = new RecordingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.TURN_BOUNDARY_LOST);
|
||||
monitor.stop();
|
||||
assertEquals(0, failTarget.calls.size());
|
||||
}
|
||||
|
||||
@Test void failTargetRetryIsBounded() {
|
||||
AlwaysThrowingFailTarget failTarget = new AlwaysThrowingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(FleetHealthMonitor.MAX_FAIL_TARGET_ATTEMPTS, failTarget.calls);
|
||||
}
|
||||
|
||||
@Test void exhaustedRetryStillDoesNotRefireOnAnUnchangedTick() {
|
||||
AlwaysThrowingFailTarget failTarget = new AlwaysThrowingFailTarget();
|
||||
FleetHealthMonitor monitor = monitorWith(failTarget);
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
int afterFirstTransition = failTarget.calls;
|
||||
monitor.reportTransition("term_a", HealthState.GONE);
|
||||
monitor.stop();
|
||||
assertEquals(afterFirstTransition, failTarget.calls);
|
||||
}
|
||||
|
||||
private record RecordedCall(String target, String reason) { }
|
||||
|
||||
private static final class RecordingFailTarget implements BiConsumer<String, String> {
|
||||
final java.util.List<RecordedCall> calls = new java.util.ArrayList<>();
|
||||
|
||||
@Override public void accept(String target, String reason) {
|
||||
calls.add(new RecordedCall(target, reason));
|
||||
}
|
||||
}
|
||||
|
||||
private static final class AlwaysThrowingFailTarget implements BiConsumer<String, String> {
|
||||
int calls = 0;
|
||||
|
||||
@Override public void accept(String target, String reason) {
|
||||
calls++;
|
||||
throw new RuntimeException("boom");
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,56 +0,0 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import org.junit.jupiter.api.Tag;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
import static org.junit.jupiter.api.Assumptions.assumeTrue;
|
||||
|
||||
/**
|
||||
* Contract test for the worker env seam against a REAL herdr. Under protocol 19 (CB-521) the
|
||||
* env map is injected at PANE CREATION ({@code tab.create}), not {@code agent.start} — and
|
||||
* {@code agent.start} now only launches supported agent kinds, so this probes the seed pane's
|
||||
* SHELL directly (never {@code claude}, so no subscription/token involvement) and always tears
|
||||
* the throwaway space down.
|
||||
*
|
||||
* <p>Tagged {@code contract}; run with {@code mvn test -Pcontract}.
|
||||
*/
|
||||
@Tag("contract")
|
||||
class AgentControlContractTest {
|
||||
|
||||
private boolean noSocket() {
|
||||
return !Files.exists(UnixSocketHerdrClient.defaultSocketPath());
|
||||
}
|
||||
|
||||
@Test
|
||||
void tabCreateInjectsEnvIntoTheSeedShell() throws Exception {
|
||||
assumeTrue(!noSocket(), "no herdr socket — skipping");
|
||||
try (UnixSocketHerdrClient herdr = UnixSocketHerdrClient.connect()) {
|
||||
WorkspaceControl spaces = new WorkspaceControl(herdr);
|
||||
Workspace space = spaces.ensureWorkspace("__fleet_env_contract__");
|
||||
Tab.Created tab = spaces.createTab(space.workspaceId(), null,
|
||||
Map.of("ANTHROPIC_BASE_URL", "http://gx00.gw:8000"));
|
||||
try {
|
||||
assertNotNull(tab.rootPaneId(), "tab.create must return the seed pane");
|
||||
Thread.sleep(1000); // let the seed shell reach its prompt
|
||||
herdr.call("pane.send_input", Map.of(
|
||||
"pane_id", tab.rootPaneId(),
|
||||
"text", "printf 'PROBE_BASE=[%s]\\n' \"$ANTHROPIC_BASE_URL\"",
|
||||
"keys", List.of("enter")));
|
||||
Thread.sleep(800);
|
||||
String visible = herdr.call("pane.read",
|
||||
Map.of("pane_id", tab.rootPaneId(), "source", "visible"))
|
||||
.path("read").path("text").asText("");
|
||||
assertTrue(visible.contains("PROBE_BASE=[http://gx00.gw:8000]"),
|
||||
"env map must reach the seed shell; saw: " + visible);
|
||||
} finally {
|
||||
spaces.closeTab(tab.tab().tabId());
|
||||
herdr.call("workspace.close", Map.of("workspace_id", space.workspaceId()));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,27 +0,0 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** Unit tests for PID → pane resolution (the herdr half of connection-based MCP identity). */
|
||||
class PaneLocatorTest {
|
||||
|
||||
private final PaneLocator loc = new PaneLocator(new FakeHerdr());
|
||||
|
||||
@Test
|
||||
void resolvesTerminalForAForegroundPid() {
|
||||
assertEquals("term_a", loc.terminalForPid(FakeHerdr.WORKER_PID));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForAPidInNoPane() {
|
||||
assertNull(loc.terminalForPid(999_999));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForNonPositivePid() {
|
||||
assertNull(loc.terminalForPid(0));
|
||||
assertNull(loc.terminalForPid(-1));
|
||||
}
|
||||
}
|
||||
@@ -1,501 +0,0 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Set;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/** Unit behaviour of the CB-106 completion resolver in isolation from the injector. */
|
||||
class CompletionResolverTest {
|
||||
|
||||
@Test
|
||||
void skipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
resolver.resolve("term_a", null); // no in-flight turn captured for this target
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a turn nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failSkipsTheScrapeWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
resolver.fail("term_a", null); // no in-flight turn, and no registered waiter to fall back to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"a wedge nobody is blocked on must not cost a transcript scrape");
|
||||
}
|
||||
|
||||
@Test
|
||||
void captureBaselineSkipsTheReadWhenNoSendIsWaiting() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ X\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous(); // no waiter opened
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
resolver.captureBaseline("term_a", TestTurnTokens.inert("term_a")); // no send to attribute a later completion to
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"with no waiting send there is no turn to baseline — skip the scrape");
|
||||
}
|
||||
|
||||
// --- CB-115 clean scrape: extract the last assistant block ----------------
|
||||
|
||||
@Test
|
||||
void extractsTheLastAssistantBlockStrippingChrome() {
|
||||
String raw = """
|
||||
⏺ Reading the file…
|
||||
|
||||
⏺ Done. The bug was an off-by-one in the loop bound.
|
||||
|
||||
╭──────────────────────────────────────╮
|
||||
│ > │
|
||||
╰──────────────────────────────────────╯
|
||||
⏵⏵ auto mode on · ? for shortcuts
|
||||
""";
|
||||
assertEquals("Done. The bug was an off-by-one in the loop bound.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void keepsMultiLineAssistantContent() {
|
||||
String raw = "⏺ Line one.\nLine two.\n❯ ";
|
||||
assertEquals("Line one.\nLine two.", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void fallsBackToRawTextWhenThereIsNoMarker() {
|
||||
String raw = "plain worker output with no glyph";
|
||||
assertEquals("plain worker output with no glyph", CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void blankScrapeYieldsEmpty() {
|
||||
assertTrue(CompletionResolver.lastAssistantBlock("").isEmpty());
|
||||
assertTrue(CompletionResolver.lastAssistantBlock(null).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void stripsSpinnerAndRuleChrome() {
|
||||
String raw = """
|
||||
⏺ Channel check confirmed — your message got through.
|
||||
|
||||
✻ Brewed for 11s
|
||||
|
||||
─────────────────────────────────────
|
||||
""";
|
||||
assertEquals("Channel check confirmed — your message got through.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
@Test
|
||||
void cutsANextTurnPromptEchoAndTrailingTipsFromTheBlock() {
|
||||
// The exact turn-2 leak: the scrape captured the settled answer, then a "✻ Cooked" spinner,
|
||||
// then the NEXT turn's echoed prompt, then a "✶ Forming…" spinner and trailing tips/warnings
|
||||
// whose lines (⎿, ⚠) are not themselves chrome-terminated. Stopping at the first boundary
|
||||
// (the ✻ spinner) is what keeps every one of those interface lines out of the reply.
|
||||
String raw = """
|
||||
⏺ Channel confirmed — the bridge reply delivered successfully.
|
||||
|
||||
✻ Cooked for 9s
|
||||
|
||||
❯ Thanks. Now a small task: what is 17 * 23? Show just the number.
|
||||
|
||||
|
||||
|
||||
✶ Forming…
|
||||
⎿ Tip: Name your conversations with /rename
|
||||
⚠ claude.ai connectors are disabled because ANTHROPIC_API_KEY is set
|
||||
""";
|
||||
assertEquals("Channel confirmed — the bridge reply delivered successfully.",
|
||||
CompletionResolver.lastAssistantBlock(raw));
|
||||
}
|
||||
|
||||
// --- CB-115 misattribution guard: suppress a stale (unchanged) completion -------
|
||||
|
||||
@Test
|
||||
void suppressesACompletionWhoseScrapeIsUnchangedFromDelivery() {
|
||||
// Rapid back-to-back turn: the pane still shows the PREVIOUS turn's answer when this turn's
|
||||
// (misattributed) completion boundary fires. The scrape == the delivery baseline, so the
|
||||
// send must NOT be resolved with the stale answer.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ 391\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
// The turn as captured at delivery: its waiter, and the previous turn's answer still on screen.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn); // scrape still "391" == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(), "a completion with no output change must not resolve the send");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesACompletionWhoseScrapeChangedSinceDelivery() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ No, 391 = 17 × 23.\n❯ "); // the worker's real answer
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
// Delivery baseline was the previous turn's "391"; the scrape now differs → resolve.
|
||||
var turn = new CompletionResolver.InFlight(waiter, "391");
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(), "a completion with new output must resolve the send");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("No, 391 = 17 × 23.", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void marksAClippedCompletionPaneTail() {
|
||||
String block = "⏺ " + "x".repeat(CompletionResolver.MAX_SCRAPE_CHARS + 1) + "\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals("x".repeat(CompletionResolver.MAX_SCRAPE_CHARS)
|
||||
+ "\n[Pane tail clipped: member did not call fleet_reply.]",
|
||||
waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void leavesAnUnclippedCompletionPaneTailUnmarked() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ complete report\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals("complete report", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesSynchronouslyBeforePostTurnContextClearing() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ previous answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.captureBaseline("term_a", new TurnToken("term_a", waiter));
|
||||
herdr.readText("⏺ answer that /clear would erase\n❯ ");
|
||||
|
||||
resolver.resolveBeforePostAction("term_a");
|
||||
|
||||
assertTrue(waiter.isDone(), "the answer is captured before the adapter sends /clear");
|
||||
assertEquals("answer that /clear would erase", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void suppressesAnUnchangedCompletionEvenWhenTheBlockExceedsTheScrapeCap() {
|
||||
// The fan-out issue-hunt finding: captureBaseline once stored the RAW (unclipped) assistant
|
||||
// block while resolve compares against a clip()'d tail. For a block longer than MAX_SCRAPE_CHARS
|
||||
// the two capped representations differ even when the pane never changed, so the CB-115
|
||||
// byte-identical guard failed to fire and a stale completion could resolve the send. Both sides
|
||||
// must clip identically. The returned-text marker is added only after this comparison, so an
|
||||
// unchanged >cap block on rapid back-to-back turns still stays suppressed.
|
||||
String longBlock = "⏺ " + "x".repeat(CompletionResolver.MAX_SCRAPE_CHARS + 500) + "\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(longBlock);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // a send is blocked on this turn
|
||||
resolver.captureBaseline("term_a", new TurnToken("term_a", waiter)); // baseline is the clipped >cap block
|
||||
var turn = resolver.inFlight("term_a");
|
||||
assertEquals(CompletionResolver.MAX_SCRAPE_CHARS, turn.baseline().length(),
|
||||
"the delivery baseline is clipped to the same cap resolve() applies to the tail");
|
||||
|
||||
resolver.resolve("term_a", turn); // scrape unchanged → clipped tail == baseline → suppress
|
||||
|
||||
assertFalse(waiter.isDone(),
|
||||
"an unchanged >cap block must still be recognised as stale and suppressed");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "the send stays waiting for a real reply");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenThereIsNoBaseline() {
|
||||
// No delivery baseline (e.g. the pre-turn read failed) ⇒ never suppress; the completion resolves.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ hello\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "with no baseline a completion resolves as before");
|
||||
assertEquals("hello", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWhenTheScrapeItselfFailsEvenWithABaselinePresent() {
|
||||
// The most important branch of the CB-115 guard: a failed read means the resolver could not
|
||||
// SEE the screen — "couldn't see", not "no change". It must still resolve the send (an empty
|
||||
// tail beats hanging until the caller's timeout), even though a baseline was captured. The
|
||||
// baseline here is "" (an empty pane at delivery), so without the !scrapeFailed clause the
|
||||
// byte-identical guard would wrongly match the empty tail and suppress.
|
||||
FakeHerdr herdr = new FakeHerdr().healthy(false); // agent.read throws HerdrException
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, ""); // empty pane baselined at delivery
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(waiter.isDone(),
|
||||
"a failed scrape must still resolve the send, not hang until the caller's timeout");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind());
|
||||
assertEquals("", waiter.getNow(null).text(), "the tail is empty because the screen was unreadable");
|
||||
}
|
||||
|
||||
// --- CB-115/CB-116 fail guard: an already-done or absent waiter is left alone ---------
|
||||
|
||||
@Test
|
||||
void failLeavesAnAlreadyResolvedWaiterUntouchedAndSkipsTheScrape() {
|
||||
// The send was already resolved (e.g. by the worker's explicit reply) before fail fired.
|
||||
// fail must not overwrite that value, and must not even scrape the worker — nobody needs it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied"));
|
||||
|
||||
resolver.fail("term_a", turn);
|
||||
|
||||
assertFalse(herdr.called("agent.read"),
|
||||
"fail must not scrape a waiter that is already done");
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"fail must not overwrite the existing resolution");
|
||||
assertEquals("already replied", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void failFallsBackToTheRegisteredWaiterWhenThereIsNoInFlightTurn() {
|
||||
// A never-delivered readiness failure has no in-flight record but still has a blocked send;
|
||||
// fail falls back to the waiter currently registered on the Rendezvous and fails it.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a"); // send registered, but no captureBaseline ever ran
|
||||
resolver.fail("term_a", null); // no in-flight turn → fall back to the registered waiter
|
||||
|
||||
assertTrue(waiter.isDone(), "fail falls back to the registered waiter when no turn is in flight");
|
||||
assertEquals(Rendezvous.Kind.FAILED, waiter.getNow(null).kind());
|
||||
assertEquals("stuck on an error screen", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void failIsLoggedAtWarnWithTheReason() {
|
||||
// CB-564: this used to be a bare DEBUG "failed send to X via turn-stall fallback" — a symptom
|
||||
// with no cause, and below the level anyone watching for member health would see. A fail that
|
||||
// resolves a caller's blocked send is at least WARN and must carry the reason.
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger resolverLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(CompletionResolver.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
resolverLog.addAppender(appender);
|
||||
resolverLog.setLevel(Level.WARN);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("stuck on an error screen");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
var waiter = rendezvous.open("term_a");
|
||||
|
||||
resolver.fail("term_a", null);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no turn-stall WARN logged");
|
||||
assertTrue(warn.contains("term_a"), "the log names the target: " + warn);
|
||||
assertTrue(warn.contains("stuck on an error screen"), "the log carries the reason: " + warn);
|
||||
assertTrue(waiter.isDone());
|
||||
} finally {
|
||||
resolverLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
// --- CB-116 waiter identity: a late completion never crosses into the next turn ---------
|
||||
|
||||
@Test
|
||||
void aLateCompletionForOneTurnNeverResolvesTheNextTurnsWaiter() {
|
||||
// The cross-turn stale reply the conversation test surfaced: turn N's completion fallback
|
||||
// fires AFTER turn N was resolved by an explicit fleet_reply and turn N+1 has opened its own
|
||||
// waiter on the same session. Resolving "whatever is waiting now" would hand turn N's stale
|
||||
// scrape to turn N+1; targeting turn N's captured waiter makes the late completion a no-op.
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ turn N answer\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiterN = rendezvous.open("term_a"); // turn N's send
|
||||
// The turn as the injector captured it at delivery (waiter + pre-turn baseline).
|
||||
var turnN = new CompletionResolver.InFlight(waiterN, "an earlier answer");
|
||||
|
||||
// Turn N is resolved by the worker's explicit reply, and its send deregisters the waiter.
|
||||
assertTrue(rendezvous.resolve("term_a", "N replied"));
|
||||
rendezvous.close("term_a", waiterN); // the sender's finally, before the next turn opens
|
||||
|
||||
// Turn N+1's send opens its own waiter on the same session (CB-548: open fails if the
|
||||
// previous waiter is still registered, so a clean turn deregisters it first as above).
|
||||
var waiterN1 = rendezvous.open("term_a");
|
||||
|
||||
resolver.resolve("term_a", turnN); // turn N's completion fallback finally fires
|
||||
|
||||
assertFalse(waiterN1.isDone(), "turn N's late completion must not resolve turn N+1's waiter");
|
||||
assertEquals(Rendezvous.Kind.REPLY, waiterN.getNow(null).kind(),
|
||||
"turn N stays resolved by its own reply");
|
||||
assertTrue(rendezvous.isWaiting("term_a"), "turn N+1 is still awaiting its own resolution");
|
||||
}
|
||||
|
||||
// --- CB-578 stage A: backend-exhausted classification ---------------------------------
|
||||
|
||||
@Test
|
||||
void classifiesAMatchingScrapeAsBackendExhaustedInsteadOfACompletedReply() {
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertTrue(waiter.isDone(), "a matching scrape still resolves the blocked send");
|
||||
assertEquals(Rendezvous.Kind.BACKEND_EXHAUSTED, waiter.getNow(null).kind(),
|
||||
"not reported as a completed reply — the classification is distinct");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theExhaustedReasonCarriesTheMatchedLine() {
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals("backend exhausted (usage limit): The usage limit has been reached. Try again later.",
|
||||
waiter.getNow(null).text(), "the reason names the real cause and carries the matched line");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWinningBackendExhaustedClassificationNotifiesTheExhaustionSink() {
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(1, notified.size(), "the sink is notified exactly once for the winning classification");
|
||||
assertTrue(notified.get(0).startsWith("term_a: "), "the sink is told which target exhausted");
|
||||
assertTrue(notified.get(0).contains("The usage limit has been reached"),
|
||||
"the sink is told the matched reason: " + notified.get(0));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aLosingBackendExhaustedClassificationNeverNotifiesTheExhaustionSink() {
|
||||
// The waiter was already resolved (e.g. by the worker's own reply) before this scrape landed —
|
||||
// resolveExhausted loses the race and must return false, so the sink must not fire either.
|
||||
String block = "⏺ Working on it...\nThe usage limit has been reached. Try again later.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
java.util.List<String> notified = new java.util.ArrayList<>();
|
||||
ExhaustionSink sink = (target, reason) -> notified.add(target + ": " + reason);
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, sink);
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
var turn = new CompletionResolver.InFlight(waiter, null);
|
||||
assertTrue(rendezvous.resolveCompletion(waiter, "already replied"));
|
||||
|
||||
resolver.resolve("term_a", turn);
|
||||
|
||||
assertTrue(notified.isEmpty(), "a classification that loses the race must not quarantine anything");
|
||||
assertEquals("already replied", waiter.getNow(null).text(), "the earlier resolution stands untouched");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingScrapeResolvesAsAnOrdinaryCompletion() {
|
||||
FakeHerdr herdr = new FakeHerdr().readText("⏺ complete report\n❯ ");
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
ExhaustedPatternLookup patterns = target -> Pattern.compile("usage limit has been reached");
|
||||
CompletionResolver resolver = new CompletionResolver(new AgentControl(herdr), rendezvous, patterns, ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"a scrape that does not match the pattern is an ordinary completion");
|
||||
assertEquals("complete report", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aProfileWithNoConfiguredPatternKeepsTodaysCompletionFallbackUnchanged() {
|
||||
// Even a scrape that WOULD have matched some other profile's pattern must resolve as a
|
||||
// plain completion when this target's own profile has none configured (CB-578 criterion 4).
|
||||
String block = "⏺ The usage limit has been reached.\n❯ ";
|
||||
FakeHerdr herdr = new FakeHerdr().readText(block);
|
||||
Rendezvous rendezvous = new Rendezvous();
|
||||
CompletionResolver resolver =
|
||||
new CompletionResolver(new AgentControl(herdr), rendezvous, ExhaustedPatternLookup.none(), ExhaustionSink.none());
|
||||
|
||||
var waiter = rendezvous.open("term_a");
|
||||
resolver.resolve("term_a", new CompletionResolver.InFlight(waiter, null));
|
||||
|
||||
assertEquals(Rendezvous.Kind.COMPLETION, waiter.getNow(null).kind(),
|
||||
"no pattern configured for this target's profile ⇒ unchanged completion-fallback behaviour");
|
||||
assertEquals("The usage limit has been reached.", waiter.getNow(null).text());
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsOffWhenNoProfileHasAPatternConfigured() {
|
||||
assertEquals("off (no profile has an exhaustedPattern configured; profiles: [terra])",
|
||||
CompletionResolver.coverage(Set.of("terra"), Set.of()));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsFullWhenEveryProfileHasAPatternConfigured() {
|
||||
assertEquals("full (all profiles configured: [gx10, terra])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra", "gx10")));
|
||||
}
|
||||
|
||||
@Test
|
||||
void coverageIsPartialAndNamesWhichProfilesAreConfigured() {
|
||||
assertEquals("partial (configured: [terra]; not configured: [gx10])",
|
||||
CompletionResolver.coverage(Set.of("terra", "gx10"), Set.of("terra")));
|
||||
}
|
||||
}
|
||||
@@ -1,46 +0,0 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/** Connection → caller-identity resolution, with the OS peer-PID lookup faked. */
|
||||
class ConnectionIdentityTest {
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
|
||||
private ConnectionIdentity with(PeerPidLookup pids) {
|
||||
return new ConnectionIdentity(new PaneLocator(herdr), pids);
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesWorkerFromLoopbackPeerPid() {
|
||||
assertEquals("term_a", with(_ -> FakeHerdr.WORKER_PID).callerTerminal("127.0.0.1", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullForOffHostCaller() {
|
||||
// A non-loopback peer can't be an on-host worker → treat as primary/unknown.
|
||||
assertNull(with(_ -> FakeHerdr.WORKER_PID).callerTerminal("10.0.0.9", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullWhenPidOwnsNoPane() {
|
||||
// e.g. the primary — its PID maps to no worker pane.
|
||||
assertNull(with(_ -> 999_999).callerTerminal("127.0.0.1", 55555));
|
||||
}
|
||||
|
||||
@Test
|
||||
void resolvesTheCallersPidAndCwd() {
|
||||
// CB-112: the primary maps to no pane, but its PID and cwd are still readable.
|
||||
ConnectionIdentity id = new ConnectionIdentity(
|
||||
new PaneLocator(herdr), _ -> 999_999, pid -> pid == 999_999 ? "/main/project" : null);
|
||||
ConnectionIdentity.Caller c = id.resolve("127.0.0.1", 55555);
|
||||
assertNull(c.terminal(), "the primary owns no worker pane");
|
||||
assertEquals(999_999, c.pid());
|
||||
assertEquals("/main/project", id.cwdForPid(c.pid()), "the primary's cwd is resolvable from its PID");
|
||||
assertNull(id.cwdForPid(-1), "no cwd for an unresolved PID");
|
||||
}
|
||||
}
|
||||
@@ -1,204 +0,0 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import dev.ltms.fleet.auth.Authz;
|
||||
import dev.ltms.fleet.auth.CallerResolver;
|
||||
import dev.ltms.fleet.auth.MemberRegistry;
|
||||
import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.auth.Role;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.PaneLocator;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.Injector;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.msg.Rendezvous;
|
||||
import dev.ltms.fleet.session.FakeWorktrees;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import io.modelcontextprotocol.spec.McpSchema;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-513 — the CB-505 authorization gate on the <strong>MCP</strong> entry path.
|
||||
*
|
||||
* <p>Why this file exists: CB-505 claimed authorization is "enforced on both entry paths", and it
|
||||
* is — but only REST was ever tested ({@code FleetAppAuthTest}). Coverage showed
|
||||
* {@code FleetMcp.deny()}, {@code principal()} and every tool-registration lambda at <em>zero</em>
|
||||
* executed lines, because no test had ever constructed a {@code FleetMcp} — the existing
|
||||
* {@code FleetMcpTest} calls only the static handler methods. An unexercised security control is
|
||||
* a claim, not a control.
|
||||
*
|
||||
* <p>These tests construct a real {@code FleetMcp} (which also exercises the constructor and the
|
||||
* tool wiring) and drive the policy half of the gate directly.
|
||||
*/
|
||||
class FleetMcpAuthzTest {
|
||||
|
||||
private final FakeHerdr herdr = new FakeHerdr();
|
||||
private final AgentControl agents = new AgentControl(herdr);
|
||||
private Metrics metrics;
|
||||
private FleetMcp mcp;
|
||||
|
||||
@AfterEach
|
||||
void close() {
|
||||
if (mcp != null) mcp.close();
|
||||
}
|
||||
|
||||
/** A fully wired FleetMcp on fakes — constructing it is itself part of what is under test. */
|
||||
private FleetMcp mcp(boolean enforce) {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN", null,
|
||||
"tab", "bridged-workers", "worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(agents, new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(),
|
||||
_ -> "tok");
|
||||
SessionManager sessions = new SessionManager(workers, new FakeWorktrees());
|
||||
MessageService messages = new MessageService(agents, new Injector(agents), new Rendezvous(),
|
||||
new InMemoryReplyInbox());
|
||||
ConnectionIdentity identity = new ConnectionIdentity(new PaneLocator(herdr), _ -> 999_999);
|
||||
metrics = FleetMetrics.create(sessions, new InMemoryReplyInbox());
|
||||
|
||||
mcp = new FleetMcp(messages, workers, sessions, identity, sessions.asPresence(),
|
||||
new PrimaryRegistry(null),
|
||||
enforce ? CallerResolver.withLeadsAndMembers(identity, false, null,
|
||||
Map::of, new MemberRegistry(null)) : null,
|
||||
metrics, FleetMcp.CapacitySource.none(), new FleetMcp.HealthCoverageSource(() -> "off"),
|
||||
FleetMcp.QuarantineSource.none());
|
||||
return mcp;
|
||||
}
|
||||
|
||||
private static final Principal PRIMARY = Principal.primary(100);
|
||||
private static final Principal WORKER_A = Principal.worker("term_a", 200);
|
||||
private static final Principal ANON = Principal.anonymous();
|
||||
private static final Principal ARCH_DESIGN = Principal.architect("lead-designer", "term_design", 400);
|
||||
|
||||
// --- the table, enforced on THIS path too ---------------------------------------------------
|
||||
|
||||
@Test
|
||||
void primaryMayOrchestrate() {
|
||||
FleetMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.SEND, Authz.Action.DRAIN, Authz.Action.READ}) {
|
||||
assertNull(m.denyFor(PRIMARY, a, "term_a"), a + " is the primary's to perform");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayNotOrchestrateOverMcp() {
|
||||
FleetMcp m = mcp(true);
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.SEND, Authz.Action.DRAIN}) {
|
||||
McpSchema.CallToolResult denied = m.denyFor(WORKER_A, a, "term_a");
|
||||
assertNotNull(denied, a + " must be refused to a worker");
|
||||
assertTrue(denied.isError(), "a refusal is returned as an MCP tool error");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWorkerMayReplyAndAskOnlyAsItself() {
|
||||
FleetMcp m = mcp(true);
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.REPLY, "term_a"), "its own session is allowed");
|
||||
assertNull(m.denyFor(WORKER_A, Authz.Action.ASK, "term_a"));
|
||||
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.REPLY, "term_b"),
|
||||
"worker A must not reply on worker B's session");
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.ASK, "term_b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void thePrimaryMayNotForgeAWorkerReplyOverMcp() {
|
||||
FleetMcp m = mcp(true);
|
||||
// A forged reply would resolve the very rendezvous the primary is blocked on.
|
||||
assertNotNull(m.denyFor(PRIMARY, Authz.Action.REPLY, "term_a"));
|
||||
assertNotNull(m.denyFor(PRIMARY, Authz.Action.ASK, "term_a"));
|
||||
}
|
||||
|
||||
// --- CB-548: the architect on this path ------------------------------------------------
|
||||
|
||||
@Test
|
||||
void anArchitectMaySendAndReadButNotOrchestrateOverMcp() {
|
||||
FleetMcp m = mcp(true);
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.SEND, "term_a"),
|
||||
"delegating a turn is the architect's job");
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.READ, null));
|
||||
|
||||
for (Authz.Action a : new Authz.Action[]{Authz.Action.SPAWN, Authz.Action.STOP,
|
||||
Authz.Action.DRAIN}) {
|
||||
McpSchema.CallToolResult denied = m.denyFor(ARCH_DESIGN, a, null);
|
||||
assertNotNull(denied, a + " must be refused to an architect");
|
||||
assertTrue(denied.isError(), "a refusal is returned as an MCP tool error");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void anArchitectMayReplyAndAskOnlyAsItsOwnPaneOverMcp() {
|
||||
FleetMcp m = mcp(true);
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.REPLY, "term_design"));
|
||||
assertNull(m.denyFor(ARCH_DESIGN, Authz.Action.ASK, "term_design"));
|
||||
|
||||
assertNotNull(m.denyFor(ARCH_DESIGN, Authz.Action.REPLY, "term_a"),
|
||||
"architect 'lead-designer' must not reply on worker term_a's session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void anonymousIsRefusedEverythingAndCountedAsUnauthenticated() {
|
||||
FleetMcp m = mcp(true);
|
||||
McpSchema.CallToolResult denied = m.denyFor(ANON, Authz.Action.READ, null);
|
||||
|
||||
assertNotNull(denied, "authenticated as nothing ⇒ authorized for nothing");
|
||||
assertEquals(1, metrics.count(FleetMetrics.AUTH_FAILURES, "reason", "unauthenticated"));
|
||||
assertEquals(0, metrics.count(FleetMetrics.AUTH_FAILURES, "reason", "forbidden"),
|
||||
"a missing credential is 401-shaped, not 403-shaped");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aWrongRoleIsCountedAsForbiddenNotUnauthenticated() {
|
||||
FleetMcp m = mcp(true);
|
||||
assertNotNull(m.denyFor(WORKER_A, Authz.Action.SPAWN, null));
|
||||
|
||||
assertEquals(1, metrics.count(FleetMetrics.AUTH_FAILURES, "reason", "forbidden"));
|
||||
assertEquals(0, metrics.count(FleetMetrics.AUTH_FAILURES, "reason", "unauthenticated"),
|
||||
"the caller IS authenticated — it is just not the right role");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theLegacyConstructorLeavesTheGateOpen() {
|
||||
// The 22 pre-existing FleetMcpTest cases rely on no authorization being enforced.
|
||||
FleetMcp m = mcp(false);
|
||||
assertNull(m.denyFor(ANON, Authz.Action.SPAWN, null),
|
||||
"no CallerResolver supplied ⇒ authorization not enforced (legacy behaviour)");
|
||||
}
|
||||
|
||||
// --- identity reconstruction from the transport context ------------------------------------
|
||||
|
||||
@Test
|
||||
void principalIsRebuiltFromTheStashedRole() {
|
||||
assertEquals(Role.WORKER, FleetMcp.principalFrom("WORKER", "term_a", 7).role());
|
||||
assertEquals("term_a", FleetMcp.principalFrom("WORKER", "term_a", 7).terminal());
|
||||
assertEquals(Role.PRIMARY, FleetMcp.principalFrom("PRIMARY", null, 7).role());
|
||||
assertEquals(Role.ANONYMOUS, FleetMcp.principalFrom("ANONYMOUS", null, -1).role());
|
||||
// CB-548: an architect round-trips through the same stash, carrying its slot name.
|
||||
Principal arch = FleetMcp.principalFrom("ARCHITECT", "term_design", 7, "lead-designer");
|
||||
assertEquals(Role.ARCHITECT, arch.role());
|
||||
assertEquals("lead-designer", arch.name());
|
||||
assertEquals("term_design", arch.terminal());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingRoleFallsBackToTheHistoricalInterpretation() {
|
||||
// Legacy path: no role stashed. A terminal means worker; its absence meant "the primary",
|
||||
// which is exactly the pre-CB-501 default CB-501 inverted — preserved only here.
|
||||
assertEquals(Role.WORKER, FleetMcp.principalFrom(null, "term_a", 7).role());
|
||||
assertEquals(Role.PRIMARY, FleetMcp.principalFrom(null, null, 7).role());
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
File diff suppressed because it is too large
Load Diff
@@ -1,726 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The composite router: profile → owning adapter for spawn/cwd/parity, pane id → owner for stop,
|
||||
* and fleet-wide union/dedup for list/reap/caps/profiles. Exercised through two real adapters —
|
||||
* claude-code + opencode — over one FakeHerdr, so each call is observed reaching the right adapter
|
||||
* (the started herdr agent name carries that adapter's {@code claude-}/{@code opencode-} prefix).
|
||||
*/
|
||||
class CompositePeerLauncherTest {
|
||||
|
||||
private ClaudeCodeLauncher claudeAdapter(FakeHerdr herdr) {
|
||||
// 12-arg back-compat Worker ctor → kind defaults to claude-code.
|
||||
FleetConfig.Profile claude = new FleetConfig.Profile("claude", "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null);
|
||||
return new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of("claude", claude), "claude", _ -> null);
|
||||
}
|
||||
|
||||
private OpenCodeLauncher opencodeAdapter(FakeHerdr herdr) {
|
||||
FleetConfig.Profile gemini = new FleetConfig.Profile("gemini", null, "google/gemini-2.5-pro",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("opencode"), "tab", "bridged-workers", "w #{n}",
|
||||
null, null, null, "GITEA_ACCESS_TOKEN", null, FleetConfig.Profile.KIND_OPENCODE);
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", gemini), "gemini", _ -> "tok");
|
||||
}
|
||||
|
||||
private CompositePeerLauncher composite(FakeHerdr herdr) {
|
||||
return new CompositePeerLauncher(
|
||||
List.of(claudeAdapter(herdr), opencodeAdapter(herdr)), "claude");
|
||||
}
|
||||
|
||||
/**
|
||||
* A minimal concrete HerdrPeerLauncher for policy tests. It either returns a fake handle for the
|
||||
* requested profile or throws, depending on {@code failProfiles}. buildLaunch is a stub; only
|
||||
* spawn/stop/list/caps/reap are exercised by the composite.
|
||||
*/
|
||||
private static final class StubLauncher extends HerdrPeerLauncher {
|
||||
private final Set<String> failProfiles;
|
||||
private final Map<String, Integer> spawnCounts = new HashMap<>();
|
||||
|
||||
StubLauncher(String prefix, FakeHerdr herdr,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Set<String> failProfiles) {
|
||||
super(prefix, new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
profiles, defaultProfile, _ -> null, 0L, System::currentTimeMillis, () -> { });
|
||||
this.failProfiles = Set.copyOf(failProfiles);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected Launch buildLaunch(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
return new Launch(Map.of(), List.of());
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String p = (req.profileName() == null || req.profileName().isBlank())
|
||||
? defaultProfile() : req.profileName();
|
||||
spawnCounts.merge(p, 1, Integer::sum);
|
||||
if (failProfiles.contains(p)) {
|
||||
throw new PeerUnreachableException(p + " is down");
|
||||
}
|
||||
return new PeerHandle() {
|
||||
@Override public String id() { return "pane-" + p; }
|
||||
@Override public String terminalId() { return "term-" + p; }
|
||||
@Override public String profile() { return p; }
|
||||
@Override public String agentSessionId() { return null; }
|
||||
@Override public CharterReceipt charterReceipt() { return null; }
|
||||
};
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) { }
|
||||
|
||||
@Override
|
||||
public List<Agent> list() { return List.of(); }
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() { return EnumSet.noneOf(Capability.class); }
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() { return 0; }
|
||||
|
||||
int spawnCount(String profile) {
|
||||
return spawnCounts.getOrDefault(profile, 0);
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile stubWorker(String profile) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile stubWorker(String profile, float weight, Integer maxLoad) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null,
|
||||
weight, maxLoad);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile stubWorker(String profile, String credentialId) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "BRIDGED_WORKER_TOKEN", List.of("claude"), "tab", "bridged-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null,
|
||||
null, null, credentialId, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* An <em>order-preserving</em> profile map. Never {@code Map.of} here: its iteration order is
|
||||
* salted per JVM run, and the weighted policy breaks an exact-weight tie on candidate order —
|
||||
* so a {@code Map.of} would make "which profile is tried first" a coin flip per run and any
|
||||
* assertion about the first attempt intermittently false.
|
||||
*/
|
||||
private static Map<String, FleetConfig.Profile> ordered(String first, FleetConfig.Profile a,
|
||||
String second, FleetConfig.Profile b) {
|
||||
Map<String, FleetConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put(first, a);
|
||||
m.put(second, b);
|
||||
return m;
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static String startedName(FakeHerdr herdr) {
|
||||
return (String) ((Map<String, Object>) herdr.lastCall("agent.start").params()).get("name");
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnRoutesEachProfileToItsOwningAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
|
||||
composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("opencode-"),
|
||||
"the gemini profile is spawned by the opencode adapter: " + startedName(herdr));
|
||||
|
||||
composite.spawn(new SpawnRequest("claude", null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"the claude profile is spawned by the claude-code adapter: " + startedName(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullProfileResolvesTheDefaultAndRoutesToItsOwner() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
composite(herdr).spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"a no-profile spawn resolves the default (claude) and routes to its adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unknownProfileIsRejected() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> composite.spawn(new SpawnRequest("nope", null, null)),
|
||||
"a profile no adapter declares is an error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void profilesAndDefaultAreExposedAcrossAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(Set.of("claude", "gemini"), composite.profiles(),
|
||||
"profiles are the union of every adapter's profiles");
|
||||
assertEquals("claude", composite.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesAreTheUnionOfEveryAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
ClaudeCodeLauncher claude = claudeAdapter(herdr);
|
||||
OpenCodeLauncher opencode = opencodeAdapter(herdr);
|
||||
PeerLauncher composite = new CompositePeerLauncher(List.of(claude, opencode), "claude");
|
||||
|
||||
assertTrue(composite.capabilities().containsAll(claude.capabilities()),
|
||||
"the fleet offers every claude-code capability");
|
||||
assertTrue(composite.capabilities().containsAll(opencode.capabilities()),
|
||||
"the fleet offers every opencode capability (incl. SELF_PR from its git-token profile)");
|
||||
}
|
||||
|
||||
@Test
|
||||
void listIsDeduplicatedByPaneIdAcrossAdaptersSharingHerdr() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
// Both adapters wrap the same herdr, so each list() returns the same global agent set;
|
||||
// the composite must return each pane once, not once per adapter.
|
||||
assertEquals(1, composite.list().size(),
|
||||
"the single herdr-tracked pane appears once, not duplicated per adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapSumsAcrossAdaptersAndEachAdapterReapsOnlyItsOwnPrefix() {
|
||||
// One foreign opencode orphan + one foreign claude orphan, from a prior daemon (different nonce).
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.withAgent("opencode-gemini-ffffff-1", "term_o", "wQ:pO", "wQ:tO")
|
||||
.withAgent("claude-claude-eeeeee-1", "term_c", "wQ:pC", "wQ:tC");
|
||||
PeerLauncher composite = composite(herdr);
|
||||
assertEquals(2, composite.reapOrphanWorkers(),
|
||||
"both orphans are reaped — one by each adapter, summed by the composite");
|
||||
}
|
||||
|
||||
@Test
|
||||
void stopTearsDownAPaneSpawnedThroughTheComposite() {
|
||||
// CB-519: handle.id() is a host-unique opaque UUID, not the herdr pane — stop(id) must
|
||||
// resolve it through the owning adapter down to the actual pane coordinate it spawned.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
assertNotEquals("w9:pRoot_1", handle.id(), "the id is decoupled from the pane coordinate");
|
||||
|
||||
composite.stop(handle.id());
|
||||
assertTrue(herdr.calls.stream()
|
||||
.anyMatch(c -> c.method().equals("pane.close")
|
||||
&& "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id"))),
|
||||
"stop routes to the spawning adapter and closes exactly that worker's pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void opencodeContextResetIsANoOpAndWarnsOnlyOnce() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
PeerHandle handle = composite.spawn(new SpawnRequest("gemini", null, null));
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(HerdrPeerLauncher.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
try {
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
assertFalse(composite.clearContext(handle.id()));
|
||||
} finally {
|
||||
logger.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertFalse(opencodeAdapter(herdr).capabilities().contains(Capability.CONTEXT_RESET));
|
||||
assertTrue(herdr.calls.stream().noneMatch(c -> "agent.prompt".equals(c.method())),
|
||||
"never type Claude's /clear into an opencode prompt");
|
||||
assertEquals(1, appender.list.stream()
|
||||
.filter(e -> e.getFormattedMessage().contains("context reset is unsupported"))
|
||||
.count(), "unsupported reset is logged once per adapter, not once per turn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAProfileClaimedByTwoAdapters() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// Two opencode adapters both declaring "gemini" — a profile-name collision.
|
||||
OpenCodeLauncher a = opencodeAdapter(herdr);
|
||||
OpenCodeLauncher b = opencodeAdapter(herdr);
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(a, b), "gemini"),
|
||||
"a profile two adapters both claim is a configuration error");
|
||||
}
|
||||
|
||||
@Test
|
||||
void constructorRejectsAnEmptyAdapterList() {
|
||||
assertThrows(IllegalArgumentException.class,
|
||||
() -> new CompositePeerLauncher(List.of(), "claude"),
|
||||
"at least one adapter must be configured");
|
||||
}
|
||||
|
||||
@Test
|
||||
void fixedDefaultIsNoOpForUnqualifiedSpawns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = composite(herdr);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertTrue(startedName(herdr).startsWith("claude-"),
|
||||
"fixed placement still routes an unqualified spawn to the default profile");
|
||||
assertEquals("claude", h.profile(), "the returned handle carries the resolved default profile");
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyGatesProfileAtMaxLoad() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
for (int i = 0; i < 5; i++) {
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "profile a is at maxLoad, so every spawn must land on b");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicyDistributesAccordingToWeightRatio() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.75f, null),
|
||||
"b", stubWorker("b", 0.25f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
int a = 0, b = 0;
|
||||
for (int i = 0; i < 40; i++) {
|
||||
String p = composite.spawn(new SpawnRequest(null, null, null)).profile();
|
||||
if ("a".equals(p)) a++;
|
||||
else if ("b".equals(p)) b++;
|
||||
}
|
||||
assertEquals(30, a, "weighted distribution should hold the 3:1 ratio");
|
||||
assertEquals(10, b);
|
||||
}
|
||||
|
||||
@Test
|
||||
void weightedPolicySkipsWeightZeroProfileOnUnqualifiedSpawn() {
|
||||
// CB-554: weight: 0 must exclude a profile from automatic placement, not coerce to 1.0.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.0f, null),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
for (int i = 0; i < 5; i++) {
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "profile a has weight 0, so every unqualified spawn must land on b");
|
||||
}
|
||||
assertEquals(0, adapter.spawnCount("a"), "a is never chosen automatically");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnStillSucceedsOnWeightZeroProfile() {
|
||||
// CB-554: weight: 0 excludes a profile from AUTOMATIC selection only — an explicit
|
||||
// fleet_spawn{profile:"a"} must still work exactly as today (e.g. `opus` on the
|
||||
// operator's own subscription, kept weight-0 so it is never picked automatically).
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 0.0f, null),
|
||||
"b", stubWorker("b", 1.0f, null));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "b", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "b", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "naming a weight-0 profile explicitly bypasses placement and still spawns it");
|
||||
assertEquals(1, adapter.spawnCount("a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverRetriesNextCandidateWhenProfileIsUnreachable() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "the spawn must fail over from unreachable a to b");
|
||||
assertEquals(1, adapter.spawnCount("a"), "a was tried once and failed");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b was tried once and succeeded");
|
||||
}
|
||||
|
||||
/**
|
||||
* Definition order — not hash order — decides an exact-weight tie. Paired with the test above
|
||||
* (same two profiles, opposite declaration order, opposite expected first attempt) this pins the
|
||||
* ordering contract from both sides: under a salted map one of the two must fail on every run.
|
||||
*/
|
||||
@Test
|
||||
void reversingDefinitionOrderReversesWhichProfileIsTriedFirst() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"b", stubWorker("b"),
|
||||
"a", stubWorker("a"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("a", h.profile(), "b is declared first and unreachable, so the spawn lands on a");
|
||||
assertEquals(1, adapter.spawnCount("b"), "b, declared first, is the one tried first");
|
||||
}
|
||||
|
||||
@Test
|
||||
void failoverBoundedByCandidateCount() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of("a", "b"));
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 0);
|
||||
|
||||
PeerUnreachableException e = assertThrows(PeerUnreachableException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("no reachable worker profile"), e.getMessage());
|
||||
assertEquals(1, adapter.spawnCount("a"));
|
||||
assertEquals(1, adapter.spawnCount("b"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnAtMaxLoadThrowsPlacementExceptionNamingProfileLiveAndCap() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 2 : 0);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("a", null, null)));
|
||||
assertTrue(e.getMessage().contains("'a'"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 live"), "message names the live count: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("2 cap"), "message names the cap: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("a"), "at cap, the spawn is refused before any delegation");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnUnderMaxLoadStillSucceeds() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 2),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), name -> "a".equals(name) ? 1 : 0);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile under its cap accepts an explicit spawn");
|
||||
assertEquals(1, adapter.spawnCount("a"), "the under-cap spawn is delegated");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnWithNullMaxLoadIsNeverCapped() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, null),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
// A deliberately absurd live count: an unset maxLoad means unlimited, so it must never refuse.
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 1000);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("a", null, null));
|
||||
assertEquals("a", h.profile(), "a profile with no maxLoad is never capped, however many live workers");
|
||||
}
|
||||
|
||||
@Test
|
||||
void explicitSpawnOnMaxLoadZeroProfileIsRefusedEvenWithZeroLiveWorkers() {
|
||||
// CB-585: before the fix, maxLoad: 0 normalised to null (unlimited) in the compact
|
||||
// constructor, so this exact case — naming a zero-cap profile explicitly, with nothing
|
||||
// live on it yet — would have spawned instead of refusing. "at most zero members" must
|
||||
// hold even when the profile is named directly, not only against automatic placement.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 0),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.fixed(), _ -> 0);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("a", null, null)));
|
||||
assertTrue(e.getMessage().contains("'a'"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("0 cap"), "message names the zero cap: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("a"), "a maxLoad: 0 profile accepts no explicit spawn");
|
||||
}
|
||||
|
||||
@Test
|
||||
void emptyCandidateSetThrowsClearException() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"a", stubWorker("a", 1.0f, 1),
|
||||
"b", stubWorker("b", 1.0f, 1));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "a", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(
|
||||
List.of(adapter), "a", profiles, PlacementPolicies.weighted(), _ -> 1);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
// ── CB-557: an unqualified spawn is placed inside its role's pool ─────────────────────────
|
||||
|
||||
/** Three profiles in definition order — pools are carved out of this set. */
|
||||
private static Map<String, FleetConfig.Profile> threeProfiles() {
|
||||
Map<String, FleetConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("opus", stubWorker("opus", 1.0f, null));
|
||||
m.put("sonnet", stubWorker("sonnet", 1.0f, null));
|
||||
m.put("terra", stubWorker("terra", 1.0f, null));
|
||||
return m;
|
||||
}
|
||||
|
||||
private static Map<String, FleetConfig.Slot> pool(String... names) {
|
||||
Map<String, FleetConfig.Slot> m = new LinkedHashMap<>();
|
||||
for (String n : names) {
|
||||
m.put(n, new FleetConfig.Slot(n));
|
||||
}
|
||||
return m;
|
||||
}
|
||||
|
||||
private static CompositePeerLauncher withPools(FakeHerdr herdr, FleetConfig.Fleet fleet) {
|
||||
Map<String, FleetConfig.Profile> profiles = threeProfiles();
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
return new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, fleet);
|
||||
}
|
||||
|
||||
/**
|
||||
* The point of the pools: a role is placed only on a backend its pool names. Before CB-557 an
|
||||
* unqualified spawn ranged over every configured profile, so a reviewer could land on the
|
||||
* architect-only one.
|
||||
*/
|
||||
@Test
|
||||
void anUnqualifiedSpawnIsPlacedInsideItsRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new FleetConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), pool("sonnet"), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.ARCHITECT)).profile());
|
||||
assertEquals("terra", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile());
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.REVIEWER)).profile());
|
||||
}
|
||||
|
||||
/** Under `fixed`, the pool's first entry wins — not the global defaultProfile. */
|
||||
@Test
|
||||
void theRolePoolOutranksTheGlobalDefaultProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new FleetConfig.Fleet(
|
||||
Map.of(), Map.of(), pool("sonnet", "terra"), Map.of(), null));
|
||||
|
||||
assertEquals("sonnet", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"the dev pool starts at sonnet, so the global default 'opus' must not win");
|
||||
}
|
||||
|
||||
/**
|
||||
* A role with no pool is unconstrained, not blocked. A config that declares pools for some roles
|
||||
* and not others must keep spawning the rest.
|
||||
*/
|
||||
@Test
|
||||
void aRoleWithNoPoolFallsBackToEveryProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new FleetConfig.Fleet(
|
||||
Map.of(), pool("sonnet"), Map.of(), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)).profile(),
|
||||
"no dev pool ⇒ all profiles are candidates, so `fixed` takes the first one");
|
||||
}
|
||||
|
||||
/** No fleet at all is the pre-CB-557 wiring, and must behave exactly as it did. */
|
||||
@Test
|
||||
void noFleetConfiguredKeepsTheOldWholeProfileListBehaviour() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, null);
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest(null, null, null)).profile());
|
||||
}
|
||||
|
||||
/**
|
||||
* An explicit profile is the operator overriding and is NOT judged against the pool. It must
|
||||
* stay that way: an unrolled `fleet_spawn{profile:"opus"}` carries no role, so it defaults to
|
||||
* DEV, and enforcing the pool here would refuse a spawn the operator asked for by name.
|
||||
*/
|
||||
@Test
|
||||
void anExplicitProfileIsNotConfinedToTheRolePool() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
CompositePeerLauncher composite = withPools(herdr, new FleetConfig.Fleet(
|
||||
Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
assertEquals("opus", composite.spawn(new SpawnRequest("opus", null, null)).profile(),
|
||||
"naming opus explicitly must work even though the dev pool holds only terra");
|
||||
}
|
||||
|
||||
/** Placement still respects maxLoad, but only across the pool — never by escaping it. */
|
||||
@Test
|
||||
void aFullPoolIsRefusedRatherThanSpilledOntoAnotherRolesProfile() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = new LinkedHashMap<>();
|
||||
profiles.put("opus", stubWorker("opus", 1.0f, null)); // architect-only, uncapped
|
||||
profiles.put("terra", stubWorker("terra", 1.0f, 1)); // the sole dev, capped at 1
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "opus", Set.of());
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "opus", profiles,
|
||||
PlacementPolicies.weighted(), name -> "terra".equals(name) ? 1 : 0,
|
||||
new FleetConfig.Fleet(Map.of(), pool("opus"), pool("terra"), Map.of(), null));
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class, () -> composite.spawn(
|
||||
new SpawnRequest(null, null, null, null, null, MemberRole.DEV)));
|
||||
assertTrue(e.getMessage().contains("maxLoad"), e.getMessage());
|
||||
}
|
||||
|
||||
// ── CB-578 stage B: a BACKEND_EXHAUSTED classification quarantines the credential ──────────
|
||||
|
||||
@Test
|
||||
void explicitSpawnOntoAQuarantinedProfileIsRefusedNamingTheCredentialAndRemainingTime() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
PlacementException e = assertThrows(PlacementException.class,
|
||||
() -> composite.spawn(new SpawnRequest("sol", null, null)));
|
||||
assertTrue(e.getMessage().contains("sol"), "message names the profile: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("shared-openai"), "message names the credential: " + e.getMessage());
|
||||
assertTrue(e.getMessage().contains("1800"), "message names roughly when it lifts: " + e.getMessage());
|
||||
assertEquals(0, adapter.spawnCount("sol"), "the quarantined profile is never delegated to");
|
||||
}
|
||||
|
||||
/**
|
||||
* The part CB-578 stage B calls out as easy to get wrong: sol and terra are two different
|
||||
* profiles sharing one OpenAI credential. Quarantining because of an exhaustion classified on
|
||||
* ONE of them must lock out the other too, or the fleet just walks onto the same dead account
|
||||
* under the sibling's name.
|
||||
*/
|
||||
@Test
|
||||
void twoProfilesSharingACredentialAreBothQuarantinedByOneExhaustionEvent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"terra", stubWorker("terra", "shared-openai"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
// Only "sol" was classified BACKEND_EXHAUSTED — but the two profiles share one credential.
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("sol", null, null)),
|
||||
"sol was the one classified exhausted");
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("terra", null, null)),
|
||||
"terra shares sol's credential, so it must be locked out too");
|
||||
assertEquals(0, adapter.spawnCount("sol"));
|
||||
assertEquals(0, adapter.spawnCount("terra"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void placementSkipsAQuarantinedProfileAndRoutesToAnUnquarantinedOne() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, quarantine);
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest(null, null, null));
|
||||
assertEquals("b", h.profile(), "sol is quarantined, so an unqualified spawn must land on b");
|
||||
assertEquals(0, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aQuarantineLiftsOnTheInjectedClockAndTheProfileBecomesSpawnableAgain() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Map<String, FleetConfig.Profile> profiles = ordered(
|
||||
"sol", stubWorker("sol", "shared-openai"),
|
||||
"b", stubWorker("b"));
|
||||
StubLauncher adapter = new StubLauncher("claude", herdr, profiles, "sol", Set.of());
|
||||
AtomicLong nowNanos = new AtomicLong(0L);
|
||||
BackendQuarantine quarantine = new BackendQuarantine(nowNanos::get, TimeUnit.MINUTES.toNanos(30));
|
||||
quarantine.quarantine("shared-openai");
|
||||
CompositePeerLauncher composite = new CompositePeerLauncher(List.of(adapter), "sol", profiles,
|
||||
PlacementPolicies.fixed(), _ -> 0, null, quarantine);
|
||||
|
||||
assertThrows(PlacementException.class, () -> composite.spawn(new SpawnRequest("sol", null, null)),
|
||||
"still inside the cooldown");
|
||||
|
||||
nowNanos.set(TimeUnit.MINUTES.toNanos(31));
|
||||
|
||||
PeerHandle h = composite.spawn(new SpawnRequest("sol", null, null));
|
||||
assertEquals("sol", h.profile(), "the cooldown expired on the injected clock — sol is spawnable again");
|
||||
assertEquals(1, adapter.spawnCount("sol"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFleetWithNoExhaustedPatternAnywhereBehavesExactlyAsBeforeQuarantineExisted() {
|
||||
// BackendQuarantine.none() is the inert stand-in every constructor already defaults to when
|
||||
// no quarantine is wired — the 2/5/6-arg constructors used throughout this file all exercise
|
||||
// it. This test pins that an explicit .none() also never refuses a spawn, for any profile.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerLauncher composite = composite(herdr);
|
||||
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("claude", null, null)));
|
||||
assertDoesNotThrow(() -> composite.spawn(new SpawnRequest("gemini", null, null)));
|
||||
}
|
||||
}
|
||||
@@ -1,211 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assumptions.assumeTrue;
|
||||
|
||||
/**
|
||||
* CB-633: the artefact here is a shell file handed to ANOTHER PROGRAM — so the only test that
|
||||
* proves anything is one that runs that program. A unit test on {@link EnvAllowListScrub}'s string
|
||||
* assembly proves nothing about zsh; this repo has shipped a green suite before whose tests checked
|
||||
* argv we build and none ran the binary that has to accept it.
|
||||
*
|
||||
* <p>This test starts a REAL login interactive zsh from a CLEAN parent ({@code env -i}), once with
|
||||
* the generated ZDOTDIR and once without (the baseline), and asserts the surviving exported NAME set
|
||||
* EQUALS the allow-list intersection of the baseline — equality, not "these names are blocked". A
|
||||
* blocked-name list can only check names somebody already thought of; that is exactly the failure
|
||||
* being fixed. Only NAMES are compared — never values.
|
||||
*
|
||||
* <p>The scrub run sources this operator's real {@code ~/.zshenv}/~/.zprofile/~/.zshrc/~/.zlogin}
|
||||
* chain, so it skips cleanly (JUnit {@code assumeTrue}) on a machine with no /bin/zsh or no real
|
||||
* shell rc files rather than failing there.
|
||||
*/
|
||||
class EnvAllowListScrubTest {
|
||||
|
||||
private static final Path ZSH = Path.of("/bin/zsh");
|
||||
|
||||
/** Env var names appearing in command output; anything else (prompts, wrapped lines) is noise. */
|
||||
private static final Pattern ENV_NAME = Pattern.compile("^([A-Za-z_][A-Za-z0-9_]*)$");
|
||||
|
||||
/**
|
||||
* The equality test. Expected survivors = baseline exports ∩ allowed — i.e. every survivor is
|
||||
* allowed AND every allowed name that existed survives. The operator's own secret-store exports
|
||||
* (~30 names on this host) are the decoys: none is on the derived list, so every one must be
|
||||
* gone from the scrub run.
|
||||
*/
|
||||
@Test
|
||||
void scrubbedLoginShellSurvivorsEqualTheDerivedAllowList(@TempDir Path tmp) throws Exception {
|
||||
assumeTrue(Files.isExecutable(ZSH), "/bin/zsh not present — nothing to prove here");
|
||||
Path homeZshrc = Path.of(System.getProperty("user.home"), ".zshrc");
|
||||
assumeTrue(Files.exists(homeZshrc), "$HOME/.zshrc does not exist — no real login chain to test against");
|
||||
|
||||
// Derived shape, zero profiles: infrastructure + LC_* rule. SSH_AUTH_SOCK deliberately NOT
|
||||
// included — blocked by default is the decision under test.
|
||||
Set<String> allowed = MemberEnvAllowList.derive(List.of());
|
||||
Path zdotdir = EnvAllowListScrub.generate(tmp, allowed);
|
||||
|
||||
Map<String, String> cleanParent = Map.of(
|
||||
"HOME", System.getProperty("user.home"),
|
||||
"PATH", "/usr/bin:/bin",
|
||||
"SHELL", "/bin/zsh",
|
||||
"USER", System.getProperty("user.name", "nobody"),
|
||||
"TMPDIR", tmp.toString());
|
||||
|
||||
Set<String> baseline = exportedNamesFromCleanParent(cleanParent, null);
|
||||
Set<String> scrubbed = exportedNamesFromCleanParent(cleanParent, zdotdir);
|
||||
|
||||
// The scrub RUN itself carries ZDOTDIR (the harness set it; it is infrastructure and MUST
|
||||
// survive, or every later login shell loses the scrub) — so the expected set starts from
|
||||
// baseline plus that one name.
|
||||
Set<String> expected = new TreeSet<>();
|
||||
for (String name : baseline) {
|
||||
if (MemberEnvAllowList.keeps(allowed, name)) {
|
||||
expected.add(name);
|
||||
}
|
||||
}
|
||||
assertTrue(MemberEnvAllowList.keeps(allowed, "ZDOTDIR"));
|
||||
expected.add("ZDOTDIR");
|
||||
assertEquals(expected, scrubbed,
|
||||
"surviving exported names must EQUAL baseline ∩ allow-list — a survivor outside the "
|
||||
+ "list is a leak; a missing allowed name means the scrub broke something it "
|
||||
+ "should have kept. Names only, values never printed.");
|
||||
}
|
||||
|
||||
/** The scrub writes its denominator report next to itself; names only, parseable. */
|
||||
@Test
|
||||
void scrubWritesAnAllowedNofMReport(@TempDir Path tmp) throws Exception {
|
||||
Set<String> allowed = MemberEnvAllowList.derive(List.of());
|
||||
Path zdotdir = EnvAllowListScrub.generate(tmp, allowed);
|
||||
|
||||
Map<String, String> cleanParent = Map.of(
|
||||
"HOME", System.getProperty("user.home"),
|
||||
"PATH", "/usr/bin:/bin",
|
||||
"SHELL", "/bin/zsh");
|
||||
exportedNamesFromCleanParent(cleanParent, zdotdir); // runs the login shell → runs the scrub
|
||||
|
||||
EnvAllowListScrub.ScrubReport report = EnvAllowListScrub.readReport(zdotdir);
|
||||
assertNotNull(report, "a completed login shell must leave a report behind");
|
||||
assertTrue(report.allowed() >= 0 && report.total() >= report.allowed(),
|
||||
"allowed N of M with N <= M — the denominator is always reported");
|
||||
}
|
||||
|
||||
/** Report parsing is lenient: absent file → null (no measurement), not an exception. */
|
||||
@Test
|
||||
void readReportReturnsNullForADirectoryWithoutOne(@TempDir Path dir) {
|
||||
assertNull(EnvAllowListScrub.readReport(dir));
|
||||
}
|
||||
|
||||
/**
|
||||
* The same equality, for a shell that is INTERACTIVE but NOT a login shell — the shape herdr
|
||||
* opens on Linux.
|
||||
*
|
||||
* <p>Why this test exists. The first version of this control put the scrub in {@code .zlogin}
|
||||
* alone. zsh reads {@code .zlogin} only for a login shell, and herdr does not open one
|
||||
* everywhere: measured on herdr 0.8.0, a macOS pane runs {@code -zsh} (login) while a Linux pane
|
||||
* runs a plain {@code /usr/bin/zsh}. So the control would have passed every test on the
|
||||
* developer's Mac and protected nothing at all on the vhost it was being built for, in silence.
|
||||
*
|
||||
* <p>This runs {@code zsh -i} — no {@code -l} — so {@code .zprofile} and {@code .zlogin} are
|
||||
* skipped exactly as they are on Linux. It therefore tests the Linux code path from a Mac,
|
||||
* which is the only place we can currently run it. Reverting the scrub to {@code .zlogin} only
|
||||
* makes this test fail while the login-shell test above still passes.
|
||||
*/
|
||||
@Test
|
||||
void scrubAlsoRunsInAnInteractiveNonLoginShell(@TempDir Path tmp) throws Exception {
|
||||
assumeTrue(Files.isExecutable(ZSH), "/bin/zsh not present — nothing to prove here");
|
||||
Path homeZshrc = Path.of(System.getProperty("user.home"), ".zshrc");
|
||||
assumeTrue(Files.exists(homeZshrc), "$HOME/.zshrc does not exist — no real chain to test against");
|
||||
|
||||
Set<String> allowed = MemberEnvAllowList.derive(List.of());
|
||||
Path zdotdir = EnvAllowListScrub.generate(tmp, allowed);
|
||||
|
||||
Map<String, String> cleanParent = Map.of(
|
||||
"HOME", System.getProperty("user.home"),
|
||||
"PATH", "/usr/bin:/bin",
|
||||
"SHELL", "/bin/zsh",
|
||||
"USER", System.getProperty("user.name", "nobody"),
|
||||
"TMPDIR", tmp.toString());
|
||||
|
||||
List<String> interactiveOnly = List.of("-i");
|
||||
Set<String> baseline = exportedNamesFromCleanParent(cleanParent, null, interactiveOnly);
|
||||
Set<String> scrubbed = exportedNamesFromCleanParent(cleanParent, zdotdir, interactiveOnly);
|
||||
|
||||
Set<String> expected = new TreeSet<>();
|
||||
for (String name : baseline) {
|
||||
if (MemberEnvAllowList.keeps(allowed, name)) {
|
||||
expected.add(name);
|
||||
}
|
||||
}
|
||||
expected.add("ZDOTDIR"); // the harness set it and it is infrastructure, so it must survive
|
||||
assertEquals(expected, scrubbed,
|
||||
"a non-login interactive zsh is what a herdr pane runs on Linux; its surviving "
|
||||
+ "exported names must EQUAL baseline \u2229 allow-list, exactly as for a login "
|
||||
+ "shell. A difference here means the scrub is dead on Linux.");
|
||||
}
|
||||
|
||||
/**
|
||||
* Run {@code /bin/zsh -l -i} from a clean parent and return the NAMES it has exported by prompt
|
||||
* time. With {@code zdotdir} non-null, {@code ZDOTDIR} points at a generated scrub directory, so
|
||||
* the login chain ends in CB-633's scrub; null gives the un-scrubbed baseline.
|
||||
*/
|
||||
private static Set<String> exportedNamesFromCleanParent(Map<String, String> cleanParent,
|
||||
Path zdotdir) throws IOException, InterruptedException {
|
||||
return exportedNamesFromCleanParent(cleanParent, zdotdir, List.of("-l", "-i"));
|
||||
}
|
||||
|
||||
private static Set<String> exportedNamesFromCleanParent(Map<String, String> cleanParent,
|
||||
Path zdotdir, List<String> shellFlags)
|
||||
throws IOException, InterruptedException {
|
||||
List<String> argv = new ArrayList<>();
|
||||
argv.add("/bin/zsh");
|
||||
argv.addAll(shellFlags);
|
||||
ProcessBuilder pb = new ProcessBuilder(argv);
|
||||
pb.environment().clear();
|
||||
pb.environment().putAll(cleanParent);
|
||||
if (zdotdir != null) {
|
||||
pb.environment().put("ZDOTDIR", zdotdir.toAbsolutePath().toString());
|
||||
}
|
||||
pb.redirectError(ProcessBuilder.Redirect.DISCARD); // prompts and rc chatter, never data
|
||||
|
||||
Process zsh = pb.start();
|
||||
// Names of exports whose VALUE is still non-empty. A blanked variable stays EXPORTED with
|
||||
// an empty value ("NAME=") — that is the control working, not surviving — so plain
|
||||
// `env | cut -d= -f1` would wrongly count blanked names as survivors.
|
||||
String probeScript = "command env | awk -F= '/^[A-Za-z_][A-Za-z0-9_]*=/ && length($0) > "
|
||||
+ "length($1)+1 { print $1 }' | command sort -u\nexit\n";
|
||||
zsh.getOutputStream().write(probeScript.getBytes(StandardCharsets.UTF_8));
|
||||
zsh.getOutputStream().flush();
|
||||
|
||||
String stdout = new String(zsh.getInputStream().readAllBytes(), StandardCharsets.UTF_8);
|
||||
assertTrue(zsh.waitFor(60, java.util.concurrent.TimeUnit.SECONDS),
|
||||
"the probe login shell did not exit within 60s");
|
||||
assertTrue(zsh.exitValue() == 0, "probe zsh exited non-zero — see test failure, values never printed");
|
||||
|
||||
Set<String> names = new HashSet<>();
|
||||
for (String line : stdout.split("\n")) {
|
||||
Matcher m = ENV_NAME.matcher(line.trim());
|
||||
if (m.matches()) {
|
||||
names.add(m.group(1));
|
||||
}
|
||||
}
|
||||
return names;
|
||||
}
|
||||
}
|
||||
-170
@@ -1,170 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-633: proves the allow-list scrub is actually WIRED INTO the spawn path — not merely that its
|
||||
* pieces work when a test calls them directly.
|
||||
*
|
||||
* <p>Why this test exists, and why it is separate from {@link EnvAllowListScrubTest}. Every other
|
||||
* test of this feature calls {@code EnvAllowListScrub} or {@code MemberEnvAllowList} itself. Those
|
||||
* prove the scrub is correct. None of them proves anyone runs it: deleting the single
|
||||
* {@code applyEnvironmentAllowListPolicy(cfg, launch)} line from {@code spawnInternal} left all 896
|
||||
* tests green while turning the control completely off. That is the recurring shape in this
|
||||
* codebase — a feature behind one call, with every test on the far side of it (CB-586, CB-611).
|
||||
*
|
||||
* <p>So this test starts a real spawn through {@link HerdrPeerLauncher#spawn} and asserts on what
|
||||
* reached herdr. It deliberately checks the pane-creation parameters rather than the launcher's own
|
||||
* map, because the map is an intermediate: {@code ZDOTDIR} only protects anything if it is in the
|
||||
* env herdr uses to create the pane, and that is the last point we can observe before the shell
|
||||
* starts.
|
||||
*/
|
||||
class HerdrPeerLauncherAllowListWiringTest {
|
||||
|
||||
/** A name the daemon itself injects — it must survive its own scrub, so it must be allowed. */
|
||||
private static final String INJECTED = "ANTHROPIC_BASE_URL";
|
||||
|
||||
@Test
|
||||
void spawningUnderAllowListPolicyGivesThePaneAGeneratedZdotdir() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList());
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
String paneParams = String.valueOf(herdr.lastCall("pane.split").params());
|
||||
assertTrue(paneParams.contains("ZDOTDIR"),
|
||||
"the spawn must hand herdr a ZDOTDIR so the pane's zsh reads our generated startup "
|
||||
+ "files; without it the scrub never runs and the member inherits the whole "
|
||||
+ "host environment. pane.split params were: " + paneParams);
|
||||
|
||||
Path dir = Path.of(launcher.env.get("ZDOTDIR"));
|
||||
assertTrue(Files.isDirectory(dir), "ZDOTDIR must point at a directory that exists: " + dir);
|
||||
// Both startup files must exist and both must source the scrub: .zlogin covers macOS panes
|
||||
// (login shells), .zshrc covers Linux panes (interactive, NOT login). Checking only one
|
||||
// would pass on the platform it was written for and ship a dead control on the other.
|
||||
for (String file : List.of(".zshrc", ".zlogin")) {
|
||||
Path f = dir.resolve(file);
|
||||
assertTrue(Files.isRegularFile(f), file + " must be generated: " + f);
|
||||
assertTrue(readAll(f).contains(EnvAllowListScrub.SCRUB_FILE),
|
||||
file + " must source " + EnvAllowListScrub.SCRUB_FILE + " — a scrub only one of "
|
||||
+ "them runs is dead on the platform that reads the other");
|
||||
}
|
||||
assertTrue(readAll(dir.resolve(EnvAllowListScrub.SCRUB_FILE)).contains(INJECTED),
|
||||
"the allow-list must include the names this very launch injects (" + INJECTED
|
||||
+ "), or the daemon's own configuration is blanked by its own control");
|
||||
}
|
||||
|
||||
/** The default policy must not generate anything — an upgrade changes nothing until asked. */
|
||||
@Test
|
||||
void spawningUnderTheDefaultPolicyGeneratesNoZdotdir() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, () -> new FleetConfig.MemberCredentials(
|
||||
null, List.of(), List.of(), null));
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"policy=deny-by-default is the shipped default; it must not silently start "
|
||||
+ "rewriting members' shell startup files");
|
||||
}
|
||||
|
||||
/**
|
||||
* A non-zsh shell cannot read {@code ZDOTDIR} at all. The launcher must fall back rather than
|
||||
* generate a directory nothing will ever read — a directory that would look like protection.
|
||||
*/
|
||||
@Test
|
||||
void aNonZshShellGeneratesNothingAndFallsBack() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList(), "/bin/bash");
|
||||
|
||||
launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
|
||||
assertFalse(launcher.env.containsKey("ZDOTDIR"),
|
||||
"bash ignores ZDOTDIR; setting it would be protection theatre");
|
||||
}
|
||||
|
||||
private static Supplier<FleetConfig.MemberCredentials> allowList() {
|
||||
return () -> new FleetConfig.MemberCredentials(
|
||||
FleetConfig.MemberCredentials.POLICY_ALLOW_LIST, List.of(), List.of(), null);
|
||||
}
|
||||
|
||||
private static String readAll(Path p) {
|
||||
try {
|
||||
return Files.readString(p);
|
||||
} catch (java.io.IOException e) {
|
||||
throw new AssertionError("cannot read " + p, e);
|
||||
}
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile profile() {
|
||||
return new FleetConfig.Profile("test", "http://gx00.gw:8000", null, null,
|
||||
"BRIDGED_WORKER_TOKEN", List.of("test"), "pane", null, null, null, null, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* A minimal peer launcher whose {@code buildLaunch} returns a MUTABLE env map holding one name
|
||||
* the daemon injects. Mutable on purpose: the policy adds {@code ZDOTDIR} to this very map, so
|
||||
* an immutable one would throw and the test would pass for the wrong reason.
|
||||
*/
|
||||
private static final class WiringLauncher extends HerdrPeerLauncher {
|
||||
private final Map<String, String> env = new HashMap<>(Map.of(INJECTED, "http://gateway"));
|
||||
|
||||
WiringLauncher(FakeHerdr herdr, Supplier<FleetConfig.MemberCredentials> creds) {
|
||||
this(herdr, creds, "/bin/zsh");
|
||||
}
|
||||
|
||||
WiringLauncher(FakeHerdr herdr, Supplier<FleetConfig.MemberCredentials> creds, String shell) {
|
||||
super("test", new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("test", profile()), "test",
|
||||
name -> "SHELL".equals(name) ? shell : null,
|
||||
0, () -> 0L, () -> { }, null, creds);
|
||||
}
|
||||
|
||||
@Override
|
||||
protected Launch buildLaunch(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
return new Launch(env, List.of("test"));
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
return Set.of();
|
||||
}
|
||||
}
|
||||
|
||||
/** The generated directory is a temp directory; make sure the test does not leave a pile. */
|
||||
@Test
|
||||
void theGeneratedDirectoryIsRemovedWhenThePaneIsStopped() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
WiringLauncher launcher = new WiringLauncher(herdr, allowList());
|
||||
|
||||
var spawned = launcher.spawn(new SpawnRequest("test", null, null, null, null, MemberRole.DEV));
|
||||
Path dir = Path.of(launcher.env.get("ZDOTDIR"));
|
||||
assertNotNull(spawned, "spawn returned nothing");
|
||||
assertTrue(Files.isDirectory(dir));
|
||||
|
||||
launcher.stop(spawned.id());
|
||||
|
||||
assertEquals(false, Files.exists(dir),
|
||||
"stopping the pane must remove its generated ZDOTDIR: " + dir);
|
||||
}
|
||||
}
|
||||
@@ -1,614 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ExecutorService;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.Future;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* The opencode adapter's launch build: a file-based MCP mount + reply-charter instructions (no
|
||||
* inline flags, no {@code ANTHROPIC_*}, no guard), the {@code -m} model flag, and the shared base
|
||||
* transport (naming, reap, readiness gate) proving the {@link HerdrPeerLauncher} SPI is neutral.
|
||||
*/
|
||||
class OpenCodeLauncherTest {
|
||||
|
||||
private static FleetConfig.Profile opencodeCfg(String model, String mcpUrl, String gitTokenEnv) {
|
||||
return new FleetConfig.Profile("gemini", null, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, gitTokenEnv, null, FleetConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
/** Gate-disabled launcher whose per-spawn config dirs land under an inspectable temp root. */
|
||||
private static OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, FleetConfig.Profile cfg) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), k -> "GITEA_ACCESS_TOKEN".equals(k) ? "tok" : null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot);
|
||||
}
|
||||
|
||||
private static OpenCodeLauncher service(FakeHerdr herdr, Path configRoot, FleetConfig.Profile cfg,
|
||||
Supplier<FleetConfig.Fleet> fleet) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot, fleet);
|
||||
}
|
||||
|
||||
private static OpenCodeLauncher serviceWithCredentials(FakeHerdr herdr, Path configRoot,
|
||||
FleetConfig.Profile cfg,
|
||||
FleetConfig.MemberCredentials creds) {
|
||||
return new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), k -> "GITEA_ACCESS_TOKEN".equals(k) ? "tok" : null,
|
||||
0, System::currentTimeMillis, () -> { }, configRoot, configRoot, null, () -> creds);
|
||||
}
|
||||
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, Object> lastStart(FakeHerdr herdr) {
|
||||
return (Map<String, Object>) herdr.lastCall("agent.start").params();
|
||||
}
|
||||
|
||||
/** Protocol 19: the worker's env is injected at pane creation (tab.create), not agent.start. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static Map<String, String> startEnv(FakeHerdr herdr) {
|
||||
Map<String, String> env =
|
||||
(Map<String, String>) ((Map<String, Object>) herdr.lastCall("tab.create").params()).get("env");
|
||||
return env == null ? Map.of() : env;
|
||||
}
|
||||
|
||||
/** Protocol 19: agent.start carries only the args after the kind-resolved executable. */
|
||||
@SuppressWarnings("unchecked")
|
||||
private static List<String> startArgs(FakeHerdr herdr) {
|
||||
return (List<String>) lastStart(herdr).get("args");
|
||||
}
|
||||
|
||||
@Test
|
||||
void writesRemoteMcpConfigAndCharterInstructionsWhenMcpUrlSet(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Fleet fleet = new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null),
|
||||
() -> fleet).spawn();
|
||||
|
||||
Map<String, String> env = startEnv(herdr);
|
||||
assertNull(env.get("ANTHROPIC_BASE_URL"), "opencode carries no ANTHROPIC_* / subscription boundary");
|
||||
String cfgPath = env.get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "OPENCODE_CONFIG points the worker at the generated config file");
|
||||
assertTrue(Path.of(cfgPath).startsWith(root), "config file is generated under the injected root");
|
||||
|
||||
// Assert on parsed structure, not substrings: the generated config is real JSON and its
|
||||
// whitespace is the formatter's business, not the contract's.
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
assertTrue(json.path("compaction").path("auto").asBoolean(),
|
||||
"spawned opencode peers explicitly enable automatic compaction");
|
||||
JsonNode mount = json.path("mcp").path("fleet");
|
||||
assertEquals("remote", mount.path("type").asText(), "the fleet MCP server is mounted as remote");
|
||||
assertEquals("http://127.0.0.1:8765/mcp", mount.path("url").asText(),
|
||||
"the profile's fleet MCP url is present");
|
||||
assertTrue(mount.path("enabled").asBoolean(), "the fleet server is enabled");
|
||||
assertTrue(json.path("mcp").path("bridge").isMissingNode(),
|
||||
"the mount is named fleet since CB-632, not bridge");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the member charter is mounted via instructions");
|
||||
|
||||
// The instructions entry is a real file path holding the composed member charter.
|
||||
Path charter = Path.of(cfgPath).resolveSibling("member-charter.md");
|
||||
assertTrue(Files.exists(charter), "the charter file the config references was written");
|
||||
assertEquals("role rule\n\n" + HerdrPeerLauncher.REPLY_CHARTER, Files.readString(charter),
|
||||
"the composed charter keeps the role rule first and the reply rule last");
|
||||
assertEquals(charter.toAbsolutePath().toString(), json.path("instructions").get(0).asText(),
|
||||
"instructions names the charter file by its absolute path");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noConfigFileWhenMcpUrlAbsent(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
assertNull(startEnv(herdr).get("OPENCODE_CONFIG"),
|
||||
"no bridge MCP url → no config file and no OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void roleCharterWithoutMcpOrCustomProviderStillWritesAConfig(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Fleet fleet = new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
Path configRoot = Files.createDirectory(root.resolve("configs"));
|
||||
Path checkout = Files.createDirectory(root.resolve("checkout"));
|
||||
service(herdr, configRoot, opencodeCfg("google/gemini-2.5-pro", null, null), () -> fleet).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a role charter needs a config even without MCP or custom provider");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
Path charter = Path.of(json.path("instructions").get(0).asText());
|
||||
assertEquals("role rule", Files.readString(charter), "the base-composed role charter is unchanged");
|
||||
assertTrue(json.path("mcp").isMissingNode(), "a charter does not add an MCP mount");
|
||||
assertTrue(charter.startsWith(configRoot), "the charter is written under the temp config root");
|
||||
try (var files = Files.walk(checkout)) {
|
||||
assertFalse(files.anyMatch(path -> path.getFileName().toString().equals("member-charter.md")),
|
||||
"the worker checkout receives no charter file");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void nullCharterWritesNoCharterFileOrInstructions(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/model", "http://127.0.0.1:8000", null),
|
||||
() -> new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(), Map.of(), null)).spawn();
|
||||
|
||||
String config = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(config, "the custom provider still needs a config");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(config).toFile());
|
||||
assertTrue(json.path("instructions").isMissingNode(), "a null charter adds no instructions entry");
|
||||
try (var files = Files.walk(root)) {
|
||||
assertFalse(files.anyMatch(path -> path.getFileName().toString().equals("member-charter.md")),
|
||||
"a null charter creates no charter file");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void concurrentSpawnsWriteSeparateCharterDirectories(@TempDir Path root) throws Exception {
|
||||
FleetConfig.Profile cfg = opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null);
|
||||
ExecutorService executor = Executors.newFixedThreadPool(2);
|
||||
try {
|
||||
Future<String> first = executor.submit(() -> spawnConfigPath(root, cfg));
|
||||
Future<String> second = executor.submit(() -> spawnConfigPath(root, cfg));
|
||||
|
||||
Path firstCharter = Path.of(first.get()).resolveSibling("member-charter.md");
|
||||
Path secondCharter = Path.of(second.get()).resolveSibling("member-charter.md");
|
||||
assertNotEquals(firstCharter.getParent(), secondCharter.getParent(),
|
||||
"each concurrent spawn owns a separate config directory");
|
||||
assertTrue(Files.exists(firstCharter));
|
||||
assertTrue(Files.exists(secondCharter));
|
||||
} finally {
|
||||
executor.shutdownNow();
|
||||
}
|
||||
}
|
||||
|
||||
private static String spawnConfigPath(Path root, FleetConfig.Profile cfg) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, cfg).spawn();
|
||||
return startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
}
|
||||
|
||||
@Test
|
||||
void passesTheModelAsDashMFlagAlongsideAutoApprove(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null)).spawn();
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
assertTrue(args.contains("--auto"),
|
||||
"--auto is present alongside -m so a spawned peer never blocks on approval");
|
||||
int m = args.indexOf("-m");
|
||||
assertTrue(m >= 0, "model is selected with -m");
|
||||
assertEquals("google/gemini-2.5-pro", args.get(m + 1), "the provider/model selector follows -m");
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoApproveIsUnconditionalWhenModelBlank(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null)).spawn();
|
||||
assertEquals(List.of("--auto"), startArgs(herdr),
|
||||
"--auto is unconditional: a model-less worker still must never block on approval");
|
||||
}
|
||||
|
||||
// --- CB-617: --agent <role> when the role has an agent-definition file --------------------
|
||||
|
||||
@Test
|
||||
void agentFlagIsPassedWhenTheRoleAgentDefinitionFileExists(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Path agentsDir = Files.createDirectories(root.resolve(".opencode/agent"));
|
||||
Files.writeString(agentsDir.resolve("dev.md"), "You are a dev.");
|
||||
OpenCodeLauncher svc = service(herdr, root, opencodeCfg(null, null, null));
|
||||
|
||||
svc.spawn(new SpawnRequest(null, root.toString(), null, null, null, MemberRole.DEV));
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
int flag = args.indexOf("--agent");
|
||||
assertTrue(flag >= 0, "--agent is passed when the role's agent file exists: " + args);
|
||||
assertEquals("dev", args.get(flag + 1));
|
||||
}
|
||||
|
||||
@Test
|
||||
void noAgentFlagWhenTheRoleAgentDefinitionFileIsAbsent(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher svc = service(herdr, root, opencodeCfg(null, null, null));
|
||||
|
||||
PeerHandle handle = svc.spawn(new SpawnRequest(null, root.toString(), null, null, null, MemberRole.DEV));
|
||||
|
||||
assertNotNull(handle, "the member still spawns with no agent-definition file");
|
||||
assertFalse(startArgs(herdr).contains("--agent"),
|
||||
"no --agent flag when the role has no agent-definition file");
|
||||
}
|
||||
|
||||
@Test
|
||||
void injectsForgeTokenWhenProfileGrantsIt(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN")).spawn();
|
||||
assertEquals("tok", startEnv(herdr).get("GITEA_TOKEN"),
|
||||
"a git-token profile gets the peer-neutral GITEA_TOKEN grant, same as Claude");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-596: the config-driven shadow lives in {@link HerdrPeerLauncher#baseEnv}, shared by every
|
||||
* adapter — this pins that the opencode path gets it too, not just Claude's. See the matching
|
||||
* tests in {@code ClaudeCodeLauncherTest} for the full rationale (gitea #82, superseding CB-592's
|
||||
* single hardcoded name).
|
||||
*/
|
||||
@Test
|
||||
void aKnownNameNotAllowedIsShadowedWithTheSentinel(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.MemberCredentials creds = new FleetConfig.MemberCredentials(
|
||||
null, List.of("AI_GATEWAY_TOKEN"), List.of("AI_GATEWAY_TOKEN", "GITEA_ACCESS_TOKEN"));
|
||||
serviceWithCredentials(herdr, root, opencodeCfg(null, null, null), creds).spawn();
|
||||
|
||||
String shadowed = startEnv(herdr).get("GITEA_ACCESS_TOKEN");
|
||||
assertNotNull(shadowed, "GITEA_ACCESS_TOKEN must be explicitly overlaid, not left unmentioned");
|
||||
assertFalse(shadowed.isBlank(), "a blank overlay value's override behaviour is unverified — must be non-blank");
|
||||
assertFalse(startEnv(herdr).containsKey("AI_GATEWAY_TOKEN"),
|
||||
"an allow-listed name must get no overlay entry at all");
|
||||
}
|
||||
|
||||
/** No {@code memberCredentials} configured (the pre-CB-596 constructor overloads) blocks nothing. */
|
||||
@Test
|
||||
void noMemberCredentialsConfiguredBlocksNothing(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null)).spawn();
|
||||
|
||||
assertNull(startEnv(herdr).get("GITEA_ACCESS_TOKEN"),
|
||||
"with no memberCredentials configured, nothing is shadowed — config must supply the policy");
|
||||
}
|
||||
|
||||
@Test
|
||||
void capabilitiesDeclareOrphanReapAndMcpAskAndConditionalSelfPr(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
assertEquals(java.util.Set.of(Capability.MID_TURN_ASK, Capability.WORKTREE, Capability.ORPHAN_REAP,
|
||||
Capability.SESSION_RESUME),
|
||||
service(herdr, root, opencodeCfg(null, null, null)).capabilities(),
|
||||
"opencode can be resumed by its own session id, so SESSION_RESUME is always declared");
|
||||
assertFalse(service(herdr, root, opencodeCfg(null, null, null))
|
||||
.capabilities().contains(Capability.SESSION_NAME),
|
||||
"opencode has no display-name flag, so SESSION_NAME must NOT be declared");
|
||||
assertTrue(service(herdr, root, opencodeCfg(null, null, "GITEA_ACCESS_TOKEN"))
|
||||
.capabilities().contains(Capability.SELF_PR),
|
||||
"a git-token profile adds SELF_PR");
|
||||
}
|
||||
|
||||
// --- CB-547: resume + post-hoc session discovery --------------------------------------------
|
||||
|
||||
@Test
|
||||
void aResumeSpawnPassesTheSessionIdAsDashS(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, "ses_41b79fc90ffeI9E8uZv6VprUn2"));
|
||||
|
||||
List<String> args = startArgs(herdr);
|
||||
int s = args.indexOf("-s");
|
||||
assertTrue(s >= 0, "a resumed spawn carries opencode's -s flag");
|
||||
assertEquals("ses_41b79fc90ffeI9E8uZv6VprUn2", args.get(s + 1),
|
||||
"the resume target id follows -s");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aFreshSpawnCarriesNoSessionFlag(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null, null, null));
|
||||
|
||||
assertFalse(startArgs(herdr).contains("-s"),
|
||||
"no resume target → a fresh session with no -s flag");
|
||||
}
|
||||
|
||||
@Test
|
||||
void theHandleDiscoversTheSessionIdForTheWorkersCwdOnlyAfterItAppears(@TempDir Path root,
|
||||
@TempDir Path discRoot)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher = new OpenCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), Map.of("gemini", opencodeCfg(null, null, null)),
|
||||
"gemini", _ -> null, 0, System::currentTimeMillis, () -> { }, root, discRoot);
|
||||
|
||||
PeerHandle handle = launcher.spawn(new SpawnRequest(null, "/work/dir", null));
|
||||
|
||||
// opencode writes the record only when the session is first persisted — the instant the
|
||||
// pane is ready it does not exist, so agentSessionId() is null (never a spawn failure).
|
||||
assertNull(handle.agentSessionId(), "no record yet → null, not a spawn-time block");
|
||||
// Once the record appears (here: same cwd), lazy discovery resolves it — the handle's
|
||||
// session id matches its own worktree, not another's.
|
||||
OpenCodeSessionDiscoveryTest.writeRecord(discRoot, "p1", "ses_a.json",
|
||||
"ses_resolved", "/work/dir", 1000L);
|
||||
assertEquals("ses_resolved", handle.agentSessionId(),
|
||||
"agentSessionId() re-scans and picks up a record that has since been written");
|
||||
}
|
||||
|
||||
@Test
|
||||
void foreignWorkerMatchesOpencodePrefixButNotClaude() {
|
||||
String nonce = "abc123";
|
||||
assertTrue(OpenCodeLauncher.isForeignWorker("opencode-gemini-def456-1", nonce),
|
||||
"an opencode pane from another process is foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("opencode-gemini-" + nonce + "-1", nonce),
|
||||
"our own opencode pane (same nonce) is not foreign");
|
||||
assertFalse(OpenCodeLauncher.isForeignWorker("claude-ltms-local-def456-1", nonce),
|
||||
"a claude pane is never reaped by the opencode adapter");
|
||||
}
|
||||
|
||||
@Test
|
||||
void productionConstructorsWireThroughToTheBase() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile cfg = opencodeCfg(null, null, null);
|
||||
// 5-arg (gate disabled) and 7-arg (gate enabled) production constructors both expose the profile.
|
||||
OpenCodeLauncher disabled = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
OpenCodeLauncher gated = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null, 5000, 100);
|
||||
assertEquals(java.util.Set.of("gemini"), disabled.profiles());
|
||||
assertEquals("gemini", gated.defaultProfile());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnGateThrowsPeerUnreachableWhenNeverInjectable(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never injectable
|
||||
long[] clock = {0};
|
||||
OpenCodeLauncher svc = new OpenCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
Map.of("gemini", opencodeCfg(null, null, null)), "gemini", _ -> null,
|
||||
1000, () -> clock[0], () -> clock[0] += 50, root, root);
|
||||
|
||||
PeerUnreachableException ex = assertThrows(PeerUnreachableException.class,
|
||||
() -> svc.spawn(new SpawnRequest(null, null, null)));
|
||||
assertTrue(clock[0] >= 1000, "the fake clock advanced past the timeout: " + clock[0]);
|
||||
long closes = herdr.calls.stream()
|
||||
.filter(c -> c.method().equals("pane.close"))
|
||||
.filter(c -> "w9:pRoot_1".equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
assertEquals(1, closes, "the worker pane was reaped on timeout (no orphan)");
|
||||
assertNotNull(ex.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void spawnReturnsHandleWhenGateDisabled(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
PeerHandle handle = service(herdr, root, opencodeCfg(null, null, null))
|
||||
.spawn(new SpawnRequest(null, null, null));
|
||||
assertNotNull(handle, "spawn returns a handle when the gate is disabled");
|
||||
assertFalse(herdr.called("agent.get"), "no polling when the gate is disabled");
|
||||
}
|
||||
|
||||
@Test
|
||||
void handleCarriesTheRealCharterReceiptNotTheInterfaceDefault(@TempDir Path root) {
|
||||
// The base's WorkerHandle computes a real CharterReceipt (CB-571), but the opencode adapter
|
||||
// wraps it in SessionAwareHandle for lazy session discovery. Before this fix that decorator
|
||||
// did not override charterReceipt(), so it silently inherited PeerHandle's `null` default
|
||||
// and the real receipt sitting on its delegate was lost.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Fleet fleet = new FleetConfig.Fleet(Map.of(), Map.of(), Map.of(), Map.of(),
|
||||
Map.of("dev", "role rule"), null);
|
||||
PeerHandle handle = service(herdr, root,
|
||||
opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null), () -> fleet)
|
||||
.spawn(new SpawnRequest(null, null, null));
|
||||
|
||||
assertNotNull(handle.charterReceipt(),
|
||||
"an opencode spawn's charterReceipt() must not silently be null");
|
||||
String composed = "role rule\n\n" + HerdrPeerLauncher.REPLY_CHARTER;
|
||||
assertEquals(CharterReceipt.digestOf(composed), handle.charterReceipt().charterSha256(),
|
||||
"the receipt on the wrapped handle must match the exact composed charter bytes");
|
||||
}
|
||||
|
||||
// --- CB-508: pinned OpenAI-compatible endpoint (e.g. a local vLLM) ---------------------------
|
||||
|
||||
/** A profile with a baseUrl but no model provider prefix cannot be resolved — fail loudly. */
|
||||
private static FleetConfig.Profile pinnedCfg(String model, String baseUrl, String mcpUrl) {
|
||||
return new FleetConfig.Profile("local", baseUrl, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
null, null, null, null, FleetConfig.Profile.KIND_OPENCODE);
|
||||
}
|
||||
|
||||
@Test
|
||||
void baseUrlDeclaresACustomOpenAiCompatibleProvider(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash", "http://127.0.0.1:8000", null))
|
||||
.spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "a pinned endpoint needs a config file even with no bridge MCP url");
|
||||
JsonNode provider = new ObjectMapper().readTree(Path.of(cfgPath).toFile())
|
||||
.path("provider").path("local-vllm");
|
||||
|
||||
assertFalse(provider.isMissingNode(), "the provider id comes from the model selector");
|
||||
assertEquals("@ai-sdk/openai-compatible", provider.path("npm").asText());
|
||||
assertEquals("http://127.0.0.1:8000/v1", provider.path("options").path("baseURL").asText(),
|
||||
"a bare host:port gets /v1 appended — that is where these servers mount the API");
|
||||
assertFalse(provider.path("options").path("apiKey").asText().isBlank(),
|
||||
"the AI SDK requires a non-empty key even when the server ignores it");
|
||||
assertFalse(provider.path("models").path("deepseek-v4-flash").isMissingNode(),
|
||||
"the model half of the selector is declared under the provider");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBaseUrlThatAlreadyCarriesAPathIsUsedVerbatim(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/m", "http://127.0.0.1:8000/openai/v1", null)).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("http://127.0.0.1:8000/openai/v1",
|
||||
json.path("provider").path("local-vllm").path("options").path("baseURL").asText(),
|
||||
"an endpoint mounted on a custom path must not have /v1 bolted on");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointRejectsAModelWithNoProviderPrefix(@TempDir Path root) {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
OpenCodeLauncher launcher =
|
||||
service(herdr, root, pinnedCfg("deepseek-v4-flash", "http://127.0.0.1:8000", null));
|
||||
|
||||
// Silently falling back to the default gateway would point the worker at the wrong LLM
|
||||
// while looking healthy — the one failure mode worth being loud about.
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, launcher::spawn);
|
||||
assertTrue(e.getMessage().contains("<provider>/<model>"), "the error says how to fix it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aPinnedEndpointAndTheFleetMcpCoexistInOneConfig(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, pinnedCfg("local-vllm/deepseek-v4-flash",
|
||||
"http://127.0.0.1:8000", "http://127.0.0.1:8766/mcp")).spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertEquals("remote", json.path("mcp").path("fleet").path("type").asText(),
|
||||
"pinning an endpoint must not drop the fleet MCP mount");
|
||||
assertFalse(json.path("provider").path("local-vllm").isMissingNode(),
|
||||
"and the provider block is still declared alongside it");
|
||||
assertTrue(json.path("instructions").isArray() && !json.path("instructions").isEmpty(),
|
||||
"the reply charter survives too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noBaseUrlDeclaresNoProviderSoTheDefaultGatewayIsUsed(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("opencode/some-free-model", "http://127.0.0.1:8766/mcp", null))
|
||||
.spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertTrue(json.path("provider").isMissingNode(),
|
||||
"without a baseUrl opencode resolves its own provider as before");
|
||||
}
|
||||
|
||||
// --- autoCompactWindow: opencode has no absolute compact-at-N knob, so this is applied as the
|
||||
// model's own limit.context, only when model: resolves to "provider/model" -----------------
|
||||
|
||||
private static FleetConfig.Profile opencodeCfgWithAutoCompactWindow(String model, String baseUrl,
|
||||
Integer window) {
|
||||
return new FleetConfig.Profile("gemini", baseUrl, model, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", null,
|
||||
null, null, null, null, FleetConfig.Profile.KIND_OPENCODE, Map.of(), null, null,
|
||||
null, null, null, null, null, null, window);
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoCompactWindowIsAppliedAsThePerModelContextLimit(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfgWithAutoCompactWindow("openai/gpt-5", null, 250_000)).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "autoCompactWindow alone must trigger config generation, with no MCP"
|
||||
+ " and no custom provider set");
|
||||
JsonNode limit = new ObjectMapper().readTree(Path.of(cfgPath).toFile())
|
||||
.path("provider").path("openai").path("models").path("gpt-5").path("limit");
|
||||
assertEquals(250_000, limit.path("context").asInt());
|
||||
assertEquals(16384, limit.path("output").asInt(),
|
||||
"opencode's limit schema requires both keys; output gets a safe documented default");
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoCompactWindowMergesIntoACustomProviderRatherThanOverwritingIt(@TempDir Path root)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfgWithAutoCompactWindow(
|
||||
"local-vllm/deepseek-v4-flash", "http://127.0.0.1:8000", 300_000)).spawn();
|
||||
|
||||
JsonNode provider = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile())
|
||||
.path("provider").path("local-vllm");
|
||||
assertEquals("@ai-sdk/openai-compatible", provider.path("npm").asText(),
|
||||
"addCustomProvider's own fields must survive the later limit merge");
|
||||
assertEquals(300_000, provider.path("models").path("deepseek-v4-flash")
|
||||
.path("limit").path("context").asInt());
|
||||
assertEquals("deepseek-v4-flash", provider.path("models").path("deepseek-v4-flash")
|
||||
.path("name").asText(),
|
||||
"the model's pre-existing 'name' field must survive the limit merge too");
|
||||
}
|
||||
|
||||
@Test
|
||||
void autoCompactWindowWithNoProviderSlashInModelGetsNoLimitAndAWarn(@TempDir Path root)
|
||||
throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
// mcpUrl set too, only so a config file gets written at all to inspect; a bare model name
|
||||
// with no other config-triggering knob would leave OPENCODE_CONFIG unset entirely, which is
|
||||
// also correct (nothing to write) but not what this test is asserting.
|
||||
service(herdr, root, new FleetConfig.Profile("gemini", null, "some-free-model", null,
|
||||
"BRIDGED_WORKER_TOKEN", List.of("opencode"), "tab", "bridged-workers",
|
||||
"opencode: {model} #{n}", "http://127.0.0.1:8765/mcp", null, null, null, null,
|
||||
FleetConfig.Profile.KIND_OPENCODE, Map.of(), null, null, null, null, null, null,
|
||||
null, null, 250_000)).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
assertTrue(json.path("provider").isMissingNode(),
|
||||
"a bare model name cannot be targeted at a specific provider/model limit entry — "
|
||||
+ "no silent no-op, but also no broken partial write");
|
||||
}
|
||||
|
||||
// --- CB-634: IDE Index MCP + guidance overlay (opencode does not read CLAUDE.local.md) -------
|
||||
|
||||
private static FleetConfig.Profile opencodeIdeCfg(String mcpUrl, String ideUrl, String cwd) {
|
||||
return new FleetConfig.Profile("gemini", null, null, null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("opencode"), "tab", "bridged-workers", "opencode: {model} #{n}", mcpUrl,
|
||||
cwd, null, null, null, FleetConfig.Profile.KIND_OPENCODE,
|
||||
null, null, null, null, null, null, ideUrl);
|
||||
}
|
||||
|
||||
@Test
|
||||
void ideMcpUrlAddsTheIntellijServerAndAnInstructionsRulesEntry(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
Path cwd = Files.createDirectory(root.resolve("checkout"));
|
||||
service(herdr, root, opencodeIdeCfg("http://127.0.0.1:8765/mcp",
|
||||
"http://127.0.0.1:29170/index-mcp/streamable-http", cwd.toString())).spawn();
|
||||
|
||||
String cfgPath = startEnv(herdr).get("OPENCODE_CONFIG");
|
||||
assertNotNull(cfgPath, "an IDE profile needs a config file");
|
||||
JsonNode json = new ObjectMapper().readTree(Path.of(cfgPath).toFile());
|
||||
JsonNode ide = json.path("mcp").path("intellij");
|
||||
assertEquals("remote", ide.path("type").asText(),
|
||||
"the IDE server uses the same remote shape as the bridge mount");
|
||||
assertEquals("http://127.0.0.1:29170/index-mcp/streamable-http", ide.path("url").asText());
|
||||
assertTrue(ide.path("enabled").asBoolean(), "the IDE server is enabled");
|
||||
assertEquals("remote", json.path("mcp").path("fleet").path("type").asText(),
|
||||
"the bridge mount still coexists with the IDE server");
|
||||
|
||||
// The instructions array gains an entry pointing at a real rules file pinning the worktree.
|
||||
String rulesContent = null;
|
||||
for (JsonNode n : json.path("instructions")) {
|
||||
Path p = Path.of(n.asText());
|
||||
if (p.getFileName().toString().equals("ide-rules.md")) {
|
||||
rulesContent = Files.readString(p);
|
||||
}
|
||||
}
|
||||
assertNotNull(rulesContent, "an ide-rules.md instructions entry is present");
|
||||
assertTrue(rulesContent.contains("project_path: \"" + cwd + "\""),
|
||||
"the rules pin every ide_* call to the worker's own cwd");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noIdeServerWhenIdeMcpUrlUnset(@TempDir Path root) throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
service(herdr, root, opencodeCfg("google/gemini-2.5-pro", "http://127.0.0.1:8765/mcp", null))
|
||||
.spawn();
|
||||
|
||||
JsonNode json = new ObjectMapper()
|
||||
.readTree(Path.of(startEnv(herdr).get("OPENCODE_CONFIG")).toFile());
|
||||
assertTrue(json.path("mcp").path("intellij").isMissingNode(),
|
||||
"no IDE server when ideMcpUrl is unset");
|
||||
}
|
||||
}
|
||||
@@ -1,90 +0,0 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.FileTime;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* {@link OpenCodeSessionDiscovery} matches an opencode session record by the worker's cwd (its
|
||||
* {@code directory}) against opencode's on-disk storage. These tests populate a TEMP storage root
|
||||
* themselves — never the operator's real {@code ~/.local/share/opencode}.
|
||||
*/
|
||||
class OpenCodeSessionDiscoveryTest {
|
||||
|
||||
/**
|
||||
* Write a session record {@code {"id":..., "directory":...}} under
|
||||
* {@code <root>/session/<projectID>/<fileName>} and stamp it with a known last-modified time,
|
||||
* so "most recently modified wins" is deterministic. Static so the launcher test can reuse it.
|
||||
*/
|
||||
static void writeRecord(Path root, String projectId, String fileName, String id,
|
||||
String directory, long lastModifiedEpochMillis) throws Exception {
|
||||
Path dir = root.resolve("session").resolve(projectId);
|
||||
Files.createDirectories(dir);
|
||||
Path file = dir.resolve(fileName);
|
||||
Files.writeString(file, "{\"id\":\"" + id + "\",\"directory\":\"" + directory
|
||||
+ "\",\"projectID\":\"" + projectId + "\",\"version\":\"1.1.31\"}");
|
||||
Files.setLastModifiedTime(file, FileTime.fromMillis(lastModifiedEpochMillis));
|
||||
}
|
||||
|
||||
@Test
|
||||
void findsTheRecordWhoseDirectoryEqualsTheCwd(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "ses_b.json", "ses_bbb", "/w/b", 2000L);
|
||||
|
||||
assertEquals("ses_bbb", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/b"),
|
||||
"the record whose directory equals the cwd is the one found");
|
||||
assertEquals("ses_aaa", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonMatchingDirectoryYieldsNullRatherThanAMismatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "ses_a.json", "ses_aaa", "/w/a", 1000L);
|
||||
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/other"),
|
||||
"no record for this cwd yet → null, not a wrong session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void prefersTheMostRecentlyModifiedRecordWhenSeveralMatch(@TempDir Path root) throws Exception {
|
||||
writeRecord(root, "p1", "old.json", "ses_old", "/w/a", 1000L);
|
||||
writeRecord(root, "p2", "new.json", "ses_new", "/w/a", 5000L);
|
||||
|
||||
assertEquals("ses_new", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"the freshest record for the cwd wins");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMissingOrEmptyStorageRootYieldsNullWithoutThrowing(@TempDir Path root) throws Exception {
|
||||
// Missing: no session dir at all under the root.
|
||||
assertNull(new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"));
|
||||
|
||||
// Present but empty: a session dir with nothing in it produces no match, not a throw.
|
||||
Path emptyRoot = root.resolve("empty");
|
||||
Files.createDirectories(emptyRoot.resolve("session"));
|
||||
assertNull(new OpenCodeSessionDiscovery(emptyRoot).sessionIdForDirectory("/w/a"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aBlankOrNullDirectoryYieldsNull(@TempDir Path root) {
|
||||
OpenCodeSessionDiscovery discovery = new OpenCodeSessionDiscovery(root);
|
||||
assertNull(discovery.sessionIdForDirectory(null));
|
||||
assertNull(discovery.sessionIdForDirectory(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aMalformedRecordIsSkippedRatherThanFatal(@TempDir Path root) throws Exception {
|
||||
// A record that fails to parse must not abort the scan of its siblings.
|
||||
Path dir = root.resolve("session").resolve("p1");
|
||||
Files.createDirectories(dir);
|
||||
Files.writeString(dir.resolve("broken.json"), "{not valid json");
|
||||
writeRecord(root, "p1", "good.json", "ses_good", "/w/a", 1000L);
|
||||
|
||||
assertEquals("ses_good", new OpenCodeSessionDiscovery(root).sessionIdForDirectory("/w/a"),
|
||||
"an unreadable record is skipped; a later valid one still matches");
|
||||
}
|
||||
}
|
||||
@@ -1,173 +0,0 @@
|
||||
package dev.ltms.fleet.msg;
|
||||
|
||||
import org.junit.jupiter.api.BeforeAll;
|
||||
import org.junit.jupiter.api.Tag;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.testcontainers.containers.RabbitMQContainer;
|
||||
import org.testcontainers.junit.jupiter.Testcontainers;
|
||||
import org.testcontainers.utility.DockerImageName;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* Contract test for {@link LeadMailbox} against a REAL broker — same approach as
|
||||
* {@code AmqpReplyInboxContractTest}, which this mirrors: a Testcontainers RabbitMQ locally, or an
|
||||
* externally-provisioned broker in CI via {@code AMQP_URI}. Tagged {@code contract} so it is
|
||||
* excluded from {@code mvn test}/{@code mvn clean install} (which stay hermetic and need no
|
||||
* Docker); run it with Docker present via {@code mvn test -Pcontract}.
|
||||
*
|
||||
* <p>Proves the mechanism this ticket adds: a {@link LeadMessage} published to
|
||||
* {@code lead.<to>.inbox} is received with {@code from}/{@code to}/{@code content} intact, and
|
||||
* {@link LeadMailbox#ack} removes it — the same publish→peek→ack roundtrip
|
||||
* {@code AmqpReplyInboxContractTest} proves for {@link AmqpReplyInbox}, adapted to this class's
|
||||
* single-owned-mailbox shape (no {@code own}/{@code release} — the mailbox for {@code selfCoordId}
|
||||
* is owned the moment {@link LeadMailbox#open} returns).
|
||||
*/
|
||||
@Tag("contract")
|
||||
// disabledWithoutDocker=false: on the CI path (AMQP_URI set) no container is started and the class
|
||||
// must still run against the external broker even though the runner has no Docker.
|
||||
@Testcontainers(disabledWithoutDocker = false)
|
||||
class LeadMailboxTest {
|
||||
|
||||
private static final String EXTERNAL_URI = System.getenv("AMQP_URI");
|
||||
|
||||
private static final RabbitMQContainer BROKER =
|
||||
new RabbitMQContainer(DockerImageName.parse("rabbitmq:3.13-management"));
|
||||
|
||||
private static final AtomicLong SEQ = new AtomicLong();
|
||||
|
||||
// No @Container: the JUnit 5 extension would force-start it even when AMQP_URI is set. Start it
|
||||
// manually only on the local (no-external-broker) path; Ryuk reaps it on JVM exit.
|
||||
@BeforeAll
|
||||
static void startBrokerUnlessExternal() {
|
||||
if (EXTERNAL_URI == null) {
|
||||
BROKER.start();
|
||||
}
|
||||
}
|
||||
|
||||
private static String uri() {
|
||||
if (EXTERNAL_URI != null) {
|
||||
return EXTERNAL_URI;
|
||||
}
|
||||
// No trailing slash: an empty path is vhost "", which does not exist — omitting it selects
|
||||
// the default vhost "/".
|
||||
return "amqp://guest:guest@" + BROKER.getHost() + ":" + BROKER.getAmqpPort();
|
||||
}
|
||||
|
||||
/** A fresh coord-id per test run so parallel/repeat runs never collide on the same queue. */
|
||||
private static String coordId(String prefix) {
|
||||
return prefix + "-" + System.nanoTime() + "-" + SEQ.incrementAndGet();
|
||||
}
|
||||
|
||||
@Test
|
||||
void publishThenPeekThenAckRoundTrip() throws Exception {
|
||||
String to = coordId("lead-to");
|
||||
String from = "lead-from";
|
||||
try (LeadMailbox inbox = LeadMailbox.open(uri(), to)) {
|
||||
LeadMessage sent = new LeadMessage("m1", from, to, "hello peer lead");
|
||||
inbox.publish(to, sent);
|
||||
|
||||
List<LeadMessage> got = awaitPeek(inbox);
|
||||
assertEquals(1, got.size(), "the published message should be held for drain");
|
||||
assertEquals("m1", got.getFirst().msgId());
|
||||
assertEquals(from, got.getFirst().from());
|
||||
assertEquals(to, got.getFirst().to());
|
||||
assertEquals("hello peer lead", got.getFirst().content());
|
||||
|
||||
inbox.ack("m1");
|
||||
assertTrue(inbox.peek().isEmpty(), "an acked message is dropped");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void duplicateMsgIdIsNotDoubleQueued() throws Exception {
|
||||
String to = coordId("lead-dedup");
|
||||
try (LeadMailbox inbox = LeadMailbox.open(uri(), to)) {
|
||||
inbox.publish(to, new LeadMessage("dup", "lead-from", to, "first"));
|
||||
awaitPeek(inbox);
|
||||
inbox.publish(to, new LeadMessage("dup", "lead-from", to, "second")); // same msgId — no-op
|
||||
|
||||
Thread.sleep(500); // give any erroneous second delivery time to land
|
||||
List<LeadMessage> got = inbox.peek();
|
||||
assertEquals(1, got.size(), "a repeated msgId must not double-queue");
|
||||
assertEquals("first", got.getFirst().content(), "the first payload wins");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void unackedMessageSurvivesRestartAndIsRedelivered() throws Exception {
|
||||
String to = coordId("lead-durable");
|
||||
|
||||
// First "process life": publish, see it held, but crash before acking.
|
||||
try (LeadMailbox first = LeadMailbox.open(uri(), to)) {
|
||||
first.publish(to, new LeadMessage("persist-1", "lead-from", to, "survive me"));
|
||||
assertEquals(1, awaitPeek(first).size());
|
||||
// no ack — simulate a java -jar bounce with the message still pending
|
||||
}
|
||||
|
||||
// Second "process life": a fresh connection owning the same mailbox must be redelivered it.
|
||||
try (LeadMailbox second = LeadMailbox.open(uri(), to)) {
|
||||
List<LeadMessage> got = awaitPeek(second);
|
||||
assertEquals(1, got.size(), "an unacked persistent message is redelivered after restart");
|
||||
assertEquals("persist-1", got.getFirst().msgId());
|
||||
assertEquals("survive me", got.getFirst().content());
|
||||
|
||||
second.ack("persist-1");
|
||||
}
|
||||
|
||||
// Third life: once acked, it is gone for good.
|
||||
try (LeadMailbox third = LeadMailbox.open(uri(), to)) {
|
||||
Thread.sleep(500);
|
||||
assertTrue(third.peek().isEmpty(), "an acked message does not come back on the next restart");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void publishDoesNotRequireTheSenderToOwnTheTargetMailbox() throws Exception {
|
||||
// CB-308 federation: a sender that never opened its own LeadMailbox for `to` can still
|
||||
// publish to it — publish must not imply ownership. Only the owner ever consumes here.
|
||||
String to = coordId("lead-federated");
|
||||
String senderId = coordId("lead-sender");
|
||||
try (LeadMailbox owner = LeadMailbox.open(uri(), to);
|
||||
LeadMailbox sender = LeadMailbox.open(uri(), senderId)) {
|
||||
sender.publish(to, new LeadMessage("m1", senderId, to, "from a federated peer"));
|
||||
|
||||
List<LeadMessage> got = awaitPeek(owner);
|
||||
assertEquals(1, got.size(), "only the owner's mailbox should receive the message");
|
||||
assertEquals(senderId, got.getFirst().from());
|
||||
assertTrue(sender.peek().isEmpty(), "the sender must not also hold a copy — it never owns `to`");
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void unroutablePublishReportsFailureNotSilentSuccess() throws Exception {
|
||||
// Publish to a coord-id whose mailbox was never opened by anyone: the queue is never
|
||||
// declared, so the default-exchange route to lead.<to>.inbox does not exist and the broker
|
||||
// must return the publish.
|
||||
String to = coordId("lead-nobody-home");
|
||||
try (LeadMailbox sender = LeadMailbox.open(uri(), coordId("lead-sender"))) {
|
||||
IllegalStateException ex = assertThrows(IllegalStateException.class,
|
||||
() -> sender.publish(to, new LeadMessage("m1", "lead-from", to, "nobody home")));
|
||||
assertTrue(ex.getMessage() != null && ex.getMessage().toLowerCase().contains("unroutable"),
|
||||
"expected an unroutable-publish failure, got: " + ex.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
/** Poll peek until at least one message is held, or ~10s elapse (broker delivery is async). */
|
||||
@SuppressWarnings("BusyWait")
|
||||
private static List<LeadMessage> awaitPeek(LeadMailbox inbox) throws InterruptedException {
|
||||
long deadline = System.nanoTime() + TimeUnit.SECONDS.toNanos(10);
|
||||
List<LeadMessage> msgs = inbox.peek();
|
||||
while (msgs.isEmpty() && System.nanoTime() < deadline) {
|
||||
Thread.sleep(50);
|
||||
msgs = inbox.peek();
|
||||
}
|
||||
return msgs;
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -1,119 +0,0 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertThrows;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-578 stage B: the credential-keyed quarantine tracker itself, isolated from placement/spawn
|
||||
* wiring (that's {@code CompositePeerLauncherTest}). The clock is a plain {@link AtomicLong} of
|
||||
* nanos so expiry is exercised without a real sleep.
|
||||
*/
|
||||
class BackendQuarantineTest {
|
||||
|
||||
@Test
|
||||
void aFreshCredentialIsNotQuarantined() {
|
||||
BackendQuarantine q = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
assertFalse(q.isQuarantined("shared-openai"));
|
||||
assertEquals(OptionalLong.empty(), q.remainingSeconds("shared-openai"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void quarantineBlocksTheCredentialForTheFullCooldown() {
|
||||
BackendQuarantine q = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
q.quarantine("shared-openai");
|
||||
|
||||
assertTrue(q.isQuarantined("shared-openai"));
|
||||
assertEquals(OptionalLong.of(1800L), q.remainingSeconds("shared-openai"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void onlyTheQuarantinedCredentialIsAffected() {
|
||||
BackendQuarantine q = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
q.quarantine("shared-openai");
|
||||
|
||||
assertFalse(q.isQuarantined("some-other-credential"),
|
||||
"an unrelated credential must not be swept into the quarantine");
|
||||
}
|
||||
|
||||
@Test
|
||||
void expiresOnTheInjectedClock() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.MINUTES.toNanos(30));
|
||||
q.quarantine("shared-openai");
|
||||
assertTrue(q.isQuarantined("shared-openai"));
|
||||
|
||||
now.set(TimeUnit.MINUTES.toNanos(29));
|
||||
assertTrue(q.isQuarantined("shared-openai"), "still inside the cooldown");
|
||||
|
||||
now.set(TimeUnit.MINUTES.toNanos(31));
|
||||
assertFalse(q.isQuarantined("shared-openai"), "the cooldown has elapsed on the injected clock");
|
||||
assertEquals(OptionalLong.empty(), q.remainingSeconds("shared-openai"));
|
||||
}
|
||||
|
||||
@Test
|
||||
void aRepeatQuarantineCallRestartsTheCooldownAtFullLength() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.MINUTES.toNanos(30));
|
||||
q.quarantine("shared-openai");
|
||||
|
||||
now.set(TimeUnit.MINUTES.toNanos(20));
|
||||
q.quarantine("shared-openai");
|
||||
|
||||
now.set(TimeUnit.MINUTES.toNanos(45)); // 25 min after the second call, 45 after the first
|
||||
assertTrue(q.isQuarantined("shared-openai"),
|
||||
"a fresh exhaustion resets the cooldown to full length, not the earlier shorter wait");
|
||||
}
|
||||
|
||||
@Test
|
||||
void activeRemainingSecondsListsOnlyStillQuarantinedCredentials() {
|
||||
AtomicLong now = new AtomicLong(0L);
|
||||
BackendQuarantine q = new BackendQuarantine(now::get, TimeUnit.MINUTES.toNanos(30));
|
||||
q.quarantine("shared-openai");
|
||||
q.quarantine("another-credential");
|
||||
|
||||
now.set(TimeUnit.MINUTES.toNanos(31));
|
||||
q.quarantine("shared-openai"); // re-quarantined after the first one expired
|
||||
|
||||
Map<String, Long> active = q.activeRemainingSeconds();
|
||||
assertEquals(Map.of("shared-openai", 1800L), active,
|
||||
"the expired credential is dropped; the re-quarantined one is reported");
|
||||
}
|
||||
|
||||
@Test
|
||||
void noneReportsNothingQuarantinedWhenNeverToldTo() {
|
||||
BackendQuarantine q = BackendQuarantine.none();
|
||||
|
||||
assertFalse(q.isQuarantined("anything"));
|
||||
assertTrue(q.activeRemainingSeconds().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void noneIgnoresAQuarantineCallInsteadOfLockingTheCredentialForever() {
|
||||
// .none() holds a clock frozen at 0, so if #quarantine recorded a deadline the credential
|
||||
// would never expire — locked out for the life of the daemon. Two production
|
||||
// CompositePeerLauncher constructors default to none(), so that failure would be silent and
|
||||
// permanent. An inert stand-in must omit the fact, never invent one.
|
||||
BackendQuarantine q = BackendQuarantine.none();
|
||||
|
||||
q.quarantine("shared-openai");
|
||||
|
||||
assertFalse(q.isQuarantined("shared-openai"), "none() must not quarantine anything");
|
||||
assertTrue(q.remainingSeconds("shared-openai").isEmpty());
|
||||
assertTrue(q.activeRemainingSeconds().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNonPositiveCooldownIsRejected() {
|
||||
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, 0L));
|
||||
assertThrows(IllegalArgumentException.class, () -> new BackendQuarantine(() -> 0L, -1L));
|
||||
}
|
||||
}
|
||||
@@ -1,625 +0,0 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.charset.StandardCharsets;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-525 acceptance test for tool-surface isolation. This is one of the few tests that drives real
|
||||
* {@code git} — the behaviour under test is precisely what {@link GitWorktrees} does to a checkout,
|
||||
* so a fake would assert nothing. Everything happens inside a {@link TempDir} throwaway repo.
|
||||
*/
|
||||
class GitWorktreesTest {
|
||||
|
||||
/** A project MCP config with servers in it — what this repo actually commits. */
|
||||
private static final String WITH_SERVERS = """
|
||||
{
|
||||
"mcpServers": {
|
||||
"jetbrains": { "type": "sse", "url": "http://localhost:64342/sse" }
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** An opencode config carrying a {@code {file:.secrets/...}} reference — CB-543's crash repro. */
|
||||
private static final String OPENCODE_WITH_FILE_REF = """
|
||||
{
|
||||
"env": {
|
||||
"CONTEXT7_TOKEN": "{file:.secrets/context7-token}"
|
||||
}
|
||||
}
|
||||
""";
|
||||
|
||||
/** A non-empty autoenv file — the form that would prompt for authorization in a worktree. */
|
||||
private static final String AUTOENV_WITH_DIRECTIVE = "export HELLO=world\n";
|
||||
|
||||
private static Path initRepo(Path dir) throws Exception {
|
||||
Files.createDirectories(dir);
|
||||
git(dir, "init", "-q", "-b", "main");
|
||||
git(dir, "config", "user.email", "test@example.invalid");
|
||||
git(dir, "config", "user.name", "Test");
|
||||
Files.writeString(dir.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(dir.resolve("README.md"), "seed\n");
|
||||
git(dir, "add", ".mcp.json", "README.md");
|
||||
git(dir, "commit", "-q", "-m", "seed");
|
||||
return dir;
|
||||
}
|
||||
|
||||
private static void git(Path cwd, String... args) throws Exception {
|
||||
List<String> cmd = new java.util.ArrayList<>(List.of("git"));
|
||||
cmd.addAll(List.of(args));
|
||||
Process p = new ProcessBuilder(cmd).directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git timed out: " + String.join(" ", cmd));
|
||||
assertEquals(0, p.exitValue(), "git " + String.join(" ", args) + " failed:\n" + out);
|
||||
}
|
||||
|
||||
/** Pending changes to {@code file} in {@code cwd}, empty when git considers it unmodified. */
|
||||
private static String status(Path cwd, String file) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "status", "--porcelain", "--", file)
|
||||
.directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git status timed out");
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Every pending change in {@code cwd} — the whole-tree porcelain status, unlike {@link #status}. */
|
||||
private static String fullStatus(Path cwd) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "status", "--porcelain")
|
||||
.directory(cwd.toFile()).redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git status timed out");
|
||||
return out;
|
||||
}
|
||||
|
||||
private static String revParse(Path cwd, String ref) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "rev-parse", ref)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git rev-parse timed out");
|
||||
assertEquals(0, p.exitValue(), "git rev-parse " + ref + " failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The recursive file list of a commit's tree — used to check what a snapshot actually committed. */
|
||||
private static String lsTree(Path cwd, String ref) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "ls-tree", "-r", "--name-only", ref)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git ls-tree timed out");
|
||||
assertEquals(0, p.exitValue(), "git ls-tree " + ref + " failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** The set of paths in {@code git diff --name-only from..to} — used to check exactly what a
|
||||
* snapshot's tree changed relative to its parent, the same shape {@code git status --porcelain}
|
||||
* reports for the worktree it was taken from. */
|
||||
private static Set<String> diffNameOnly(Path cwd, String from, String to) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "diff", "--name-only", from, to)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git diff timed out");
|
||||
assertEquals(0, p.exitValue(), "git diff " + from + ".." + to + " failed:\n" + out);
|
||||
Set<String> paths = new HashSet<>();
|
||||
for (String line : out.split("\\R")) {
|
||||
if (!line.isBlank()) {
|
||||
paths.add(line.trim());
|
||||
}
|
||||
}
|
||||
return paths;
|
||||
}
|
||||
|
||||
/** The set of paths a {@code git status --porcelain} listing names, stripping the two-char status
|
||||
* code prefix each line carries. */
|
||||
private static Set<String> porcelainPaths(String porcelain) {
|
||||
Set<String> paths = new HashSet<>();
|
||||
for (String line : porcelain.split("\\R")) {
|
||||
if (!line.isBlank()) {
|
||||
paths.add(line.substring(3).trim());
|
||||
}
|
||||
}
|
||||
return paths;
|
||||
}
|
||||
|
||||
private static String forEachRef(Path cwd, String pattern) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "for-each-ref", pattern)
|
||||
.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes());
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git for-each-ref timed out");
|
||||
assertEquals(0, p.exitValue(), "git for-each-ref " + pattern + " failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Write {@code content} as a blob into the object database; returns its sha. */
|
||||
private static String blobOf(Path cwd, String content) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "hash-object", "-w", "--stdin")
|
||||
.redirectErrorStream(true).start();
|
||||
p.getOutputStream().write(content.getBytes(StandardCharsets.UTF_8));
|
||||
p.getOutputStream().close();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git hash-object timed out");
|
||||
assertEquals(0, p.exitValue(), "git hash-object failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Build a single-file tree object from {@code blob}; returns the tree's sha. */
|
||||
private static String treeOf(Path cwd, String path, String blob) throws Exception {
|
||||
Process p = new ProcessBuilder("git", "-C", cwd.toString(), "mktree")
|
||||
.redirectErrorStream(true).start();
|
||||
p.getOutputStream().write(("100644 blob " + blob + "\t" + path + "\n").getBytes(StandardCharsets.UTF_8));
|
||||
p.getOutputStream().close();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git mktree timed out");
|
||||
assertEquals(0, p.exitValue(), "git mktree failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** {@code git commit-tree} rooted at {@code tree} with a chosen committer date; returns the sha. */
|
||||
private static String commitTree(Path cwd, String tree, String parent, String committerDate,
|
||||
String message) throws Exception {
|
||||
ProcessBuilder pb = new ProcessBuilder("git", "-C", cwd.toString(), "commit-tree",
|
||||
tree, "-p", parent, "-m", message);
|
||||
pb.environment().put("GIT_COMMITTER_DATE", committerDate);
|
||||
Process p = pb.redirectErrorStream(true).start();
|
||||
String out = new String(p.getInputStream().readAllBytes()).trim();
|
||||
assertTrue(p.waitFor(30, TimeUnit.SECONDS), "git commit-tree timed out");
|
||||
assertEquals(0, p.exitValue(), "git commit-tree failed:\n" + out);
|
||||
return out;
|
||||
}
|
||||
|
||||
/** {@code git update-ref <ref> <sha>} — create the snapshot ref directly. */
|
||||
private static void updateRef(Path cwd, String ref, String sha) throws Exception {
|
||||
git(cwd, "update-ref", ref, sha);
|
||||
}
|
||||
|
||||
/** True when {@code ref} exists in the repo (for-each-ref on a missing ref is empty, not an error). */
|
||||
private static boolean refExists(Path cwd, String ref) throws Exception {
|
||||
return !forEachRef(cwd, ref).trim().isEmpty();
|
||||
}
|
||||
|
||||
/**
|
||||
* The heart of CB-525: a provisioned worktree must not inherit the primary's MCP servers. Without
|
||||
* the isolation step the checked-out {@code .mcp.json} carries them in, and a worker navigating
|
||||
* through the primary's IDE servers edits the primary's tree while building its own.
|
||||
*/
|
||||
@Test
|
||||
void aProvisionedWorktreeInheritsNoMcpServers(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-a", "HEAD");
|
||||
|
||||
Path mcp = Path.of(wt).resolve(".mcp.json");
|
||||
assertTrue(Files.exists(mcp), ".mcp.json must still exist — present and explicitly empty");
|
||||
String body = Files.readString(mcp);
|
||||
assertFalse(body.contains("jetbrains"), "worktree inherited the primary's MCP servers:\n" + body);
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
"expected an explicitly empty server map, got:\n" + body);
|
||||
}
|
||||
|
||||
/** Neutralizing must not look like work in progress, or a worker would commit it into its PR. */
|
||||
@Test
|
||||
void theNeutralizedConfigIsNotAPendingLocalModification(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-b", "HEAD");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"),
|
||||
"the neutralized .mcp.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** Isolation is the worktree's business only; the primary's own checkout must be untouched. */
|
||||
@Test
|
||||
void thePrimaryCheckoutIsLeftAlone(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
new GitWorktrees(tmp.resolve("wts").toString()).add(repo.toString(), "cb-525-c", "HEAD");
|
||||
|
||||
assertEquals(WITH_SERVERS, Files.readString(repo.resolve(".mcp.json")),
|
||||
"the primary's .mcp.json was rewritten — isolation reached out of the worktree");
|
||||
}
|
||||
|
||||
/** A repo that commits no {@code .mcp.json} still gets one, so nothing can be inherited later. */
|
||||
@Test
|
||||
void aRepoWithoutAnMcpConfigStillGetsANeutralOne(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-525-d", "HEAD");
|
||||
|
||||
// Untracked is the normal case here, so the --skip-worktree branch must be skipped rather
|
||||
// than run and fail: `update-index --skip-worktree` on an unknown path exits non-zero.
|
||||
String body = Files.readString(Path.of(wt).resolve(".mcp.json"));
|
||||
assertTrue(body.replaceAll("\\s+", "").contains("\"mcpServers\":{}"), body);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-543's repro: a tracked {@code opencode.json} carries a {@code {file:.secrets/...}} reference
|
||||
* to a gitignored secret that never reaches a worktree, and opencode refuses to start on it. The
|
||||
* worktree's copy must be neutralized and hidden like {@code .mcp.json}.
|
||||
*/
|
||||
@Test
|
||||
void aTrackedOpencodeConfigIsNeutralizedAndHidden(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-a", "HEAD");
|
||||
|
||||
String body = Files.readString(Path.of(wt).resolve("opencode.json"));
|
||||
assertFalse(body.contains(".secrets"),
|
||||
"worktree kept a dangling {file:...} secret reference:\n" + body);
|
||||
assertEquals("{}", body.replaceAll("\\s+", ""),
|
||||
"expected an empty JSON object stub, got:\n" + body);
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"),
|
||||
"the neutralized opencode.json shows as modified — --skip-worktree did not take");
|
||||
}
|
||||
|
||||
/** A config the repo does not carry must be skipped — no stub invented, provisioning still succeeds. */
|
||||
@Test
|
||||
void anAbsentConfigIsSkippedWithoutError(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo")); // only .mcp.json + README are committed
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-b", "HEAD");
|
||||
|
||||
assertFalse(Files.exists(Path.of(wt).resolve("opencode.json")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
assertFalse(Files.exists(Path.of(wt).resolve(".autoenv")),
|
||||
"a stub was invented for a config the repo does not carry");
|
||||
// .mcp.json's long-standing create-always behaviour must be unchanged.
|
||||
assertTrue(Files.exists(Path.of(wt).resolve(".mcp.json")), ".mcp.json stub was dropped");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-576. {@code hasUncommitted} must treat a freshly-provisioned worktree as clean, but a
|
||||
* worktree holding a brand-new, never-added file as dirty. The untracked-file-only shape is
|
||||
* exactly the work lost in the incident — a worker's draft that compiled but was never
|
||||
* committed because it stopped to ask its lead a question.
|
||||
*/
|
||||
@Test
|
||||
void anUntrackedOnlyWorktreeCountsAsDirty(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String wt = gitWorktrees.add(repo.toString(), "cb-576-u", "HEAD");
|
||||
|
||||
assertFalse(gitWorktrees.hasUncommitted(wt),
|
||||
"a freshly provisioned worktree must read as clean");
|
||||
|
||||
Files.writeString(Path.of(wt).resolve("brand-new.txt"), "draft that was never added\n");
|
||||
|
||||
assertTrue(gitWorktrees.hasUncommitted(wt),
|
||||
"an untracked-only file must count as dirty");
|
||||
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
assertTrue(gitWorktrees.hasUncommitted(wt),
|
||||
"a tracked modification must also count as dirty");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-576 review. {@code hasUncommitted} must tolerate a missing worktree exactly like
|
||||
* {@code remove}: an already-gone directory holds no work to lose, and throwing here would
|
||||
* break teardown — SessionManager.release() calls it before stopping the pane, so an
|
||||
* exception would orphan a live pane and skip the release notification (CB-516).
|
||||
*/
|
||||
@Test
|
||||
void hasUncommittedOnAMissingWorktreeReturnsFalseWithoutThrowing(@TempDir Path tmp) {
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String gone = tmp.resolve("wts").resolve("does-not-exist").toString();
|
||||
|
||||
assertFalse(gitWorktrees.hasUncommitted(gone),
|
||||
"a missing worktree is reported clean, not an error");
|
||||
}
|
||||
|
||||
/** All three protected configs are covered: each one present in a worktree is neutralized and hidden. */
|
||||
@Test
|
||||
void allThreeConfigsAreNeutralizedWhenPresent(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".mcp.json"), WITH_SERVERS);
|
||||
Files.writeString(repo.resolve("opencode.json"), OPENCODE_WITH_FILE_REF);
|
||||
Files.writeString(repo.resolve(".autoenv"), AUTOENV_WITH_DIRECTIVE);
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".mcp.json", "opencode.json", ".autoenv", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
String wt = new GitWorktrees(tmp.resolve("wts").toString())
|
||||
.add(repo.toString(), "cb-543-c", "HEAD");
|
||||
|
||||
assertTrue(Files.readString(Path.of(wt).resolve(".mcp.json"))
|
||||
.replaceAll("\\s+", "").contains("\"mcpServers\":{}"),
|
||||
".mcp.json was not neutralized");
|
||||
assertEquals("{}", Files.readString(Path.of(wt).resolve("opencode.json")).replaceAll("\\s+", ""),
|
||||
"opencode.json was not neutralized");
|
||||
assertEquals("", Files.readString(Path.of(wt).resolve(".autoenv")),
|
||||
".autoenv was not neutralized");
|
||||
|
||||
assertEquals("", status(Path.of(wt), ".mcp.json"), ".mcp.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), "opencode.json"), "opencode.json still shows as modified");
|
||||
assertEquals("", status(Path.of(wt), ".autoenv"), ".autoenv still shows as modified");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, acceptance criterion 1. A dirty worktree — a tracked edit plus a brand-new
|
||||
* untracked file, exactly the shape lost in CB-576 — must land in {@code refs/wip/<branch>}'s
|
||||
* tree, and that ref must live outside {@code refs/heads} so it never shows up in
|
||||
* {@code git branch} or gets swept by a branch cleanup.
|
||||
*/
|
||||
@Test
|
||||
void dirtySnapshotCreatesARefWhoseTreeContainsUntrackedAndTrackedChanges(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-578-c-a";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertTrue(ref.isPresent(), "a dirty worktree snapshot returns a commit sha");
|
||||
String tree = lsTree(repo, "refs/wip/" + branch);
|
||||
assertTrue(tree.contains("untracked.txt"), "snapshot tree must include the untracked file:\n" + tree);
|
||||
assertTrue(tree.contains("README.md"), "snapshot tree must include the tracked edit:\n" + tree);
|
||||
String heads = forEachRef(repo, "refs/heads");
|
||||
assertFalse(heads.contains("refs/wip/"), "the snapshot ref must not live under refs/heads:\n" + heads);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, acceptance criterion 2. Building the commit through a temporary
|
||||
* {@code GIT_INDEX_FILE} must leave the worker's own index, working tree, and HEAD exactly as
|
||||
* they were — the worker may still be mid-write, and staging into the real index would corrupt
|
||||
* that.
|
||||
*/
|
||||
@Test
|
||||
void snapshotDoesNotTouchTheWorkersOwnIndexWorkingTreeOrHead(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-578-c-b";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
String headBefore = revParse(Path.of(wt), "HEAD");
|
||||
String statusBefore = fullStatus(Path.of(wt));
|
||||
|
||||
gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertEquals(headBefore, revParse(Path.of(wt), "HEAD"), "snapshot must not move HEAD");
|
||||
assertEquals(statusBefore, fullStatus(Path.of(wt)),
|
||||
"snapshot must not change the worker's own index or working tree status");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-578 stage C, acceptance criterion 3. {@code add -A} (never {@code -f}) respects
|
||||
* {@code .gitignore}, and that is the only thing keeping a gitignored file (secrets, local
|
||||
* config) out of a snapshot commit — an untracked file that is NOT ignored must still be
|
||||
* included, so this isn't just "untracked files are dropped".
|
||||
*/
|
||||
@Test
|
||||
void gitignoredFileIsExcludedButOtherUntrackedFilesAreNot(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve(".gitignore"), ".env\n");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", ".gitignore", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-578-c-c";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve(".env"), "SECRET=shh\n");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertTrue(ref.isPresent());
|
||||
String tree = lsTree(repo, "refs/wip/" + branch);
|
||||
assertFalse(tree.contains(".env"), "a gitignored file must never enter the snapshot:\n" + tree);
|
||||
assertTrue(tree.contains("untracked.txt"),
|
||||
"a non-ignored untracked file must still be included:\n" + tree);
|
||||
}
|
||||
|
||||
/** CB-578 stage C. A clean worktree still produces a valid, if tree-identical, commit — the caller
|
||||
* (SessionManager) is the one that decides not to call this on a clean worktree. */
|
||||
@Test
|
||||
void snapshotOfAMissingWorktreeReturnsEmptyWithoutThrowing(@TempDir Path tmp) {
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String gone = tmp.resolve("wts").resolve("does-not-exist").toString();
|
||||
|
||||
assertTrue(gitWorktrees.snapshot(gone, "some-branch", "msg").isEmpty(),
|
||||
"a missing worktree has nothing to snapshot, and must not throw");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-587, acceptance criteria 1 and 2. A file with a REAL {@code --skip-worktree} bit set in the
|
||||
* worktree's own index must never enter the snapshot, even though its on-disk content has locally
|
||||
* diverged from what is committed — that is exactly the divergence {@code --skip-worktree} exists
|
||||
* to hide from {@code git status}, and a snapshot built from a fresh empty temp index (the bug)
|
||||
* stages that local content anyway because the fresh index carries none of the real index's flags.
|
||||
* The snapshot's diff against its parent must list exactly what {@code git status --porcelain}
|
||||
* reports for the worktree — no more, no less.
|
||||
*/
|
||||
@Test
|
||||
void dirtySnapshotHonoursARealSkipWorktreeBit(@TempDir Path tmp) throws Exception {
|
||||
Path repo = tmp.resolve("repo");
|
||||
Files.createDirectories(repo);
|
||||
git(repo, "init", "-q", "-b", "main");
|
||||
git(repo, "config", "user.email", "test@example.invalid");
|
||||
git(repo, "config", "user.name", "Test");
|
||||
Files.writeString(repo.resolve("protected.cfg"), "committed-value\n");
|
||||
Files.writeString(repo.resolve("README.md"), "seed\n");
|
||||
git(repo, "add", "protected.cfg", "README.md");
|
||||
git(repo, "commit", "-q", "-m", "seed");
|
||||
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-587-a";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
|
||||
git(Path.of(wt), "update-index", "--skip-worktree", "protected.cfg");
|
||||
Files.writeString(Path.of(wt).resolve("protected.cfg"), "locally-diverged-never-commit\n");
|
||||
Files.writeString(Path.of(wt).resolve("untracked.txt"), "draft that was never added\n");
|
||||
Files.writeString(Path.of(wt).resolve("README.md"), "edited tracked file\n");
|
||||
|
||||
String porcelain = fullStatus(Path.of(wt));
|
||||
assertFalse(porcelain.contains("protected.cfg"),
|
||||
"test setup invalid — protected.cfg must not show in git status once skip-worktree is set:\n"
|
||||
+ porcelain);
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "test snapshot");
|
||||
|
||||
assertTrue(ref.isPresent(), "a dirty worktree snapshot returns a commit sha");
|
||||
Set<String> diffPaths = diffNameOnly(repo, "HEAD", "refs/wip/" + branch);
|
||||
assertFalse(diffPaths.contains("protected.cfg"),
|
||||
"a --skip-worktree file's local drift leaked into the snapshot:\n" + diffPaths);
|
||||
assertEquals(porcelainPaths(porcelain), diffPaths,
|
||||
"snapshot diff must list exactly what git status --porcelain reports, no more, no less");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-587, acceptance criterion 6. If the worktree's real index cannot be resolved/read, snapshot
|
||||
* must fail loudly (throw) rather than silently falling back to an empty temp index and producing
|
||||
* a wrong snapshot. SessionManager's caller already catches and WARNs on any exception here — this
|
||||
* only needs to confirm the failure is not swallowed inside snapshot() itself.
|
||||
*/
|
||||
@Test
|
||||
void snapshotThrowsWhenTheWorktreesRealIndexCannotBeResolved(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-587-b";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
|
||||
// Break git's ability to resolve the worktree's real index by removing the linked worktree's
|
||||
// `.git` file (which normally points at the main repo's worktrees/<name>/ directory).
|
||||
Files.delete(Path.of(wt).resolve(".git"));
|
||||
|
||||
assertThrows(WorktreeException.class, () -> gitWorktrees.snapshot(wt, branch, "test snapshot"),
|
||||
"an unresolvable real index must fail loudly, not silently snapshot from an empty index");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, criterion 2. A snapshot whose content is NOT reachable from {@code main} is the last
|
||||
* copy of a worker's work, and must never be deleted automatically — even when it is old and
|
||||
* even when the caller passes a zero age floor. Uses the real snapshot path on a dirty worktree,
|
||||
* so the unreachable tree is exactly the shape CB-576/CB-578 stage C exist to protect.
|
||||
*/
|
||||
@Test
|
||||
void anUnreachableSnapshotIsNeverPruned(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
String branch = "cb-586-unreachable";
|
||||
String wt = gitWorktrees.add(repo.toString(), branch, "HEAD");
|
||||
Files.writeString(Path.of(wt).resolve("worker-draft.txt"), "work that exists nowhere else\n");
|
||||
|
||||
Optional<String> ref = gitWorktrees.snapshot(wt, branch, "snapshot with unreachable content");
|
||||
assertTrue(ref.isPresent());
|
||||
|
||||
// Age floor 0 makes age a non-issue: only reachability can save it — and it must.
|
||||
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), 0),
|
||||
"the unreachable snapshot is the last copy and must not be pruned");
|
||||
assertTrue(refExists(repo, "refs/wip/" + branch),
|
||||
"an unreachable snapshot must survive the sweep");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, criterion 1 (the reachable half). A snapshot whose tree content IS already reachable
|
||||
* from {@code main} and which is older than the age floor is pure duplication — the work is
|
||||
* recovered — so it must be pruned.
|
||||
*/
|
||||
@Test
|
||||
void aReachableSnapshotOlderThanTheFloorIsPruned(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
// A snapshot whose tree is exactly main's current tree: fully reachable from main.
|
||||
String mainTree = revParse(repo, "main^{tree}");
|
||||
String old = commitTree(repo, mainTree, revParse(repo, "HEAD"), "2020-01-01T00:00:00", "snapshot");
|
||||
updateRef(repo, "refs/wip/recovered", old);
|
||||
|
||||
assertEquals(1, gitWorktrees.pruneWipRefs(repo.toString(), TimeUnit.HOURS.toMillis(24)),
|
||||
"an old, main-reachable snapshot must be pruned");
|
||||
assertFalse(refExists(repo, "refs/wip/recovered"),
|
||||
"the reachable snapshot's ref must be gone after the sweep");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, the age floor. A snapshot whose content IS reachable from {@code main} but which is
|
||||
* younger than the age floor must not be swept — a lead may still be looking at it.
|
||||
*/
|
||||
@Test
|
||||
void aReachableButRecentSnapshotIsNotPruned(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
// Reachable from main, but committed "now" — a fresh snapshot. The 24h floor must protect it.
|
||||
String mainTree = revParse(repo, "main^{tree}");
|
||||
String fresh = commitTree(repo, mainTree, revParse(repo, "HEAD"),
|
||||
"2038-01-01T00:00:00", "snapshot just taken");
|
||||
updateRef(repo, "refs/wip/fresh", fresh);
|
||||
|
||||
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), TimeUnit.HOURS.toMillis(24)),
|
||||
"a recent snapshot must be kept even when reachable");
|
||||
assertTrue(refExists(repo, "refs/wip/fresh"),
|
||||
"the recent reachable snapshot must survive the sweep");
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-586, criterion 5. A fleet that has never snapshotted anything has no {@code refs/wip/*},
|
||||
* so a sweep is a no-op and the census reports none — identical to before CB-586 existed.
|
||||
*/
|
||||
@Test
|
||||
void aFleetWithNoSnapshotsPrunesNothingAndReportsNothing(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
assertEquals(0, gitWorktrees.pruneWipRefs(repo.toString(), 0),
|
||||
"no snapshot refs means nothing to prune");
|
||||
Worktrees.WipRefStats stats = gitWorktrees.wipRefs(repo.toString());
|
||||
assertEquals(0, stats.count(), "a never-snapshotted fleet has zero refs/wip refs");
|
||||
assertEquals(0L, stats.costBytes(), "a never-snapshotted fleet costs zero bytes");
|
||||
}
|
||||
|
||||
/** CB-586, criterion 4: the census reports how many refs exist and roughly what they cost. */
|
||||
@Test
|
||||
void wipRefsReportsCountAndCost(@TempDir Path tmp) throws Exception {
|
||||
Path repo = initRepo(tmp.resolve("repo"));
|
||||
GitWorktrees gitWorktrees = new GitWorktrees(tmp.resolve("wts").toString());
|
||||
|
||||
String blob = blobOf(repo, "a recoverable snapshot's worth of content");
|
||||
String tree = treeOf(repo, "snapshot.txt", blob);
|
||||
updateRef(repo, "refs/wip/one", commitTree(repo, tree, revParse(repo, "HEAD"),
|
||||
"2020-01-01T00:00:00", "snapshot"));
|
||||
updateRef(repo, "refs/wip/two", commitTree(repo, tree, revParse(repo, "HEAD"),
|
||||
"2020-01-02T00:00:00", "snapshot"));
|
||||
|
||||
Worktrees.WipRefStats stats = gitWorktrees.wipRefs(repo.toString());
|
||||
assertEquals(2, stats.count(), "two snapshot refs are reported");
|
||||
assertTrue(stats.costBytes() > 0, "the cost of the snapshots is a positive byte count");
|
||||
}
|
||||
}
|
||||
@@ -1,930 +0,0 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.LoggerContext;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.msg.TestTurnTokens;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.*;
|
||||
|
||||
/**
|
||||
* CB-301 / CB-303 acceptance tests for the authoritative session registry, one-shot lifecycle FSM,
|
||||
* and configurable lifecycle limits (idle TTL, context cap, drain).
|
||||
* No live herdr — everything runs against the same {@link FakeHerdr} the rest of the project uses.
|
||||
*/
|
||||
class SessionManagerTest {
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr) {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock) {
|
||||
return sessionManager(herdr, clock, 0);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, Worktrees worktrees) {
|
||||
return sessionManager(herdr, worktrees, System::nanoTime);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, Worktrees worktrees, LongSupplier clock) {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers, worktrees, clock);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-581: a {@link Worktrees} test double whose {@code hasUncommitted} and {@code remove} can
|
||||
* be told to throw, so {@link SessionManager#release} can be exercised against exactly the
|
||||
* failure {@code GitWorktrees} produces when {@code git status}/{@code git worktree remove}
|
||||
* exits non-zero.
|
||||
*/
|
||||
private static final class RecordingWorktrees implements Worktrees {
|
||||
private final List<String> removeCalls = new java.util.ArrayList<>();
|
||||
private final List<String> snapshotCalls = new java.util.ArrayList<>();
|
||||
private final java.util.Set<String> failRemoveFor = new java.util.HashSet<>();
|
||||
private volatile boolean dirty = false;
|
||||
private volatile RuntimeException hasUncommittedFailure;
|
||||
private volatile RuntimeException snapshotFailure;
|
||||
private final java.util.concurrent.atomic.AtomicLong snapshotSeq = new java.util.concurrent.atomic.AtomicLong();
|
||||
|
||||
RecordingWorktrees dirty(boolean dirty) {
|
||||
this.dirty = dirty;
|
||||
return this;
|
||||
}
|
||||
|
||||
RecordingWorktrees failHasUncommittedWith(RuntimeException e) {
|
||||
this.hasUncommittedFailure = e;
|
||||
return this;
|
||||
}
|
||||
|
||||
RecordingWorktrees failRemoveFor(String worktreePath) {
|
||||
failRemoveFor.add(worktreePath);
|
||||
return this;
|
||||
}
|
||||
|
||||
RecordingWorktrees failSnapshotWith(RuntimeException e) {
|
||||
this.snapshotFailure = e;
|
||||
return this;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String add(String repoRoot, String branch, String baseRef) {
|
||||
return "/wt/" + branch.replace('/', '_');
|
||||
}
|
||||
|
||||
@Override
|
||||
public void remove(String repoRoot, String worktreePath) {
|
||||
if (failRemoveFor.contains(worktreePath)) {
|
||||
throw new WorktreeException("simulated remove failure for " + worktreePath);
|
||||
}
|
||||
removeCalls.add(worktreePath);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasUncommitted(String worktreePath) {
|
||||
if (hasUncommittedFailure != null) {
|
||||
throw hasUncommittedFailure;
|
||||
}
|
||||
return dirty;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void overlayParity(String repoRoot, String worktreePath, List<String> overlay) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public String repoRoot(String cwd) {
|
||||
return "/repo";
|
||||
}
|
||||
|
||||
@Override
|
||||
public java.util.Optional<String> snapshot(String worktreePath, String branch, String message) {
|
||||
snapshotCalls.add(worktreePath);
|
||||
if (snapshotFailure != null) {
|
||||
throw snapshotFailure;
|
||||
}
|
||||
return java.util.Optional.of("wip" + snapshotSeq.incrementAndGet());
|
||||
}
|
||||
|
||||
@Override
|
||||
public WipRefStats wipRefs(String repoRoot) {
|
||||
return new WipRefStats(0, 0L);
|
||||
}
|
||||
|
||||
@Override
|
||||
public int pruneWipRefs(String repoRoot, long minAgeMillis) {
|
||||
return 0;
|
||||
}
|
||||
|
||||
List<String> removeCalls() {
|
||||
return List.copyOf(removeCalls);
|
||||
}
|
||||
|
||||
List<String> snapshotCalls() {
|
||||
return List.copyOf(snapshotCalls);
|
||||
}
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap) {
|
||||
return sessionManager(herdr, clock, contextCap, false);
|
||||
}
|
||||
|
||||
private SessionManager sessionManager(FakeHerdr herdr, LongSupplier clock, int contextCap,
|
||||
boolean clearAfterTurn) {
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null);
|
||||
return new SessionManager(workers, new GitWorktrees(), clock, contextCap, clearAfterTurn);
|
||||
}
|
||||
|
||||
@Test
|
||||
void primaryContactWithNoTerminalIsNotAReadinessSignal() {
|
||||
// The MCP context extractor calls presence.markPresent(p.terminal()) on EVERY request,
|
||||
// and the primary's terminal is null — the presence bridge must treat that as a no-op,
|
||||
// not feed it into the READY transition (which NPEd on the first real primary contact).
|
||||
SessionManager sessions = sessionManager(new FakeHerdr());
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null));
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(" "));
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireRegistersSpawningSessionWithDistinctPaneId() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession a = sessions.acquire("ltms-local", "/work/a", "/caller/a", "term_primary");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/work/b", "/caller/b", "term_primary");
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, a.state(), "fresh session starts spawning");
|
||||
assertEquals("ltms-local", a.profile());
|
||||
assertEquals("/work/a", a.cwd(), "explicit requested cwd is recorded");
|
||||
assertEquals("term_primary", a.ownerTerminal());
|
||||
assertTrue(a.spawnedAtNanos() > 0);
|
||||
assertNotNull(a.paneId());
|
||||
assertNotNull(a.terminalId());
|
||||
|
||||
assertNotEquals(a.paneId(), b.paneId(), "no pane reuse");
|
||||
assertNotEquals(a.terminalId(), b.terminalId(), "no terminal reuse");
|
||||
assertEquals(2, sessions.roster().size(), "both sessions are registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void rosterViewExposesTheCharterReceiptButNeverTheCharterText() {
|
||||
// The roster (fleet_list and GET /members both render through rosterView) must let a lead
|
||||
// see which charter a member got, without ever carrying the charter prose itself (CB-571).
|
||||
MemberSession s = new MemberSession("p1", "term1", "prof", MemberRole.DEV, "/cwd", null,
|
||||
0, 0, 0, MemberSession.State.READY, null, null,
|
||||
CharterReceipt.compose(MemberRole.DEV, "prof", "role charter", "role charter\n\nreply"), null);
|
||||
|
||||
Map<String, Object> view = SessionManager.rosterView(s, null);
|
||||
|
||||
assertEquals("fleet.charters.dev", view.get("charterSource"),
|
||||
"the config key that supplied the role charter is reported");
|
||||
assertEquals(CharterReceipt.digestOf("role charter\n\nreply"), view.get("charterSha256"),
|
||||
"the digest of the exact composed charter bytes is reported");
|
||||
assertFalse(view.values().toString().contains("role charter"),
|
||||
"the roster row must not embed the charter text itself");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aNullTerminalFromThePrimaryIsANoOpEvenWithSessionsRegistered() {
|
||||
// The primary resolves to a Principal with no terminal, and FleetMcp's context extractor
|
||||
// forwards that null into markPresent on EVERY MCP call. It only reached the registry scan
|
||||
// once a session existed, so this NPE'd the primary's second spawn while the first passed.
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.asPresence().markPresent(null),
|
||||
"the primary's null terminal must not blow up an unrelated tool call");
|
||||
assertDoesNotThrow(() -> sessions.onDelivered(null, TestTurnTokens.inert(null)));
|
||||
assertDoesNotThrow(() -> sessions.onTurnComplete(null));
|
||||
assertDoesNotThrow(() -> sessions.onTurnFailed(null));
|
||||
|
||||
assertEquals(MemberSession.State.SPAWNING, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"and must not transition any registered session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void presenceMovesSpawningToReadyAndDeliveredTurnMovesToDone() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
assertEquals(MemberSession.State.READY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"MCP presence moves SPAWNING → READY");
|
||||
assertTrue(sessions.asPresence().isPresent(terminal), "presence is also recorded");
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
assertEquals(MemberSession.State.BUSY, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"delivery moves READY → BUSY");
|
||||
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"turn completion moves BUSY → DONE");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseTearsDownWorkerAndRemovesFromRosterAndIsIdempotent() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
String paneId = session.paneId();
|
||||
|
||||
sessions.release(paneId);
|
||||
|
||||
assertTrue(herdr.called("pane.close"), "release tears the worker pane down");
|
||||
assertTrue(sessions.get(paneId).isEmpty(), "released session is no longer retrievable");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session is no longer in the roster");
|
||||
|
||||
assertDoesNotThrow(() -> sessions.release(paneId), "a second release is harmless");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedMovesSessionToFailed() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
|
||||
sessions.onTurnFailed(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.FAILED, updated.state(), "turn failure moves to FAILED");
|
||||
assertTrue(sessions.roster().contains(updated), "FAILED is still in acquired-minus-released roster");
|
||||
}
|
||||
|
||||
@Test
|
||||
void onTurnFailedIsLoggedAtWarnWithThePriorState() {
|
||||
// CB-564: this transition used to be a bare DEBUG "session marked failed" — a symptom with no
|
||||
// cause. A member that can no longer be delegated to must be at least WARN, and should name
|
||||
// what stage it failed at (here: BUSY, i.e. a turn was in flight and never resolved).
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
try {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
|
||||
sessions.onTurnFailed(terminal);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.findFirst()
|
||||
.orElse("no turn-failed WARN logged");
|
||||
assertTrue(warn.contains(terminal), "the log names the member: " + warn);
|
||||
assertTrue(warn.contains("BUSY"), "the log names the stage it failed at: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void rosterReflectsAcquiredMinusReleased() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
MemberSession a = sessions.acquire("ltms-local", "/a", "/caller", "ownerA");
|
||||
MemberSession b = sessions.acquire("ltms-local", "/b", "/caller", "ownerB");
|
||||
|
||||
assertEquals(2, sessions.roster().size());
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(a.paneId())));
|
||||
assertTrue(sessions.roster().stream().anyMatch(s -> s.paneId().equals(b.paneId())));
|
||||
|
||||
sessions.release(a.paneId());
|
||||
|
||||
assertEquals(1, sessions.roster().size());
|
||||
assertEquals(b.paneId(), sessions.roster().getFirst().paneId());
|
||||
}
|
||||
|
||||
// --- CB-303 lifecycle limits ----------------------------------------------------
|
||||
|
||||
@Test
|
||||
void reapIdleDoesNothingWhenNoSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L);
|
||||
|
||||
assertEquals(0, sessions.reapIdle(10));
|
||||
assertTrue(sessions.roster().isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 11;
|
||||
assertEquals(1, sessions.reapIdle(10), "READY session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "reaped session is removed from registry");
|
||||
assertTrue(herdr.called("pane.close"), "reaped session tears the pane down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void readySessionWithinIdleTtlSurvives() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
clock[0] = 5;
|
||||
assertEquals(0, sessions.reapIdle(10), "READY session within TTL is not reaped");
|
||||
assertEquals(MemberSession.State.READY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"READY session survives");
|
||||
}
|
||||
|
||||
@Test
|
||||
void busySessionPastIdleTtlIsNotReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
|
||||
clock[0] = 100;
|
||||
assertEquals(0, sessions.reapIdle(10), "BUSY session past TTL is never reaped");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"BUSY session remains");
|
||||
}
|
||||
|
||||
@Test
|
||||
void doneSessionPastIdleTtlIsReaped() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
clock[0] = 21;
|
||||
assertEquals(1, sessions.reapIdle(20), "DONE session past TTL is reaped");
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "DONE session is removed");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIdleReturnsCorrectCountAndSkipsBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "owner1");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "owner2");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId(), TestTurnTokens.inert(busy.terminalId()));
|
||||
|
||||
clock[0] = 50;
|
||||
assertEquals(1, sessions.reapIdle(30), "only READY past TTL is reaped");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "READY session is gone");
|
||||
assertEquals(MemberSession.State.BUSY,
|
||||
sessions.get(busy.paneId()).orElseThrow().state(),
|
||||
"BUSY session is still registered");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapDisabledSessionSurvivesMultipleTurns() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(MemberSession.State.DONE, updated.state(), "session finishes second turn");
|
||||
assertEquals(2, updated.turnCount(), "turn count tracks both deliveries");
|
||||
long releaseCloseCount = paneCloseCallsFor(herdr, "w9:pRoot_1"); // the real pane coordinate
|
||||
assertEquals(0, releaseCloseCount, "cap disabled — no forced release of the worker pane");
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapTwoReleasesAfterSecondComplete() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 2);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
String terminal = session.terminalId();
|
||||
sessions.asPresence().markPresent(terminal);
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
assertEquals(MemberSession.State.DONE,
|
||||
sessions.get(session.paneId()).orElseThrow().state(),
|
||||
"first turn completes without release");
|
||||
|
||||
sessions.onDelivered(terminal, TestTurnTokens.inert(terminal));
|
||||
sessions.onTurnComplete(terminal);
|
||||
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty(), "session released after cap reached");
|
||||
assertTrue(sessions.roster().isEmpty(), "released session leaves roster");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"forced release tears the worker pane down exactly once");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnResetsContextWithoutDoubleCountingTheTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
|
||||
sessions.onDelivered(session.terminalId(), TestTurnTokens.inert(session.terminalId()));
|
||||
assertTrue(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
|
||||
MemberSession updated = sessions.get(session.paneId()).orElseThrow();
|
||||
assertEquals(1, updated.turnCount(), "the reset is housekeeping, not a second delegation");
|
||||
assertEquals(List.of("/clear"), promptTexts(herdr));
|
||||
}
|
||||
|
||||
@Test
|
||||
void contextCapReleaseWinsOverClearAfterTurn() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 1, true);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId(), TestTurnTokens.inert(session.terminalId()));
|
||||
|
||||
assertFalse(sessions.hasPostTurnAction(session.terminalId()),
|
||||
"a session at its cap will be released, not reset for reuse");
|
||||
assertFalse(sessions.onTurnCompleteWithPostAction(session.terminalId()));
|
||||
assertTrue(sessions.get(session.paneId()).isEmpty());
|
||||
assertTrue(promptTexts(herdr).isEmpty(), "never send /clear into a worker being torn down");
|
||||
}
|
||||
|
||||
@Test
|
||||
void clearAfterTurnFalsePreservesCompletionWithoutAControlPrompt() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> 0L, 0, false);
|
||||
MemberSession session = sessions.acquire("ltms-local", null, "/caller", "term_primary");
|
||||
sessions.asPresence().markPresent(session.terminalId());
|
||||
sessions.onDelivered(session.terminalId(), TestTurnTokens.inert(session.terminalId()));
|
||||
|
||||
sessions.onTurnComplete(session.terminalId());
|
||||
|
||||
assertEquals(MemberSession.State.DONE, sessions.get(session.paneId()).orElseThrow().state());
|
||||
assertTrue(promptTexts(herdr).isEmpty());
|
||||
}
|
||||
|
||||
@Test
|
||||
void drainAllReleasesBusyAndReadySessionsAndWaitsForBusy() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr, () -> clock[0]);
|
||||
|
||||
MemberSession ready = sessions.acquire("ltms-local", "/ready", "/caller", "ownerR");
|
||||
MemberSession busy = sessions.acquire("ltms-local", "/busy", "/caller", "ownerB");
|
||||
sessions.asPresence().markPresent(ready.terminalId());
|
||||
sessions.asPresence().markPresent(busy.terminalId());
|
||||
sessions.onDelivered(busy.terminalId(), TestTurnTokens.inert(busy.terminalId()));
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(sessions.roster().isEmpty(), "drain clears the roster");
|
||||
assertTrue(sessions.get(ready.paneId()).isEmpty(), "ready session is released");
|
||||
assertTrue(sessions.get(busy.paneId()).isEmpty(), "busy session is released after timeout");
|
||||
// ready is the first spawn → pane w9:pRoot_1, busy the second → w9:pRoot_2 (FakeHerdr order).
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"ready worker pane is torn down");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_2"),
|
||||
"busy worker pane is torn down");
|
||||
}
|
||||
|
||||
private static long paneCloseCallsFor(FakeHerdr herdr, String paneId) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "pane.close".equals(c.method()))
|
||||
.filter(c -> paneId.equals(((Map<?, ?>) c.params()).get("pane_id")))
|
||||
.count();
|
||||
}
|
||||
|
||||
private static List<String> promptTexts(FakeHerdr herdr) {
|
||||
return herdr.calls.stream()
|
||||
.filter(c -> "agent.prompt".equals(c.method()))
|
||||
.map(c -> String.valueOf(((Map<?, ?>) c.params()).get("text")))
|
||||
.toList();
|
||||
}
|
||||
|
||||
// --- CB-306 spawn-readiness gate: no half-registered session on timeout ----------------
|
||||
|
||||
@Test
|
||||
void acquireThrowsPeerUnreachableWhenGateTimesOutAndRegistersNoSession() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
herdr.agentStatus("unknown"); // never becomes injectable
|
||||
long[] clock = {0};
|
||||
|
||||
// Gate-enabled launcher (1 ms timeout + no-op sleeper that advances clock past deadline)
|
||||
FleetConfig.Profile cfg = new FleetConfig.Profile(
|
||||
"ltms-local", "http://gx00.gw:8000", "coder", null, "BRIDGED_WORKER_TOKEN",
|
||||
List.of("ccs", "ltms-local"), "tab", "bridged-workers",
|
||||
"worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher workers = new ClaudeCodeLauncher(
|
||||
new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(cfg.profile(), cfg), cfg.profile(), _ -> null,
|
||||
1, () -> clock[0], () -> clock[0] += 10);
|
||||
SessionManager sessions = new SessionManager(workers, new GitWorktrees(), () -> 0L, 0);
|
||||
|
||||
assertThrows(PeerUnreachableException.class,
|
||||
() -> sessions.acquire("ltms-local", null, "/caller", "term_primary"),
|
||||
"acquire must throw PeerUnreachableException when spawn times out");
|
||||
|
||||
// No half-registered session — the error happened inside spawn, before
|
||||
// SessionManager could put() anything into the registry.
|
||||
assertTrue(sessions.roster().isEmpty(),
|
||||
"no session is registered when spawn times out (roster empty)");
|
||||
}
|
||||
|
||||
// --- CB-516: release must notify, so a blocked send can be failed --------------------------
|
||||
|
||||
@Test
|
||||
void releaseNotifiesTheListenerWithTheReleasedTerminal() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(detail -> released.add(detail.terminalId()));
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(java.util.List.of(s.terminalId()), released,
|
||||
"every teardown path funnels through release, so one hook must see the terminal");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releasingAnUnknownPaneNotifiesNobody() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(detail -> released.add(detail.terminalId()));
|
||||
|
||||
sessions.release("w9:p404"); // idempotent teardown of something already gone
|
||||
|
||||
assertTrue(released.isEmpty(), "no session removed ⇒ no send was waiting on it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void aThrowingReleaseListenerDoesNotBlockTheTeardown() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
sessions.onRelease(_ -> {
|
||||
throw new IllegalStateException("listener blew up");
|
||||
});
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller", null);
|
||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||
"a listener failure must never prevent the teardown it is reacting to");
|
||||
assertTrue(sessions.get(s.paneId()).isEmpty(), "and the session is still deregistered");
|
||||
}
|
||||
|
||||
// --- CB-581: a throw inside release() must not orphan the pane or abort reapIdle -----------
|
||||
|
||||
@Test
|
||||
void releasePreservesWorktreeWhenDirtyCheckThrows() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581a", null));
|
||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
try {
|
||||
assertDoesNotThrow(() -> sessions.release(s.paneId()),
|
||||
"a throwing dirty check must not abort the release");
|
||||
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"the worktree is preserved when its dirty state cannot be determined");
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains(s.worktree()))
|
||||
.findFirst()
|
||||
.orElse("no warn logged naming the worktree");
|
||||
assertTrue(warn.contains(s.paneId()), "the WARN names the pane: " + warn);
|
||||
assertTrue(warn.contains(s.terminalId()), "the WARN names the terminal: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseStillStopsThePaneWhenDirtyCheckThrows() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581b", null));
|
||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"),
|
||||
"the pane is stopped exactly once even though the dirty check threw");
|
||||
}
|
||||
|
||||
@Test
|
||||
void releaseStillNotifiesTheListenerWhenDirtyCheckThrows() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
java.util.List<String> released = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
sessions.onRelease(detail -> released.add(detail.terminalId()));
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581c", null));
|
||||
worktrees.failHasUncommittedWith(new WorktreeException("git status exited 128"));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(java.util.List.of(s.terminalId()), released,
|
||||
"a blocked caller must still be told the terminal was released, even though the "
|
||||
+ "dirty check threw");
|
||||
}
|
||||
|
||||
@Test
|
||||
void reapIdleSurvivesOneSessionThatFailsToRelease() {
|
||||
long[] clock = {0};
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees, () -> clock[0]);
|
||||
MemberSession a = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581d", null));
|
||||
MemberSession b = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581e", null));
|
||||
MemberSession c = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581f", null));
|
||||
sessions.asPresence().markPresent(a.terminalId());
|
||||
sessions.asPresence().markPresent(b.terminalId());
|
||||
sessions.asPresence().markPresent(c.terminalId());
|
||||
// The middle session's worktree removal fails — release() propagates that, so this is the
|
||||
// one call reapIdle's per-session guard must survive without skipping the rest of the pass.
|
||||
worktrees.failRemoveFor(b.worktree());
|
||||
|
||||
LoggerContext ctx = (LoggerContext) LoggerFactory.getILoggerFactory();
|
||||
ch.qos.logback.classic.Logger sessionLog =
|
||||
(ch.qos.logback.classic.Logger) LoggerFactory.getLogger(SessionManager.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.setContext(ctx);
|
||||
appender.start();
|
||||
sessionLog.addAppender(appender);
|
||||
sessionLog.setLevel(Level.WARN);
|
||||
int reaped;
|
||||
try {
|
||||
clock[0] = 100;
|
||||
reaped = sessions.reapIdle(10);
|
||||
|
||||
String warn = appender.list.stream()
|
||||
.filter(e -> e.getLevel().equals(Level.WARN))
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.contains(b.paneId()))
|
||||
.findFirst()
|
||||
.orElse("no reap-failure WARN logged");
|
||||
assertTrue(warn.contains(b.terminalId()), "the WARN names the failed session's terminal: " + warn);
|
||||
assertTrue(warn.contains(b.worktree()), "the WARN names the failed session's worktree: " + warn);
|
||||
} finally {
|
||||
sessionLog.detachAppender(appender);
|
||||
}
|
||||
|
||||
assertEquals(2, reaped, "the middle session's failure is logged, not counted as reaped");
|
||||
assertTrue(sessions.get(a.paneId()).isEmpty(), "the first session is still released");
|
||||
assertTrue(sessions.get(c.paneId()).isEmpty(), "the third session is still released");
|
||||
assertTrue(sessions.get(b.paneId()).isEmpty(),
|
||||
"the middle session is still deregistered even though its worktree removal threw");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_1"), "the first pane is stopped");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_2"),
|
||||
"the middle pane is still stopped even though its worktree removal failed");
|
||||
assertEquals(1, paneCloseCallsFor(herdr, "w9:pRoot_3"), "the third pane is stopped");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unchangedRegressionCleanCompletedReleaseStillRemovesTheWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581g", null));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertEquals(List.of(s.worktree()), worktrees.removeCalls(),
|
||||
"COMPLETED release of a clean worktree still removes it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unchangedRegressionDirtyCompletedReleaseStillPreservesTheWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees().dirty(true);
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581h", null));
|
||||
|
||||
sessions.release(s.paneId());
|
||||
|
||||
assertTrue(worktrees.removeCalls().isEmpty(),
|
||||
"COMPLETED release of a dirty worktree still preserves it");
|
||||
}
|
||||
|
||||
@Test
|
||||
void unchangedRegressionShutdownDrainStillPreservesTheWorktree() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
RecordingWorktrees worktrees = new RecordingWorktrees();
|
||||
SessionManager sessions = sessionManager(herdr, worktrees);
|
||||
MemberSession s = sessions.acquire("ltms-local", null, "/caller/proj", null,
|
||||
new WorktreeRequest("cb-581i", null));
|
||||
sessions.asPresence().markPresent(s.terminalId());
|
||||
|
||||
sessions.drainAll(TimeUnit.MILLISECONDS.toNanos(100));
|
||||
|
||||
assertTrue(worktrees.removeCalls().isEmpty(), "SHUTDOWN drain still preserves the worktree");
|
||||
}
|
||||
|
||||
// ── CB-584: agentSessionId recorded at acquire, and gated by Capability.SESSION_RESUME ─────
|
||||
|
||||
/**
|
||||
* A minimal {@link PeerLauncher} whose capabilities exclude SESSION_RESUME — the "does not
|
||||
* support it" case. {@link #requireResumeCapability} throws before ever reaching {@link #spawn},
|
||||
* so every method beyond {@link #capabilitiesFor} is unreachable in these tests and left unimplemented.
|
||||
*/
|
||||
private static final class NoResumeLauncher implements PeerLauncher {
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
return Set.of(Capability.WORKTREE); // deliberately no SESSION_RESUME
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
return capabilities();
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — the capability check refuses first");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return Set.of("stub-profile");
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return "stub-profile";
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — the capability check refuses first");
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<?> list() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireRecordsTheAgentSessionIdFromTheHandle() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", MemberRole.DEV, null, null, null, null,
|
||||
"my-session-name", null);
|
||||
|
||||
assertNotNull(s.agentSessionId(), "a spawn that asked for session identity gets one back");
|
||||
Map<String, Object> view = SessionManager.rosterView(s, null);
|
||||
assertEquals(s.agentSessionId(), view.get("agentSessionId"),
|
||||
"the roster exposes the same id the session recorded");
|
||||
}
|
||||
|
||||
@Test
|
||||
void acquireWithNeitherSessionFieldLeavesAgentSessionIdNull() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", null, null, null);
|
||||
|
||||
assertNull(s.agentSessionId(), "no identity requested — unchanged from before CB-584");
|
||||
assertFalse(SessionManager.rosterView(s, null).containsKey("agentSessionId"),
|
||||
"a null id is omitted from the roster, like charterSha256 for a receipt-less session");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSessionIdResumesTheSameConversationOnASupportingAdapter() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
MemberSession s = sessions.acquire("ltms-local", MemberRole.DEV, null, null, null, null,
|
||||
null, "cb-resume-77");
|
||||
|
||||
assertEquals("cb-resume-77", s.agentSessionId(),
|
||||
"a resume adopts the prior id as its own agentSessionId");
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSessionIdWithoutAnExplicitProfileIsRefused() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
SessionManager sessions = sessionManager(herdr);
|
||||
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, () ->
|
||||
sessions.acquire(null, MemberRole.DEV, null, null, null, null, null, "cb-resume-1"));
|
||||
|
||||
assertTrue(e.getMessage().contains("explicit profile"), e.getMessage());
|
||||
}
|
||||
|
||||
@Test
|
||||
void resumeSessionIdOnAnAdapterWithoutTheCapabilityIsRefusedNamingIt() {
|
||||
SessionManager sessions = new SessionManager(new NoResumeLauncher());
|
||||
|
||||
IllegalArgumentException e = assertThrows(IllegalArgumentException.class, () ->
|
||||
sessions.acquire("stub-profile", MemberRole.DEV, null, null, null, null, null, "cb-resume-1"));
|
||||
|
||||
assertTrue(e.getMessage().contains("SESSION_RESUME"), e.getMessage());
|
||||
assertTrue(e.getMessage().contains("stub-profile"), e.getMessage());
|
||||
}
|
||||
}
|
||||
@@ -1,61 +0,0 @@
|
||||
# CB-504 — systemd unit for bridged (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.bridged.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
#
|
||||
# Install (user service — bridged drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/bridged.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now bridged
|
||||
# journalctl --user -u bridged -f
|
||||
|
||||
[Unit]
|
||||
Description=bridged — claude-bridge message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# bridged retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take bridged down with it.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/bridged
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/bridged.jar bridged.yaml
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit bridged → [Service] / Environment=BRIDGED_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/bridged/env
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes bridged fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=bridged
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -1,36 +1,36 @@
|
||||
<?xml version="1.0" encoding="UTF-8"?>
|
||||
<!DOCTYPE plist PUBLIC "-//Apple//DTD PLIST 1.0//EN" "http://www.apple.com/DTDs/PropertyList-1.0.dtd">
|
||||
<!--
|
||||
CB-504 / CB-594 — launchd agent for bridged (macOS).
|
||||
CB-504 / CB-594 — launchd agent for fleetd (macOS).
|
||||
|
||||
This is the real supervision target today: the dogfooded daemon runs on macOS, where there is
|
||||
no systemd. A systemd unit ships alongside (deploy/bridged.service) for the Linux gateways
|
||||
no systemd. A systemd unit ships alongside (deploy/fleetd.service) for the Linux gateways
|
||||
CB-308 introduces.
|
||||
|
||||
Install:
|
||||
cp deploy/dev.ltms.bridged.plist ~/Library/LaunchAgents/
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.bridged.plist
|
||||
launchctl list | grep bridged
|
||||
cp deploy/dev.ltms.fleetd.plist ~/Library/LaunchAgents/
|
||||
launchctl load -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist
|
||||
launchctl list | grep fleetd
|
||||
|
||||
The paths below are already filled in for this host (resolved 2026-08-16 from
|
||||
`/usr/libexec/java_home`... except that reported the system Applet-plugin JVM, not the jenv-
|
||||
managed JDK 25 actually used to build/run bridged, so JAVA_HOME here is the real one:
|
||||
managed JDK 25 actually used to build/run fleetd, so JAVA_HOME here is the real one:
|
||||
`JENV_VERSION=25.0.3 java -XshowSettings:properties -version 2>&1 | grep java.home`; `which mvn`;
|
||||
`echo $HOME`). If this file is copied to a different host, re-resolve all three paths and check
|
||||
no placeholder path is left behind; scripts/redeploy-bridged.sh's check mode does not (and
|
||||
no placeholder path is left behind; scripts/redeploy-fleetd.sh's check mode does not (and
|
||||
cannot) check this file for you.
|
||||
|
||||
CB-594 — launchd cannot run a login shell (see the PATH comment on EnvironmentVariables below,
|
||||
and scripts/bridged-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
|
||||
and scripts/fleetd-launchd-wrapper.sh for the fix): ProgramArguments below execs THAT wrapper,
|
||||
not java directly, so WORKER_GITEA_TOKEN and AI_GATEWAY_TOKEN still get sourced from
|
||||
${SHARED_ENV}/tools/secrets.sh even though launchd itself never sources anything.
|
||||
|
||||
Note on ordering: launchd has no "start after herdr" primitive for user agents, and neither
|
||||
does systemd in a way that survives a socket appearing late. bridged retries the herdr socket
|
||||
does systemd in a way that survives a socket appearing late. fleetd retries the herdr socket
|
||||
on startup instead, so an agent that comes up before herdr converges rather than dying — that
|
||||
retry is the actual fix; KeepAlive below is the backstop.
|
||||
|
||||
CB-594 — KeepAlive vs. scripts/redeploy-bridged.sh: a bare SIGTERM makes this JVM exit 143 even
|
||||
CB-594 — KeepAlive vs. scripts/redeploy-fleetd.sh: a bare SIGTERM makes this JVM exit 143 even
|
||||
with its shutdown hook running to completion (measured, see the CB-594 report), which
|
||||
SuccessfulExit:false below reads as a crash and races to restart the OLD jar. The redeploy
|
||||
script now detects a loaded agent and uses `launchctl unload`/`load` instead of a raw kill, so
|
||||
@@ -40,20 +40,20 @@
|
||||
<plist version="1.0">
|
||||
<dict>
|
||||
<key>Label</key>
|
||||
<string>dev.ltms.bridged</string>
|
||||
<string>dev.ltms.fleetd</string>
|
||||
|
||||
<key>ProgramArguments</key>
|
||||
<array>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/bridged-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/scripts/fleetd-launchd-wrapper.sh</string>
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin/java</string>
|
||||
<string>-jar</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/target/bridged.jar</string>
|
||||
<string>bridged.yaml</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/target/fleetd.jar</string>
|
||||
<string>fleetd.yaml</string>
|
||||
</array>
|
||||
|
||||
<!-- Config path in ProgramArguments is relative, so the working directory must be the module. -->
|
||||
<key>WorkingDirectory</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd</string>
|
||||
|
||||
<key>EnvironmentVariables</key>
|
||||
<dict>
|
||||
@@ -62,7 +62,7 @@
|
||||
<key>HERDR_SOCKET_PATH</key>
|
||||
<string>/Users/dai.ha/.config/herdr/herdr.sock</string>
|
||||
<!--
|
||||
PATH matters more than it looks (CB-511): bridged propagates its own PATH to every worker
|
||||
PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker
|
||||
it spawns, so this line decides whether the fleet can run a build at all. launchd does NOT
|
||||
source .zprofile/.zshrc, so without this the daemon (and therefore every worker) gets a
|
||||
bare /usr/bin:/bin and no JDK or Maven. Keep the toolchain entries first.
|
||||
@@ -71,10 +71,10 @@
|
||||
<string>/Users/dai.ha/Softwares/jdks/jdk-25.0.3.jdk/Contents/Home/bin:/Users/dai.ha/Softwares/apache-maven/bin:/opt/homebrew/bin:/usr/local/bin:/usr/bin:/bin:/usr/sbin:/sbin</string>
|
||||
<!--
|
||||
Worker/API tokens are NOT set here: this file is committed. CB-594 —
|
||||
scripts/bridged-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
|
||||
scripts/fleetd-launchd-wrapper.sh (named in ProgramArguments above) is what supplies
|
||||
them, by execing a login shell that sources ${SHARED_ENV}/tools/secrets.sh before the
|
||||
daemon itself starts. bridged also reads the API token from the env var named by
|
||||
auth.tokenEnv (default BRIDGED_API_TOKEN) and only in auth.mode: token — the wrapper
|
||||
daemon itself starts. fleetd also reads the API token from the env var named by
|
||||
auth.tokenEnv (default FLEETD_API_TOKEN) and only in auth.mode: token — the wrapper
|
||||
covers that one too, since it is the same login shell.
|
||||
-->
|
||||
</dict>
|
||||
@@ -84,21 +84,21 @@
|
||||
|
||||
<!--
|
||||
CB-600 — read this before assuming ThrottleInterval bounds anything. It paces restarts to at
|
||||
most one per 10s; it does NOT cap how many times launchd retries. If bridged fails fast on
|
||||
every start — a bad bridged.yaml, for example auth.mode: token with the token env var unset,
|
||||
most one per 10s; it does NOT cap how many times launchd retries. If fleetd fails fast on
|
||||
every start — a bad fleetd.yaml, for example auth.mode: token with the token env var unset,
|
||||
which throws in main() before the daemon ever binds a port — launchd restarts it forever,
|
||||
once every 10s, until a human intervenes. LaunchAgents have no "give up after N attempts"
|
||||
primitive, so this is not something a config change here can fix.
|
||||
|
||||
That loop stops only two ways: (1) `launchctl unload -w ~/Library/LaunchAgents/dev.ltms.bridged.plist`,
|
||||
That loop stops only two ways: (1) `launchctl unload -w ~/Library/LaunchAgents/dev.ltms.fleetd.plist`,
|
||||
or (2) the underlying cause gets fixed, so the process starts successfully and stays up (no
|
||||
more exits to restart). scripts/redeploy-bridged.sh does not add a third way — it does not
|
||||
make bridged self-disable on a config error, on purpose: a fail-fast exit path that
|
||||
more exits to restart). scripts/redeploy-fleetd.sh does not add a third way — it does not
|
||||
make fleetd self-disable on a config error, on purpose: a fail-fast exit path that
|
||||
sometimes decides "this is unrecoverable, stop trying" is one more thing that can misfire,
|
||||
and a wrongly self-disabled daemon needs the exact same manual `launchctl load -w` recovery
|
||||
this comment already names — so it buys nothing an operator watching for the crash loop
|
||||
doesn't already have, at the cost of a new way to be silently down. Watch for it with
|
||||
`launchctl list dev.ltms.bridged` (a high restart count) or by tailing bridged.out for the
|
||||
`launchctl list dev.ltms.fleetd` (a high restart count) or by tailing fleetd.out for the
|
||||
same startup error repeating every ~10s.
|
||||
-->
|
||||
<key>KeepAlive</key>
|
||||
@@ -110,16 +110,16 @@
|
||||
<integer>10</integer>
|
||||
|
||||
<!--
|
||||
CB-594 — same file scripts/redeploy-bridged.sh already tails ($BRIDGED/bridged.out), and both
|
||||
CB-594 — same file scripts/redeploy-fleetd.sh already tails ($BRIDGED/fleetd.out), and both
|
||||
streams point at it, not two separate log files: the script's fresh-line / ERROR-count checks
|
||||
after a restart read this one path regardless of whether launchd or the script started the
|
||||
process, and a stdout/stderr split would make half of what happens during a launchd-driven
|
||||
restart invisible to it.
|
||||
-->
|
||||
<key>StandardOutPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/bridged.out</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
<key>StandardErrorPath</key>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/bridged/bridged.out</string>
|
||||
<string>/Users/dai.ha/LTMS/claude-bridge/fleetd/fleetd.out</string>
|
||||
|
||||
<key>ProcessType</key>
|
||||
<string>Background</string>
|
||||
@@ -0,0 +1,85 @@
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# fleetd #360: the previous version of this file started clean and broke the daemon in three ways
|
||||
# that nothing logs (see the DO NOT block and the ExecStart/PrivateTmp comments below for what and
|
||||
# why). The unit below, plus its companion deploy/herdr.service, is the version that has actually
|
||||
# run on fleet01 without those failures. Do not "improve" it back toward the old shape without
|
||||
# re-reading why each line is the way it is.
|
||||
#
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/fleetd.service deploy/herdr.service ~/.config/systemd/user/
|
||||
# # edit WorkingDirectory / ExecStart below for your host's paths and java location
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now herdr fleetd
|
||||
# loginctl enable-linger $USER # REQUIRED -- see below
|
||||
# journalctl --user -u fleetd -f
|
||||
#
|
||||
# `loginctl enable-linger` is not optional and is easy to miss, because leaving it out looks like
|
||||
# success: `systemctl --user enable` reports "enabled" and both units run for as long as you stay
|
||||
# logged in. A user manager without lingering starts at your first login and stops at your last
|
||||
# logout, so the fleet simply does not come back after a reboot -- which is the whole reason to
|
||||
# use systemd here rather than the setsid scripts these units replaced. Check it with
|
||||
# `loginctl show-user $USER -p Linger`; the answer must be `Linger=yes`.
|
||||
#
|
||||
# Secrets (AI_GATEWAY_TOKEN, WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI, ...) are not set
|
||||
# here and need no systemd drop-in: ExecStart runs a login shell, so they come from wherever your
|
||||
# login shell already sources them (this host: ~/.fleet/secrets.sh via ~/.zprofile). If a token is
|
||||
# missing there, fleetd still starts — the daemon reports every secret a configured profile
|
||||
# references, by name, never by value:
|
||||
# journalctl --user -u fleetd | grep 'startup secret'
|
||||
# A resolved one logs "startup secret NAME: set (profile 'x' tokenEnv)"; a missing one logs
|
||||
# "startup secret NAME: MISSING" at WARN and the daemon starts anyway — the first visible symptom
|
||||
# is a member that cannot open a pull request, hours later and in a different component.
|
||||
|
||||
[Unit]
|
||||
Description=fleetd — fleet message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only. fleetd retries the herdr socket rather than exiting, which is what actually makes
|
||||
# a late socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down too.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/LTMS/fleetd/fleetd
|
||||
|
||||
# A LOGIN shell, not java directly. Every secret this daemon needs (AI_GATEWAY_TOKEN,
|
||||
# WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI) lives in ~/.fleet/secrets.sh, which only
|
||||
# ~/.zprofile sources. systemd runs no login shell. Started any other way the daemon boots fine
|
||||
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
||||
# request. exec keeps it one process, so systemd tracks the right PID.
|
||||
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
||||
ExecStart=/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"
|
||||
|
||||
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
||||
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
||||
# unit) has to read them. A private /tmp turns the credential scrub into a silent no-op.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config makes fleetd fail fast by design. Give up rather than restart-loop forever.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# DO NOT add ProtectSystem=, ProtectHome=, ProtectKernelTunables= or ProtectControlGroups=.
|
||||
# Measured on fleet01 2026-09-05: each of those gives the unit its own mount namespace, and
|
||||
# fleetd resolves a caller role by running lsof to find the loopback peer PID
|
||||
# (mcp/LsofPeerPidLookup). Inside such a namespace lsof returns nothing, every caller falls back
|
||||
# to ANONYMOUS, and the primary is refused every orchestration call with
|
||||
# "unauthenticated: anonymous may not SPAWN".
|
||||
# The daemon still starts, healthz still returns ok and the secrets still resolve - the only
|
||||
# symptom is that the fleet cannot be driven at all. Verified by bisecting the directives:
|
||||
# no sandbox 3 lsof lines | ProtectSystem=strict 0 | ProtectHome=read-only 0
|
||||
# ProtectKernelTunables 0 | ProtectControlGroups 0 | RestrictSUIDSGID 3 | NoNewPrivileges 3
|
||||
# The two below add no mount namespace and are safe.
|
||||
NoNewPrivileges=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=fleetd
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/bin/zsh
|
||||
# fleetd #360 — template for the script deploy/herdr.service's ExecStart wraps in a pty.
|
||||
#
|
||||
# `script -qfec <this> /dev/null` needs a real command to run, and that command has to be a LOGIN
|
||||
# shell script: herdr itself needs the same secrets fleetd.service's login shell picks up (this
|
||||
# host: ~/.fleet/secrets.sh via ~/.zprofile), because members it spawns inherit its environment.
|
||||
# systemd's own Environment= lines in herdr.service are not enough for that -- they set TERM and a
|
||||
# bare PATH so the pty starts at all, nothing more.
|
||||
#
|
||||
# Copy this file to the path deploy/herdr.service's ExecStart names
|
||||
# (%h/LTMS/fleetd/fleetd-run/herdr-inner.sh by default) and `chmod +x` it. Not committed under
|
||||
# that path itself because the session name below is host-specific.
|
||||
|
||||
# A 0x0 pty makes every pane spawn fail with "ghostty error -2" (see herdr-multi-instance-facts /
|
||||
# fleet01-headless-herdr-standup) -- give it a real size before herdr ever touches it.
|
||||
stty rows 50 cols 200
|
||||
|
||||
# -l: login shell, so herdr and everything it spawns gets the real secrets and PATH.
|
||||
exec zsh -lc 'exec herdr --session <name>'
|
||||
@@ -0,0 +1,40 @@
|
||||
# fleetd #360 — systemd unit for herdr (Linux), the terminal multiplexer fleetd drives.
|
||||
#
|
||||
# This is fleetd.service's companion: fleetd.service's After=/Wants=herdr.service assumes this
|
||||
# unit exists. Before this ticket it did not, so on a fresh host fleetd started against a herdr
|
||||
# that systemd never supervised at all.
|
||||
#
|
||||
# Install: see deploy/fleetd.service's header comment (both units install the same way).
|
||||
#
|
||||
# ExecStart below runs deploy/herdr-inner.sh (copy the template of that name from this directory
|
||||
# to the path in ExecStart, or point ExecStart at wherever you keep it, and make it executable).
|
||||
# It is a separate file rather than an inline command because it must itself be a login shell (see
|
||||
# its own header for why) and systemd's ExecStart does not run one.
|
||||
|
||||
[Unit]
|
||||
Description=herdr terminal multiplexer (fleet session)
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
# script(1) gives herdr a real pty. Without it the client reports a 0x0 window and every pane
|
||||
# spawn fails with "ghostty error -2" -- which surfaces as a fleetd spawn failure, not a herdr one.
|
||||
ExecStart=/usr/bin/script -qfec %h/LTMS/fleetd/fleetd-run/herdr-inner.sh /dev/null
|
||||
StandardInput=null
|
||||
Environment=TERM=xterm-256color
|
||||
Environment=PATH=%h/.local/bin:/usr/local/bin:/usr/bin:/bin
|
||||
|
||||
# PrivateTmp MUST stay false. fleetd writes the member ZDOTDIR scrub dir and the opencode config
|
||||
# dir under its own java.io.tmpdir, and the member pane -- a child of THIS process -- has to read
|
||||
# them. A private /tmp here silently breaks the credential scrub instead of failing loudly.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=5s
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=herdr
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -1,8 +1,8 @@
|
||||
# LavinMQ — the AMQP broker behind bridged's durable ReplyInbox (CB-307 Stage 2).
|
||||
# LavinMQ — the AMQP broker behind fleetd's durable ReplyInbox (CB-307 Stage 2).
|
||||
#
|
||||
# Why this file exists: the broker was previously run ad hoc and simply vanished from the host,
|
||||
# which takes bridged down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Bridged.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# which takes fleetd down with it — AmqpReplyInbox.open throws on an unreachable broker and
|
||||
# Fleetd.java:187 does not guard it, so a missing broker is a hard startup failure, not a
|
||||
# degraded mode. This pins the version, keeps the data, and brings itself back after a reboot.
|
||||
#
|
||||
# Usage:
|
||||
@@ -14,27 +14,27 @@
|
||||
#
|
||||
# Management UI: http://127.0.0.1:15672 (guest / guest)
|
||||
#
|
||||
# This is bridged's OWN broker. Do not point bridged at any other AMQP server on this host —
|
||||
# This is fleetd's OWN broker. Do not point fleetd at any other AMQP server on this host —
|
||||
# notably not the `local-rabbitmq` container, which belongs to a different project and would end
|
||||
# up carrying this project's queues.
|
||||
|
||||
name: bridged-broker
|
||||
name: fleetd-broker
|
||||
|
||||
services:
|
||||
lavinmq:
|
||||
# Pinned deliberately: :latest silently moves the broker under a running daemon.
|
||||
image: cloudamqp/lavinmq:2.9.1
|
||||
container_name: bridged-lavinmq
|
||||
container_name: fleetd-lavinmq
|
||||
|
||||
# The failure this deployment exists to prevent — survive reboots and Docker restarts, but
|
||||
# stay down if it was stopped on purpose.
|
||||
restart: unless-stopped
|
||||
|
||||
# Loopback-bound on purpose. LavinMQ ships a default guest/guest account, which is only
|
||||
# acceptable because nothing off-host can reach it. bridged connects over 127.0.0.1, and
|
||||
# acceptable because nothing off-host can reach it. fleetd connects over 127.0.0.1, and
|
||||
# binding 0.0.0.0 here would expose a broker with default credentials to the network.
|
||||
ports:
|
||||
- "127.0.0.1:5672:5672" # AMQP — bridged.yaml broker.uri points here
|
||||
- "127.0.0.1:5672:5672" # AMQP — fleetd.yaml broker.uri points here
|
||||
- "127.0.0.1:15672:15672" # HTTP management API + UI
|
||||
|
||||
# The whole point of Stage 2. Held-but-unacked replies live here; without a named volume a
|
||||
@@ -57,4 +57,4 @@ services:
|
||||
|
||||
volumes:
|
||||
lavinmq-data:
|
||||
name: bridged-lavinmq-data
|
||||
name: fleetd-lavinmq-data
|
||||
|
||||
@@ -0,0 +1,531 @@
|
||||
# CB-201 and CB-227 refinement
|
||||
|
||||
Date: 2026-09-03
|
||||
|
||||
## Decision
|
||||
|
||||
#201 and #227 are one delivery program, but they are not one implementation unit.
|
||||
|
||||
#201 has a real seam: `CompletionResolver` can publish a typed backend-error event only after its
|
||||
waiter resolution wins. #227 can consume that event without knowing any pane text. The classifier
|
||||
must land before the final #227 wiring. However, the policy engine, roster state, and lead nudge can
|
||||
be built in parallel with the classifier.
|
||||
|
||||
I propose five units. Units 1 to 4 own separate files and can run in parallel. Unit 5 owns all
|
||||
composition files and lands after them. It also depends on the #234 defect 2 fix named in the task.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
U1["Unit 1: typed backend-error classification"] --> U5["Unit 5: wire policy, spawn gate, and fleet views"]
|
||||
U2["Unit 2: credential outage policy"] --> U5
|
||||
U3["Unit 3: lead outage nudge"] --> U5
|
||||
U4["Unit 4: durable member outcome"] --> U5
|
||||
D234["#234 defect 2: fail-loud target resolution"] --> U5
|
||||
```
|
||||
|
||||
*Figure 1. Four file-disjoint foundations feed one composition unit.*
|
||||
|
||||
This split keeps `Fleetd.java` under one owner. It also keeps every other production file under one
|
||||
unit in this plan.
|
||||
|
||||
## Evidence checked in the current branch
|
||||
|
||||
I read both issue pages in full. Each page reports zero comments.
|
||||
|
||||
| Evidence | What the code says now |
|
||||
|---|---|
|
||||
| `inject/CompletionResolver.java:229-237` | A turn below two seconds fails before normal scrape classification. A matching fast backend error is therefore only a generic failure today. |
|
||||
| `inject/CompletionResolver.java:260-276` and `:332-367` | #211 already added raw-screen classification when `lastAssistantBlock` is empty. The “dead-code question” in #201 is stale on this branch. |
|
||||
| `inject/CompletionResolver.java:288-317` | Exhaustion wins before the hard-coded `API Error:` match. A backend error then goes through generic `fail(...)`. |
|
||||
| `inject/CompletionResolver.java:311-316` | The code admits that the pattern is a heuristic. A member report which quotes an API error may match it. |
|
||||
| `inject/CompletionResolver.java:449-467` | Startup coverage exists only for `exhaustedPattern`. |
|
||||
| `inject/ExhaustedPatternLookup.java:13-25` | The current lookup and explicit `none()` value are a good shape for the new classifier seam. |
|
||||
| `Fleetd.java:196-207` | One `BackendQuarantine` is shared by placement and the exhaustion sink. Its cooldown comes from `quarantineCooldownSeconds`. |
|
||||
| `Fleetd.java:322-363` | Pattern compilation, target-to-profile lookup, and the live `ExhaustionSink` are composed in `Fleetd.main`. The sink on this branch still ends in `.ifPresent(...)`. This plan assumes #234 replaces that silent path. |
|
||||
| `placement/BackendQuarantine.java:60-87` | A repeated exhaustion restarts one long quarantine. The store is credential-keyed and uses an injected monotonic clock. |
|
||||
| `member/CompositePeerLauncher.java:260-317` | Explicit and policy-selected spawns have separate gates. Both paths must learn about outage cool-off. |
|
||||
| `member/CompositePeerLauncher.java:347-379` | Exhaustion refusal already checks a credential for explicit spawns and filters policy candidates. Its error text says “exhausted”. |
|
||||
| `placement/PlacementContext.java:10-22` and `PlacementPolicyUtil.java:14-83` | Automatic placement has only one transient exclusion set named `quarantined`. Reusing it would make outage errors say “backend exhausted”. |
|
||||
| `mcp/FleetMcp.java:913-1025` | `fleet_list` sets `free: 0` and adds `credentialId` plus `quarantinedForSeconds` when quarantine is active. |
|
||||
| `session/MemberSession.java:51-59` | The roster has `DONE` and generic `FAILED`, but no backend-error state or stored reason. |
|
||||
| `session/SessionManager.java:695-773` | A normal boundary moves `BUSY` to `DONE`. A failure moves any non-released session to `FAILED`. The async completion resolver can race the `DONE` update. |
|
||||
| `session/SessionManager.java:648-687` | `rosterView` reports the session state, but it reports no terminal reason. |
|
||||
| `msg/MessageService.java:922-940` | CB-588 already nudges for every terminal async ticket, including failures. Current code would report failed tickets, but it would not report one correlated outage. |
|
||||
| `msg/ReplyPushLoop.java:20-48` | Replies, terminal tickets, and questions share one per-lead schedule. This prevents two push sources from injecting competing turns. |
|
||||
| `msg/ReplyPushLoop.java:305-395` | Each push entry point resolves the owning lead through `PrimaryRegistry`. Missing ownership is logged and the durable or pending item remains the backstop. |
|
||||
| `msg/ReplyPushLoop.java:496-547` | One tick builds one combined nudge. Pending items have separate reminder counts. |
|
||||
| `health/FleetHealthMonitor.java:91-143` | Health is a slow periodic observer of members and message-layer facts. It does not receive completion classifications. |
|
||||
| `health/FleetHealthMonitor.java:206-208` | `healthCoverage` means health enabled plus webhook configured. It does not describe lead-pane alerts. |
|
||||
| `Fleetd.java:465-486` | Health stays `detection-only` without the webhook notification setting. |
|
||||
|
||||
I also read the related unit tests for `CompletionResolver`, `ReplyPushLoop`, `BackendQuarantine`,
|
||||
`CompositePeerLauncher`, `PlacementPolicyUtil`, `SessionManager`, `MessageService`, and `FleetMcp`.
|
||||
|
||||
I did not inspect the in-progress #234 branch. I only used the two measured facts in the task. No
|
||||
peer architect was named, so I did not exchange a design with one.
|
||||
|
||||
## Required behaviour
|
||||
|
||||
The policy should use these first values:
|
||||
|
||||
- Threshold: **2** classified backend errors.
|
||||
- Window: **60 seconds**, measured from the first error to the second.
|
||||
- Cool-off: **60 seconds**, starting when the threshold is reached.
|
||||
- Correlation key: `credentialId`, never profile name and never error text.
|
||||
- Incident rule: one active incident per credential. Errors during its cool-off do not extend it and
|
||||
do not create more lead notices.
|
||||
- Rearm rule: after cool-off ends, two fresh errors are needed for another incident.
|
||||
|
||||
Two errors are the smallest threshold which protects the honest one-turn failure. A 60-second window
|
||||
fits the measured two-member outage. A 60-second cool-off blocks immediate repeat spawns without
|
||||
turning a short backend fault into the default 1,800-second exhaustion quarantine.
|
||||
|
||||
A single classified error still fails its send and marks its member `backend_error`. It does not
|
||||
cool a credential and does not send an outage notice. This is what “a single error changes nothing”
|
||||
must mean at the credential level. It cannot mean that the failed member still looks successful.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant R1 as Resolver for member A
|
||||
participant R2 as Resolver for member B
|
||||
participant P as Outage policy
|
||||
participant S as Spawn gate
|
||||
participant N as Lead push loop
|
||||
participant L as Lead pane
|
||||
|
||||
R1->>P: backend error for credential C
|
||||
Note over P: Count 1, no cool-off
|
||||
R2->>P: backend error for credential C within 60s
|
||||
P->>P: Start one 60s incident
|
||||
P->>S: Credential C is cooling off
|
||||
P->>N: Queue one incident notice
|
||||
N->>L: Inject when lead is idle, blocked, or done
|
||||
L->>S: Request another spawn on credential C
|
||||
S-->>L: Refuse and report remaining cool-off
|
||||
```
|
||||
|
||||
*Figure 2. The second independent classification creates the fleet-level event.*
|
||||
|
||||
Against the 2026-09-01 case, the second failed member would start cool-off. `fleet_list` would show
|
||||
zero free capacity and both members as `backend_error`. The push loop would inject one outage notice
|
||||
even if the lead had not polled either ticket yet. The design reports the outage. It does not recover
|
||||
uncommitted work from the members.
|
||||
|
||||
## Unit 1 — Typed backend-error classification
|
||||
|
||||
### Scope
|
||||
|
||||
Replace the direct hard-coded check inside `CompletionResolver` with a lookup and a sink. Keep the
|
||||
public send result as a failed send. The typed internal event is the seam #227 consumes.
|
||||
|
||||
The lookup returns the pattern for a target. The sink receives the target, matched line, and full
|
||||
failure reason. It fires only after `Rendezvous.resolveFailure(...)` wins for that exact captured
|
||||
waiter. This copies the race rule already used by `ExhaustionSink`.
|
||||
|
||||
The classifier must run in all three current paths:
|
||||
|
||||
1. a normal non-empty assistant block;
|
||||
2. the #211 raw scrape fallback;
|
||||
3. a turn inside `MIN_TURN_NANOS`, before it becomes a generic too-fast failure.
|
||||
|
||||
In every path, the order stays: stale-baseline guard, exhaustion, backend error, then generic
|
||||
failure or completion. A fast turn still fails when no configured pattern matches.
|
||||
|
||||
Keep `(?i)\bAPI Error\s*:` as a compatibility pattern for profiles without `errorPattern` until the
|
||||
operator config is updated. Do not call this full coverage. Startup reporting in Unit 5 must name
|
||||
profiles using this weaker legacy default.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorPatternLookup.java`.
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorSink.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/inject/CompletionResolverTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. A target-specific error pattern matches a normal assistant block and resolves the send as failed.
|
||||
2. The same match calls `BackendErrorSink` exactly once after the waiter resolution wins.
|
||||
3. A late classification which loses to `fleet_reply` does not call the sink.
|
||||
4. An exhausted line that also matches the generic error pattern stays `BACKEND_EXHAUSTED`. It calls
|
||||
only `ExhaustionSink`.
|
||||
5. A raw pane with leading Terminal User Interface (TUI) chrome and no assistant marker still uses
|
||||
the #211 fallback and calls the backend-error sink.
|
||||
6. A matching error inside the two-second floor is typed and sent to the sink. A non-matching fast
|
||||
turn stays a generic failure.
|
||||
7. An unchanged delivery baseline which contains old backend-error text is suppressed. It never
|
||||
increments outage evidence.
|
||||
8. A non-match keeps the existing completion result and text.
|
||||
9. Constructors used by current callers keep compiling. They use the legacy default lookup and an
|
||||
explicit inert sink until Unit 5 supplies the production objects.
|
||||
10. Unit tests pass. The developer runs the focused test first, then `mvn clean install` from
|
||||
`fleetd/`.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 1 can run with Units 2, 3, and 4.
|
||||
|
||||
Unit 5 depends on its new lookup, sink, and constructor.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact classifier order in all three paths.
|
||||
- The focused test command and result.
|
||||
- The test which proves a losing waiter race does not publish an event.
|
||||
- The test which proves a fast matching failure is typed.
|
||||
- The final `mvn clean install` result.
|
||||
- Any constructor kept only for transition and where Unit 5 replaces it.
|
||||
|
||||
## Unit 2 — Credential outage policy
|
||||
|
||||
### Scope
|
||||
|
||||
Build a small credential-keyed state machine. It accepts already-classified backend-error events.
|
||||
It does not read pane text, profiles, sessions, or lead state.
|
||||
|
||||
Use an injected monotonic clock. A call records `credentialId`, target, and reason. It returns a new
|
||||
incident only on the threshold crossing. The incident contains a stable event id, credential id,
|
||||
the distinct affected targets, evidence count, window, and remaining cool-off.
|
||||
|
||||
This class owns both correlation and short cool-off. Keeping them together makes threshold crossing
|
||||
and the cool-off deadline one atomic state change.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/placement/BackendOutagePolicy.java`.
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/placement/BackendOutagePolicyTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One error creates no incident and no cool-off.
|
||||
2. Two errors for one credential within 60 seconds create exactly one incident and a 60-second
|
||||
cool-off.
|
||||
3. Two errors more than 60 seconds apart do not create an incident.
|
||||
4. The exact 60-second boundary has a pinned result. Use inclusive `<= 60s` so scheduler delay does
|
||||
not discard evidence at the boundary.
|
||||
5. Different credentials never share evidence.
|
||||
6. Different profiles which supply the same credential id do share evidence. The policy itself only
|
||||
sees the credential id.
|
||||
7. More errors during active cool-off do not extend its deadline and do not return another incident.
|
||||
8. After expiry, old evidence is cleared. Two fresh errors are needed to create the next incident.
|
||||
9. Remaining seconds round up, matching `BackendQuarantine` reporting.
|
||||
10. Concurrent second and third errors cannot return two incidents.
|
||||
11. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 2 can run with Units 1, 3, and 4.
|
||||
|
||||
Unit 5 depends on the policy API.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The state transition table and locking method.
|
||||
- The exact threshold, window, cool-off, and boundary rule.
|
||||
- The test which proves one incident under concurrent calls.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 3 — Lead outage nudge
|
||||
|
||||
### Scope
|
||||
|
||||
Add backend incidents as a fourth pending source in `ReplyPushLoop`. Do not create another scheduler
|
||||
or call `AgentControl.send` from `Fleetd`. The existing combined per-lead schedule is the control
|
||||
which prevents competing injected turns.
|
||||
|
||||
The entry point takes an incident id, affected worker targets, credential id, affected profile
|
||||
names, and remaining cool-off. It resolves distinct owning leads through `PrimaryRegistry`.
|
||||
|
||||
Each `(incidentId, lead)` item is one-shot. It waits while the lead is not injectable. After one
|
||||
successful `agents.send`, remove it. A send exception keeps it pending for a bounded retry. It never
|
||||
uses the repeated reminder behaviour of an uncollected ticket.
|
||||
|
||||
Also add a fail-loud entry point for a classified target that Unit 5 cannot map to a credential. It
|
||||
uses `PrimaryRegistry.nudgeTargetFor(target)` and says that correlation could not run. If no lead is
|
||||
known, log at `WARN`, not `DEBUG`.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/msg/ReplyPushLoop.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/msg/ReplyPushLoopTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One incident affecting two workers owned by one lead causes one successful pane injection.
|
||||
2. Two affected workers owned by two leads cause one successful injection per affected lead. This
|
||||
is one notice per event per lead, not one notice per member.
|
||||
3. Repeating the same incident id is idempotent.
|
||||
4. A busy or unknown lead is not injected. The item stays pending until the lead becomes injectable
|
||||
or its attempt cap is reached.
|
||||
5. After one successful injection, later ticks do not mention that incident again.
|
||||
6. A failed `agents.send` is retried within the existing bound. A successful retry still gives only
|
||||
one successful send.
|
||||
7. A pending ticket and an outage incident for one lead appear in one combined nudge, not two
|
||||
competing turns.
|
||||
8. The text names the credential, profiles, affected workers, and remaining cool-off. It tells the
|
||||
lead to run `fleet_list`.
|
||||
9. An unmapped target produces a direct warning notice when a lead is known. If no lead is known,
|
||||
the code logs a `WARN` naming the target and reason.
|
||||
10. `stop()` clears incident state as it clears other push state.
|
||||
11. Existing reply, ticket, and question tests stay green. The focused tests and
|
||||
`mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. The API uses plain values, not the Unit 2 incident class. This lets Unit 3 run in parallel.
|
||||
|
||||
Unit 5 adapts the Unit 2 incident into this entry point.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact one-shot and retry rules.
|
||||
- The test showing one combined nudge with a failed ticket.
|
||||
- The test showing one successful send for two affected workers.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 4 — Durable member backend-error outcome
|
||||
|
||||
### Scope
|
||||
|
||||
Make a classified backend failure remain visible after its ticket is collected or expires.
|
||||
|
||||
Add `BACKEND_ERROR` to `MemberSession.State`. Add a nullable failure detail to `MemberSession` and
|
||||
render it as `failureReason` in `SessionManager.rosterView`. Add
|
||||
`SessionManager.onBackendError(target, reason)`.
|
||||
|
||||
The transition must handle both completion orderings:
|
||||
|
||||
- `BUSY -> BACKEND_ERROR` when classification wins before the normal completion state update;
|
||||
- `DONE -> BACKEND_ERROR` when the async resolver runs after `SessionManager.onTurnComplete`.
|
||||
|
||||
It must use a compare-and-set retry or another atomic update. `RELEASED` must never return to the
|
||||
roster. A backend-error member is terminal and cannot accept another delivery.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/MemberSession.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/SessionManager.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/session/SessionManagerTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `onBackendError` moves a `BUSY` member to `BACKEND_ERROR` and stores the reason.
|
||||
2. It also moves `DONE` to `BACKEND_ERROR`, covering the resolver race.
|
||||
3. A later normal `onTurnComplete` cannot change `BACKEND_ERROR` back to `DONE`.
|
||||
4. A released or unknown member is not recreated. The unknown case logs at `WARN` and returns an
|
||||
explicit false result to its caller.
|
||||
5. `onDelivered` refuses a `BACKEND_ERROR` member, just as it refuses generic `FAILED`.
|
||||
6. `rosterView` reports `state: backend_error` and `failureReason` after the send ticket is gone.
|
||||
7. Ordinary members do not gain a blank or invented `failureReason` field.
|
||||
8. Existing constructors keep source compatibility for tests and adapters.
|
||||
9. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 4 can run with Units 1, 2, and 3.
|
||||
|
||||
Unit 5 calls the new session method from the production sink.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The two race orderings and the tests for both.
|
||||
- The exact roster JSON shape.
|
||||
- The unknown-target result and log level.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 5 — Production wiring, spawn gate, and fleet views
|
||||
|
||||
### Scope
|
||||
|
||||
Compose Units 1 to 4 in production. This is the only unit which edits `Fleetd.java`.
|
||||
|
||||
Add per-profile `errorPattern` config beside `exhaustedPattern`. Compile both once at startup. A
|
||||
configured pattern wins over the legacy default. Report configured profiles and legacy-default
|
||||
profiles separately at startup. A bad regex must stop startup with the profile and key in the
|
||||
message.
|
||||
|
||||
Wire one production `BackendErrorSink` with this order:
|
||||
|
||||
1. mark the member `backend_error` with its reason;
|
||||
2. resolve the profile and its current `effectiveCredentialId()` through the fail-loud #234 seam;
|
||||
3. record the error in `BackendOutagePolicy`;
|
||||
4. on a new incident, submit one event to `ReplyPushLoop`.
|
||||
|
||||
If target metadata cannot be resolved, do not end in `Optional.ifPresent`. Log an error and call the
|
||||
Unit 3 unmapped-target notice. The failed send still reaches its ticket through CB-588.
|
||||
|
||||
Teach both spawn paths about a separate cool-off source. Exhaustion quarantine has priority when
|
||||
both states are active. Automatic placement needs a distinct `coolingOff` set so its refusal does
|
||||
not say “exhausted”.
|
||||
|
||||
Extend the MCP (Model Context Protocol) views:
|
||||
|
||||
- A cooling profile has `free: 0`, `credentialId`, and `coolingOffForSeconds` in `fleet_list`.
|
||||
- It does not have `quarantinedForSeconds` unless exhaustion quarantine is also active.
|
||||
- `fleet_profiles` has a separate `coolingOff` map, not an entry in `quarantined`.
|
||||
- A direct spawn refusal says the credential is cooling off after repeated backend errors and gives
|
||||
the remaining seconds.
|
||||
|
||||
Do not change `FleetHealthMonitor.coverage`. It still describes the periodic health webhook path.
|
||||
Lead-pane outage delivery is a separate capability.
|
||||
|
||||
### Files owned
|
||||
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/Fleetd.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/ConfigRef.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/member/CompositePeerLauncher.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementContext.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementPolicyUtil.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/FleetConfigTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/ConfigRefTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/member/CompositePeerLauncherTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/placement/PlacementPolicyTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/mcp/FleetMcpTest.java`
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/BackendOutageFlowTest.java`.
|
||||
- `fleetd/fleetd.example.yaml`
|
||||
- `CLAUDE.md`
|
||||
|
||||
No earlier unit edits these files.
|
||||
|
||||
The lead, not a worker, must update `wiki/7-Use-Cases.md`, `wiki/9-Implementation.md`, and
|
||||
`wiki/11-Features.md`. Project rules forbid workers from committing `wiki/`. The portable block in
|
||||
`CLAUDE.md` and `wiki/7-Use-Cases.md` must remain byte-identical.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `errorPattern` binds per profile. Blank uses the legacy default and is reported as degraded
|
||||
coverage. A config reload which changes it is reported as deferred because patterns are compiled
|
||||
at startup.
|
||||
2. A malformed `errorPattern` stops startup and names `profiles.<name>.errorPattern`.
|
||||
3. The real production sink never silently drops an unknown target. A test captures its error log
|
||||
and the fallback notice call.
|
||||
4. One real backend-error classification through `CompletionResolver` marks only its member. It does
|
||||
not cool the credential and does not send an outage notice.
|
||||
5. Two real classifications for one credential within 60 seconds start one incident.
|
||||
6. The integration test then calls the real explicit-profile spawn gate. It is refused before any
|
||||
adapter spawn call, with “cooling off” and remaining seconds in the message.
|
||||
7. The same test calls an automatic placement path. A cooling candidate is skipped. If every
|
||||
candidate is cooling, the error names cool-off rather than exhaustion.
|
||||
8. Profiles sharing the credential are all blocked. A profile on another credential stays usable.
|
||||
9. `fleet_list` from the same fixture shows both members as `backend_error`, preserves each failure
|
||||
reason, and reports `free: 0`, the credential, and `coolingOffForSeconds`.
|
||||
10. `fleet_profiles` reports cool-off separately from quarantine.
|
||||
11. The real `ReplyPushLoop` receives one incident and makes one successful lead-pane send. Existing
|
||||
failed-ticket notice content may share that same combined send.
|
||||
12. Exhaustion still wins when a line matches both patterns. A simultaneous exhaustion quarantine
|
||||
also wins in spawn errors and fleet views.
|
||||
13. After the 60-second cool-off, spawn is allowed again. A new incident needs two fresh errors.
|
||||
14. `healthCoverage` has the same value before and after this change for the same health config.
|
||||
15. `fleetd.example.yaml` explains `errorPattern`, the legacy fallback, 2/60/60 policy, and the
|
||||
difference between cool-off and exhaustion quarantine.
|
||||
16. `CLAUDE.md` tells leads how `fleet_profiles` and `fleet_list` report cool-off. The lead later
|
||||
applies the matching wiki updates and runs the documented byte-sync check.
|
||||
17. The developer records the new end-to-end test failing before implementation, then passing. The
|
||||
focused suites and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
Unit 5 starts only after Units 1 to 4 are merged or rebased into its branch. It also starts after the
|
||||
#234 defect 2 fix lands, because both areas touch the same target-resolution control path.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact commits used for Units 1 to 4 and #234.
|
||||
- The startup coverage line with one configured and one legacy-default profile.
|
||||
- The red test output before the implementation and its green result after.
|
||||
- The explicit and automatic spawn refusal text.
|
||||
- Sample `fleet_list` and `fleet_profiles` JSON for cool-off and exhaustion.
|
||||
- The number and text of lead-pane sends in the real-path test.
|
||||
- The focused test commands and final `mvn clean install` result.
|
||||
- The exact `CLAUDE.md` change and the wiki edits the lead must apply.
|
||||
|
||||
## File ownership summary
|
||||
|
||||
| Area | Unit | Shared edit risk |
|
||||
|---|---:|---|
|
||||
| Completion classification | 1 | Only Unit 1 edits `CompletionResolver` and its test. |
|
||||
| Correlation and cool-off state | 2 | New files only. |
|
||||
| Lead push scheduling | 3 | Only Unit 3 edits `ReplyPushLoop` and its test. |
|
||||
| Member terminal state | 4 | Only Unit 4 edits `MemberSession`, `SessionManager`, and their test. |
|
||||
| Main composition, config, placement, MCP views, shipped prompt | 5 | Only Unit 5 edits `Fleetd`, `FleetConfig`, `CompositePeerLauncher`, placement context, `FleetMcp`, and `CLAUDE.md`. |
|
||||
| Wiki propagation | Lead after Unit 5 | Workers do not commit the wiki submodule. |
|
||||
|
||||
## What I would not build
|
||||
|
||||
1. **Do not reuse `BackendQuarantine` for outages.** Its repeat call restarts a long credential
|
||||
quarantine. Its fields and errors say “exhausted”. That is wrong for a short outage.
|
||||
2. **Do not merge the exhaustion and generic error patterns.** Exhaustion must win because it has a
|
||||
different policy and duration.
|
||||
3. **Do not group by error string.** One outage can produce different text. The shared operational
|
||||
limit is the credential.
|
||||
4. **Do not mark a profile unusable until config changes.** The current classifier cannot safely
|
||||
tell a permanent malformed request from a transient service fault. A permanent state would need
|
||||
a stronger error taxonomy first.
|
||||
5. **Do not quarantine on the first generic backend error.** That would turn one bad request or one
|
||||
false pattern match into a fleet-wide capacity loss.
|
||||
6. **Do not add this to `FleetHealthMonitor`.** The monitor samples slow member health. The exact
|
||||
backend event already exists at completion resolution, and moving it to polling would lose type
|
||||
and time.
|
||||
7. **Do not add another direct lead injector.** `ReplyPushLoop` already owns status gating,
|
||||
per-lead coalescing, retry bounds, and heartbeat stand-down.
|
||||
8. **Do not change `healthCoverage` to `full`.** That field still means a webhook notification sink
|
||||
exists for periodic health. A backend outage nudge does not make every health event visible.
|
||||
9. **Do not persist incident history across daemon restart in this work.** Existing exhaustion
|
||||
quarantine is also in memory. A 60-second state does not justify a new durable store.
|
||||
10. **Do not build work recovery.** The PR-body survival story proves why checkpoint-first work is
|
||||
useful, but these tickets are about detection, capacity, and signalling.
|
||||
11. **Do not remove the legacy `API Error:` fallback in the first release.** Doing so would turn an
|
||||
unedited config back into a false successful completion. Report it as degraded coverage instead.
|
||||
12. **Do not reorder or add the old `visibleTurn` fallback from #201.** #211 already implemented the
|
||||
narrow raw-scrape fallback at `CompletionResolver.classifyRawScrapeFallback`.
|
||||
|
||||
## Riskiest assumption and cheapest experiment
|
||||
|
||||
The riskiest assumption is that a configured error regex means “the backend failed this turn”. The
|
||||
current code and test already show the counterexample: a worker may quote `API Error:` while writing
|
||||
a valid report. Two such false matches on one credential would now remove capacity for 60 seconds.
|
||||
|
||||
The cheapest experiment is a replay corpus before Unit 5 ships:
|
||||
|
||||
1. Save the full pane text from the measured 2026-09-01 outage.
|
||||
2. Produce one safe failure per backend with a disposable invalid endpoint or request.
|
||||
3. Save one valid member report which quotes each error line.
|
||||
4. Replay all samples through the real `CompletionResolver` test fixture.
|
||||
5. Require outage samples to match and quoted-report samples not to match after assistant-block
|
||||
extraction and baseline checks.
|
||||
|
||||
This costs no outage deployment and no real sleep. If quoted reports still match, narrow the profile
|
||||
patterns before enabling correlation. Do not raise the threshold to hide a bad classifier.
|
||||
|
||||
## Sequencing with three developers
|
||||
|
||||
First wave:
|
||||
|
||||
1. Developer A: Unit 1, typed classification.
|
||||
2. Developer B: Unit 2, credential outage policy.
|
||||
3. Developer C: Unit 3, lead outage nudge.
|
||||
|
||||
As soon as one slot is free, start Unit 4. It is file-disjoint from every first-wave unit. Merge and
|
||||
review Units 1 to 4 independently.
|
||||
|
||||
Start Unit 5 only after all four foundations and #234 are available. Unit 5 is the only high-conflict
|
||||
integration branch, so no other active unit should touch its file list.
|
||||
|
||||
## Checks performed for this refinement
|
||||
|
||||
- Read issue #201 and issue #227 through their Gitea pages. Both showed zero comments.
|
||||
- Read the source and tests named in the evidence section.
|
||||
- Ran `git status --short --branch`; the branch was clean before this document was added.
|
||||
- Ran `git log --oneline -12` to identify the branch base.
|
||||
- I did not run Maven because this change adds only a design document.
|
||||
- Rendered both Mermaid blocks with `npx @mermaid-js/mermaid-cli`; both commands succeeded.
|
||||
@@ -79,7 +79,7 @@ Unit tests (add to the existing `ClaudeCodeLauncher` test):
|
||||
|
||||
## 5. Config
|
||||
|
||||
Add to the launcher-level config (a bridged-level knob, not per-profile) in `bridged.yaml` +
|
||||
Add to the launcher-level config (a fleetd-level knob, not per-profile) in `fleetd.yaml` +
|
||||
`FleetConfig`:
|
||||
|
||||
```yaml
|
||||
|
||||
@@ -140,7 +140,7 @@ Stage-2 AMQP adapter: absent → in-memory, present → AMQP).
|
||||
|
||||
## 5. Build & verification (worker side)
|
||||
|
||||
- Build with Maven from the worktree's `bridged/` dir. **Capture the exit code without a masking pipe**
|
||||
- Build with Maven from the worktree's `fleetd/` dir. **Capture the exit code without a masking pipe**
|
||||
(`mvn clean install; echo "MVN_EXIT=$?"` — never `mvn … | tail`, which hides failures).
|
||||
- Read the real test totals from `target/surefire-reports/TEST-*.xml`, not from stdout scroll.
|
||||
- You do **not** have IDE MCP access — do not claim `ide_diagnostics` results. The **primary** runs the
|
||||
|
||||
@@ -208,7 +208,7 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
never an argument* — across the broker: cross-host, identity comes from the key. Complements
|
||||
(not replaces) per-gateway broker logins over TLS.
|
||||
2. **Profiles are owned by the worker's host.** `fleet_spawn(profile, host)` resolves the name in
|
||||
the *target* gateway's `bridged.yaml`. Gateways advertise their profile names in presence
|
||||
the *target* gateway's `fleetd.yaml`. Gateways advertise their profile names in presence
|
||||
heartbeats, so a leader sees what each host offers before spawning; an unknown name is a clear
|
||||
error from the target. Secrets (base URLs, tokens) never leave the host that uses them.
|
||||
3. **Repo provisioning — clone from the forge, pinned.** A cross-host spawn names the repo URL and
|
||||
@@ -275,11 +275,11 @@ they overlap (notably: the envelope is no longer optional, and dedup is split by
|
||||
- **Gateway discovery:** how gateways find the broker and each other (static config vs. discovery).
|
||||
- **Control authorization — THE GATE ON U4.** Signing (§7.1) settles *who sent it*; authorization
|
||||
is *who may do what*. **Cross-host spawn must not land before the minimal version exists**: a
|
||||
per-host allowlist in `bridged.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
per-host allowlist in `fleetd.yaml` — beside the peer public keys — of gateway ids permitted to
|
||||
publish control to this host, checked against the verified signature. A few lines of config and
|
||||
check; without them, any principal holding broker credentials can start processes on every host
|
||||
in the fleet.
|
||||
- **Key distribution & rotation:** static config (host → public key in each `bridged.yaml`) is
|
||||
- **Key distribution & rotation:** static config (host → public key in each `fleetd.yaml`) is
|
||||
fine at the current 2–3 host scale; rotation is manual. A refinement, not a blocker.
|
||||
- **Gateway death mid-turn:** the roster reaps it by missed heartbeat, and in-flight primary-bound
|
||||
messages survive by broker durability; still open is reconciling *worker* state when the dead
|
||||
|
||||
@@ -11,7 +11,7 @@ All five increments of §4 are done, including increment 5 (the §5 live checkli
|
||||
|
||||
## 1. Goal
|
||||
|
||||
Prove the [`PeerLauncher`](../bridged/src/main/java/dev/ltms/fleet/peer/PeerLauncher.java) SPI
|
||||
Prove the [`PeerLauncher`](../fleetd/src/main/java/dev/ltms/fleet/peer/PeerLauncher.java) SPI
|
||||
actually holds for a **non-Claude** coding agent by shipping a second, first-class in-tree
|
||||
adapter: **opencode** (`opencode` 1.1.31, a provider-agnostic terminal coding agent).
|
||||
|
||||
|
||||
@@ -9,7 +9,7 @@ with no terminator, and our third-party members cannot detect it (§7.2).
|
||||
· **Upstream issue:** [systems/vms#31](https://git.ltms.dev/systems/vms/issues/31)
|
||||
|
||||
The gateway went live on 2026-08-15 and replaced Bifrost. This plan says what that means for a
|
||||
**member definition** in `bridged.yaml`, because that is the part of this repo the change actually
|
||||
**member definition** in `fleetd.yaml`, because that is the part of this repo the change actually
|
||||
touches.
|
||||
|
||||
---
|
||||
@@ -134,16 +134,16 @@ own provider credentials. So 3b needs **no allowlist change**; only 3a does.
|
||||
`HerdrPeerLauncher.resolveEnv` → `env.apply(name)`, which reads the **daemon's own process
|
||||
environment**. The running `fleetd` inherited its environment when it started, so a variable added to
|
||||
`secrets.sh` afterwards is simply not there — the launcher would inject an empty token and the
|
||||
gateway would answer 401. This is the same failure as trap 1 in `scripts/redeploy-bridged.sh`
|
||||
gateway would answer 401. This is the same failure as trap 1 in `scripts/redeploy-fleetd.sh`
|
||||
(`WORKER_GITEA_TOKEN`), and it has the same fix: **restart from a login shell**, and use
|
||||
`scripts/redeploy-bridged.sh --check` to confirm the name resolves before restarting anything.
|
||||
`scripts/redeploy-fleetd.sh --check` to confirm the name resolves before restarting anything.
|
||||
|
||||
### 3c. What this does to `ccs`
|
||||
|
||||
Once a profile carries `baseUrl`, `tokenEnv` and `model` itself, the ccs instance stops being what
|
||||
routes a member. Be precise about what is left, though: `configDir` still supplies **folder trust**
|
||||
and `settings.json`, and dropping it is what produced the trust dialog and the wrong-model error
|
||||
recorded in `bridged.yaml`. So ccs goes from *deciding where the tokens go* to *holding client-side
|
||||
recorded in `fleetd.yaml`. So ccs goes from *deciding where the tokens go* to *holding client-side
|
||||
state*. Less load-bearing, not removable.
|
||||
|
||||
### Why `/anthropic` and never `/v1/chat/completions`
|
||||
@@ -217,7 +217,7 @@ defensible; picking one is not mine to do.
|
||||
### D3 — token scope
|
||||
|
||||
One consumer, `claude-bridge`, its token in `${SHARED_ENV}/tools/secrets.sh` as `AI_GATEWAY_TOKEN`,
|
||||
referenced by name only. Never the literal value in `bridged.yaml` — `tokenEnv` exists for this.
|
||||
referenced by name only. Never the literal value in `fleetd.yaml` — `tokenEnv` exists for this.
|
||||
|
||||
---
|
||||
|
||||
@@ -226,7 +226,7 @@ referenced by name only. Never the literal value in `bridged.yaml` — `tokenEnv
|
||||
```mermaid
|
||||
flowchart TB
|
||||
U1["U1 · consumer token<br/>issue via cockpit, add to secrets.sh"]
|
||||
U2["U2 · profile + guard<br/>bridged.yaml, restart"]
|
||||
U2["U2 · profile + guard<br/>fleetd.yaml, restart"]
|
||||
U3["U3 · verify live<br/>spawn, prove thinking survives"]
|
||||
U4["U4 · context7 via gateway<br/>.mcp.json + opencode.json"]
|
||||
U5["U5 · docs<br/>CLAUDE.md, wiki 11-Features"]
|
||||
@@ -238,7 +238,7 @@ flowchart TB
|
||||
| # | Scope | Who | Why |
|
||||
|---|---|---|---|
|
||||
| U1 | Issue the `claude-bridge` consumer at `auth.ltms.dev`; store as `AI_GATEWAY_TOKEN` | **operator** | touches secrets and a host we do not own |
|
||||
| U2a | New `gx` opencode profile on `/v1` — pure config, no guard change | **lead** | `bridged.yaml` is gitignored, so a worker cannot see or edit it |
|
||||
| U2a | New `gx` opencode profile on `/v1` — pure config, no guard change | **lead** | `fleetd.yaml` is gitignored, so a worker cannot see or edit it |
|
||||
| U2b | `local` → `/anthropic`; add `local-direct` weight 0; add `llm.ltms.dev` to the guard allowlist | **lead** | same |
|
||||
| U2c | One restart from a **login shell**, after U2a and U2b | **lead** | picks up `AI_GATEWAY_TOKEN` into the daemon env *and* the guard allowlist, in one stop |
|
||||
| U3 | Live spawn on both new profiles; confirm reasoning survives on each surface | **lead** | needs real spawns and the running daemon |
|
||||
@@ -372,7 +372,7 @@ one turn; this one costs the whole task and is indistinguishable from a slow wor
|
||||
|
||||
> **Diagnosing a stuck opencode member.** Do not read its pane. The launcher writes its config to a
|
||||
> temp dir and passes it as `OPENCODE_CONFIG` — find it with
|
||||
> `ls -dt /var/folders/*/*/T/bridged-opencode-* | head -1`, check the provider block and the key's
|
||||
> `ls -dt /var/folders/*/*/T/fleetd-opencode-* | head -1`, check the provider block and the key's
|
||||
> length and prefix (never its value), then reproduce with `opencode run --auto -m <provider>/<model>`
|
||||
> using the same `OPENCODE_CONFIG`. That is what turned "it hangs" into a one-line error.
|
||||
|
||||
|
||||
@@ -67,7 +67,7 @@ it will be disabled and the stage is wasted. So:
|
||||
auth:
|
||||
mode: loopback-trust # default — behaves exactly like today: loopback ⇒ PRIMARY, no token needed
|
||||
# mode: token # every non-worker caller must present a valid bearer token
|
||||
# tokenEnv: BRIDGED_API_TOKEN # host env var holding the token; never the literal value
|
||||
# tokenEnv: FLEETD_API_TOKEN # host env var holding the token; never the literal value
|
||||
```
|
||||
|
||||
`mode: loopback-trust` is the current behaviour, named honestly and now *chosen* rather than
|
||||
@@ -114,7 +114,7 @@ not found), and the daemon that has been dogfooded for weeks runs as a bare fore
|
||||
|
||||
- `deploy/dev.ltms.fleet.plist` — launchd agent, the *actual* runtime here, with `KeepAlive` and
|
||||
ordered start after herdr.
|
||||
- `deploy/bridged.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
- `deploy/fleetd.service` — systemd unit for the Linux gateways CB-308 introduces.
|
||||
|
||||
Ordering after herdr is advisory in both: the herdr socket may not exist at boot, so the daemon
|
||||
must **retry the socket rather than exit** — supervision ordering is a nicety, socket-retry is the
|
||||
@@ -192,7 +192,7 @@ Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ sent\|exhausted — a rising `exhausted` means the primary is not draining. `sent` was called `delivered` until fleetd #365; it counts the herdr paste-and-submit call returning, never a confirmation the pane read it |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
|
||||
@@ -807,7 +807,7 @@ Checked against `main` at `e09cac6` on 2026-08-15. Unit 2 was written as one blo
|
||||
have since been built by separate CB tickets. Read this before planning the rest, or that work gets
|
||||
done twice.
|
||||
|
||||
The check was a symbol survey of `bridged/src/main/java` plus the merge history. It tells you whether
|
||||
The check was a symbol survey of `fleetd/src/main/java` plus the merge history. It tells you whether
|
||||
the machinery exists at all. It is **not** a line-by-line audit of whether each criterion is fully
|
||||
met, and I did not run one.
|
||||
|
||||
|
||||
+123
-324
@@ -1,311 +1,162 @@
|
||||
# MCP Contract — `fleetd`'s unified gateway
|
||||
# MCP flows and error model — `fleetd`
|
||||
|
||||
> **Status: 🔴 HISTORICAL DESIGN — do NOT use as the tool reference.** Written 2026-07-14, before
|
||||
> any MCP code existed. The system shipped and this page never caught up, so **its tool names,
|
||||
> parameter names and REST paths are wrong today**. Audited 2026-08-17; the specific drift:
|
||||
> **What this page is.** The **flows**: how a delegation, a clarification, a detached task and a
|
||||
> silent member each travel through `fleetd`. These shapes are what shipped, and they are hard to
|
||||
> read off the code because they span the MCP face, the rendezvous registry, the `Injector` and
|
||||
> herdr.
|
||||
>
|
||||
> - **Tools it names that do not exist:** `bridge_read`, `bridge_cancel`.
|
||||
> - **Shipped tools it omits:** `bridge_poll`, `bridge_ack`, `bridge_profiles`, `bridge_whoami`.
|
||||
> - **Parameter names are wrong nearly everywhere** — it says `message`/`target`/`timeout_seconds`/
|
||||
> `block` where the code takes `content`/`sessionId`/`timeoutMs`/`wait`; `text` where
|
||||
> `bridge_reply` takes `content`; `target` where `bridge_stop` takes `paneId`.
|
||||
> - **REST paths are wrong:** it says `POST /workers` and `DELETE /workers/{paneId}`; the daemon
|
||||
> serves `POST /members` and `DELETE /members/{paneId}`.
|
||||
> **What this page is NOT: a tool reference.** It deliberately holds no tool catalogue, no
|
||||
> parameter tables and no REST paths. **The live MCP schema is the authority** — each tool's own
|
||||
> description and parameters, as mounted — with the intent→tool table in `CLAUDE.md` as the short
|
||||
> form.
|
||||
>
|
||||
> **The authoritative tool surface is the live MCP schema** (each tool's own description and
|
||||
> parameters, as mounted), with the intent→tool table in `CLAUDE.md` as the short form. Both were
|
||||
> checked against `mcp/FleetMcp.java` on 2026-08-17 and are accurate.
|
||||
> That absence is the fix for fleetd #114 (CB-609), and it is worth stating why. This page used to
|
||||
> carry a full tool catalogue written in July 2026, before any MCP code existed. The code shipped;
|
||||
> the page did not follow. By August it named two tools that do not exist, omitted five that do,
|
||||
> had the wrong name for nearly every parameter, pointed at REST paths the daemon does not serve,
|
||||
> and — worst — still described an identity model (*"any connection that does not map to a known
|
||||
> worker is treated as a primary"*) that was a real privilege bug, fixed since by the ancestry
|
||||
> walk in fleetd #161. Every one of those errors is the same error: **a second, hand-maintained
|
||||
> copy of something the code already states**. So the second copy is gone rather than corrected.
|
||||
> Only the flows remain, because a flow is a shape rather than a name, and shapes are what this
|
||||
> page was ever good for.
|
||||
>
|
||||
> What is still worth reading here is **§6 — the flows and the error model** (rendezvous,
|
||||
> `fleet_ask`, detached delivery, the turn-done fallback). The shapes it describes are the ones
|
||||
> that shipped; only the names around them drifted. Rewriting this page is tracked as **CB-609**.
|
||||
|
||||
`fleetd` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
> The names that do appear below are checked by `McpContractDocTest`, which fails if this page
|
||||
> names a `fleet_*` tool the server does not register. That test is the whole reason it is safe to
|
||||
> write a tool name here at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
## 1. Rendezvous flows
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
### 1.1 Delegation — happy path
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `fleetd` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `fleetd` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`fleetd` owns policy; herdr owns PTYs.** MCP tools express *intent*; `fleetd`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["fleetd — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>bridge_send · bridge_reply<br/>bridge_ask · bridge_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"bridge_send (blocks)"| MCP
|
||||
W -.->|"bridge_reply / bridge_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `fleetd` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `fleetd` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `bridge_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `bridge_reply` / `bridge_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`fleetd` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http fleetd http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`bridge_send`](#bridge_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`bridge_reply`](#bridge_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`bridge_ask`](#bridge_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`bridge_status`](#bridge_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`bridge_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`bridge_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`bridge_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`bridge_read`](#bridge_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`bridge_cancel`](#bridge_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `bridge_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `bridge_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `bridge_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `bridge_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `bridge_status` on a split-host primary.
|
||||
|
||||
#### `bridge_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `fleetd` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `bridge_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `bridge_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `bridge_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`bridge_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`bridge_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`bridge_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `bridge_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `bridge_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `bridge_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
One blocking call, zero polls. The lead's call is held open by `fleetd` until the member answers.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as fleetd (MCP + Injector)
|
||||
participant P as "Lead (primary)"
|
||||
participant B as "fleetd (MCP + Injector)"
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: fleet_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B->>B: "register waiter(sessionId)"
|
||||
B->>H: "agent.send — only in an injectable window"
|
||||
H-->>W: "prompt injected"
|
||||
W->>W: "works the turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B->>B: "resolve waiter"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`fleet_ask`)
|
||||
**The cap that matters:** a blocking `fleet_send` is bounded by the *caller's own* MCP client
|
||||
timeout, about 60 seconds — not by the task. Anything slower than that must use the detached flow
|
||||
in §1.3, or the lead's call returns while the member is still working.
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
### 1.2 Clarification — reverse rendezvous
|
||||
|
||||
The member pauses mid-turn to ask, the lead answers, and the member resumes **the same turn** with
|
||||
its context intact.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: fleet_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: fleet_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve fleet_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: fleet_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B-->>W: "content injected"
|
||||
W->>B: "fleet_ask{question} — member blocks"
|
||||
B-->>P: "{ outcome: question, turnId }"
|
||||
P->>B: "fleet_send{turnId, content} — answers THIS turn"
|
||||
B-->>W: "fleet_ask returns the answer"
|
||||
W->>W: "resumes the same turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
**Answer with `turnId`, never `sessionId`.** A `sessionId` send starts a new turn; it does not
|
||||
resolve the waiting `fleet_ask`.
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
**The window is about 55 seconds and no nudge extends it.** So never brief a member to "ask me":
|
||||
decide before delegating, or give the member an explicit default to fall back on.
|
||||
|
||||
### 1.3 Detached delegation — the lead does not block
|
||||
|
||||
The lead gets a ticket immediately and collects the answer later. This is the flow for any real
|
||||
task, because of the ~60s cap in §1.1.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: fleet_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
P->>B: "fleet_send{sessionId, content, wait:false}"
|
||||
B-->>P: "accepted — ticket"
|
||||
P->>P: "continues its own work"
|
||||
W->>B: "fleet_reply{content}"
|
||||
Note over B: "no waiter is blocked — the reply is held"
|
||||
B->>B: "nudge the lead's own pane (status-gated)"
|
||||
P->>B: "fleet_poll{ticket}"
|
||||
B-->>P: "the member's report"
|
||||
P->>B: "fleet_ack{target, msgId}"
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
A terminal ticket nudges the lead's pane by itself, so a detached task does not need watching. The
|
||||
nudge needs an injectable lead pane and is capped, so it is a convenience rather than a guarantee.
|
||||
|
||||
A worker that never calls `fleet_reply` still returns a result: `fleetd` reads its terminal
|
||||
tail when the turn completes.
|
||||
### 1.4 The member never replies — turn-done fallback
|
||||
|
||||
A member that ends its turn without `fleet_reply` still produces something: `fleetd` reads its
|
||||
pane tail. This is a **fallback, not a channel** — it is lossy in three separate ways, and every
|
||||
one of them has produced a wrong answer in practice.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls fleet_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
P->>B: "fleet_send — blocks or detaches"
|
||||
B-->>W: "content injected"
|
||||
W->>W: "works, never calls fleet_reply"
|
||||
B->>B: "StatusPoller sees the turn end"
|
||||
B->>B: "read the pane tail"
|
||||
B->>B: "classify: exhausted? echoed brief? real report?"
|
||||
B-->>P: "{ outcome: turn_done } or a named failure"
|
||||
```
|
||||
|
||||
The three ways it goes wrong, and what each looks like now:
|
||||
|
||||
| What happened | What the lead used to get | What it gets today |
|
||||
|---|---|---|
|
||||
| The report is longer than the scrape window | The **end** silently cut off | Still clipped, but marked partial |
|
||||
| The member never started — spent credential | The lead's **own brief** echoed back as a report | A named failure: backend exhausted |
|
||||
| The member is simply slow | A tail of work in progress | Unchanged — read it as a hint, not a result |
|
||||
|
||||
The echoed-brief case is the one to remember: it reads as a long, on-topic report with nothing in
|
||||
it from the member. It is suppressed now, but the general rule stands — **check the member's
|
||||
worktree with `git log` before believing a report you did not watch arrive.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
## 2. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `bridge_send` is simply its producer.
|
||||
Delivery only happens in a safe window. `fleet_send` is a producer for the `Injector`, which
|
||||
already enforces this through `AgentStatus.injectable()`.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
IDLE --> WORKING: "message delivered, picked up"
|
||||
WORKING --> IDLE: "turn done"
|
||||
WORKING --> BLOCKED: "awaits input"
|
||||
BLOCKED --> WORKING: "input delivered"
|
||||
IDLE --> UNKNOWN: "detection glitch"
|
||||
BLOCKED --> UNKNOWN: "detection glitch"
|
||||
UNKNOWN --> IDLE: "re-detected"
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
@@ -321,69 +172,17 @@ stateDiagram-v2
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
**At most one message per turn.** After a send, the `Injector` waits for a `WORKING` pickup before
|
||||
delivering the next, with a grace-poll fallback for turns that finish faster than the poll
|
||||
interval.
|
||||
|
||||
---
|
||||
Two consequences a lead feels directly:
|
||||
|
||||
## 8. Error model
|
||||
- **A second send to a busy member never lands.** It reports as queued and times out. The member
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
|
||||
| Condition | `bridge_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `bridge_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`bridge_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `bridge_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `bridge_reply` / `bridge_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `bridge_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `bridge_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `bridge_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `FleetApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`bridge_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `bridge_send` (keeps the catalog
|
||||
small) vs. a separate `bridge_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `bridge_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `bridge_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `bridge_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `bridge_reply` / `bridge_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`bridge_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `bridge_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
unreadable pane as a ready one.
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
# Wiki audit for #168
|
||||
|
||||
**Source checked:** `.wiki-snapshot/` at `68e32c6` (2026-08-31). I did not use
|
||||
`wiki/`. Code references below are from the current `fleetd` source tree. A quoted
|
||||
line is a concrete claim that needs correction, unless the table says `KEEP`.
|
||||
|
||||
| Page | Verdict | One-line reason |
|
||||
|---|---|---|
|
||||
| `Home.md` | REVISE | Good overview, but it still names the retired product. |
|
||||
| `_Sidebar.md` | REVISE | The heading still says `claude-bridge`. |
|
||||
| `1-Architecture.md` | REBUILD | Its component contract mixes current names with removed tools, routes, and planned backends. |
|
||||
| `2-Message-Server.md` | REBUILD | The claimed MCP schema, mount command, REST/SSE surface, and fallback paths are pre-build design. |
|
||||
| `3-Approaches.md` | REVISE | Useful research history, but it presents unbuilt AgentAPI as a selectable fallback. |
|
||||
| `4-Setup.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `5-Operations.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `6-Team.md` | REBUILD | It teaches role-addressed sends and a Claude-only team model that the shipped API does not have. |
|
||||
| `7-Use-Cases.md` | REBUILD | Its flagship flow depends on removed `ccs` profiles and removed send parameters. |
|
||||
| `8-Roadmap.md` | REBUILD | It is a historical plan, but it presents old implementation choices and planned work as the current stack. |
|
||||
| `9-Implementation.md` | REBUILD | Its package, class, endpoint, and outcome map has drifted from the source. |
|
||||
| `10-Cross-Host-Messaging.md` | REVISE | It labels most federation work proposed, but misses the shipped `coordinator:` lead channel. |
|
||||
| `11-Features.md` | REVISE | It is the right catalogue, but code-path names are old and it misses the second-herdr-daemon capability. |
|
||||
| `12-Claude-to-OpenCode.md` | REVISE | The porting guide is mostly current, but calls the product and spawned-member path a bridge. |
|
||||
| `13-User-Guide.md` | REVISE | It is the best operator page, but needs the product rename and the second-herdr-daemon setup. |
|
||||
|
||||
## Pages needing work
|
||||
|
||||
### `Home.md` — REVISE
|
||||
|
||||
- Quote: `# claude-bridge` (line 1) and `` `claude-bridge` keeps`` (line 11).
|
||||
The product is `fleet` / `fleetd`. The MCP server identifies itself as `fleet` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:313-315`.
|
||||
- Quote: `AgentAPI ... swappable fallback injector` (lines 73-76).
|
||||
There is no AgentAPI implementation under `fleetd/src/main/java`; the actual
|
||||
launchers are selected by `Profile.kind` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java:265-270`.
|
||||
|
||||
### `_Sidebar.md` — REVISE
|
||||
|
||||
- Quote: `### 📖 claude-bridge` (line 1).
|
||||
Rename it to `fleet`. `FleetMcp` registers the current product-facing tool set at
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:301-326`.
|
||||
|
||||
### `1-Architecture.md` — REBUILD
|
||||
|
||||
- Quote: `` `claude-bridge` lets`` (line 3). The product was renamed; the MCP
|
||||
server name is `fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: ``fleet_read`` in the tool list (line 102). No such tool is registered.
|
||||
The complete registered list is `fleet_send` through `fleet_whoami` at
|
||||
`FleetMcp.java:301-326`; `fleet_read` is absent.
|
||||
- Quote: `SSE (GET /events)` (line 143). `FleetApp.build()` registers no `/events`
|
||||
route; its routes are listed at `FleetApp.java:143-159`.
|
||||
- Quote: `Redis Streams / NATS JetStream, or an embedded queue` (line 106).
|
||||
The shipped durable inbox is AMQP, configured by `broker`, at
|
||||
`FleetConfig.java:49-50` and `FleetConfig.java:655-714`.
|
||||
- Quote: `AgentAPI (fallback)` (line 107). No AgentAPI adapter exists; shipped
|
||||
launcher kinds are `claude-code` and `opencode` (`FleetConfig.java:265-270`).
|
||||
|
||||
### `2-Message-Server.md` — REBUILD
|
||||
|
||||
- Quote: `claude mcp add --transport http bridge http://127.0.0.1:8080/mcp`
|
||||
(line 67). The daemon defaults to port `8765` in `FleetConfig.java:183-187`,
|
||||
and identifies its server as `fleet` at `FleetMcp.java:313-315`.
|
||||
- Quote: ``fleet_send(message, target?, {block, timeout_seconds, auto_spawn,
|
||||
turn_id})`` (line 80). The real parameters are `sessionId`, `content`,
|
||||
`timeoutMs`, `wait`, `turnId`, and `coordId` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: ``fleet_read(target, source)`` (line 85). It is not registered; see the
|
||||
complete registration at `FleetMcp.java:301-326`.
|
||||
- Quote: `docs/MCP-Contract.md ... normative` (lines 87-88). That is not a valid
|
||||
reference: only §6 is current, as the current operator guide itself says at
|
||||
`.wiki-snapshot/13-User-Guide.md:466`.
|
||||
- Quote: `SSE (GET /events)` (line 45). No route exists in the built REST surface,
|
||||
`FleetApp.java:143-159`.
|
||||
|
||||
### `3-Approaches.md` — REVISE
|
||||
|
||||
- Quote: `AgentAPI ... remains a swappable fallback injector` (lines 78-84).
|
||||
It was never built. The shipped adapter selection is only `claude-code` or
|
||||
`opencode` (`FleetConfig.java:265-270`). Keep it as discarded research, not an
|
||||
operational fallback.
|
||||
- Quote: `claude-bridge` (line 109). Rename the product to `fleet`; the runtime
|
||||
package is `dev.ltms.fleet`, for example `FleetMcp.java:1`.
|
||||
|
||||
### `4-Setup.md` — RETIRE
|
||||
|
||||
It is a 25-line redirect and says its procedure was never written (lines 3-9).
|
||||
Chapter 13 is the maintained install procedure. Keeping a second navigation page
|
||||
adds no working documentation.
|
||||
|
||||
### `5-Operations.md` — RETIRE
|
||||
|
||||
It is a 35-line redirect and says its runbook was never written (lines 3-14).
|
||||
Chapter 13 now owns run and recovery instructions.
|
||||
|
||||
### `6-Team.md` — REBUILD
|
||||
|
||||
- Quote: `fleet_send {role: w-claude, prompt: A}` (line 98). `fleet_send` accepts
|
||||
`sessionId` and `content`, not `role` or `prompt` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: `some on Claude, some on the remote local LLM` (lines 3-5) and `Every
|
||||
worker is ... Claude Code` (line 25). `opencode` is a first-class launcher kind,
|
||||
not a Claude worker (`FleetConfig.java:265-270`).
|
||||
- Quote: `fleetd's concurrency policy` (line 121). The configured capacity control
|
||||
is per-profile `maxLoad` (`FleetConfig.java:251-264`), not the role routing model
|
||||
described here.
|
||||
|
||||
### `7-Use-Cases.md` — REBUILD
|
||||
|
||||
- Quote: `ccs profile` (line 10), `ccs + herdr` (line 22), and `ccs-spawn`
|
||||
(line 45). The configuration has `profiles` and `fleet`, not `ccs`:
|
||||
`FleetConfig.java:34-58` and `FleetConfig.java:81-101`.
|
||||
- Quote: `fleet_send({"to", "kind", "body", "block"})` (lines 55-62).
|
||||
None of those are the shipped send parameters. The schema is
|
||||
`FleetMcp.java:1096-1108`.
|
||||
- Quote: `fleet_list() → { "profiles": ... }` (lines 74-80). `fleet_list` is a
|
||||
roster view; `fleet_profiles` is the configured-backend view, as registered at
|
||||
`FleetMcp.java:307-311` and described at `FleetMcp.java:1176-1182`.
|
||||
|
||||
### `8-Roadmap.md` — REBUILD
|
||||
|
||||
- Quote: `Java 21+` (line 43). The current project guidance and source use Java 25;
|
||||
the `FleetConfig` source itself uses Java 25 unnamed lambda parameters, for
|
||||
example `FleetConfig.java:102`.
|
||||
- Quote: `herdr 0.7.0 / protocol 14` (line 46). The current REST health endpoint
|
||||
reports the live protocol returned by herdr (`FleetApp.java:240-244`), while the
|
||||
current operator guide records protocol 19 at
|
||||
`.wiki-snapshot/13-User-Guide.md:76-85`.
|
||||
- Quote: `ccs <profile> claude` and `ccs env <profile>` (lines 47-48). Shipped
|
||||
configuration uses `Profile` records and launcher `kind`,
|
||||
`FleetConfig.java:313-330` and `FleetConfig.java:265-270`.
|
||||
- Quote: `Redis Streams via Lettuce` (line 50). The actual durable inbox is AMQP
|
||||
`broker`, `FleetConfig.java:655-714`.
|
||||
|
||||
### `9-Implementation.md` — REBUILD
|
||||
|
||||
- Quote: `rest.FleetdApp` and `mcp.BridgeMcp` (lines 29-30). The classes are
|
||||
`rest.FleetApp` and `mcp.FleetMcp` (`FleetApp.java:46`; `FleetMcp.java:67`).
|
||||
- Quote: `dev.ltms.fleetd` (line 67). The source package is `dev.ltms.fleet`
|
||||
(`FleetMcp.java:1`).
|
||||
- Quote: `WorkerPresence` (line 110). The current class is `MemberPresence`, as
|
||||
imported and used by `FleetMcp` at `FleetMcp.java:12` and `465-469`.
|
||||
- Quote: the outcome list ending in `STALE_TURN` (lines 128-131). The code also
|
||||
has `BACKEND_EXHAUSTED` (`FleetMcp.java:550-554`) and async `ASKING` handling
|
||||
(`FleetMcp.java:664-668`).
|
||||
- Quote: `FleetdApp` (line 207) and `FleetdConfig` (line 211). These names do not
|
||||
resolve; current classes are `FleetApp` and `FleetConfig`.
|
||||
|
||||
### `10-Cross-Host-Messaging.md` — REVISE
|
||||
|
||||
- Quote: the chapter says the cross-host fabric is proposed except for the
|
||||
single-host inbox (lines 3-8). Cross-host **lead-to-lead** delivery shipped:
|
||||
`fleet_send` accepts `coordId` (`FleetMcp.java:1094-1107`) and publishes it at
|
||||
`FleetMcp.java:616-641`; configuration has `coordinator` at
|
||||
`FleetConfig.java:74-78` and `99-101`.
|
||||
- Quote: `bridge.dlx` (line 90). This product name is stale. The shipped lead path
|
||||
uses `LeadChannel`, not the proposed exchange flow (`FleetMcp.java:95-96` and
|
||||
`616-641`). Keep the proposed federation design, but add a clear shipped/proposed
|
||||
boundary for CB-637.
|
||||
|
||||
### `11-Features.md` — REVISE
|
||||
|
||||
- Quote: `mcp/BridgeMcp` (line 22), `config/FleetdConfig` (lines 25-27), and other
|
||||
index references. These paths no longer resolve; the source classes are
|
||||
`mcp/FleetMcp` (`FleetMcp.java:67`) and `config/FleetConfig`
|
||||
(`FleetConfig.java:81`).
|
||||
- Quote: `fleet_whoami` returns only `primary` or `worker` (lines 99-100).
|
||||
It also returns `architect` (`FleetMcp.java:1235-1244`).
|
||||
- The page needs the missing separate member-herdr-daemon feature listed below.
|
||||
|
||||
### `12-Claude-to-OpenCode.md` — REVISE
|
||||
|
||||
- Quote: `same bridge mount` (line 5) and `a bridge-spawned worker` (line 94).
|
||||
Rename the product path to `fleet`. The daemon exposes the MCP server as `fleet`
|
||||
(`FleetMcp.java:313-315`), and profiles select OpenCode with `kind: opencode`
|
||||
(`FleetConfig.java:332-335`).
|
||||
- Quote: the sample mount name is `fleetd` (line 67). The server name is `fleet`;
|
||||
update the sample to avoid teaching a second product name.
|
||||
|
||||
### `13-User-Guide.md` — REVISE
|
||||
|
||||
- Quote: `The bridge is the only channel` (line 63). The invariant is correct, but
|
||||
the product term needs the `fleet` rename. The daemon's MCP server name is
|
||||
`fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: it describes one herdr socket (lines 72-85). It needs the optional
|
||||
`memberHerdrSocket` setup and two-daemon health meaning. The config key is in
|
||||
`FleetConfig.java:34-37`, and `/healthz` checks both daemons when configured at
|
||||
`FleetApp.java:210-245`.
|
||||
|
||||
## MISSING
|
||||
|
||||
`11-Features.md` has a body section for **routing members through a separate herdr daemon**
|
||||
(`## memberHerdrSocket`, line 2174), but **no row in the index table** at the top of the page
|
||||
(lines 20-95). That table is how the page is meant to be read, so a capability absent from it is
|
||||
effectively undiscoverable. Lead note: this is my own omission — I added the section on 2026-08-31
|
||||
and did not add the matching row. Fixed in the wiki at `68e32c6`'s successor.
|
||||
|
||||
The original audit stated the feature had no entry at all. That was wrong: the section exists. The
|
||||
gap is the index row. Recorded here rather than silently corrected, because the difference matters —
|
||||
"undocumented" and "documented but unindexed" are different jobs.
|
||||
|
||||
Evidence for the feature itself: `FleetConfig.java:34-37` and `FleetApp.java:103-115`, `210-245`,
|
||||
and `247-263`.
|
||||
|
||||
## Audit method and coverage
|
||||
|
||||
I checked all 15 pages. I checked concrete tool, route, config, class, file, and
|
||||
product-name claims claim-by-claim on 11 pages: Home, Sidebar, 1, 2, 4, 5, 6, 7, 9,
|
||||
11, and 13. I skimmed the remaining four long historical or research pages (3, 8, 10,
|
||||
12), then checked their concrete claims that affect the verdict. This is an audit of
|
||||
the supplied snapshot, not a wiki rewrite.
|
||||
+3
-3
@@ -2,7 +2,7 @@
|
||||
|
||||
A standard, repeatable **live** end-to-end test of the two-way channel: it drives a real
|
||||
multi-turn conversation between a primary and an off-subscription worker **through the
|
||||
running `bridged` daemon**, captures the full transcript, and grades the channel.
|
||||
running `fleetd` daemon**, captures the full transcript, and grades the channel.
|
||||
|
||||
This is the committed form of the ad-hoc channel test that discovered the CB-115 gaps
|
||||
(herdr `unknown` misclassification wedging delivery, dirty completion scrapes, and workers
|
||||
@@ -20,7 +20,7 @@ construction** — it only calls the bridge's loopback REST face.
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant T as conversation_test.py
|
||||
participant B as bridged (REST)
|
||||
participant B as fleetd (REST)
|
||||
participant W as worker (off-sub)
|
||||
T->>B: POST /workers (spawn)
|
||||
T->>B: GET /sessions/{id}/status (await ready)
|
||||
@@ -37,7 +37,7 @@ sequenceDiagram
|
||||
|
||||
## Prerequisites
|
||||
|
||||
- `bridged` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
- `fleetd` is running (default REST on `http://127.0.0.1:8765`) with at least one worker
|
||||
profile configured and its backend reachable.
|
||||
- herdr is up (the daemon needs it).
|
||||
- Python 3 (standard library only — no pip installs).
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Sustained back-and-forth bridge test — ONE primary, ONE worker, many dependent turns
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `bridged` daemon.
|
||||
over a fixed wall-clock window (default 5 minutes), through the running `fleetd` daemon.
|
||||
|
||||
Where conversation_test.py proves a handful of turns work and issue_hunt_test.py proves
|
||||
fan-out isolation, this proves the channel stays healthy under a *sustained, stateful*
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge conversation test — a multi-turn primary↔worker exchange through
|
||||
the running `bridged` daemon, fully captured, with automatic gap analysis.
|
||||
the running `fleetd` daemon, fully captured, with automatic gap analysis.
|
||||
|
||||
This is the repeatable form of the ad-hoc channel test that surfaced the CB-115 gaps
|
||||
(herdr `unknown` misclassification, dirty completion scrape, workers not calling
|
||||
|
||||
@@ -24,7 +24,7 @@ Like the rest of the suite it talks ONLY to the bridge's REST face on loopback
|
||||
ANTHROPIC_BASE_URL and never touches herdr, so it is subscription-safe by construction.
|
||||
|
||||
Usage:
|
||||
python3 bridge_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
python3 fleet_ask_test.py [--base URL] [--profile NAME] [--repo DIR] [--out DIR]
|
||||
[--send-timeout SECS] [--answer-timeout SECS] [--keep-worker]
|
||||
|
||||
--base bridge REST base URL (default http://127.0.0.1:8765)
|
||||
@@ -1,12 +1,12 @@
|
||||
# Live bridge_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
# Live fleet_ask — reverse rendezvous — 2026-07-16 16:30
|
||||
|
||||
One worker paused its delegated turn to ask the primary, then resumed with the answer (profile `default`). Result: **`OK`**.
|
||||
|
||||
## Round-trip
|
||||
|
||||
1. **primary → worker** (delegation): the ask-forcing task.
|
||||
2. **worker → primary** (`bridge_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
2. **worker → primary** (`fleet_ask`, 6.6s): 'PICK A COLOR: red or blue?' — surfaced on the primary's blocked send as a `question` with `turnId=term_656bb47d2c42a9e#1`.
|
||||
3. **primary → worker** (answer on that turn): `blue`.
|
||||
4. **worker → primary** (`bridge_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
4. **worker → primary** (`fleet_reply`, 7.9s, source=reply): 'CHOSEN=BLUE'
|
||||
|
||||
> **OK:** asked, resumed the same turn, and the reply reflected the primary's answer
|
||||
@@ -1,6 +1,6 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Standard bridge fan-out test — ONE primary vs MANY workers, concurrently, for
|
||||
issue hunting through the running `bridged` daemon, fully captured, with gap analysis.
|
||||
issue hunting through the running `fleetd` daemon, fully captured, with gap analysis.
|
||||
|
||||
Where conversation_test.py exercises a single worker over multiple turns, this drives
|
||||
the path that only appears under fan-out: the primary spawns N workers, sends each a
|
||||
@@ -53,11 +53,11 @@ REPO_ROOT = HERE.parent
|
||||
# hot files this project has been iterating on, so a real issue is plausible to find.
|
||||
DEFAULT_ASSIGNMENTS = [
|
||||
{"id": "completion", "probe": "CompletionResolver",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/inject/CompletionResolver.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java"},
|
||||
{"id": "worker", "probe": "WorkerService",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/worker/WorkerService.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/worker/WorkerService.java"},
|
||||
{"id": "rendezvous", "probe": "Rendezvous",
|
||||
"target": "bridged/src/main/java/dev/ltms/bridged/msg/Rendezvous.java"},
|
||||
"target": "fleetd/src/main/java/dev/ltms/fleet/msg/Rendezvous.java"},
|
||||
]
|
||||
|
||||
PROMPT_TMPL = (
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
target/
|
||||
dependency-reduced-pom.xml
|
||||
|
||||
# Local runtime config (copy from fleetd.example.yaml). Both names: the live file is still
|
||||
# bridged.yaml until the cutover, and Fleetd reads either one.
|
||||
# Local runtime config (copy from fleetd.example.yaml). Both names are ignored: fleetd.yaml is
|
||||
# the current name, and bridged.yaml is the legacy name Fleetd still falls back to.
|
||||
fleetd.yaml
|
||||
bridged.yaml
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
# bridged configuration (example). Copy to bridged.yaml and adjust.
|
||||
# fleetd configuration (example). Copy to fleetd.yaml and adjust.
|
||||
#
|
||||
# bridged is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# fleetd is the sole gateway between primary/worker Claude sessions and herdr.
|
||||
# It is NOT a Claude process and must never carry ANTHROPIC_BASE_URL.
|
||||
|
||||
# REST + MCP listen address. Keep it on loopback unless you also switch auth.mode to `token`
|
||||
# below — bridged REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
# below — fleetd REFUSES TO START on a non-loopback bind under loopback-trust (see auth).
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
@@ -21,7 +21,7 @@ bind:
|
||||
# bind — the daemon fails fast otherwise, because "unauthenticated ⇒
|
||||
# primary" on a reachable port would hand spawn/stop/send to anyone.
|
||||
# tokenEnv → host env var holding the token (never the literal value). Default
|
||||
# BRIDGED_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
# FLEETD_API_TOKEN. Read only in token mode; empty ⇒ startup fails.
|
||||
#
|
||||
# TLS is deliberately NOT terminated in the daemon (CB-501 D3): run a reverse proxy in front and
|
||||
# let it own certificate lifecycle, e.g.
|
||||
@@ -29,7 +29,7 @@ bind:
|
||||
# The broker link gets TLS from its own URI (amqps://…) — see `broker` below.
|
||||
# auth:
|
||||
# mode: token
|
||||
# tokenEnv: BRIDGED_API_TOKEN
|
||||
# tokenEnv: FLEETD_API_TOKEN
|
||||
|
||||
# Optional pinned primary terminal (CB-307). Names the herdr pane the PRIMARY itself runs in:
|
||||
# a caller whose connection maps to this pane resolves as the primary (no credential needed —
|
||||
@@ -48,10 +48,10 @@ bind:
|
||||
# every orchestration call. List each lead's pane here and all of them resolve as leads.
|
||||
#
|
||||
# tab → the ONLY field identity depends on (CB-579); the exact label of the tab hosting the lead.
|
||||
# Label the tab yourself, or let bridged label one it launches — see `fleet.leaders:` below.
|
||||
# Label the tab yourself, or let fleetd label one it launches — see `fleet.leaders:` below.
|
||||
# kind/model → descriptive; they document what runs in the pane and are echoed by fleet_whoami
|
||||
#
|
||||
# A lead's tab must already carry its label (or be launched by bridged, which labels it) — there is
|
||||
# A lead's tab must already carry its label (or be launched by fleetd, which labels it) — there is
|
||||
# no terminal id to paste in and nothing to re-pin when the session restarts: the tab survives, so
|
||||
# the same label resolves the same lead again on the next scan.
|
||||
# `fleet_whoami` reports `{"role":"primary","leader":"<name>"}`; role stays "primary" because a lead
|
||||
@@ -63,7 +63,7 @@ bind:
|
||||
# Leads are configured under `fleet.leaders:` — see THE FLEET further down.
|
||||
#
|
||||
# Two things stop the tab-name convention from becoming a way to claim leadership: the configured
|
||||
# member spaces are excluded from the scan, so nothing bridged places can land in a matching tab;
|
||||
# member spaces are excluded from the scan, so nothing fleetd places can land in a matching tab;
|
||||
# and startup REFUSES a `tabPrefix` that the fleet tabLabel template, or any per-profile `tabLabel`
|
||||
# override, also matches — so the two namespaces cannot overlap by accident. The label is a NAME,
|
||||
# never a capability: what a pane may do is decided by the role the daemon resolves for it.
|
||||
@@ -88,33 +88,93 @@ bind:
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Lead rollover (fleetd #480): replace a lead session that has decided it is ready to be replaced,
|
||||
# without an operator doing it by hand. A lead writes a handover file, then asks fleetd to clear its
|
||||
# own pane and bootstrap a fresh session against that file.
|
||||
#
|
||||
# Opt-in on purpose — it clears the lead's own pane on request, so upgrading the daemon must never
|
||||
# acquire that ability for you. Absent block = feature off, and nothing is constructed at all. Even
|
||||
# once present, nothing but an explicit confirm() call — one that passes every check — can ever
|
||||
# cause a /clear: there is no recurring timer, heartbeat or scheduler anywhere in this feature that
|
||||
# fires one on its own initiative. confirm() itself is called FROM the calling lead's own turn, so
|
||||
# it cannot clear the pane inline (that pane is still WORKING); instead it schedules a one-shot
|
||||
# continuation that waits for the SAME confirm() call's turn to end, then does the actual work. See
|
||||
# dev.ltms.fleet.lead.LeadRollover's class javadoc for the exact order (fleetd #480 correction).
|
||||
#
|
||||
# handoverPath: REQUIRED when this block is present — where the handover file a fresh lead session
|
||||
# reads must live. No default (an operator-specific path); a present block with no
|
||||
# handoverPath refuses to start. May be relative: it then resolves against the
|
||||
# CALLING lead's own fleet.leaders.<name>.cwd (falling back to the daemon's own
|
||||
# working directory when that lead has none configured) — never against whatever
|
||||
# directory the daemon process happens to have been started in. An absolute path is
|
||||
# used unchanged. Prefer an absolute path if the daemon and the lead's pane might not
|
||||
# share a working directory (fleetd #480 follow-up).
|
||||
# requireOperatorConfirm: true # default true — confirm() refuses unless the caller also passes
|
||||
# # operatorConfirmed: true
|
||||
# maxDocAgeSeconds: 3600 # default 3600 — refuse a handover file older than this
|
||||
# turnSettleSeconds: 20 # default 20 — how long the deferred roll waits for the CALLING
|
||||
# # lead's own turn to end (its pane to report injectable again)
|
||||
# # before sending /clear at all. If this elapses, /clear is NEVER
|
||||
# # sent — a lead that never goes idle is still doing real work.
|
||||
# clearSettleSeconds: 20 # default 20 — how long to wait for the pane to become injectable
|
||||
# # again AFTER /clear before giving up (never sends bootstrapText
|
||||
# # if this elapses). A separate, second wait from turnSettleSeconds.
|
||||
# bootstrapText: "..." # default names the RESOLVED (absolute) handoverPath — sent to
|
||||
# # the lead once its pane settles after /clear
|
||||
# leadRollover:
|
||||
# handoverPath: /path/to/handover.md
|
||||
# requireOperatorConfirm: true
|
||||
# maxDocAgeSeconds: 3600
|
||||
# turnSettleSeconds: 20
|
||||
# clearSettleSeconds: 20
|
||||
# bootstrapText: "Fresh lead session: read the handover file and carry on."
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
# per tick.
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
# Math.max(15, intervalSeconds), so a lower value is silently raised, not
|
||||
# rejected.
|
||||
# workingSuspectAfterSeconds, paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by
|
||||
# anything — the dormant monitor only consumes intervalSeconds today (CB-573
|
||||
# shipped ahead of the evidence publishers these two knobs are for). Setting
|
||||
# them changes nothing right now, and no minimum is enforced on either, because
|
||||
# nothing reads them to enforce one. They exist so a later build can start
|
||||
# honouring them without another config-shape change.
|
||||
# workingSuspectAfterSeconds → age before a BUSY member is suspected of a stall (default 600).
|
||||
# ENFORCED floor of 300: a lower value is silently raised.
|
||||
# paneProbeIntervalSeconds → accepted and parsed, but NOT YET READ by anything. Setting it changes
|
||||
# nothing right now. It exists so a later build can start honouring it without
|
||||
# another config-shape change.
|
||||
# notifications.mode → "webhook" flips what fleet_list REPORTS (healthCoverage: "full" instead
|
||||
# of "detection-only") — it does NOT make bridged send any webhook call; no
|
||||
# of "detection-only") — it does NOT make fleetd send any webhook call; no
|
||||
# delivery mechanism is implemented yet. Any other value, or omitting the
|
||||
# block, reports "detection-only".
|
||||
# health:
|
||||
# enabled: true
|
||||
# intervalSeconds: 30
|
||||
# workingSuspectAfterSeconds: 600
|
||||
# paneProbeIntervalSeconds: 60
|
||||
# intervalSeconds: 30 # floor 15
|
||||
# workingSuspectAfterSeconds: 600 # floor 300 — how long BUSY with no activity means STALL_SUSPECTED
|
||||
# paneProbeIntervalSeconds: 60 # parsed, but nothing reads it yet — changing it changes nothing
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# Idle-sleep guard: while at least one member is live, hold an OS-level assertion against idle
|
||||
# sleep (macOS only — a `caffeinate -i` child; a no-op elsewhere or if caffeinate is missing), so
|
||||
# an unattended host does not idle-sleep out from under a member's long turn. Unlike health/
|
||||
# configReload above, this is ON BY DEFAULT — omitting the block entirely leaves it enabled, the
|
||||
# same as `enabled: true`. Uncomment only to turn it off:
|
||||
# idleSleepGuard:
|
||||
# enabled: false
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# Optional socket for member panes. Omit this to use herdrSocket for both leads and members.
|
||||
# memberHerdrSocket: /Users/member/.config/herdr/herdr.sock
|
||||
|
||||
# fleetd #213: the login shell the member OS user (memberHerdrSocket above) actually runs. ONLY
|
||||
# read when memberHerdrSocket is set — fleetd's own $SHELL says nothing about a pane running
|
||||
# under a different OS user, and there is no channel to ask herdr for that user's shell, so this
|
||||
# must be told rather than guessed. Absent, blank, or anything not ending in "zsh" is treated the
|
||||
# same as "not zsh": the memberCredentials.policy: allow-list ZDOTDIR scrub (see worktreeGroup
|
||||
# below) is skipped in favour of the weaker CB-596 sentinel overlay — a degraded control, never a
|
||||
# refusal to spawn. When memberHerdrSocket is absent this key is never consulted at all.
|
||||
# memberLoginShell: /bin/zsh
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
@@ -124,16 +184,16 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# Shared knobs (placement/workspace) can be repeated per profile; they usually match.
|
||||
# placement: tab → each worker lands in its OWN tab in a dedicated worker space (default).
|
||||
# Use `pane` for the legacy behaviour (split the focused tab).
|
||||
# mcpUrl → bridged mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# mcpUrl → fleetd mounts the bridge MCP (--mcp-config, inline) + reply charter
|
||||
# (--append-system-prompt) as launch flags; nothing is written to the profile.
|
||||
# ideMcpUrl → opt-in (CB-634), default off. When set, bridged mounts the IDE Index MCP as a
|
||||
# ideMcpUrl → opt-in (CB-634), default off. When set, fleetd mounts the IDE Index MCP as a
|
||||
# second inline server named `intellij`, and adds an IDE charter that pins every
|
||||
# ide_* call to the member's own worktree. A URL, not a boolean — host and port
|
||||
# are host-specific. Set it only on a host where the IDE actually runs.
|
||||
# ideProjectDir → repo-relative module dir the IDE opens and the overlay pins (CB-634). Only read
|
||||
# when ideMcpUrl is set. This repo's Maven pom lives in `bridged/`, not at the
|
||||
# when ideMcpUrl is set. This repo's Maven pom lives in `fleetd/`, not at the
|
||||
# worktree root, so opening the root imports no module and ide_* resolves nothing;
|
||||
# set this to `bridged`. Omit for a repo whose project is the worktree root.
|
||||
# set this to `fleetd`. Omit for a repo whose project is the worktree root.
|
||||
# ideOpenCommand → host command that opens ideProjectDir in the IDE at spawn (CB-634 auto-open).
|
||||
# Only read when ideMcpUrl is set. `{dir}` is replaced with the absolute module
|
||||
# dir and the command runs through `/bin/sh -c`, so set env inline if needed —
|
||||
@@ -151,7 +211,7 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# already), so this is instead applied as the model's `limit.context` in the
|
||||
# generated opencode.json — the member compacts WITHIN this window, not exactly
|
||||
# at it — and only when this profile's `model:` is in `provider/model` form; if it
|
||||
# isn't, bridged logs a WARN naming the profile rather than silently doing nothing.
|
||||
# isn't, fleetd logs a WARN naming the profile rather than silently doing nothing.
|
||||
# tokenEnv → host env var holding the worker's auth token (value never stored in config);
|
||||
# omit for a backend that needs no token (e.g. a local ollama).
|
||||
# cwd → pin this profile's working directory (CB-112). Omit to inherit the primary's
|
||||
@@ -161,7 +221,11 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.env, .envrc]. (.claude/settings.local.json is NOT in the default — it
|
||||
# [.env] only (CB-148). .envrc is left out of the default on purpose: it is
|
||||
# executable shell that direnv runs on every cd, so copying it carries
|
||||
# behaviour into the worker, not just values, unlike .env. An operator who
|
||||
# wants it copied can still write parityOverlay: [.env, .envrc] explicitly.
|
||||
# (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
@@ -170,7 +234,7 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# checkout, so its navigation returned paths OUTSIDE its own worktree: one
|
||||
# worker made all 59 of its edits in the primary tree while compiling its
|
||||
# worktree, and every build it ran was of code that did not contain them.
|
||||
# bridged neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# fleetd neutralizes a provisioned worktree's .mcp.json for this reason;
|
||||
# listing it here would copy the primary's back over that.
|
||||
# gitTokenEnv → host env var holding the git-forge API token. When set, its value is injected
|
||||
# as GITEA_TOKEN so the worker can open its OWN PR at checkpoint (CB-302).
|
||||
@@ -183,9 +247,13 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# refused on a subscription usage limit, rather than a real answer. Opt-in —
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into bridged itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# vendor string baked into fleetd itself.
|
||||
# HOT (fleetd #446): read live, cached by profile name, at every
|
||||
# completion-fallback check AND by fleet_profiles' exhaustionDetectionArmed —
|
||||
# editing it and reloading arms or disarms usage-limit detection for this
|
||||
# profile with no daemon restart. (Before fleetd #446 this was DEFERRED,
|
||||
# compiled once into a startup pattern map like model/baseUrl/argv still are —
|
||||
# see errorPattern below, which is still deferred that way on purpose.)
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
# BACKEND_EXHAUSTED classification fires. Two profiles that set the SAME
|
||||
# credentialId share one quarantine — the case this exists for is two models
|
||||
@@ -195,18 +263,52 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# errorPattern → fleetd #201 / #227: regex matched against a completion-fallback scrape to
|
||||
# classify a turn that ended with no fleet_reply as a BACKEND ERROR — a
|
||||
# credential outage or a provider 5xx — rather than a real answer or a
|
||||
# usage-limit exhaustion (exhaustedPattern above always wins when a line
|
||||
# matches both). Opt-in. Omit it and this profile falls back to fleetd's
|
||||
# built-in legacy pattern `(?i)\bAPI Error\s*:` — classification still
|
||||
# happens, just without a profile-specific match; every backend words its
|
||||
# failure differently, so a hardcoded sentence would only ever match one
|
||||
# of them.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart. Unlike exhaustedPattern above (made hot by fleetd #446),
|
||||
# errorPattern was scoped out of that ticket on purpose and stays deferred.
|
||||
# # errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage
|
||||
#
|
||||
# What happens once a match fires (BackendOutagePolicy, credentialId-keyed,
|
||||
# SEPARATE from the CB-578 stage B quarantine above and never merged with it):
|
||||
# - threshold 2 — TWO DISTINCT TARGETS (never raw events) on the same
|
||||
# effective credential inside a 60-second window start an "incident" and a
|
||||
# 60-second cool-off for that credential. One member repeating the same
|
||||
# classified line twice never cools anything off — a real outage hits
|
||||
# every target on that credential, so requiring a second, independent
|
||||
# target loses nothing against the case this guards against, while
|
||||
# protecting against a heuristic misfire on one flaky member.
|
||||
# - a fresh error while a credential is already cooling off is ignored
|
||||
# outright: it neither extends the 60s deadline nor starts a new incident.
|
||||
# - `fleet_list`/`fleet_profiles` report a cooling credential with
|
||||
# `coolingOffForSeconds` (never `quarantinedForSeconds`, unless CB-578
|
||||
# exhaustion quarantine is ALSO independently active for the same
|
||||
# credential — the two checks can both fire at once). A spawn onto a
|
||||
# cooling profile is refused with a message naming the credential and
|
||||
# remaining seconds — "cooling off", never "exhausted", so an operator can
|
||||
# tell a short transient fault from a spent subscription at a glance.
|
||||
# - the lead gets ONE nudge per incident (not one per affected target), via
|
||||
# the same push loop that already delivers ticket/question reminders.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
# A worker's environment does NOT come from your shell. bridged hands herdr an
|
||||
# A worker's environment does NOT come from your shell. fleetd hands herdr an
|
||||
# explicit env map and herdr merges it into ITS OWN process env — so before
|
||||
# CB-511 a worker inherited whatever PATH the herdr server happened to be
|
||||
# started with, which on a long-lived herdr can predate your toolchain entirely
|
||||
# and leave workers unable to run `mvn` or `java` at all.
|
||||
# bridged now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# fleetd now propagates ITS OWN PATH to every worker by default; set `env:`
|
||||
# only to override that or add more (JAVA_HOME, …). Since the default is the
|
||||
# daemon's PATH, make sure the daemon is started with a good one — see the PATH
|
||||
# lines in deploy/dev.ltms.fleet.plist and deploy/bridged.service.
|
||||
# lines in deploy/dev.ltms.fleet.plist and deploy/fleetd.service.
|
||||
#
|
||||
# Adapter-owned variables always win over `env:`: ANTHROPIC_BASE_URL and the
|
||||
# rest of the ANTHROPIC_*/CLAUDE_* wiring are applied after it, so an `env:`
|
||||
@@ -219,10 +321,10 @@ profiles:
|
||||
baseUrl: http://gx01.gw:8000 # the vLLM host this profile targets (gx00.gw / gx01.gw)
|
||||
model: coder
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
tokenEnv: BRIDGED_WORKER_TOKEN
|
||||
tokenEnv: FLEETD_WORKER_TOKEN
|
||||
argv: ["ccs", "gx10"]
|
||||
# weight: relative selection weight for automatic placement (weighted, round-robin, and
|
||||
# fixed's fallback walk). Absent defaults to 1.0. An explicit 0 or negative value means
|
||||
@@ -258,21 +360,57 @@ profiles:
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
#
|
||||
# GOTCHA 3 (fleetd #176, corrected by fleetd #257) — `maxLoad` counts members, never the lead
|
||||
# itself. The lead is a live `claude` session on this SAME account (a lead is never moved
|
||||
# off-subscription, whatever its own profile says), so it already holds one seat before any
|
||||
# member spawns. If a lead's `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER
|
||||
# `subscription: true` profile that shares this one's account (see THE SENTINEL, just below,
|
||||
# next to `credentialId:`) — `fleet_list` reports that seat count under `leadSeats`; see
|
||||
# `profile:` under THE FLEET below. `free` itself is NEVER reduced by `leadSeats`: `free` means
|
||||
# "what the real placement gate (`CompositePeerLauncher#enforceMaxLoad`) will actually grant a
|
||||
# fresh `fleet_spawn` right now", and that gate only ever compares live members against
|
||||
# `maxLoad` — it has no notion of the lead's own seat. An earlier cut of this feature
|
||||
# subtracted `leadSeats` from `free` on the theory it made `free` describe the true ceiling on
|
||||
# the account, but no backend seat ceiling shared with the lead has ever actually been
|
||||
# measured, and the subtraction just made `free` disagree with the one thing it is supposed to
|
||||
# describe — the fleetd #257 fix. `maxLoad: 3` means 3 member slots, full stop; a lead sharing
|
||||
# the account is a fact you can see in `leadSeats`, not a reason `free` undercounts spawns that
|
||||
# will, in practice, succeed.
|
||||
#
|
||||
# THE SENTINEL (fleetd #176 stage 2, correcting an inert stage 1 fix): every `subscription:
|
||||
# true` profile that leaves `credentialId` unset shares ONE implicit account-wide credential
|
||||
# id with every other such profile on this host — because a subscription profile doesn't
|
||||
# authenticate with a credential of its own, it authenticates as the operator's own Claude
|
||||
# login, and there is exactly one of those. So on a typical host, `opus` (the lead's profile)
|
||||
# and `sonnet` (the members' profile) are linked automatically, with NOTHING to set here — that
|
||||
# is what makes GOTCHA 3 above work without also writing matching `credentialId:` values on
|
||||
# both. This linkage is not just cosmetic: it is the same key `BackendQuarantine`/cool-off use,
|
||||
# so a usage-limit hit on `opus` now quarantines `sonnet` too (and vice versa) — correct, since
|
||||
# they are one Claude account, but worth knowing before you wonder why an unrelated-looking
|
||||
# profile went quarantined.
|
||||
#
|
||||
# WHEN TO OVERRIDE — set explicit, DIFFERENT `credentialId:` values on two `subscription: true`
|
||||
# profiles only when they are genuinely two separate Claude logins on the same host (a real,
|
||||
# if unusual, setup). An explicit `credentialId` always wins over the sentinel, so this is the
|
||||
# one way to keep two subscription profiles from being treated as one account for lead-seat
|
||||
# counting AND for quarantine/cool-off grouping alike.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage (fleetd #201/#227) — see the key doc above
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".env", ".envrc"] # the default; never add .mcp.json or .claude/settings.local.json — see above
|
||||
# parityOverlay: [".env"] # the default; add ".envrc" explicitly if you want it copied too (CB-148) — never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: bridged # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in bridged/)
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
# autoCompactWindow: 250000 # opt-in: bound member context; claude-code compacts AT this, opencode within it (model limit.context)
|
||||
gx11: # a second backend, so `placement: weighted` has a choice
|
||||
baseUrl: http://gx01.gw:8000 # self-hosted; ccs handles the model + token
|
||||
placement: tab
|
||||
workspace: bridged-workers
|
||||
workspace: fleetd-workers
|
||||
# tabLabel: an optional per-profile override; the fleet template usually covers it
|
||||
mcpUrl: http://127.0.0.1:8765/mcp
|
||||
argv: ["ccs", "gx11"]
|
||||
@@ -298,7 +436,7 @@ profiles:
|
||||
# kind: opencode
|
||||
# model: opencode/north-mini-code-free # `provider/model` selector, injected as `-m`
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
@@ -323,7 +461,7 @@ profiles:
|
||||
# baseUrl: http://127.0.0.1:8000
|
||||
# model: local-vllm/deepseek-v4-flash
|
||||
# placement: tab
|
||||
# workspace: bridged-workers
|
||||
# workspace: fleetd-workers
|
||||
# tabLabel: "opencode: {profile} #{n}"
|
||||
# mcpUrl: http://127.0.0.1:8765/mcp
|
||||
# argv: ["opencode"]
|
||||
@@ -355,14 +493,29 @@ placement: weighted
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
#
|
||||
# fleetd #466: this is now only the BASE of an escalating backoff, not a flat retry rate. A
|
||||
# credential quarantined again within one base cooldown of the previous quarantine ending (still
|
||||
# reporting exhausted — e.g. a weekly subscription limit that hasn't reset) backs off further:
|
||||
# cooldown doubles each such time, capped at 12x this value (~6 hours at the 1800s default). A
|
||||
# quarantine that starts after a base-cooldown's worth of quiet resets back to this value. Not
|
||||
# configurable per se — the multiplier and ceiling are constants in BackendQuarantine, not new
|
||||
# YAML keys; see its class doc for the exact formula and why there is no automatic probe to clear
|
||||
# it early (the operator's own design constraint — a probe spends the quota it's measuring).
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
#
|
||||
# This does NOT govern the fleetd #201 / #227 backend-error cool-off documented under errorPattern
|
||||
# above — that mechanism is a separate, shorter-lived, NOT-configurable policy (threshold 2 distinct
|
||||
# targets, 60-second window, 60-second cool-off), on purpose: it exists to survive a brief transient
|
||||
# fault, not to replace this 30-minute exhaustion quarantine. Do not conflate the two when reading
|
||||
# fleet_list/fleet_profiles — coolingOffForSeconds and quarantinedForSeconds are independent facts.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
# upgraded bridged keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. bridged checks the file's modified time on a timer and
|
||||
# upgraded fleetd keeps the old behaviour: the file is read once at boot and never again.
|
||||
# enabled → turn the watch on. fleetd checks the file's modified time on a timer and
|
||||
# reloads when it moves.
|
||||
# intervalSeconds → how often to check (default 10). One `stat` per tick, so this is cheap.
|
||||
#
|
||||
@@ -370,9 +523,10 @@ placement: weighted
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# / credentialId / exhaustedPattern. Those are hot because the placement policy (and,
|
||||
# for credentialId, the CB-578 stage B quarantine check; for exhaustedPattern, fleetd
|
||||
# #446's LiveExhaustedPatterns) reads them through a supplier — being config is not by
|
||||
# itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Fleetd.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
@@ -384,8 +538,10 @@ placement: weighted
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# configDir, mcpUrl, tabLabel, errorPattern (fleetd #201 / #227 — compiled once into
|
||||
# a startup pattern map; exhaustedPattern used to be compiled the same way until
|
||||
# fleetd #446 made it hot — see above). The launcher takes a copy of `profiles:` at
|
||||
# startup and resolves every spawn out of that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
@@ -446,6 +602,23 @@ fleet:
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `profile:` has a SECOND job as of fleetd #176, even for a recognise-only lead you never want
|
||||
# auto-launched: it is also how fleetd learns which account this lead's own session shares. A
|
||||
# `subscription: true` profile bills the operator's Claude account, and the lead itself is always
|
||||
# a live `claude` session on that same account — `maxLoad` never counted that seat. If a lead
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list` reports the
|
||||
# lead's live seat(s) on that worker profile under `leadSeats` — informational only, as of fleetd
|
||||
# #257 it is NEVER subtracted from `free` (see GOTCHA 3, next to `maxLoad:`, in THE WORKERS above,
|
||||
# for why). "Shares the account" is decided by matching `effectiveCredentialId()`, which (fleetd
|
||||
# #176 stage 2 — see THE SENTINEL, next to `credentialId:`, in THE WORKERS above) means: an
|
||||
# explicit, matching `credentialId:` on both, OR — the common case, needing NO extra config — both
|
||||
# being `subscription: true` with `credentialId` left unset, since those all share one implicit
|
||||
# account-wide id. A lead on `opus` and workers on `sonnet` link automatically this way; they do
|
||||
# NOT need the same profile name. Setting `profile:` on an already-running, recognise-only lead is
|
||||
# safe — the daemon only launches the SHORTFALL below `instances`, so naming a profile here does
|
||||
# not, by itself, start anything. Omit it and fleetd has no way to derive the sharing — there is
|
||||
# no other reliable signal on the daemon's side — so that lead's seat never appears in `leadSeats`.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
@@ -477,7 +650,7 @@ fleet:
|
||||
# workspace: leads # where a launched lead's tab is created (default "leads").
|
||||
# # MUST NOT be a member workspace — those are excluded from the
|
||||
# # scan, so a lead placed in one is never found again.
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: bridged's own)
|
||||
# cwd: /path/to/repo # the launched lead's working directory (default: fleetd's own)
|
||||
# kind: claude # descriptive; reported by fleet_whoami
|
||||
# gpt-sol-5.6:
|
||||
# tab: "lead: gpt-sol-5.6"
|
||||
@@ -506,7 +679,7 @@ guard:
|
||||
|
||||
# Member credential policy (CB-596, gitea issue #82). A herdr pane runs a LOGIN shell, and that
|
||||
# shell re-sources the operator's own secret store — so a spawned member inherits every credential
|
||||
# the operator's shell holds, not just the ones bridged means to give it. Measured on this host:
|
||||
# the operator's shell holds, not just the ones fleetd means to give it. Measured on this host:
|
||||
# 31 credential names, all set, with only ONE (GITEA_ACCESS_TOKEN) blocked before this — and that
|
||||
# block was a single name hardcoded in HerdrPeerLauncher.java, not driven by this file. This block
|
||||
# replaces that hardcoded shadow with a config-driven list of names.
|
||||
@@ -567,20 +740,35 @@ guard:
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAuthSock → whether SSH_AUTH_SOCK may pass through under allow-list ("allow") or must be
|
||||
# blanked like any other non-derived name ("block", the default). This is a decision you
|
||||
# have to make explicitly: SSH_AUTH_SOCK is a handle to YOUR ssh-agent, and a member
|
||||
# holding it can sign with your keys — it sits in no secret file and looks like no
|
||||
# credential, which is why it slipped past three earlier tickets (gitea #110). Blocking
|
||||
# it breaks git over SSH inside members (push/fetch authenticate as you); use HTTPS
|
||||
# remotes or scoped deploy keys instead of allowing it lightly.
|
||||
# sshAgentEnv → whether SSH_AUTH_SOCK may pass through under allow-list ("inherit") or is omitted
|
||||
# from the member environment ("omit", the default). Omitting it only omits the
|
||||
# inherited ssh-agent path. It discourages automatic use of the operator's agent.
|
||||
# It does not deny same-user access to that socket. It also does not block SSH keys that
|
||||
# are readable on disk. Git over SSH may still work from inside a member. Keep the block:
|
||||
# it is correct and costs nothing, but it is not a control. A member runs as the same OS
|
||||
# user as the lead. Inside one uid, ordinary Unix permissions provide no meaningful
|
||||
# confidentiality boundary. A real boundary needs a different OS user or OS-level
|
||||
# confinement, such as a container or VM. That is the open question in fleetd #184.
|
||||
#
|
||||
# Still do not set this to "inherit" casually. SSH_AUTH_SOCK is a live handle to YOUR
|
||||
# ssh-agent, so a member holding it can sign with EVERY key the agent holds. It sits in
|
||||
# no secret file and looks like no credential, which is why it slipped past three
|
||||
# earlier tickets (gitea #110). Blocking it does not contain a member, but allowing it
|
||||
# hands one a signing capability for no gain — the block costs nothing, so keep it.
|
||||
#
|
||||
# Both halves of this are measured, not argued. 2026-08-28: a member with
|
||||
# SSH_AUTH_SOCK blanked pushed to the forge over SSH successfully, because `ssh -G`
|
||||
# resolves an IdentityFile outside ~/.ssh that is readable and has no passphrase. An
|
||||
# earlier version of this comment claimed blocking the socket BREAKS git over SSH. It
|
||||
# does not. That claim came from looking only in ~/.ssh, which holds nothing but four
|
||||
# `Include` lines — looking in one place and concluding about the whole host.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAuthSock: block # allow-list only; see the sshAuthSock note above
|
||||
# sshAgentEnv: omit # allow-list only; see the sshAgentEnv note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
@@ -632,6 +820,46 @@ guard:
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Worktree group sharing (fleetd #185 stage 3). OPTIONAL, off by default. Names an OS group
|
||||
# that a provisioned worktree's repo is made group-writable for (git config
|
||||
# core.sharedRepository group, plus a one-time chgrp/chmod/setgid fix-up), so a member spawned
|
||||
# under a DIFFERENT OS user (see memberHerdrSocket) can write its own worktree, its
|
||||
# per-worktree git metadata, and its own commit objects — without it, every file GitWorktrees
|
||||
# creates is owned by fleetd's own uid and unwritable by another user.
|
||||
# CAUTION: this isolates credentials, not the repository — a member in the group can still
|
||||
# write the operator's git objects and refs in the shared repo. The operator running fleetd
|
||||
# must already be a member of the named group, or every provisioning spawn fails loudly.
|
||||
#
|
||||
# fleetd #213: this is also the ONE group the memberCredentials.policy: allow-list ZDOTDIR scrub
|
||||
# reuses when memberHerdrSocket is set — deliberately not a second config key. Under
|
||||
# memberHerdrSocket, the scrub directory is generated under worktreeRoot (never java.io.tmpdir,
|
||||
# which the member OS user cannot reach) and shared read-only with this group. If worktreeGroup
|
||||
# is unset while memberHerdrSocket is set, the scrub cannot be guaranteed reachable by the member,
|
||||
# so fleetd falls back to the weaker CB-596 sentinel overlay instead (a WARN names the gap).
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# fleetd #362: a directory of skill folders (each a subdirectory holding a SKILL.md, the same
|
||||
# shape as this repo's own .claude/skills/) copied into every PROVISIONED worktree's
|
||||
# .claude/skills/, so a member spawned against ANY repo — not only one that already ships its own
|
||||
# copy — can load a bridge skill (e.g. implementer). Unset (the default): no worktree is touched
|
||||
# beyond today's behaviour. A skill folder the target repo already carries under
|
||||
# .claude/skills/<name> is never overwritten — the repo's own copy always wins. Best-effort like
|
||||
# worktreeGroup above: a missing/unreadable directory here is logged and skipped, never a failed
|
||||
# spawn. Every non-hidden subdirectory of this directory is copied wholesale, with no per-file
|
||||
# allowlist — don't park scratch files or drafts alongside the real skill folders, they will be
|
||||
# copied into every provisioned worktree too.
|
||||
#
|
||||
# fleetd #393: which member KINDS actually consume this once it is copied. kind: claude-code —
|
||||
# the Claude Code CLI discovers .claude/skills/ on its own; nothing else is needed. kind: opencode
|
||||
# — opencode has no such discovery, so OpenCodeLauncher reads whatever landed under
|
||||
# .claude/skills/ and appends each seeded skill's SKILL.md to the generated instructions[] file
|
||||
# (opencode's only channel for static guidance text; unlike Claude Code's Skill tool, the content
|
||||
# is always part of the system prompt, not loaded on demand). Both kinds are covered as of #393 —
|
||||
# earlier builds copied the files for every kind but only claude-code could read them, and the
|
||||
# seeding log said "N of M" regardless. Check the per-spawn launcher log (not just the seeding
|
||||
# log) to see what a given member actually got.
|
||||
# memberSkills: /path/to/fleetd/checkout/.claude/skills
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
@@ -679,10 +907,17 @@ guard:
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# peers → fleetd #361: the coord-ids of the OTHER daemons on this vhost, declared by the
|
||||
# operator (the daemon never guesses). fleet_list reports each one's live reachability
|
||||
# (a passive queue check, never a presence protocol) alongside this daemon's own
|
||||
# mailbox state. Omit, or leave empty, for a daemon with no known peers yet — an
|
||||
# undeclared peer can still reach you and be reached by fleet_send, it just will not
|
||||
# show up as a row in fleet_list.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
# peers: [fleet01-lead]
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
@@ -695,7 +930,7 @@ guard:
|
||||
#
|
||||
# REQUIRED (CB-522) if the primary itself runs inside a herdr pane. Caller
|
||||
# identity resolves a loopback PID to its herdr pane, and PaneLocator scans
|
||||
# EVERY pane — not just bridged-spawned ones — so such a primary is otherwise
|
||||
# EVERY pane — not just fleetd-spawned ones — so such a primary is otherwise
|
||||
# classified as a WORKER and refused SPAWN/SEND/STOP. That failure is
|
||||
# self-locking: the learned terminal is populated by the very orchestration
|
||||
# calls being refused, so only this pinned value can break the cycle. Read the
|
||||
@@ -707,3 +942,32 @@ guard:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
|
||||
# Central allow-list of models any profiles: entry may name. Nothing checked a profile's model:
|
||||
# value before this block existed — it was a free-form string handed straight to the backend
|
||||
# adapter, and a withdrawn or misspelled name failed silently instead of at config load (opencode
|
||||
# falls back to a default model rather than erroring on an unknown -m).
|
||||
#
|
||||
# Absent, or present with an empty allow:, is OFF: no profile's model: is checked, exactly like
|
||||
# before this block existed. fleetd.yaml is gitignored on every host, so an upgrade must not force
|
||||
# every operator to enumerate their models before the daemon will start.
|
||||
#
|
||||
# The list is the authority; profiles: is checked against it, never the reverse — adding or
|
||||
# editing a profiles: entry cannot, by itself, widen what is permitted here.
|
||||
#
|
||||
# Enforcement is at CONFIG LOAD only (a bad model: fails the daemon at startup, naming both the
|
||||
# model and the profile). There is no spawn-time enforcement, no runtime on/off switch, and no
|
||||
# interaction with BackendQuarantine — those are separate, later units.
|
||||
#
|
||||
# allow → the permitted models. Each entry is its own block (not a bare string) so a later unit
|
||||
# can add an on/off state or a load limit per model without changing this shape.
|
||||
# model → the model id exactly as a profiles: entry's model: field would write it. One flat,
|
||||
# opaque-string namespace: a bare Claude id (claude-sonnet-5) and an opencode
|
||||
# provider-prefixed id (openai/gpt-5.6-terra) both fit here unchanged — the check is a
|
||||
# plain string match, never a parse of the provider prefix or a branch on kind:.
|
||||
# models:
|
||||
# allow:
|
||||
# - model: claude-sonnet-5
|
||||
# - model: claude-opus-5
|
||||
# - model: openai/gpt-5.6-terra
|
||||
# - model: amazon.nova-pro-v1:0
|
||||
@@ -28,6 +28,8 @@
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
<archunit.version>1.5.0</archunit.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -44,6 +46,12 @@
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
|
||||
fleetd #206: org.xerial:sqlite-jdbc 3.53.4.0 (added for OpenCodeSessionDiscovery) — the
|
||||
only known advisory against this artifact is CVE-2023-32697 (RCE via an attacker-controlled
|
||||
JDBC URL), fixed in 3.41.2.2; 3.53.4.0 is well past that fix and OSV.dev reports no open
|
||||
advisory against it. Checked via the OSV.dev API (no Mend.io/JetBrains IDE MCP mount
|
||||
available from this worktree) on 2026-08-31.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
@@ -106,7 +114,7 @@
|
||||
</dependency>
|
||||
|
||||
<!-- MCP server: the SERVER face. Streamable-HTTP servlet mounted on Javalin's Jetty at
|
||||
/mcp, exposing bridge_send/bridge_reply/bridge_status as thin adapters over REST. -->
|
||||
/mcp, exposing fleet_send/fleet_reply/fleet_status as thin adapters over REST. -->
|
||||
<dependency>
|
||||
<groupId>io.modelcontextprotocol.sdk</groupId>
|
||||
<artifactId>mcp</artifactId>
|
||||
@@ -123,6 +131,17 @@
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #206: opencode moved its session store from a JSON tree to SQLite
|
||||
(opencode.db). This is the JDBC driver OpenCodeSessionDiscovery uses to read it
|
||||
read-only. Ships bundled native libraries (linux/mac/windows, several archs), so it
|
||||
is a heavier jar than most deps here — see the pom's dependency-security note below
|
||||
for the size/CVE tradeoff actually measured. -->
|
||||
<dependency>
|
||||
<groupId>org.xerial</groupId>
|
||||
<artifactId>sqlite-jdbc</artifactId>
|
||||
<version>${sqlite-jdbc.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -158,13 +177,21 @@
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #131: package-boundary and cycle enforcement (PackageCyclesTest). -->
|
||||
<dependency>
|
||||
<groupId>com.tngtech.archunit</groupId>
|
||||
<artifactId>archunit-junit5</artifactId>
|
||||
<version>${archunit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
<!-- CB-632: stays 'bridged' until the cutover renames the module dir and the launchd
|
||||
plist together. The installed plist names bridged/target/bridged.jar and
|
||||
KeepAlive is armed, so renaming the jar alone strands a restart. -->
|
||||
<finalName>bridged</finalName>
|
||||
<!-- CB-634: the cutover renamed the module dir (bridged/ -> fleetd/), the jar, and the
|
||||
launchd plist together. The installed plist names fleetd/target/fleetd.jar and
|
||||
KeepAlive is armed, so this name, the plist, and the wrapper must move as one. -->
|
||||
<finalName>fleetd</finalName>
|
||||
<plugins>
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
@@ -206,7 +233,7 @@
|
||||
</configuration>
|
||||
</plugin>
|
||||
|
||||
<!-- Runnable fat jar: java -jar target/bridged.jar -->
|
||||
<!-- Runnable fat jar: java -jar target/fleetd.jar -->
|
||||
<plugin>
|
||||
<groupId>org.apache.maven.plugins</groupId>
|
||||
<artifactId>maven-shade-plugin</artifactId>
|
||||
File diff suppressed because it is too large
Load Diff
+26
-2
@@ -30,8 +30,27 @@ public final class Authz {
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
READ,
|
||||
/**
|
||||
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
||||
*
|
||||
* <p>Deliberately <strong>not</strong> folded into {@link #READ}. {@code READ}'s grant
|
||||
* rests on "the roster carries no secrets" (see its case below) — a lead-to-lead body is
|
||||
* not the roster; it is where leads discuss host shapes, credentials and unmerged work.
|
||||
* Mapping this to {@code READ} would let any worker read every peer lead's mail in full
|
||||
* and would silently falsify that comment for every other {@code READ} caller.
|
||||
*/
|
||||
COORD_READ,
|
||||
/** Scrape the metrics endpoint. */
|
||||
METRICS
|
||||
METRICS,
|
||||
/**
|
||||
* Drive the lead-rollover executor ({@code fleet_handover}: open/confirm/cancel a
|
||||
* self-replace, fleetd #480 Unit C). Primary-only, same as {@link #SPAWN}/{@link #STOP}/
|
||||
* {@link #DRAIN} — and, unlike those, the terminal it acts on is never even an argument:
|
||||
* {@code LeadRollover#open}/{@code #confirm} are always called with the CALLER's own
|
||||
* connection-resolved terminal (see {@code dev.ltms.fleet.lead.LeadRollover}'s class
|
||||
* javadoc, fleetd #480 correction 2), so a primary can only ever roll itself.
|
||||
*/
|
||||
HANDOVER
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -50,7 +69,7 @@ public final class Authz {
|
||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
||||
// escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN -> caller.isPrimary();
|
||||
case SPAWN, STOP, DRAIN, HANDOVER -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
@@ -68,6 +87,11 @@ public final class Authz {
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
|
||||
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
||||
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
||||
// is coordination between leads, not observation of the roster.
|
||||
case COORD_READ -> caller.isPrimary();
|
||||
};
|
||||
}
|
||||
|
||||
+26
-6
@@ -236,7 +236,23 @@ public final class CallerResolver {
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
//
|
||||
// fleetd #317: "not a worker" must not be conflated with "identity unresolved". The real
|
||||
// primary is a real process — its pid resolves (c.resolved()), it just owns no herdr pane.
|
||||
// A caller whose peer-PID lookup failed (LsofPeerPidLookup's -1 sentinel — on any failure,
|
||||
// silently including "lsof found no match") has no such pid, and PaneLocator's own javadoc
|
||||
// already names what happens if that case is handed the primary role: a worker→primary
|
||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||
//
|
||||
// fleetd #505: the OTHER way a real pid can wrongly reach here with a null terminal — not a
|
||||
// failed lsof lookup, but a herdr error partway through PaneLocator's pane scan. c.resolved()
|
||||
// says nothing about that; it only tests the lsof sentinel (by design — see
|
||||
// ConnectionIdentity.Caller#resolved). c.scanComplete() is the separate signal: a scan that
|
||||
// could not check every pane must not be read as "checked everywhere, no match" — the pane it
|
||||
// could not check might have been the caller's own. So both must hold before this promotes.
|
||||
return isLoopback(remoteAddr) && c.resolved() && c.scanComplete()
|
||||
? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
@@ -262,11 +278,15 @@ public final class CallerResolver {
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #305: delegates to {@link ConnectionIdentity#isLoopback}. This used to be a second,
|
||||
* independent copy of the same rule, and the two drifted: this one accepted all of
|
||||
* {@code 127.0.0.0/8}, {@code ConnectionIdentity}'s accepted only {@code 127.0.0.1}. A caller
|
||||
* from {@code 127.0.0.2} therefore had its identity skipped (so it had no terminal) and was
|
||||
* then read as loopback here — which under loopback-trust is the primary. Sharing the inputs
|
||||
* would not have prevented that; only sharing the computation does.
|
||||
*/
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
return ConnectionIdentity.isLoopback(remoteAddr);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,74 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
/** A slot held before a member process starts. */
|
||||
record SlotReservation(String slot, String profile) { }
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return role; // no registry configured — nothing to bind against, so the request stands
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
// no registry configured — nothing to validate against, so nothing is refused
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||
*
|
||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||
* normal operation this fallback should not happen once a reservation has been bound.
|
||||
* It remains the honest answer if a caller has no reservation, or if binding a
|
||||
* reservation unexpectedly fails.
|
||||
*/
|
||||
MemberRole acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
|
||||
/**
|
||||
* Refuse an acquire before anything spawns when {@code role} requires a live slot binding and
|
||||
* no configured slot carries {@code profile} (CB-619 / fleetd #123). A no-op for a role with
|
||||
* no slot-binding semantics.
|
||||
*
|
||||
* @throws IllegalArgumentException naming the role, the profile, and the pools that do carry it
|
||||
*/
|
||||
void requireSlotFor(MemberRole role, String profile);
|
||||
|
||||
/** Reserve a matching slot before launch, or refuse before a charter can be delivered. */
|
||||
SlotReservation reserve(MemberRole role, String profile);
|
||||
|
||||
/** Convert a reservation into a live terminal binding. */
|
||||
boolean bind(SlotReservation reservation, String terminal);
|
||||
|
||||
/** Return an unbound reservation after a failed launch. */
|
||||
void release(SlotReservation reservation);
|
||||
}
|
||||
@@ -0,0 +1,410 @@
|
||||
package dev.ltms.fleet.auth;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
* profile it points at, plus the <em>live</em> bindings from a live architect's herdr terminal to
|
||||
* its slot.
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/{@code reviewers}
|
||||
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
||||
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
||||
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
||||
* {@code CompositePeerLauncher}'s (see {@code ConfigRef}'s class doc) — so a config reload
|
||||
* that removes or adds an architect slot governs the <em>next</em> spawn with no restart.
|
||||
* Only {@link #MemberRegistry(FleetConfig.Fleet)} freezes the pool at construction, and that
|
||||
* constructor exists for tests and for the (rare) case of wiring a fixed, code-built config.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry, initially <em>empty</em>, and
|
||||
* <strong>never</strong> touched by a reload. Config declares no architect terminal, so at
|
||||
* startup every slot is idle and nothing resolves to an architect; a session only becomes one
|
||||
* when the spawn lifecycle {@linkplain #bind(String, String) binds} its terminal to a slot.
|
||||
* {@link CallerResolver} reads this through {@link #snapshot()} to turn a pane into an
|
||||
* {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The binding rule (fleetd #424): config governs what a bound slot still grants, as
|
||||
* well as what may be bound next.</strong> Removing a slot from config revokes it — that is the
|
||||
* ticket's entire point ("Revoking an architect slot does not revoke it"). Revoking it means an
|
||||
* architect already bound to that slot loses the ARCHITECT privilege on its very next request:
|
||||
* {@link #roleForSlot} and {@link #nameForSlot} read {@link #slots()} directly, with no cache, so
|
||||
* the moment a slot drops out of config, {@link CallerResolver#resolve} (which calls both on every
|
||||
* request from a bound pane, {@code CallerResolver.java:220}) can no longer confirm the pane's slot
|
||||
* is an architect slot, and the pane falls through to {@code Principal.worker(...)}. What does
|
||||
* <em>not</em> change is the {@code terminalToSlot} <em>occupancy</em> — the binding created by
|
||||
* {@link #bind} is untouched by a reload, on purpose: unbinding it here would double-book the slot
|
||||
* key (a second terminal could then bind to the "freed" key while the first is still the terminal
|
||||
* the operator actually meant to demote) and would silently break {@link #unbind}'s compare-safe
|
||||
* contract, which needs the original {@code terminal → slot} pair intact to remove it cleanly. So
|
||||
* the demoted session keeps occupying its slot — {@link #slotForTerminal} and {@link #snapshot()}
|
||||
* still name it — it just no longer resolves as an architect through that occupancy, and a fresh
|
||||
* spawn still cannot bind to the same key while it is occupied ({@link #reserve}/
|
||||
* {@link #requireSlotFor} refuse it anyway, since it is gone from {@link #slots()}). The demoted
|
||||
* session's own turn is unaffected: {@code fleet_reply}'s authorization
|
||||
* ({@code Authz.Action.REPLY}) is {@code caller.ownsSession(targetSession)} — identity by terminal,
|
||||
* not by role — so a demoted architect can still end its own turn normally.
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
*/
|
||||
public final class MemberRegistry implements MemberLifecycle {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(MemberRegistry.class);
|
||||
|
||||
/**
|
||||
* One flattened {@code fleet:} entry.
|
||||
*
|
||||
* <p>Flattened because a slot name is unique only <em>within</em> its pool — {@code sonnet} may
|
||||
* legitimately be both a developer and a reviewer — while a terminal binds to exactly one thing.
|
||||
* The qualified {@link #key()} is what that binding uses.
|
||||
*
|
||||
* @param name the slot's key inside its pool
|
||||
* @param role the pool it came from
|
||||
* @param profile the backend it runs on
|
||||
*/
|
||||
public record Entry(String name, MemberRole role, String profile) {
|
||||
/** {@code "architect:opus"} — unique across pools, unlike {@link #name()}. */
|
||||
public String key() {
|
||||
return role.wireName() + ":" + name;
|
||||
}
|
||||
}
|
||||
|
||||
private final Supplier<FleetConfig.Fleet> fleet;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/**
|
||||
* Freeze the pool at construction — for tests, and for the rare case of wiring a fixed,
|
||||
* code-built config. Production wiring should prefer {@link #live}, which re-reads
|
||||
* {@code fleet:} on every call.
|
||||
*/
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
this(() -> fleet);
|
||||
}
|
||||
|
||||
private MemberRegistry(Supplier<FleetConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
}
|
||||
|
||||
/**
|
||||
* Live variant (fleetd #424): {@code fleet} is read fresh on every {@link #slots()} call — pass
|
||||
* {@code () -> config.get().fleet()}, the same supplier shape {@code CompositePeerLauncher}
|
||||
* already uses for placement — so a reload that adds or removes an architect slot governs the
|
||||
* next spawn's {@link #reserve}/{@link #requireSlotFor} check with no restart. A separate,
|
||||
* private constructor rather than a same-arity public overload of
|
||||
* {@link #MemberRegistry(FleetConfig.Fleet)}: a {@code FleetConfig.Fleet} and a
|
||||
* {@code Supplier<FleetConfig.Fleet>} overload are ambiguous for a literal {@code null} — the
|
||||
* same reason {@code CallerResolver.withLeads} is a static factory rather than a fourth
|
||||
* constructor overload.
|
||||
*/
|
||||
public static MemberRegistry live(Supplier<FleetConfig.Fleet> fleet) {
|
||||
return new MemberRegistry(Objects.requireNonNull(fleet, "fleet"));
|
||||
}
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one map. Leaders are not members. */
|
||||
private static Map<String, Entry> flatten(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
fleet.pool(role).forEach((name, slot) -> {
|
||||
if (slot != null) {
|
||||
Entry e = new Entry(name, role, slot.profile());
|
||||
flat.put(e.key(), e);
|
||||
}
|
||||
});
|
||||
}
|
||||
}
|
||||
return Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/**
|
||||
* The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot of
|
||||
* {@code fleet:} <em>as of this call</em> — see the class doc for which constructor makes that
|
||||
* live versus frozen.
|
||||
*/
|
||||
public Map<String, Entry> slots() {
|
||||
return flatten(fleet.get());
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order, as of this call. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots().forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
});
|
||||
return Collections.unmodifiableMap(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* An immutable copy of the live {@code terminal_id → slot name} bindings.
|
||||
*
|
||||
* <p>Passed to {@link CallerResolver} as the source of architect identity, and what
|
||||
* {@code fleet_whoami}/the roster will read to say which slot a pane hosts. Empty until the
|
||||
* spawn lifecycle binds a slot.
|
||||
*/
|
||||
public Map<String, String> snapshot() {
|
||||
synchronized (terminalToSlot) {
|
||||
return Map.copyOf(terminalToSlot);
|
||||
}
|
||||
}
|
||||
|
||||
/** The slot a live terminal is bound to, or {@code null} if it is not an architect slot. */
|
||||
public String slotForTerminal(String terminal) {
|
||||
if (terminal == null) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
return terminalToSlot.get(terminal);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under, as of this call.
|
||||
*
|
||||
* <p>Nothing in {@code src/main} calls this (fleetd #431 — grepped both the {@code
|
||||
* .profileForSlot(} and the {@code ::profileForSlot} form). This javadoc used to say "what the
|
||||
* spawn lifecycle reads", and that seam does not exist: the spawn lifecycle takes its profile
|
||||
* from the {@link MemberLifecycle.SlotReservation} that {@code reserve} returns, never from
|
||||
* here. Kept and pinned rather than deleted because it is the natural accessor for that seam
|
||||
* if one is added; live for the same reason as {@link #roleForSlot}, so a reload cannot leave
|
||||
* it answering for the old config.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots().get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/**
|
||||
* The role a qualified slot key belongs to, or {@code null} when the key is not currently
|
||||
* configured. Deliberately live, with no cache (fleetd #424, see the class doc's binding rule):
|
||||
* removing a slot from config must make {@link CallerResolver#resolve} stop granting the
|
||||
* ARCHITECT role for it on the very next request from a terminal that was bound to it, which is
|
||||
* the ticket's whole point — revoking a slot must actually revoke it, not just refuse the next
|
||||
* spawn.
|
||||
*/
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/**
|
||||
* The unqualified configured name for a slot, or {@code null} if it is not currently configured.
|
||||
* Live for the same reason as {@link #roleForSlot} — see the class doc's binding rule.
|
||||
*/
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots().containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind {@code terminal} to {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it stands a slot up. The bind is atomic and preserves
|
||||
* the two cardinality invariants: a terminal may occupy at most one slot, and a slot may host at
|
||||
* most one terminal. Binding the same terminal to the same slot again is a harmless no-op.
|
||||
*
|
||||
* @param slot a configured slot name, or the bind is refused
|
||||
* @param terminal the pane that will act as this architect
|
||||
* @return {@code true} if the binding is now {@code terminal → slot}; {@code false} if it was
|
||||
* refused — an unknown slot, a terminal already bound to a different slot, or a slot
|
||||
* already hosting a different terminal
|
||||
*/
|
||||
public boolean bind(String slot, String terminal) {
|
||||
if (slot == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!isSlot(slot)) {
|
||||
return false; // unknown slot — nothing to bind to
|
||||
}
|
||||
String existingSlot = terminalToSlot.get(terminal);
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot) || reservedSlots.contains(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Compare-safe unbind of {@code expectedTerminal} from {@code slot} (CB-548).
|
||||
*
|
||||
* <p>The spawn lifecycle calls this when it tears a slot down. Only the exact binding
|
||||
* {@code expectedTerminal → slot} is removed; if that terminal was since rebound to a different
|
||||
* slot (or the slot to a different terminal), the call is a no-op returning {@code false} — a
|
||||
* stale unbind must never remove a replacement.
|
||||
*
|
||||
* @param slot the slot the caller believes the terminal is bound to
|
||||
* @param expectedTerminal the terminal it expects to be bound there
|
||||
* @return {@code true} if {@code expectedTerminal → slot} was removed; {@code false} if nothing
|
||||
* was (no such binding, or the binding had already moved)
|
||||
*/
|
||||
public boolean unbind(String slot, String expectedTerminal) {
|
||||
if (slot == null || expectedTerminal == null) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
String current = terminalToSlot.get(expectedTerminal);
|
||||
if (current == null || !slot.equals(current)) {
|
||||
return false; // absent, or a replacement/moved binding — leave it in place
|
||||
}
|
||||
terminalToSlot.remove(expectedTerminal);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Bind only architect sessions to a free slot with the resolved profile.
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*
|
||||
* <p>CB-619 / fleetd #123: the return value is the role this session actually holds, and the
|
||||
* caller is required to record THAT — never the requested {@code role} — on the session. Before
|
||||
* this fix the caller kept the requested role regardless of whether the bind below succeeded, so
|
||||
* a demoted session's {@code GET /members} row still said {@code "architect"} while
|
||||
* {@code fleet_whoami} (which reads the live binding, not the request) correctly said
|
||||
* {@code "worker"} — three sources of truth that disagreed about one live member, silently.
|
||||
*/
|
||||
@Override
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return role;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return MemberRole.ARCHITECT;
|
||||
}
|
||||
}
|
||||
// fleetd #123: at least WARN — a role downgrade that the roster must now also reflect is
|
||||
// not routine bookkeeping. requireSlotFor already refuses the config-gap case (no slot at
|
||||
// all carries this profile) before a process ever spawns; reaching here means the config DID
|
||||
// carry a matching slot but every one of them was already bound to a different terminal — a
|
||||
// race this pre-spawn check cannot close on its own (see requireSlotFor's javadoc).
|
||||
log.warn("member slot: no free architect slot for profile={} terminal={}; holding the session "
|
||||
+ "as {} instead of the architect it asked for — every configured slot for this "
|
||||
+ "profile is already bound to a different terminal", profile, terminal,
|
||||
MemberRole.DEV.wireName());
|
||||
return MemberRole.DEV;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||
* documented operator override, not a defect.
|
||||
*
|
||||
* <p>This closes the config-gap case, not the live-capacity case: a profile that DOES carry a
|
||||
* slot can still lose the race to a concurrent spawn between this check and the actual
|
||||
* {@link #bind}, which is why {@link #acquired} must still answer honestly even after this
|
||||
* check has passed.
|
||||
*/
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return;
|
||||
}
|
||||
boolean hasSlot = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.anyMatch(e -> Objects.equals(profile, e.profile()));
|
||||
if (hasSlot) {
|
||||
return;
|
||||
}
|
||||
List<String> pools = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.map(Entry::profile)
|
||||
.distinct()
|
||||
.toList();
|
||||
throw new IllegalArgumentException(
|
||||
"no " + role.wireName() + " slot for profile '" + profile + "' — an architect's "
|
||||
+ "identity IS the slot it is bound to, so there is nothing to bind this "
|
||||
+ "session's identity to. fleet." + role.configKey() + " carries profiles: "
|
||||
+ (pools.isEmpty() ? "(none configured)" : String.join(", ", pools))
|
||||
+ "; add profile '" + profile + "' there, or spawn " + role.wireName()
|
||||
+ " on one of those profiles instead");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if ((profile == null || profile.isBlank() || Objects.equals(profile, entry.profile()))
|
||||
&& !terminalToSlot.containsValue(entry.key()) && reservedSlots.add(entry.key())) {
|
||||
return new SlotReservation(entry.key(), entry.profile());
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("no free architect slot for profile '" + profile
|
||||
+ "' — every matching slot is already bound or reserved");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
if (reservation == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!reservedSlots.remove(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
if (!isSlot(reservation.slot()) || terminalToSlot.containsKey(terminal)
|
||||
|| terminalToSlot.containsValue(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
terminalToSlot.put(terminal, reservation.slot());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
if (reservation != null) {
|
||||
synchronized (terminalToSlot) {
|
||||
reservedSlots.remove(reservation.slot());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
String slot = slotForTerminal(terminal);
|
||||
if (slot != null) {
|
||||
unbind(slot, terminal);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,725 @@
|
||||
package dev.ltms.fleet.config;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The daemon's live configuration, re-readable without a restart (CB-559).
|
||||
*
|
||||
* <p>Consumers hold this, not a {@link FleetConfig}, and read through {@link #get()} at the point
|
||||
* of use. A component that captures {@code ref.get()} into a field at construction has opted out of
|
||||
* reload — which is sometimes right (see <em>deferred</em> below), but it must then be a deliberate
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into four classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
||||
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
||||
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
||||
* reads it live for placement (which profile an unqualified architect spawn may land on), and
|
||||
* {@code MemberRegistry} separately reads it live, through its own instance of the same
|
||||
* supplier shape, for identity — both which slot a spawn may bind to <em>and</em> what a slot
|
||||
* already bound still grants. Removing an architect slot from config therefore revokes the
|
||||
* {@link dev.ltms.fleet.auth.Role#ARCHITECT} role on the bound pane's very next request; only
|
||||
* the slot <em>occupancy</em> survives, so the demoted session still holds its slot key until
|
||||
* it unbinds. See {@code MemberRegistry}'s class doc for that binding rule.
|
||||
* <strong>But {@code fleet:} as a whole is NOT in this class</strong>: {@code fleet.leaders}
|
||||
* inside the same key is frozen, which is exactly what makes {@code fleet:} split rather than
|
||||
* hot — see below. {@code models:} (fleetd #422) joined this class whole: {@link
|
||||
* FleetConfig#validateModels()} re-runs fully against the fresh config on every {@link
|
||||
* #reload()} (via {@link FleetConfig#validateAll()}), refusing a bad edit outright rather than
|
||||
* caching a stale copy anywhere, and the on/off half added by fleetd #422 is read live both by
|
||||
* {@code CompositePeerLauncher}'s spawn gate ({@code enforceModelEnabled} and its candidate
|
||||
* filter) and by {@code fleet_profiles}/{@code GET /profiles} (via
|
||||
* {@code PeerLauncher.disabledModels()}). Nothing about {@code models:} is baked into an
|
||||
* object built at startup, so — unlike the deferred keys below — there is no frozen half left
|
||||
* to report; it moved here from deferred rather than joining split. An existing profile's
|
||||
* {@code exhaustedPattern} (fleetd #446) joined this class the same way: it used to be
|
||||
* compiled once into {@code Fleetd.main}'s startup pattern map (see the Deferred bullet's old
|
||||
* wording, and {@code LiveExhaustedPatterns}'s class doc for the history), and is now read
|
||||
* live, cached by profile name, by both {@code CompletionResolver}'s classification (via
|
||||
* {@code LiveExhaustedPatterns.patternFor}) and {@code fleet_profiles}'s {@code
|
||||
* exhaustionDetectionArmed} (via {@code LiveExhaustedPatterns.armed}) — the one live object
|
||||
* both read, so a reload that arms or disarms a profile's usage-limit detection takes effect
|
||||
* on the next check with no restart. {@code errorPattern}, {@code exhaustedPattern}'s sibling
|
||||
* key for backend-error (not usage-limit) classification, was deliberately left OUT of this
|
||||
* fleetd #446 change and stays deferred below — the ticket scoped it out explicitly.
|
||||
* {@code leadRollover:} (fleetd #480) joined this class whole, the same shape as
|
||||
* {@code models:} above: {@code dev.ltms.fleet.lead.LeadRollover} holds a
|
||||
* {@code Supplier<FleetConfig.LeadRollover>} (the same {@code () -> config.get().x()} shape)
|
||||
* and reads {@code handoverPath}/{@code requireOperatorConfirm}/{@code maxDocAgeSeconds}/
|
||||
* {@code turnSettleSeconds}/{@code clearSettleSeconds}/{@code bootstrapText} fresh on every
|
||||
* {@code open()}/{@code confirm()} call (and on the deferred post-{@code confirm()}
|
||||
* continuation fleetd #480's correction added — see {@code LeadRollover}'s class doc) rather
|
||||
* than capturing them into fields at construction — unlike its closest
|
||||
* structural cousin {@code leadHeartbeat:}, whose {@code LeadHeartbeatLoop} bakes
|
||||
* {@code idleAfterNanos}/{@code backoffMs}/{@code quietNudgeCap} into final fields. The one
|
||||
* restart-only edge is structural, not a stale value: {@code Fleetd.java} decides whether to
|
||||
* construct the {@code LeadRollover} object at all off the startup snapshot (the same
|
||||
* presence gate {@code leadHeartbeat:} uses), so a block ADDED where it was absent at boot
|
||||
* needs a restart before anything exists to call — the same fact already true of adding a
|
||||
* brand-new {@code profiles:} entry.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code idleSleepGuard:} ({@code Fleetd.java} reads it once, at startup, to decide whether
|
||||
* to construct an {@code IdleSleepGuard} and wire {@code SessionManager}'s
|
||||
* {@code onAcquire}/{@code onRelease} hooks to it — neither is rebuilt on reload, so a
|
||||
* running daemon keeps whatever this was at startup regardless of a later edit),
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, {@code worktreeGroup:} and {@code memberSkills:}
|
||||
* (all three of the latter baked once into the {@code GitWorktrees} built at
|
||||
* {@code Fleetd.java:251} and never rebuilt — fleetd #323 instance 2 found
|
||||
* {@code worktreeGroup} missing from this list and from {@link #changedDeferredKeys};
|
||||
* {@code memberSkills} (fleetd #362) followed the same shape), {@code primary:} (fleetd #326 — {@code Fleetd.java:506, 519,
|
||||
* 520} read {@code cfg.primary()} only off the startup snapshot to build {@code
|
||||
* PrimaryRegistry} and size {@code ReplyPushLoop}'s reminder cap/backoff, and neither is
|
||||
* rebuilt on reload. Say the consequence exactly: {@code primary.terminal} is DEPRECATED
|
||||
* (CB-532, and {@code Fleetd.java:511} warns about it at startup) — a lead's identity comes
|
||||
* from {@code leaders:}/{@code leadScan:}, so changing this pin does not demote or promote a
|
||||
* lead that uses those. What a changed pin still does not take effect on until a restart is
|
||||
* the fallback nudge destination the pin remains, the deprecated identity path for an operator
|
||||
* who still relies on it, and {@code pushReminders}/{@code pushBackoffMs}), {@code configReload:} (fleetd #326 — {@code
|
||||
* Fleetd.java:679-680} read it only at startup to decide whether to build a {@code
|
||||
* ConfigWatcher} at all and with what interval; the watcher that would apply a later change is
|
||||
* itself built once, so a running watcher keeps polling on its original enabled flag and
|
||||
* interval regardless of what a reload changes it to, the same shape as {@code lifecycle} —
|
||||
* not cold, because no already-open resource goes inconsistent with the new value, the watcher
|
||||
* (if any) simply keeps its old settings), adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code errorPattern} (fleetd #201 Unit 5 — compiled once into
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup; deliberately NOT made hot
|
||||
* alongside {@code exhaustedPattern} by fleetd #446 — that ticket scoped {@code errorPattern}
|
||||
* and cooling-off out explicitly),
|
||||
* {@code ideProjectDir} / {@code ideOpenCommand} / {@code autoCompactWindow} (fleetd #323
|
||||
* instance 1 — all three are read at spawn off the same frozen profile map and were missing
|
||||
* from {@link #sameLaunchSettings}), and the rest of {@link #sameLaunchSettings}.
|
||||
* {@code credentialId} (CB-578 stage B) and {@code exhaustedPattern} (fleetd #446) are NOT on
|
||||
* this list — both are read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so both are hot instead
|
||||
* (see the Hot bullet above for {@code exhaustedPattern}'s history).
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Split</strong> (fleetd #330; extended to a third key by fleetd #333) — read
|
||||
* <em>both</em> ways at different sites, so the key does not fit any class above as a whole:
|
||||
* {@code health:}, {@code coordinator:} and {@code fleet:}. Each is read off the startup
|
||||
* snapshot to build a long-lived object, and read live off {@link #get()} at a different,
|
||||
* unrelated site — so half of a reload's effect already applies while the other half waits
|
||||
* for a restart, and a bare "config reloaded" would under-claim by exactly that half.
|
||||
* <ul>
|
||||
* <li>{@code health:} — the monitor itself ({@code enabled}, {@code intervalSeconds},
|
||||
* {@code workingSuspectAfterSeconds}) is built once at {@code Fleetd.java:556-563}
|
||||
* and never rebuilt, so a changed value needs a restart to actually start, stop, or
|
||||
* retime it. The coverage string {@code fleet_profiles} reports
|
||||
* ({@code Fleetd.java:648-650}) is read live off {@link #get()} on every call, so it
|
||||
* already reflects the new value.</li>
|
||||
* <li>{@code coordinator:} — the {@code LeadMailbox} connection ({@code uri},
|
||||
* {@code uriEnv}, {@code selfId}, {@code prefetch}) is opened once at
|
||||
* {@code Fleetd.java:502} and never reopened, so a changed value needs a restart —
|
||||
* {@code selfId} in particular names this daemon's own AMQP inbox queue, and a peer
|
||||
* lead that learned the old name would not discover a new one on its own. The broker
|
||||
* URI env-var <em>name</em> that {@code MemberEnvAllowList} keeps out of a member's
|
||||
* environment is read live off {@link #get()} on every spawn
|
||||
* ({@code HerdrPeerLauncher.java:1530}), so it already applies.</li>
|
||||
* <li>{@code fleet:} (fleetd #333) — {@code fleet.leaders} is the frozen half:
|
||||
* {@code Fleetd.java:281} reads {@code cfg.fleet().leaders()} off the startup snapshot
|
||||
* for two long-lived objects built right after it and never rebuilt — the
|
||||
* {@code LeadTabScanner}'s {@code tab label → lead name} map ({@code Fleetd.java:301},
|
||||
* wired into {@code CallerResolver.withLeadsAndMembers} at {@code Fleetd.java:620/624},
|
||||
* which is how a caller's pane is recognised as a lead at all) and, when herdr answered
|
||||
* at startup, {@code LeadLauncher(...).ensureLeads()} ({@code Fleetd.java:315}), which
|
||||
* auto-launches each declared lead up to its {@code instances} count. So a lead added,
|
||||
* removed, or given a new {@code tab:} label under {@code fleet.leaders} needs a
|
||||
* restart — until then it is invisible to identity resolution, and this is exactly the
|
||||
* scenario fleetd #333 named: an operator edits a lead's {@code tab:} to match a
|
||||
* renamed pane, sees "config reloaded", and the pane keeps resolving as a worker,
|
||||
* because {@code CallerResolver} is still matching against the old label. The live
|
||||
* half is the rest of {@code fleet:} — {@code architects}/{@code developers}/
|
||||
* {@code reviewers}, {@code charters}, {@code tabLabel} — read live through the same
|
||||
* {@code CompositePeerLauncher} supplier the Hot bullet above names, so a reload that
|
||||
* only touches those already applies with nothing to report. Because the hot and frozen
|
||||
* halves of {@code fleet:} are disjoint sub-fields rather than the same fields read two
|
||||
* ways (contrast {@code coordinator.uriEnv} above), {@link #changedSplitKeys} compares
|
||||
* {@code fleet.leaders} alone, not the whole {@code Fleet} record — comparing the whole
|
||||
* record would report "split" for a {@code tabLabel}-only change that is actually fully
|
||||
* hot, over-claiming in exactly the direction this class exists to avoid under-claiming
|
||||
* in.</li>
|
||||
* </ul>
|
||||
* A split change is still accepted — {@link Outcome#applied()} stays {@code true}, the same
|
||||
* as a deferred change — because the live half genuinely took effect; refusing the whole
|
||||
* reload would leave the operator worse off than today. {@link Outcome#split()} names the
|
||||
* key and says which half is which each time, rather than trying to score "how changed" a
|
||||
* mixed key is or handle "both halves changed in one reload" as a special case.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon. All five of
|
||||
* {@link #COLD_KEYS}: {@code bind:}, {@code herdrSocket:}, {@code memberHerdrSocket:},
|
||||
* {@code broker:} and {@code auth:}. The sockets are already connected, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening. This bullet omitted {@code memberHerdrSocket:} until fleetd #333 — say "all
|
||||
* five of COLD_KEYS" rather than re-listing them, so prose and set cannot drift again.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The denominator, measured on 2026-09-04 (fleetd #330; recounted for fleetd #333);
|
||||
* recounted again for fleetd #362, again after {@code idleSleepGuard:} was added, again after
|
||||
* {@code models:} was added as deferred, again for fleetd #422, which moved {@code models:}
|
||||
* from deferred to hot-excluded once its on/off half was read live everywhere, and again after
|
||||
* {@code leadRollover:} was added (fleetd #480).</strong>
|
||||
* {@code FleetConfig} has 26 top-level record components: 5 cold, 13 deferred, 3 split, 5
|
||||
* hot-excluded. Five of them are named nowhere in this file, and the reason is the same for all
|
||||
* five: {@code placement}, {@code memberCredentials}, {@code memberLoginShell}, {@code models} and
|
||||
* {@code leadRollover} are <strong>hot</strong> and correctly absent — all five are read live off
|
||||
* {@code config.get()} (placement through the {@code CompositePeerLauncher} supplier the Hot bullet
|
||||
* names; {@code memberCredentials}/{@code memberLoginShell} at spawn time, {@code Fleetd.java:198,
|
||||
* 205, 729} and {@code HerdrPeerLauncher#configuredMemberLoginShell}; {@code models} the same way,
|
||||
* through the Hot bullet's {@code models:} paragraph; {@code leadRollover} through the Hot bullet's
|
||||
* {@code leadRollover:} paragraph), so a reload takes effect on the next spawn (or, for
|
||||
* {@code models}, the next reported status; for {@code leadRollover}, the next {@code open()}/
|
||||
* {@code confirm()} call) with no entry needed here.
|
||||
* {@code health} and {@code coordinator} used to be a third kind — <strong>undecided</strong>, not
|
||||
* hot — until fleetd #330 added the <strong>split</strong> class above and gave them a home. A
|
||||
* reload touching either used to report a bare "config reloaded", which under-claimed; now it names
|
||||
* the key and says which half is which. {@code fleet} was the same story in reverse: fleetd #330's
|
||||
* own fact-find named {@code health}/{@code coordinator} as "the complete set of split-shaped keys"
|
||||
* and filed {@code fleet.leaders}'s restart requirement as a documented caveat sitting in the
|
||||
* <strong>hot-excluded</strong> escape hatch instead — correctly documented, but in the one bucket
|
||||
* this file's own coverage test cannot check the truth of (see that test's javadoc). fleetd #333
|
||||
* moved it into <strong>split</strong>, where {@link #changedSplitKeys} actually reports it.
|
||||
* <p>The point of writing the count down: "not mentioned in this file" looks identical for a key
|
||||
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
||||
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
||||
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
||||
* {@link ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
||||
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
||||
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
||||
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
||||
* any reporting code behind it at all — {@code ConfigRefTopLevelReportingCoverageTest} is what
|
||||
* fleetd #333 added for that, after measuring that a {@code SPLIT_KEYS} entry with its reporting
|
||||
* branch deleted passes both this file's own "kept in step" assert and
|
||||
* {@code ConfigRefTopLevelCoverageTest} unchanged. fleetd #337 extended it to {@code DEFERRED_KEYS}
|
||||
* after measuring the same one-way gap there directly: dropping {@code guard}'s branch out of
|
||||
* {@link #changedDeferredKeys} while {@code "guard"} stayed in the set left the whole suite green.
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart. A split change does <em>not</em> refuse, for a different reason than a
|
||||
* deferred change does not: its live half genuinely took effect, so refusing would throw that away
|
||||
* and leave the operator worse off than the partial-but-honest report {@link Outcome#split()} gives.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*
|
||||
* <p><strong>fleetd #474</strong> — {@link FleetConfig#validateAll()} is not the only gate startup
|
||||
* runs before a config takes effect: {@code Fleetd.main} also calls {@code
|
||||
* dev.ltms.fleet.mcp.CharterToolSurface#assertChartersNameOnlyRegisteredTools}, right after {@code
|
||||
* cfg.validateAll()}, to refuse a charter that names an MCP tool the server does not register. That
|
||||
* check cannot live inside {@link FleetConfig} — {@code CharterToolSurface} lives in the {@code mcp}
|
||||
* package because the canonical tool set ({@code FleetTool}) does, and config is loaded before the
|
||||
* MCP server exists, so {@code FleetConfig} must not gain a dependency on {@code mcp}. {@link
|
||||
* #reload} cannot import {@code mcp} either, for the same reason applied one layer up: {@code
|
||||
* dev.ltms.fleet.config} is loaded before {@code dev.ltms.fleet.mcp} exists, same as {@code
|
||||
* FleetConfig}. So this class accepts the check as a {@code Consumer<FleetConfig>} —
|
||||
* {@link #extraValidation} — supplied by whichever caller already sits at the seam that holds both
|
||||
* a loaded {@code FleetConfig} and the {@code mcp} package: {@code Fleetd.main}. It is invoked
|
||||
* inside the same try/catch as {@code fresh.validateAll()}, so a charter that would have refused to
|
||||
* boot refuses a reload too, and keeps the running config exactly like any other {@code
|
||||
* validateAll()} failure. A ref built through the two-argument constructor (every test fixture that
|
||||
* does not care about this check, and {@link #fixed}) gets a no-op consumer, so nothing outside
|
||||
* {@code Fleetd.main} needs to know this hook exists.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/**
|
||||
* Keys that cannot change under a running daemon — see the class doc.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} can fold it
|
||||
* into the top-level triage it checks, the same way it reads {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
/**
|
||||
* Keys read BOTH off the startup snapshot and live off {@link #get()} at different sites, so
|
||||
* neither the hot, deferred nor cold class fits them as a whole — see the class doc's Split
|
||||
* bullet (fleetd #330). A changed split key is accepted ({@link Outcome#applied()} stays
|
||||
* {@code true}) and reported by name, with a message naming which half is live and which needs
|
||||
* a restart.
|
||||
*/
|
||||
static final Set<String> SPLIT_KEYS = Set.of("health", "coordinator", "fleet");
|
||||
|
||||
/**
|
||||
* Top-level keys {@link #changedDeferredKeys} compares — see the class doc's Deferred bullet.
|
||||
* Promoted here from a test-side copy in {@code ConfigRefTopLevelCoverageTest} by fleetd #337,
|
||||
* the same reason {@link #COLD_KEYS} and {@link #SPLIT_KEYS} live here rather than in a test: a
|
||||
* second, hand-maintained copy of this set is exactly the kind of thing that silently drifts
|
||||
* from the method it is supposed to describe. {@code spawnReadyTimeoutMs} and
|
||||
* {@code spawnReadyPollMs} are compared together in one branch and reported under the combined
|
||||
* label {@code "spawnReady*"}; {@code profiles} is compared twice over (added/removed names,
|
||||
* then an existing profile's launch settings) — see {@link #changedDeferredKeys}.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} and
|
||||
* {@code ConfigRefTopLevelReportingCoverageTest} can both read it, the same way they already
|
||||
* read {@link #COLD_KEYS} and {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> DEFERRED_KEYS = Set.of(
|
||||
"guard", "worktreeRoot", "worktreeGroup", "memberSkills", "primary", "configReload",
|
||||
"leadHeartbeat", "lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs",
|
||||
"quarantineCooldownSeconds", "profiles", "idleSleepGuard");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
private final Consumer<FleetConfig> extraValidation;
|
||||
|
||||
/** Equivalent to the three-argument constructor with a no-op {@code extraValidation}. */
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this(path, initial, cfg -> { });
|
||||
}
|
||||
|
||||
/**
|
||||
* @param extraValidation run on every {@link #reload} candidate, inside the same try/catch as
|
||||
* {@code fresh.validateAll()} — see the class doc's fleetd #474 note.
|
||||
* {@code Fleetd.main} passes {@code
|
||||
* Fleetd::assertChartersNameOnlyRegisteredTools} (a package-private
|
||||
* {@code FleetConfig -> void} adapter over {@code
|
||||
* CharterToolSurface#assertChartersNameOnlyRegisteredTools}), so a reload
|
||||
* runs the same gate startup does without this class depending on the
|
||||
* {@code mcp} package.
|
||||
*/
|
||||
public ConfigRef(Path path, FleetConfig initial, Consumer<FleetConfig> extraValidation) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
this.extraValidation = Objects.requireNonNull(extraValidation, "extraValidation");
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
public static ConfigRef fixed(FleetConfig cfg) {
|
||||
return new ConfigRef(null, cfg);
|
||||
}
|
||||
|
||||
/** The live configuration. Read this per use; do not cache it in a field. */
|
||||
@Override
|
||||
public FleetConfig get() {
|
||||
return current.get();
|
||||
}
|
||||
|
||||
/** The file this ref reloads from, or {@code null} for a {@link #fixed} ref. */
|
||||
public Path path() {
|
||||
return path;
|
||||
}
|
||||
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* <p>{@code split} is a separate field from {@code deferred} rather than a differently-worded
|
||||
* entry inside it, because the two carry different guarantees for any caller that branches on
|
||||
* them rather than just printing {@link #summary()}: every {@code deferred} entry means "this
|
||||
* key's whole change waits for a restart", while every {@code split} entry means "part of this
|
||||
* key's change already applied, and the message says which part" — collapsing them would force
|
||||
* a caller to re-parse the message to tell those apart. See the class doc's Split bullet
|
||||
* (fleetd #330) for why the key needs this at all.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits entirely on a
|
||||
* restart
|
||||
* @param split split keys that changed and were accepted, each named with which half of it
|
||||
* is already live and which half waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
List<String> split, String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
split = List.copyOf(split);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
public String summary() {
|
||||
if (error != null) {
|
||||
return "config reload refused — " + error;
|
||||
}
|
||||
if (!applied) {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (deferred.isEmpty() && split.isEmpty()) {
|
||||
return "config reloaded";
|
||||
}
|
||||
StringBuilder out = new StringBuilder("config reloaded");
|
||||
if (!deferred.isEmpty()) {
|
||||
out.append("; these changes need a restart to take effect: ")
|
||||
.append(String.join(", ", deferred));
|
||||
}
|
||||
if (!split.isEmpty()) {
|
||||
out.append("; partially live — ").append(String.join(" | ", split));
|
||||
}
|
||||
return out.toString();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the file, validate it, and swap it in when nothing cold changed.
|
||||
*
|
||||
* <p>Never throws: a reload is a best-effort operation on a daemon that is already serving, and
|
||||
* a bad edit must not take it down. Every failure path leaves the previous config live and is
|
||||
* reported through the returned {@link Outcome}.
|
||||
*/
|
||||
public Outcome reload() {
|
||||
if (path == null) {
|
||||
return Outcome.failed("this config was built in code and has no file to reload from");
|
||||
}
|
||||
FleetConfig old = current.get();
|
||||
FleetConfig fresh;
|
||||
try {
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug. fleetd ticket "central allow-list
|
||||
// of usable models" follow-up: this used to be six individual validateXxx() calls, and
|
||||
// mutation testing found two of the six unpinned here even though startup pinned nothing
|
||||
// at all — see FleetConfig#validateAll's javadoc for why the fix is one reflective call,
|
||||
// not a longer hand-maintained list.
|
||||
fresh.validateAll();
|
||||
// fleetd #474: validateAll() does not cover everything startup refuses on — the charter
|
||||
// tool-surface check (Fleetd.main, right after cfg.validateAll()) lives outside
|
||||
// FleetConfig on purpose (see this class's doc) and is supplied here as extraValidation.
|
||||
// Same try/catch as validateAll() above, on purpose: either failure must refuse the whole
|
||||
// reload and keep the running config the same way.
|
||||
extraValidation.accept(fresh);
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
return Outcome.failed(msg);
|
||||
}
|
||||
|
||||
List<String> cold = changedColdKeys(old, fresh);
|
||||
if (!cold.isEmpty()) {
|
||||
Outcome out = Outcome.refusedCold(cold);
|
||||
log.warn(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
List<String> split = changedSplitKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, split, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cold keys whose value differs between the running config and the candidate.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333).
|
||||
*/
|
||||
static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
}
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.memberHerdrSocket(), fresh.memberHerdrSocket())) {
|
||||
changed.add("memberHerdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
if (!Objects.equals(old.auth(), fresh.auth())) {
|
||||
changed.add("auth");
|
||||
}
|
||||
// Kept in step with COLD_KEYS so the doc and the code cannot drift apart silently.
|
||||
assert COLD_KEYS.containsAll(changed) : "a cold key was reported that COLD_KEYS omits";
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Changed keys that were accepted but whose effect waits for a restart.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #changedColdKeys} and {@link #changedSplitKeys} already are (fleetd #333, extended to
|
||||
* this method by fleetd #337 — membership in {@link #DEFERRED_KEYS} proved nothing about this
|
||||
* method on its own until then; see that test's class doc).
|
||||
*/
|
||||
static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
}
|
||||
if (!Objects.equals(old.leadHeartbeat(), fresh.leadHeartbeat())) {
|
||||
changed.add("leadHeartbeat");
|
||||
}
|
||||
if (!Objects.equals(old.guard(), fresh.guard())) {
|
||||
changed.add("guard");
|
||||
}
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
// Baked into the same GitWorktrees as worktreeRoot (Fleetd.java:251) and never rebuilt
|
||||
// either — see the class doc. Missing this check was fleetd #323 instance 2: a reload
|
||||
// that changed only worktreeGroup reported "config reloaded" with nothing deferred, and
|
||||
// newly provisioned worktrees kept the old sharing behaviour.
|
||||
if (!Objects.equals(old.worktreeGroup(), fresh.worktreeGroup())) {
|
||||
changed.add("worktreeGroup");
|
||||
}
|
||||
// fleetd #362: baked into the same GitWorktrees as worktreeRoot/worktreeGroup
|
||||
// (Fleetd.java:251) and never rebuilt either — a reload that changes only memberSkills
|
||||
// must be reported the same way, or a newly provisioned worktree keeps seeding from (or
|
||||
// skipping) the old source directory with nothing telling the operator why.
|
||||
if (!Objects.equals(old.memberSkills(), fresh.memberSkills())) {
|
||||
changed.add("memberSkills");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:506, 519, 520 read cfg.primary() only off the startup snapshot
|
||||
// (PrimaryRegistry's pinned terminal, ReplyPushLoop's reminder cap and backoff) — neither is
|
||||
// rebuilt on reload, so a changed value needs a restart. Note what it does NOT mean:
|
||||
// primary.terminal is deprecated (CB-532), identity comes from leaders:/leadScan:, so a lead
|
||||
// using those is unaffected by this pin either way. See the class doc for the exact scope.
|
||||
if (!Objects.equals(old.primary(), fresh.primary())) {
|
||||
changed.add("primary");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:679-680 read cfg.configReload() only at startup to decide whether
|
||||
// to build a ConfigWatcher at all and with what interval — the watcher that would apply a
|
||||
// later change is itself built once, so a running watcher keeps its original enabled flag and
|
||||
// interval regardless of what a reload changes it to. Not cold: no already-open resource goes
|
||||
// inconsistent with the new value, a watcher (if any) simply keeps polling on the old settings.
|
||||
if (!Objects.equals(old.configReload(), fresh.configReload())) {
|
||||
changed.add("configReload");
|
||||
}
|
||||
// Fleetd.java reads cfg.idleSleepGuard() once, at startup, to decide whether to construct
|
||||
// an IdleSleepGuard at all and wire SessionManager's onAcquire/onRelease hooks to it —
|
||||
// neither is rebuilt on reload, so a running daemon keeps whatever this was at startup
|
||||
// (armed or not) regardless of a later edit here. Not cold: nothing already-open goes
|
||||
// inconsistent with the new value, an armed-or-not guard just keeps its original answer.
|
||||
if (!Objects.equals(old.idleSleepGuard(), fresh.idleSleepGuard())) {
|
||||
changed.add("idleSleepGuard");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
}
|
||||
// CB-578 stage B: baked once into the BackendQuarantine built at startup — a running
|
||||
// quarantine keeps its original cooldown regardless, and a new cooldown only applies to a
|
||||
// quarantine that starts after a restart.
|
||||
if (!Objects.equals(old.quarantineCooldownSeconds(), fresh.quarantineCooldownSeconds())) {
|
||||
changed.add("quarantineCooldownSeconds");
|
||||
}
|
||||
Map<String, FleetConfig.Profile> before =
|
||||
old.profiles() == null ? Map.of() : old.profiles();
|
||||
Map<String, FleetConfig.Profile> after =
|
||||
fresh.profiles() == null ? Map.of() : fresh.profiles();
|
||||
// Adding or removing a profile is deferred: a new backend needs its own launcher, and
|
||||
// launchers are built once at startup.
|
||||
if (!before.keySet().equals(after.keySet())) {
|
||||
Set<String> diff = new LinkedHashSet<>(before.keySet());
|
||||
diff.addAll(after.keySet());
|
||||
diff.removeIf(p -> before.containsKey(p) && after.containsKey(p));
|
||||
changed.add("profiles (added/removed: " + String.join(", ", diff) + ")");
|
||||
}
|
||||
// An EXISTING profile's launch settings are deferred too, and this is easy to get wrong:
|
||||
// `HerdrPeerLauncher` takes `Map.copyOf(profiles)` at construction and `spawn` resolves the
|
||||
// profile out of that snapshot, so a reloaded model/baseUrl/argv/env never reaches a launch.
|
||||
// Only weight and maxLoad are genuinely hot, because placement reads them through the
|
||||
// supplier on the composite rather than from the adapter's copy. Without this check a
|
||||
// changed model would report "config reloaded" and silently do nothing — the worst outcome
|
||||
// a reload can produce, because the operator has no reason to doubt it.
|
||||
List<String> relaunch = new ArrayList<>();
|
||||
before.forEach((name, was) -> {
|
||||
FleetConfig.Profile now = after.get(name);
|
||||
if (now != null && !sameLaunchSettings(was, now)) {
|
||||
relaunch.add(name);
|
||||
}
|
||||
});
|
||||
if (!relaunch.isEmpty()) {
|
||||
changed.add("profiles." + String.join("/", relaunch) + " launch settings "
|
||||
+ "(model, baseUrl, argv, env, …) — the launcher holds a startup snapshot");
|
||||
}
|
||||
return changed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Split keys whose value differs between the running config and the candidate — see the class
|
||||
* doc's Split bullet (fleetd #330, extended for {@code fleet:} by fleetd #333). Unlike
|
||||
* {@link #changedDeferredKeys}, this does not try to tell which sub-field moved for {@code
|
||||
* health:} or {@code coordinator:}: any change to either gets the same fixed message, because
|
||||
* the message already names both halves every time, so there is no "which half changed"
|
||||
* question left for the caller to answer. {@code fleet:} is different on purpose — see below.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333). That
|
||||
* test exists because membership in {@link #SPLIT_KEYS} proves nothing about this method on its
|
||||
* own: fleetd #333 measured that dropping the {@code coordinator} branch out of this method
|
||||
* while leaving {@code "coordinator"} in {@code SPLIT_KEYS} left the whole suite green except a
|
||||
* hand-written {@code ConfigRefTest} case — neither {@code ConfigRefTopLevelCoverageTest} (it
|
||||
* only reads the set) nor the "kept in step" assert below (it only checks the reported keys are
|
||||
* a SUBSET of {@code SPLIT_KEYS}, never that every {@code SPLIT_KEYS} member has a branch here)
|
||||
* would have caught it.
|
||||
*/
|
||||
static List<String> changedSplitKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.health(), fresh.health())) {
|
||||
changed.add("health: the monitor itself (enabled, interval, workingSuspectAfter) is "
|
||||
+ "frozen at startup and needs a restart; the coverage status fleet_profiles "
|
||||
+ "reports is read live and already applied");
|
||||
}
|
||||
if (!Objects.equals(old.coordinator(), fresh.coordinator())) {
|
||||
changed.add("coordinator: the LeadMailbox connection (uri, uriEnv, selfId, prefetch) is "
|
||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||
+ "member's environment is read live on every spawn and already applied");
|
||||
}
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, reviewers,
|
||||
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
||||
// its consumers, not just the one this comment used to name: CompositePeerLauncher reads it
|
||||
// live for PLACEMENT through the () -> config.get().fleet() supplier named in the class doc's
|
||||
// Hot bullet, and MemberRegistry separately reads it live for IDENTITY (which slot a spawn
|
||||
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
||||
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
||||
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds. Only
|
||||
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
||||
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
||||
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
||||
// LeadLauncher(...).ensureLeads() at Fleetd.java:315, which auto-launches each lead up to its
|
||||
// `instances` count; neither is rebuilt on reload). So this compares fleet.leaders alone, not
|
||||
// the whole Fleet record: comparing the whole record would report "split" for a tabLabel-only
|
||||
// or architects-only change that is actually fully hot, which is the over-claim mirror of the
|
||||
// under-claim bug this class exists to prevent.
|
||||
if (!Objects.equals(leadersOf(old), leadersOf(fresh))) {
|
||||
changed.add("fleet: fleet.leaders (each lead's tab, workspace, cwd, profile and "
|
||||
+ "instances count) is read once at startup to build the LeadTabScanner's "
|
||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||
+ "lead; the rest of fleet: (developers, reviewers, charters, tabLabel) is read "
|
||||
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
||||
+ "live through that same supplier for placement AND through a separate supplier "
|
||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
||||
// member with no branch above at all, only a branch whose message is mis-worded relative to
|
||||
// the set. ConfigRefTopLevelReportingCoverageTest is what proves the former.
|
||||
assert changed.stream().allMatch(m -> SPLIT_KEYS.stream().anyMatch(k -> m.startsWith(k + ":")))
|
||||
: "a split entry was reported that does not start with a SPLIT_KEYS name: " + changed;
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** {@code cfg.fleet().leaders()}, defensively, in case a caller hands in a non-defaulted config. */
|
||||
private static Map<String, FleetConfig.Leader> leadersOf(FleetConfig cfg) {
|
||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link FleetConfig.Profile} record components deliberately left out of
|
||||
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
||||
* the class doc's <em>Hot</em> bullet. {@code weight} and {@code maxLoad} are read live by the
|
||||
* placement policy on every spawn; {@code credentialId} is read live by
|
||||
* {@code CompositePeerLauncher} and the CB-578 stage B exhaustion sink; {@code exhaustedPattern}
|
||||
* (fleetd #446) is read live, cached by profile name, by {@code LiveExhaustedPatterns} — see
|
||||
* that class's doc and the class doc's <em>Hot</em> bullet for the history (it used to be
|
||||
* compared here, deferred, like its sibling {@code errorPattern} still is). Nothing else is
|
||||
* excluded — see {@code sameLaunchSettingsComparesEveryProfileComponentOrExcludesIt} in
|
||||
* {@code ConfigRefProfileCoverageTest}, which enumerates every {@code Profile} record component
|
||||
* by reflection and fails the build if one is neither compared below nor named here.
|
||||
*/
|
||||
static final Set<String> LAUNCH_SETTINGS_EXCLUDED =
|
||||
Set.of("weight", "maxLoad", "credentialId", "exhaustedPattern");
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically.
|
||||
*
|
||||
* <p>This must compare every {@link FleetConfig.Profile} record component except the three in
|
||||
* {@link #LAUNCH_SETTINGS_EXCLUDED}. That is not a claim this javadoc can make good on by
|
||||
* itself — a javadoc saying "compares every component" is exactly what fleetd #323 found to be
|
||||
* false for three fields (and a sibling method's field list, for a fourth). The actual
|
||||
* guarantee comes from {@code ConfigRefProfileCoverageTest}: it enumerates every record
|
||||
* component of {@code FleetConfig.Profile} by reflection, mutates each one not in
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} on a base profile, and asserts this method reports a
|
||||
* difference — so a new component that is neither compared here nor added to
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} (with a reason) fails that test by name, rather than
|
||||
* silently reporting "config reloaded" for a value the daemon never picked up.
|
||||
*/
|
||||
static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.profile(), b.profile())
|
||||
&& Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
&& Objects.equals(a.argv(), b.argv())
|
||||
&& Objects.equals(a.placement(), b.placement())
|
||||
&& Objects.equals(a.workspace(), b.workspace())
|
||||
&& Objects.equals(a.tabLabel(), b.tabLabel())
|
||||
&& Objects.equals(a.mcpUrl(), b.mcpUrl())
|
||||
// CB-634: the IDE MCP mount is a launch flag, fixed at spawn like mcpUrl — a
|
||||
// reload changes it only for members spawned after, so a changed value is deferred.
|
||||
&& Objects.equals(a.ideMcpUrl(), b.ideMcpUrl())
|
||||
&& Objects.equals(a.cwd(), b.cwd())
|
||||
&& Objects.equals(a.parityOverlay(), b.parityOverlay())
|
||||
&& Objects.equals(a.gitTokenEnv(), b.gitTokenEnv())
|
||||
&& Objects.equals(a.gitHostEnv(), b.gitHostEnv())
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// fleetd #446: exhaustedPattern moved to LAUNCH_SETTINGS_EXCLUDED — it is now read
|
||||
// live, cached by profile name, through LiveExhaustedPatterns (see that class's doc
|
||||
// and ConfigRef's class doc Hot bullet), so it must NOT be compared here any more: a
|
||||
// reload that changes only exhaustedPattern must report "config reloaded", not
|
||||
// "these changes need a restart".
|
||||
// fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error
|
||||
// pattern map at startup (see BackendErrorPatternLookup wiring) — unlike its sibling
|
||||
// exhaustedPattern above (fleetd #446), a reload still never re-reads it; fleetd #446
|
||||
// scoped errorPattern out on purpose (see the class doc's Hot bullet).
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern())
|
||||
// fleetd #323 instance 1: ideProjectDir and ideOpenCommand are read at spawn off the
|
||||
// same frozen profile map as ideMcpUrl above (ClaudeCodeLauncher.java:267/269,
|
||||
// OpenCodeLauncher.java:474/480/486) and were missing from this comparison.
|
||||
&& Objects.equals(a.ideProjectDir(), b.ideProjectDir())
|
||||
&& Objects.equals(a.ideOpenCommand(), b.ideOpenCommand())
|
||||
// fleetd #323 instance 1: autoCompactWindow is read at spawn the same way
|
||||
// (ClaudeCodeLauncher.java:926, OpenCodeLauncher.java:650). Comparing it here only
|
||||
// makes the reload REPORT that a restart is needed — it deliberately does not make
|
||||
// autoCompactWindow take effect live, which is a separate, larger change.
|
||||
&& Objects.equals(a.autoCompactWindow(), b.autoCompactWindow());
|
||||
}
|
||||
}
|
||||
+1
-1
@@ -11,7 +11,7 @@ import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
/**
|
||||
* Polls {@code bridged.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* Polls {@code fleetd.yaml}'s modified time and asks {@link ConfigRef} to reload when it moves
|
||||
* (CB-559). Opt-in through {@code configReload.enabled}.
|
||||
*
|
||||
* <p><strong>Why polling and not a filesystem watch.</strong> {@code WatchService} on macOS has no
|
||||
+801
-69
File diff suppressed because it is too large
Load Diff
+1
-1
@@ -16,7 +16,7 @@ import java.util.Set;
|
||||
* means its traffic would leave the subscription. That is a hard stop.</li>
|
||||
* </ul>
|
||||
*
|
||||
* Both checks throw {@link GuardException} on violation. {@code bridged} calls
|
||||
* Both checks throw {@link GuardException} on violation. {@code fleetd} calls
|
||||
* {@link #assertWorker} before spawning a worker and {@link #assertPrimaryClean}
|
||||
* against its own environment at startup.
|
||||
*/
|
||||
@@ -0,0 +1,380 @@
|
||||
package dev.ltms.fleet.health;
|
||||
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.HashMap;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/** Slow whole-fleet evidence collection. It is deliberately separate from the delivery poller. */
|
||||
public final class FleetHealthMonitor {
|
||||
private static final Logger log = LoggerFactory.getLogger(FleetHealthMonitor.class);
|
||||
|
||||
/** Bounded attempts to run {@link #failTarget} for one transition. Never retried tick-to-tick (CB-580). */
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
/**
|
||||
* fleetd #280: how long after a terminal transition to wait before the one bounded re-check
|
||||
* fires. Must exceed the worst-case reverse-rendezvous {@code fleet_ask} window (55-115s, see
|
||||
* {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS}) so that, if the
|
||||
* target was genuinely {@code ASKING} when {@code state} was first observed, its own ask has had
|
||||
* time to lapse (clearing {@code Task#question} back to {@code null}) before this fires.
|
||||
*/
|
||||
static final long ASK_LAPSE_RECHECK_DELAY_SECONDS = 120;
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code System.nanoTime()} (or whatever {@link #clock} is) does not advance while
|
||||
* macOS sleeps, so a raw {@code nowNanos - lastActivityAtNanos} comparison freezes with the
|
||||
* host and can never cross {@link #workingSuspectAfterNanos}. This is a second, wall-clock
|
||||
* source used ONLY inside the stall check ({@link #stallElapsedNanos}) to detect and correct
|
||||
* for that freeze. Nothing else in this class reads it — every other decision (readiness grace,
|
||||
* the fault classification itself) stays exactly on {@link #clock}, as the ticket requires.
|
||||
*/
|
||||
private static final LongSupplier DEFAULT_REALTIME_CLOCK =
|
||||
() -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis());
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final LongSupplier realtimeClock;
|
||||
private final long intervalSeconds;
|
||||
private final long tickIntervalNanos;
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
/**
|
||||
* fleetd #386 clock-drift bookkeeping. {@code haveClockBaseline}/{@code lastTickMonoNanos}/
|
||||
* {@code lastTickRealNanos} track the previous tick's pair of readings so each new tick can
|
||||
* measure how far the two clocks moved apart since then. {@code accumulatedDriftNanos} is the
|
||||
* running total of every such divergence observed since this monitor started (never decreases —
|
||||
* the monotonic clock can only lag real time, never lead it). {@code busyDriftBaselineNanos}/
|
||||
* {@code busyBaselineActivityNanos} record, per target, the value of {@code accumulatedDriftNanos}
|
||||
* at the moment this monitor first saw that target's CURRENT {@code lastActivityAtNanos} while
|
||||
* BUSY — so {@link #stallElapsedNanos} adds back only the drift observed DURING this BUSY span,
|
||||
* never drift from a sleep that happened before the member went busy. All five fields are touched
|
||||
* only from {@code tick()}, like {@link #priors}.
|
||||
*/
|
||||
private boolean haveClockBaseline = false;
|
||||
private long lastTickMonoNanos;
|
||||
private long lastTickRealNanos;
|
||||
private long accumulatedDriftNanos = 0;
|
||||
private final Map<String, Long> busyDriftBaselineNanos = new HashMap<>();
|
||||
private final Map<String, Long> busyBaselineActivityNanos = new HashMap<>();
|
||||
/**
|
||||
* The live classification per member, and the only one of this class's three maps that more
|
||||
* than one scheduler task touches. {@code tick} writes it (and prunes it to the roster);
|
||||
* fleetd #280's delayed {@link #recheckTerminalTarget} reads it from its own separate scheduled
|
||||
* task. Both run on the single-threaded scheduler {@code Fleetd} passes in today, so they are
|
||||
* serialised — but nothing in this class enforces that, and an unsynchronised {@link HashMap}
|
||||
* read racing a resize can spin a CPU forever rather than fail visibly. {@code priors} and
|
||||
* {@code orphanStreaks} stay plain maps because {@code tick} is still their only toucher.
|
||||
*/
|
||||
private final Map<String, HealthState> states = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
* for one tick during an ordinary race — an async ticket exists before its virtual thread has
|
||||
* reached {@code rendezvous.open()}, so for that instant nothing is accepted or queued behind
|
||||
* it. {@code decide} maps the field straight to {@code DELEGATION_ORPHANED} with no cross-tick
|
||||
* smoothing of its own, so a single racy read would log a fault that clears on the next tick.
|
||||
* Requiring two consecutive observations costs one interval of latency on a real orphan and
|
||||
* removes that false positive entirely.
|
||||
*/
|
||||
private final Map<String, Integer> orphanStreaks = new HashMap<>();
|
||||
|
||||
/** How many consecutive ticks a target must look orphaned before health reports it (CB-643). */
|
||||
static final int ORPHAN_CONFIRM_TICKS = 2;
|
||||
|
||||
// CB-643: every HealthSnapshot field now carries real evidence. The NOT_YET_OBSERVED placeholder
|
||||
// that stood in for 7 of the 12 is gone, and with it the reason 8 of the 9 fault states were
|
||||
// unreachable — GONE and NEVER_READY included, which is what kept CB-580's failTarget from ever
|
||||
// firing. Do not reintroduce a constant here: a field with no publisher is a dead state, and the
|
||||
// tests pass either way, so nothing else will tell you.
|
||||
|
||||
/**
|
||||
* @param failTarget CB-568's idempotent target-wide failure operation (e.g. {@code messages::abandon}),
|
||||
* invoked once when a member transitions into a terminal health state. Required —
|
||||
* there is deliberately no defaulting overload; a caller that does not want the
|
||||
* fail-tickets-on-terminal-health behavior must pass an explicit inert value (see
|
||||
* {@code TestTurnTokens.inert} / {@code FleetMcp.CapacitySource.none()} for the pattern).
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
long workingSuspectAfterSeconds, BiConsumer<String, String> failTarget) {
|
||||
this(agents, roster, messages, scheduler, clock, DEFAULT_REALTIME_CLOCK, intervalSeconds,
|
||||
workingSuspectAfterSeconds, failTarget);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param realtimeClock fleetd #386: a wall-clock nanosecond source (e.g.
|
||||
* {@code System.currentTimeMillis()} converted to nanos) that keeps
|
||||
* advancing while {@code clock} is frozen by a host sleep. Used only to
|
||||
* correct the stall check — see the class-level javadoc on the
|
||||
* clock-drift fields.
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, LongSupplier realtimeClock,
|
||||
long intervalSeconds, long workingSuspectAfterSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.realtimeClock = Objects.requireNonNull(realtimeClock, "realtimeClock");
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.tickIntervalNanos = TimeUnit.SECONDS.toNanos(intervalSeconds);
|
||||
this.workingSuspectAfterNanos = TimeUnit.SECONDS.toNanos(workingSuspectAfterSeconds);
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
|
||||
/** Pure per-member decision seam. */
|
||||
static HealthDecision decide(HealthSnapshot snapshot, HealthPrior prior, long nowNanos) {
|
||||
return FleetHealth.decide(snapshot, prior, nowNanos);
|
||||
}
|
||||
|
||||
public void start() { scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS); }
|
||||
public void stop() { scheduler.shutdownNow(); }
|
||||
|
||||
// Package-private so tests can run one tick without waiting.
|
||||
void tick() {
|
||||
try {
|
||||
List<MemberSession> rosterNow = roster.get(); // One in-memory roster snapshot for this tick.
|
||||
List<Agent> agentsNow;
|
||||
boolean controlLinkDown = false;
|
||||
try {
|
||||
agentsNow = agents.list(); // Exactly one list call for this complete observation.
|
||||
} catch (HerdrException error) {
|
||||
agentsNow = List.of();
|
||||
controlLinkDown = true;
|
||||
log.warn("fleet health control link unavailable; classifying roster", error);
|
||||
}
|
||||
Map<String, Agent> live = new HashMap<>();
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
long nowNanos = clock.getAsLong();
|
||||
long driftBeforeThisTick = observeClockDrift(nowNanos);
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
AgentStatus status = agent == null ? AgentStatus.UNKNOWN : agent.status();
|
||||
boolean accepted = messages.hasAcceptedDelivery(session.terminalId());
|
||||
boolean present = agent != null;
|
||||
boolean targetNotFound = !controlLinkDown && !present
|
||||
&& session.state() != MemberSession.State.SPAWNING;
|
||||
boolean readinessGraceElapsed = nowNanos - session.spawnedAtNanos() >= READINESS_GRACE_NANOS;
|
||||
boolean stalled = session.state() == MemberSession.State.BUSY
|
||||
&& stallElapsedNanos(session, nowNanos, driftBeforeThisTick) >= workingSuspectAfterNanos;
|
||||
// CB-643: the three message-layer facts CB-640 published. Read them here rather than
|
||||
// leaving them false — that constant is what made 8 of the 9 fault states dead.
|
||||
boolean queuedDelivery = messages.hasQueuedDelivery(session.terminalId());
|
||||
boolean replyStranded = messages.hasStrandedReply(session.terminalId());
|
||||
boolean orphanedDelegation = confirmOrphan(session.terminalId(),
|
||||
messages.hasOrphanedDelegation(session.terminalId()));
|
||||
HealthSnapshot snapshot = new HealthSnapshot(session.state(), status, accepted, queuedDelivery,
|
||||
messages.hasInboxMessage(session.terminalId()), present, targetNotFound, controlLinkDown,
|
||||
readinessGraceElapsed, orphanedDelegation, replyStranded, stalled);
|
||||
HealthDecision decision = decide(snapshot, priors.getOrDefault(session.terminalId(), HealthPrior.NONE),
|
||||
nowNanos);
|
||||
priors.put(session.terminalId(), decision.prior());
|
||||
reportTransition(session.terminalId(), decision.state());
|
||||
}
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
orphanStreaks.keySet().retainAll(current);
|
||||
busyDriftBaselineNanos.keySet().retainAll(current);
|
||||
busyBaselineActivityNanos.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// Any unclassified collection failure must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
} finally {
|
||||
if (!scheduler.isShutdown()) {
|
||||
scheduler.schedule(this::tick, intervalSeconds, TimeUnit.SECONDS);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Debounce {@link MessageService#hasOrphanedDelegation} across ticks (CB-643). Returns true only
|
||||
* once {@code observed} has held for {@link #ORPHAN_CONFIRM_TICKS} consecutive ticks; a single
|
||||
* false reading resets the streak, so a transient race never reaches the classifier.
|
||||
*/
|
||||
private boolean confirmOrphan(String target, boolean observed) {
|
||||
if (!observed) {
|
||||
orphanStreaks.remove(target);
|
||||
return false;
|
||||
}
|
||||
int streak = orphanStreaks.merge(target, 1, Integer::sum);
|
||||
return streak >= ORPHAN_CONFIRM_TICKS;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: compare this tick's monotonic and real-time readings against the previous
|
||||
* tick's, and fold any positive divergence into {@link #accumulatedDriftNanos} (a ratchet — it
|
||||
* never decreases, since the monotonic clock can only fall behind real time, never ahead of
|
||||
* it). Logs once, at WARN, when that single tick's divergence exceeds one full tick interval —
|
||||
* the signature of a host that slept between the two ticks (a tick literally cannot run while
|
||||
* the process itself is suspended, so the whole sleep duration lands inside one tick's gap).
|
||||
*
|
||||
* @return {@link #accumulatedDriftNanos} as it stood BEFORE this tick's divergence was folded
|
||||
* in — the baseline {@link #stallElapsedNanos} needs when a target is observed BUSY
|
||||
* for the first time this tick, so a sleep that happened before this member went busy
|
||||
* is not attributed to it.
|
||||
*/
|
||||
private long observeClockDrift(long nowNanos) {
|
||||
long nowRealNanos = realtimeClock.getAsLong();
|
||||
long driftBeforeThisTick = accumulatedDriftNanos;
|
||||
if (haveClockBaseline) {
|
||||
long monoDelta = nowNanos - lastTickMonoNanos;
|
||||
long realDelta = nowRealNanos - lastTickRealNanos;
|
||||
long tickDrift = realDelta - monoDelta;
|
||||
if (tickDrift > tickIntervalNanos) {
|
||||
log.warn("fleet health: the monotonic clock did not advance for about {}s that the "
|
||||
+ "real clock did since the last tick (host likely slept); the stall "
|
||||
+ "detector could not see that time", TimeUnit.NANOSECONDS.toSeconds(tickDrift));
|
||||
}
|
||||
if (tickDrift > 0) {
|
||||
accumulatedDriftNanos = driftBeforeThisTick + tickDrift;
|
||||
}
|
||||
}
|
||||
lastTickMonoNanos = nowNanos;
|
||||
lastTickRealNanos = nowRealNanos;
|
||||
haveClockBaseline = true;
|
||||
return driftBeforeThisTick;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code nowNanos - lastActivityAtNanos} alone freezes across a host sleep, since
|
||||
* both come from the monotonic {@link #clock}. This adds back the real-time drift observed
|
||||
* since this BUSY span started — not the monitor's whole lifetime, so a sleep that happened
|
||||
* before this member went busy never leaks into its stall reading (see the class-level javadoc
|
||||
* on the drift fields). The baseline resets whenever {@code lastActivityAtNanos} changes (a new
|
||||
* turn) or the member is not currently BUSY.
|
||||
*/
|
||||
private long stallElapsedNanos(MemberSession session, long nowNanos, long driftBeforeThisTick) {
|
||||
String target = session.terminalId();
|
||||
if (session.state() != MemberSession.State.BUSY) {
|
||||
busyDriftBaselineNanos.remove(target);
|
||||
busyBaselineActivityNanos.remove(target);
|
||||
return nowNanos - session.lastActivityAtNanos();
|
||||
}
|
||||
Long baselineActivity = busyBaselineActivityNanos.get(target);
|
||||
if (baselineActivity == null || baselineActivity != session.lastActivityAtNanos()) {
|
||||
busyBaselineActivityNanos.put(target, session.lastActivityAtNanos());
|
||||
busyDriftBaselineNanos.put(target, driftBeforeThisTick);
|
||||
}
|
||||
long driftSinceBusyStart = accumulatedDriftNanos - busyDriftBaselineNanos.get(target);
|
||||
return (nowNanos - session.lastActivityAtNanos()) + driftSinceBusyStart;
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
if (fault(next)) {
|
||||
log.warn("fleet health member={} state={} previous={}", target, next, previous);
|
||||
} else if (previous != null && fault(previous)) {
|
||||
log.info("fleet health member={} recovered state={} previous={}", target, next, previous);
|
||||
}
|
||||
// CB-580: a member entering GONE/NEVER_READY must not leave its waiting tickets pending
|
||||
// forever. Fire exactly once per transition — never on a tick where the state is unchanged,
|
||||
// which is what made the rejected commit call abandon() once per tick for as long as a
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
scheduleTerminalRecheck(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
private void failTerminalTarget(String target, HealthState state) {
|
||||
String reason = "fleet health: member reached terminal state " + state.name();
|
||||
RuntimeException last = null;
|
||||
for (int attempt = 1; attempt <= MAX_FAIL_TARGET_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
failTarget.accept(target, reason);
|
||||
return;
|
||||
} catch (RuntimeException error) {
|
||||
last = error;
|
||||
log.warn("fleet health: failTarget attempt {}/{} failed for member={} state={}",
|
||||
attempt, MAX_FAIL_TARGET_ATTEMPTS, target, state, error);
|
||||
}
|
||||
}
|
||||
log.warn("fleet health: giving up on failTarget for member={} state={} after {} attempts",
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: schedule the one bounded, delayed follow-up for a terminal transition — never a
|
||||
* per-tick retry (CB-580 rejected that shape; {@link #reportTransition} still fires
|
||||
* {@link #failTerminalTarget} exactly once per transition, unconditionally on the tick loop).
|
||||
* This is a single one-shot task, scheduled once per transition into GONE/NEVER_READY, so a
|
||||
* member stuck terminal for the rest of its life gets exactly one extra attempt, not one per
|
||||
* tick. See {@link #recheckTerminalTarget} for why the extra attempt is safe.
|
||||
*/
|
||||
private void scheduleTerminalRecheck(String target, HealthState state) {
|
||||
if (scheduler.isShutdown()) return;
|
||||
try {
|
||||
scheduler.schedule(() -> recheckTerminalTarget(target, state),
|
||||
ASK_LAPSE_RECHECK_DELAY_SECONDS, TimeUnit.SECONDS);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("fleet health: could not schedule terminal re-check for member={} state={}",
|
||||
target, state, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: the delayed re-check {@link #scheduleTerminalRecheck} scheduled for one terminal
|
||||
* transition. By now, a {@code fleet_ask} that was still open when {@code state} was first
|
||||
* observed has had time to lapse on its own (see {@link #ASK_LAPSE_RECHECK_DELAY_SECONDS}),
|
||||
* clearing {@code Task#question} back to {@code null} — which is exactly what
|
||||
* {@link MessageService#abandon(String, String, boolean)}'s {@code sweepAsking=false} filter
|
||||
* needs to finally match it. Calling {@link #failTerminalTarget} again is safe only because
|
||||
* {@code sweepAsking} stays {@code false}: a task genuinely still {@code ASKING} is skipped
|
||||
* exactly as it was on the very first attempt — this never fails a ticket whose ask has not yet
|
||||
* lapsed.
|
||||
*
|
||||
* <p><strong>Guarded on "target is still classified {@code state}."</strong> Without this guard,
|
||||
* a member that recovered (or was released and dropped from the roster) between the transition
|
||||
* and this re-check would still take a blind {@code failTarget} call — reaching into whatever
|
||||
* brand-new, unrelated turn it has since picked up and failing it too. {@link #states} already
|
||||
* carries the live classification (updated every tick, pruned to the current roster on release),
|
||||
* so a stale or recovered target simply reads as a mismatch here and this is a no-op.
|
||||
*/
|
||||
void recheckTerminalTarget(String target, HealthState state) {
|
||||
if (states.get(target) != state) return;
|
||||
failTerminalTarget(target, state);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
private static boolean fault(HealthState state) {
|
||||
return switch (state) {
|
||||
case NEVER_READY, GONE, TURN_BOUNDARY_LOST, ERROR_ON_SCREEN, STALL_SUSPECTED,
|
||||
MUTE, REPLY_STRANDED, DELEGATION_ORPHANED, CONTROL_LINK_DOWN -> true;
|
||||
default -> false;
|
||||
};
|
||||
}
|
||||
|
||||
public static String coverage(boolean enabled, boolean notificationConfigured) {
|
||||
return !enabled ? "off" : notificationConfigured ? "full" : "detection-only";
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user