Compare commits
477 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| e966cbadf9 | |||
| c87cc25aa6 | |||
| 3fb331145a | |||
| 60fa958da2 | |||
| 9950361bc9 | |||
| 9f74b6619a | |||
| f687046450 | |||
| c6058652be | |||
| 2a95da2ff4 | |||
| 7f9137fcb6 | |||
| 3bf3968bc7 | |||
| 261aa056f9 | |||
| 042b8c99dd | |||
| 4bfab6b718 | |||
| 008a457557 | |||
| 9494a6b99a | |||
| eb0557621e | |||
| a0eed6f01b | |||
| e2a91e883e | |||
| 62646957ea | |||
| 802c0ab701 | |||
| 4bab23e241 | |||
| a94262271b | |||
| 5c12865c25 | |||
| 3f8c956325 | |||
| 2051ceeb26 | |||
| 325d0771a4 | |||
| 7b97aae85b | |||
| 49a404ddf3 | |||
| 72d6a6878b | |||
| 4466ee0ef2 | |||
| 25ba7f16bb | |||
| 5ba69c9cf2 | |||
| 17052bb515 | |||
| e95ed99bf7 | |||
| 1477e4358a | |||
| 9e4e423ad6 | |||
| 6c2d6e93cb | |||
| 789b6a8716 | |||
| 01462c9695 | |||
| 8b4ff78546 | |||
| d4a2cd720c | |||
| 5a467e1f8b | |||
| 1fb6176783 | |||
| 49df79203c | |||
| 235644c0f0 | |||
| 7772b41993 | |||
| eccd0548ce | |||
| c4e23eebad | |||
| 7df503dfc2 | |||
| f5e02fedd6 | |||
| 29e7a06c49 | |||
| 92a96fcbd8 | |||
| 13482872bb | |||
| c1ca6273fc | |||
| 5f1b260c81 | |||
| 30d6872779 | |||
| 9d1306d442 | |||
| e54e3d87ea | |||
| bf027f10b9 | |||
| 3c5873dfe2 | |||
| e29227d5f4 | |||
| 45aca9eb3e | |||
| 2af13ab1ff | |||
| 0788d84be8 | |||
| 20c1094cbf | |||
| 9011c59b9f | |||
| c11ad71ed0 | |||
| bdcf285265 | |||
| d4f93a7b13 | |||
| cfebc575ea | |||
| 822327eed5 | |||
| 5289eb509f | |||
| e4c703a51a | |||
| 703a05db41 | |||
| 3f036b2a62 | |||
| 82fae94c55 | |||
| b1f34c2e6b | |||
| 3f807d9f1b | |||
| e70263062c | |||
| 1a1e586b62 | |||
| c16d118f09 | |||
| 4b10d02207 | |||
| c5fbfdbf4a | |||
| 12cff28abb | |||
| 77a6a7142e | |||
| 9f3671b801 | |||
| 84034b34d1 | |||
| 1e9b2c9b7e | |||
| 5d422f85fa | |||
| ed2027b202 | |||
| 6b0a99b2b7 | |||
| 6b7caba248 | |||
| f8b0d42a5c | |||
| d1e7d71eee | |||
| 7fd914df1a | |||
| b066eb1903 | |||
| a196d34455 | |||
| 051d320ea0 | |||
| 766772763f | |||
| eab8185d7b | |||
| 7f672f0fb8 | |||
| e2801b9bbc | |||
| ea02c7b248 | |||
| ce74e164c6 | |||
| e60f892efd | |||
| ce05886831 | |||
| be123d0ac7 | |||
| 2d09c8b027 | |||
| ed54f0224e | |||
| 8d5bc3ee89 | |||
| ab0cc71aa4 | |||
| 4cd9046353 | |||
| 4e98a74047 | |||
| a502ba53e0 | |||
| 446cc11d1d | |||
| ddd81fe174 | |||
| ae74cc081f | |||
| cf8da1d5fa | |||
| b8aedeafcb | |||
| 3d61af6f6f | |||
| ed99c209ac | |||
| af4c88d54b | |||
| 44c735f6f5 | |||
| b540a1744b | |||
| 0f51d53098 | |||
| c670792ffe | |||
| 3982ace544 | |||
| 7180b1aad0 | |||
| c26f695402 | |||
| 48877315ca | |||
| 5c08054533 | |||
| b6db9c31f5 | |||
| e7b33fe3a0 | |||
| e3e403e5c8 | |||
| 799014e99d | |||
| 69e09b10fa | |||
| b9d09e044e | |||
| fd8650cda4 | |||
| 11050e24ed | |||
| 769f282408 | |||
| 6f71f40047 | |||
| fb36c5238f | |||
| cc9cdc938b | |||
| 5fede82468 | |||
| 1515025804 | |||
| 7754f53662 | |||
| a507f7b31b | |||
| 71c322f104 | |||
| fde2c15627 | |||
| 2830735644 | |||
| 4e3ac91a22 | |||
| 29d3f0b41f | |||
| cbc732444f | |||
| 6f828b8c38 | |||
| 145a8c8862 | |||
| 7057291739 | |||
| 2302b3bc11 | |||
| 3a004dc1b3 | |||
| a9a3c12232 | |||
| 94ec77a1bc | |||
| 49f285cfda | |||
| 5a12ae7930 | |||
| 127e6832a9 | |||
| 6442a583ae | |||
| 24b96d29ae | |||
| 4721771052 | |||
| 2f71a30bd7 | |||
| 22cdebbdbe | |||
| 154971c2b8 | |||
| fef287c346 | |||
| a37acd5ee3 | |||
| d6ef0c8013 | |||
| dfb70871b4 | |||
| 4ee7b16929 | |||
| 8557289dc0 | |||
| dd2efd8541 | |||
| 5af786d135 | |||
| 380eb63277 | |||
| 6c61355f8f | |||
| 96c406b968 | |||
| a5d81c3f70 | |||
| 92c0f164f1 | |||
| 9e813ec179 | |||
| 395b3b5c46 | |||
| 6e9e464d62 | |||
| 105c065615 | |||
| 01492059d4 | |||
| f84824ee29 | |||
| c4d40fbc2b | |||
| 7c684e40d3 | |||
| 6938f52155 | |||
| 0df34f3220 | |||
| 457458437f | |||
| 3759c41f99 | |||
| 09159f2857 | |||
| 29cd1194c2 | |||
| 815e8f8b23 | |||
| 1e60ac0745 | |||
| 650a4c146b | |||
| 23f299e105 | |||
| dbf6fef0e9 | |||
| d292522d00 | |||
| 0241e0d3a8 | |||
| e4973eb8a4 | |||
| b6b88c5f1c | |||
| 86dddfe240 | |||
| 0d5944af63 | |||
| 4a5030a5c6 | |||
| c1c8794c48 | |||
| 65a78932c1 | |||
| f429ca1a50 | |||
| 73aab3f83e | |||
| 887aca0183 | |||
| 3fd23ecafa | |||
| f379847942 | |||
| ea12107497 | |||
| 591df91de1 | |||
| 6a814176f0 | |||
| d11d1d157c | |||
| d057d56156 | |||
| d703ce1313 | |||
| b32a30fd47 | |||
| 464dbc0930 | |||
| a8cadd9150 | |||
| 57b8c0b56d | |||
| 147f50c19e | |||
| eee4d576a2 | |||
| b4f9d7f53a | |||
| 3aca53b967 | |||
| 4aa1fae296 | |||
| 0c865032f9 | |||
| ea41bbf6b9 | |||
| 7b918c51ff | |||
| 554395b104 | |||
| 823976c1b5 | |||
| c8388a7f92 | |||
| 02e6aef98c | |||
| 6d493bc7bb | |||
| e5cb51a90e | |||
| e545c08082 | |||
| efa0deb9b2 | |||
| ca47e90c01 | |||
| fa1f49675b | |||
| 8426c3528f | |||
| 65f98ba910 | |||
| d05205d1eb | |||
| 667254df47 | |||
| 2926cd1784 | |||
| c801851c66 | |||
| de70aa38f1 | |||
| 53a533afb4 | |||
| 77ad88631b | |||
| d88017807b | |||
| 2159a5a94a | |||
| b9c2cf69f4 | |||
| 8beae50fe7 | |||
| b2a58cb966 | |||
| b8b25cf74c | |||
| f159ca7d27 | |||
| 83f2aea60f | |||
| 002329adb5 | |||
| a49671ceb9 | |||
| 76672ff016 | |||
| 9379f92c23 | |||
| 21ff63b11d | |||
| 21844b54d7 | |||
| 4769481515 | |||
| bbbb4c1eb3 | |||
| 38c248e617 | |||
| 9debc0de27 | |||
| f34361b263 | |||
| be5ba22c75 | |||
| 85c90d440a | |||
| e97502d550 | |||
| aef14ff46e | |||
| cba516bda4 | |||
| ba51e0c6cc | |||
| 086c59848e | |||
| 0c10079755 | |||
| ece2091b53 | |||
| b3f917e6f5 | |||
| fa39a5f55e | |||
| d5128a1d35 | |||
| 61097e5cf0 | |||
| ef507bcd12 | |||
| 94f50e507a | |||
| 75b15086b0 | |||
| dab9645906 | |||
| e93b5f6512 | |||
| bfabe13e8f | |||
| 7e49c6eca2 | |||
| 11cbfa79b4 | |||
| c2c2746922 | |||
| 6ed70700a0 | |||
| 0087645da4 | |||
| 19cacf5b62 | |||
| 4dd12083ab | |||
| 30e21adec7 | |||
| 719b79f892 | |||
| 66e5247b6d | |||
| 1e41bd63b4 | |||
| 4887d03d88 | |||
| 7d5434455d | |||
| f71ee4926e | |||
| 282a2fc2b8 | |||
| f04e934b94 | |||
| 3fae35c357 | |||
| 18aecbfe67 | |||
| 5d75f72473 | |||
| e028a0ae54 | |||
| 2fa673d4c0 | |||
| 9020d01b40 | |||
| 1006805027 | |||
| 27aefbf9a0 | |||
| d42c2bc204 | |||
| 3916adc372 | |||
| fa97f598dd | |||
| ea9aa4fd77 | |||
| a2b8caf6b5 | |||
| b5ddbe5757 | |||
| d223a93039 | |||
| 96d8191149 | |||
| f0e7ac73d6 | |||
| e2fe861b4d | |||
| 279d6f5fbd | |||
| 34480cebef | |||
| 9d0bf14c46 | |||
| 5a3ab5764c | |||
| 3bad9f5785 | |||
| 8308c0b68f | |||
| 3b3063eb2b | |||
| df9086263d | |||
| eb568ff451 | |||
| ac790e4cce | |||
| 6b5f3f472f | |||
| 38dec72152 | |||
| 0e8bfb74fc | |||
| 51f7b0a3ca | |||
| 21c539f22e | |||
| bd2774b5f1 | |||
| c50f5b2d61 | |||
| 01a840cc14 | |||
| c4deef08be | |||
| c796eac09c | |||
| e897e5257b | |||
| 2afa3652bb | |||
| 80092ff359 | |||
| 9d37f3aa29 | |||
| eaf89abaf6 | |||
| d895f02bc1 | |||
| 43206cac2f | |||
| 8bba3a8184 | |||
| 5cf3ca9a89 | |||
| ac474981e4 | |||
| ba04b2359b | |||
| 2e5b63f6f6 | |||
| 3437d6313d | |||
| 743377d6cd | |||
| 5952d559c7 | |||
| 205ad823b0 | |||
| d654ccb818 | |||
| a89dcc9b7e | |||
| 0d7b4fb026 | |||
| 321d8dcbb5 | |||
| ef8c97871e | |||
| 26bafe824b | |||
| 838a701109 | |||
| e5eb3534c7 | |||
| 959c83534f | |||
| 31b028e860 | |||
| 7840e9adf6 | |||
| c3672f5472 | |||
| cf54aed451 | |||
| bbf68f3e3c | |||
| 776743cbe2 | |||
| 826e0aeb2a | |||
| c935b181dd | |||
| ee932fd85b | |||
| fe2e5ede34 | |||
| c325054242 | |||
| 7662e2d0c8 | |||
| 4877992a70 | |||
| 1c051c4e47 | |||
| adb7a67880 | |||
| 723fe494e9 | |||
| 0f08b93659 | |||
| 432c1d92d1 | |||
| 60b7e67b42 | |||
| 3fbd43fe3f | |||
| 32ebf065ac | |||
| 1178b3f684 | |||
| 2a434ced2f | |||
| cc919aa2b6 | |||
| 6417b0edd9 | |||
| 39c7ce76f3 | |||
| b40f477210 | |||
| 5c56cb347f | |||
| e694deace3 | |||
| 748367b7d6 | |||
| dcf5fb3be3 | |||
| f9d2ee2a2b | |||
| 866c7f2e9a | |||
| d9168de43e | |||
| c3fa1136d4 | |||
| cabcd87b66 | |||
| bf0e09b1a2 | |||
| e1eb50ce65 | |||
| 049e7d9d54 | |||
| ff3b49cd1e | |||
| 0373b6c41b | |||
| 445a45f6e1 | |||
| 966c58a3b8 | |||
| de026b8f8a | |||
| 97f6c33a45 | |||
| 735c837604 | |||
| 457dc0330d | |||
| a1a9015217 | |||
| 5a811a3695 | |||
| 1fdaa74eb3 | |||
| 7d4a4339c2 | |||
| 847e8bd3fa | |||
| f9fb387427 | |||
| a55079afbd | |||
| e18ad4723b | |||
| 6de8ac8972 | |||
| a052975420 | |||
| 6d82ca95a4 | |||
| 8067ee4ec4 | |||
| 3743789e8d | |||
| 388aba7632 | |||
| ad587eafa3 | |||
| 08ce9aef11 | |||
| ee5f8b932b | |||
| 4ac688b6d9 | |||
| a814d1ef00 | |||
| ea98856130 | |||
| a49e96835a | |||
| 63c19dcba7 | |||
| 2823349c8e | |||
| 6fc301d62c | |||
| bc99d64786 | |||
| 615af4ed0a | |||
| 045d229728 | |||
| d1fd5700f5 | |||
| 2d55b0b9a5 | |||
| d89ae94a2e | |||
| e6193c4098 | |||
| b66f0677ed | |||
| 23ada1981e | |||
| a22480c117 | |||
| 24f404f989 | |||
| a237fbff9d | |||
| 31d5516991 | |||
| 17af61e8dd | |||
| 6af87b6ad6 | |||
| 5ba05d0bdb | |||
| fc655e78c2 | |||
| 25726a5ae7 | |||
| 11c3ff67b6 | |||
| d867c87100 | |||
| b0c4cedfab | |||
| 85417d5215 | |||
| 4accc746bd | |||
| 21c4c8cbef | |||
| 430f5b0dae | |||
| 42731833d0 | |||
| 65acf066ad | |||
| ee8f570fd7 | |||
| fa3f910d44 | |||
| 46ac6e4e38 | |||
| 3bfa82839b | |||
| 65ccf2e4ad | |||
| 7a3b27f76f | |||
| 82e7be564c | |||
| c5e24197bf | |||
| bcb402b688 |
@@ -1,15 +1,15 @@
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleetd",
|
||||
"description": "Tooling for orchestrating a fleet of delegated coding agents through the fleetd MCP gateway.",
|
||||
"owner": {
|
||||
"name": "LTMS"
|
||||
},
|
||||
"plugins": [
|
||||
{
|
||||
"name": "claude-bridge",
|
||||
"name": "fleet",
|
||||
"source": "./plugin",
|
||||
"description": "Make a project bridge-ready: mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.1.0",
|
||||
"description": "Mount the fleetd MCP gateway and apply standard Claude Code settings so a session can orchestrate delegated workers. Ships no credentials.",
|
||||
"version": "0.2.0",
|
||||
"author": {
|
||||
"name": "LTMS"
|
||||
}
|
||||
|
||||
@@ -113,8 +113,19 @@ PY
|
||||
|
||||
**What this tier cannot see:** it proves facts only about the Mac daemon at `127.0.0.1:8765`.
|
||||
It cannot show the fleet01 daemon, broker queue depth, or broker consumers. The fleet01 REST service
|
||||
at `10.10.20.13:8765` is not reachable from the Mac, and SSH as `dai.ha@10.10.20.13` is denied.
|
||||
Say this in the report rather than omitting fleet01.
|
||||
at `10.10.20.13:8765` is not reachable from the Mac. Say this in the report rather than omitting
|
||||
fleet01.
|
||||
|
||||
**But fleet01 IS reachable over SSH — checked 2026-08-28.** An older version of this line said SSH
|
||||
was denied. That is true only for the user `dai.ha`. The host alias `fleet01` maps to user `ltms`,
|
||||
and `ssh fleet01` works with key auth:
|
||||
|
||||
```bash
|
||||
ssh -o BatchMode=yes -o ConnectTimeout=6 fleet01 'echo $(id -un)@$(hostname)'
|
||||
```
|
||||
|
||||
So fleet01's daemon PID, uptime, jar and `/healthz` **can** be reported — over SSH, not over REST.
|
||||
Do that rather than writing `not reachable`. `ltms` also has passwordless sudo there.
|
||||
|
||||
## 3. Tier 2 — the shared broker (run when management access exists)
|
||||
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
---
|
||||
name: handover
|
||||
description: Procedure for an outgoing lead to write the handover file that a fresh lead session inherits. Load this when your context is filling up and you are about to be replaced, whether you hand off by hand or fleetd does it for you. The file is the new lead's only inheritance — follow it exactly.
|
||||
---
|
||||
|
||||
# Handover — write the file the next lead depends on
|
||||
|
||||
A lead session fills up its context and has to be replaced by a fresh one. The outgoing lead
|
||||
writes a handover file, and the new session reads that file and carries on.
|
||||
|
||||
**There are two ways to hand off, and the file is the same either way.**
|
||||
|
||||
- **By hand.** You write the file, then tell the operator where it is. The operator starts the new
|
||||
session and points it at the file. This always works.
|
||||
- **With `fleet_handover`** (fleetd #480, merged 2026-09-11). You ask fleetd to do the swap: it
|
||||
checks the file, clears your pane, and tells the fresh session to read it. This needs
|
||||
`leadRollover:` in `fleetd.yaml`; without it every action answers a clean refusal naming
|
||||
`NOT_CONFIGURED`, and you fall back to the manual path. Section 11 below is the procedure.
|
||||
|
||||
Nothing else in this skill changes between the two. Only who performs the swap changes.
|
||||
|
||||
**The new lead's only inheritance is that file.** It does not see your conversation, your plan,
|
||||
or your screen. If the file is thin or wrong, the new lead re-derives what you already knew, and
|
||||
that wastes hours. Writing a good handover file is real work. It is not paperwork you rush
|
||||
through at the end of a session.
|
||||
|
||||
This skill is the procedure for writing it. Every rule below earned its place because a past
|
||||
handover got it wrong.
|
||||
|
||||
## 1. Confirm you are the right session to write this
|
||||
|
||||
Run `fleet_whoami` first. It must answer `primary`. Only a primary (lead) session writes a
|
||||
handover file. A worker's job ends with its own pull request, not a fleet-wide handoff.
|
||||
|
||||
## 2. Every number needs a command, run in this turn
|
||||
|
||||
A number is a claim: a count, a commit hash, a process id, a percentage, a queue depth. Before
|
||||
you write one, run the command that produces it — now, in this turn, against the live state.
|
||||
|
||||
Never take a number from:
|
||||
|
||||
- earlier in your own conversation — the state has moved since then,
|
||||
- a peer lead's report — that is their measurement, not yours,
|
||||
- your own memory of an earlier session.
|
||||
|
||||
Put the command, or its real output, next to the number. That lets the next lead re-run it and
|
||||
check it still matches. If you cannot measure something yourself, say so instead of guessing:
|
||||
"the fleet01 lead reports 91 commits behind; I have not checked this myself."
|
||||
|
||||
## 3. Say what you measured and what you did not
|
||||
|
||||
Mark every claim as one of two things:
|
||||
|
||||
- **"I checked this myself, in the code or on this host, at `<time>`."**
|
||||
- **"I did not check this myself; `<who>` reported it."**
|
||||
|
||||
Never present someone else's measurement as your own. This matters most for cross-host claims —
|
||||
a peer lead's daemon, a worker's report, or something the operator said earlier that you cannot
|
||||
re-verify from here.
|
||||
|
||||
## 4. Record open decisions, and who owns them
|
||||
|
||||
List three things:
|
||||
|
||||
- what the operator actually asked for, in their own words where you have them,
|
||||
- what is still unanswered,
|
||||
- any question you decided yourself instead of asking, with your reason.
|
||||
|
||||
Write the decision so it cannot be mistaken for the operator's instruction. Say plainly: "the
|
||||
operator never answered X; I decided Y, because Z." Without this, the next lead either silently
|
||||
reopens a closed question or assumes the operator chose something they never did.
|
||||
|
||||
## 5. Record live hazards
|
||||
|
||||
List anything that will break if the next lead does the obvious thing next. This includes:
|
||||
|
||||
- unpushed commits or unmerged branches,
|
||||
- a build, a spawn, or a redeploy still running,
|
||||
- code merged to `main` but not yet redeployed to the live daemon,
|
||||
- any trap that looks safe and is not — say what goes wrong and why, not only that something is
|
||||
"tricky."
|
||||
|
||||
## 6. Record what is explicitly not owed
|
||||
|
||||
List work that is finished, and work that another party has said they do not want touched. Name
|
||||
who said so and when. Without this line, the next lead re-does closed work or reopens a question
|
||||
a peer already declined to revisit.
|
||||
|
||||
## 7. Open the file with three re-measurement commands
|
||||
|
||||
The file's own first section must give the next lead three concrete commands to run before
|
||||
acting on anything else in the file:
|
||||
|
||||
1. confirm role — for example `fleet_whoami`,
|
||||
2. confirm the state of the working tree — for example `git status` and
|
||||
`git rev-list origin/main..HEAD`,
|
||||
3. read the live fleet — for example `fleet_list`.
|
||||
|
||||
Record what each command answered when you wrote the file, and tell the reader to run it again
|
||||
rather than trust your answer. The point of this section is that the reader checks live state
|
||||
before acting on any claim in the rest of the file, including yours.
|
||||
|
||||
## 8. Stamp the file with time and commit
|
||||
|
||||
Near the top of the file, write:
|
||||
|
||||
- the date and time you wrote it,
|
||||
- the commit the tree was on (`git rev-parse HEAD`),
|
||||
- whether the tree was clean (`git status`).
|
||||
|
||||
Without this, nobody can tell how old the file is, or which code it describes.
|
||||
|
||||
## 9. State plainly that the file goes stale fast
|
||||
|
||||
Say near the top: **re-measure anything you act on.** The file goes stale the moment anyone
|
||||
merges a branch, spawns a member, or restarts the daemon. Everything in the file is a snapshot
|
||||
of one moment, not a live fact.
|
||||
|
||||
## 10. What to leave out
|
||||
|
||||
Do not include:
|
||||
|
||||
- narration of how the session felt, or how hard something was,
|
||||
- anything the repo already records — code structure, git history, or a rule already written in
|
||||
`CLAUDE.md`. Point at it instead of repeating it,
|
||||
- advice that is only true for the session that is ending — a half-open terminal, a local
|
||||
variable, a train of thought with no state behind it.
|
||||
|
||||
A handover file is a record of state and decisions. It is not a diary.
|
||||
|
||||
## 11. Using `fleet_handover` (only if `leadRollover:` is configured)
|
||||
|
||||
**Run the three steps in this order. The order is not a style choice — the wrong order is
|
||||
refused.**
|
||||
|
||||
1. **`fleet_handover{action: "open", reason: "<why now>"}`.** It returns a `token` and the
|
||||
`handoverPath` you must write to. Nothing has happened to your pane yet.
|
||||
|
||||
**Write to exactly that path, and do not resolve it yourself.** It is always absolute, even when
|
||||
the operator configured a relative `handoverPath`: fleetd resolves a relative one against your
|
||||
own workspace before it hands it to you. The daemon and your pane can run in different
|
||||
directories, so a path you resolve yourself can point at a different file from the one the daemon
|
||||
will check.
|
||||
|
||||
**Check that the path is ignored by git before you write to it (#491).** A relative
|
||||
`handoverPath` resolves inside YOUR workspace, which is usually a repository — and usually not
|
||||
the `fleetd` one, so an ignore rule added to `fleetd` does not protect it. Run
|
||||
`grep -n handover <your workspace>/.gitignore`. No output means the file you are about to write
|
||||
will show up as untracked content in that repo. The file is a snapshot of live state and must
|
||||
never be committed, so tell the operator rather than committing it or silently editing their
|
||||
`.gitignore`.
|
||||
2. **Write the handover file at that path**, following sections 1–10 above.
|
||||
3. **Ask the operator, then `fleet_handover{action: "confirm", token, operatorConfirmed: true}`.**
|
||||
|
||||
Why that order: `confirm` refuses with `HANDOVER_STALE` unless the file was modified **after** the
|
||||
`open` request. That check stops a leftover file from an earlier session being accepted as this
|
||||
one's handover. So writing the file first and then calling `open` — the obvious order — always
|
||||
fails.
|
||||
|
||||
`{action: "cancel", token}` drops a pending request without rolling.
|
||||
|
||||
**Things that will surprise you:**
|
||||
|
||||
- **`accepted` does not mean your pane has been cleared.** It means every gate passed and the roll
|
||||
is scheduled to run once your current turn ends. Say your goodbye in the same turn — you will not
|
||||
get another one.
|
||||
- **There is no terminal or session parameter, on purpose.** The pane is always your own, resolved
|
||||
from your connection, so you can only ever roll yourself.
|
||||
- **`operatorConfirmed` is your report of what a human told you.** Do not pass `true` because you
|
||||
are confident. Ask, wait for the answer, then pass what they said. `requireOperatorConfirm`
|
||||
defaults to `true` and this is the only thing standing between a judgement call and a wiped
|
||||
session.
|
||||
- **The roll can still refuse after `confirm` returns**, and by then there is no caller to tell.
|
||||
Those outcomes are logged only, as `lead-rollover:` lines in the daemon log.
|
||||
- **The bootstrap prompt has never yet landed, and the fix is unproven (fleetd #489).** The first
|
||||
real rollover, on 2026-09-12, joined `/clear` and the bootstrap text into one line and Claude Code
|
||||
refused it as `Unknown command: /clearFresh`. The pane was never cleared and no context was lost,
|
||||
so the failure was safe — the roll simply did nothing. PR #490 fixed the cause and is deployed,
|
||||
but no roll has bootstrapped a fresh session end to end yet. **Assume it may still fail, and tell
|
||||
the operator so before you confirm.** The recovery is the same either way: the file is already
|
||||
written, so the operator starts a session and points it at the file. That is why you write the
|
||||
file before you confirm, and never the other way round.
|
||||
|
||||
## Writing style
|
||||
|
||||
Write in plain English. Use everyday words, one idea per sentence, and active voice. Keep every
|
||||
class, method, file, flag, and config key exactly as it appears in the code — replacing a
|
||||
precise term with a vague one makes the sentence wrong, not simpler. Explain an abbreviation the
|
||||
first time you use it.
|
||||
|
||||
If a diagram genuinely helps, put it in the `.md` file as a fenced ` ```mermaid ` block with no
|
||||
hardcoded colors, so it stays readable on light and dark backgrounds. Quote any label that has
|
||||
brackets, colons, or slashes.
|
||||
|
||||
## Template
|
||||
|
||||
```markdown
|
||||
# Handover — <fleet name> lead session, <date and time>
|
||||
|
||||
Written at commit `<output of git rev-parse HEAD>`. Tree was <clean, or dirty: `<git status
|
||||
summary>`>. Re-measure anything you act on — this file goes stale the moment anyone merges,
|
||||
spawns, or restarts.
|
||||
|
||||
## 0. Do these three things first
|
||||
|
||||
1. Confirm your role: `fleet_whoami` — must answer `primary`. (Answered `<result>` at `<time>`.)
|
||||
2. Confirm tree state: `git status`, `git rev-list origin/main..HEAD`. (`<result>` at `<time>`.)
|
||||
3. Read the live fleet: `fleet_list`. (`<result>` at `<time>`.)
|
||||
|
||||
## 1. What the operator asked for
|
||||
|
||||
<the live instructions, in their words where you have them; what is still open; any decision
|
||||
you made yourself, and why>
|
||||
|
||||
## 2. Open decisions, and who owns them
|
||||
|
||||
<one line per decision: who owns it, what is unanswered>
|
||||
|
||||
## 3. Live hazards
|
||||
|
||||
<one entry per hazard: what breaks, and why, if the next lead does the obvious thing>
|
||||
|
||||
## 4. What is not owed
|
||||
|
||||
<finished work, and work another party has declined; name who said so and when>
|
||||
```
|
||||
@@ -0,0 +1,102 @@
|
||||
---
|
||||
name: hunter
|
||||
description: Defect-hunt procedure for a fleetd worker — sweep an assigned package for real bugs and report several ranked findings without fixing anything. Load this when the lead asks you to hunt or audit a scope rather than review one diff. Do NOT load `reviewer` for this; the two want different output.
|
||||
---
|
||||
|
||||
# Hunter worker — procedure
|
||||
|
||||
The turn contract (one `fleet_reply`, `fleet_ask` for the lead's decisions, honest reporting,
|
||||
never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and already applies.
|
||||
|
||||
**This skill is not `reviewer`.** `reviewer` judges one diff and reports the *single* most
|
||||
important issue in about 90 words. A hunt sweeps a whole package and reports *several* findings
|
||||
in a long structured form. Loading both gives you two contradictory output contracts, and the
|
||||
usual result is a worker that writes a good report into its terminal and ends the turn without
|
||||
sending it. Load exactly one.
|
||||
|
||||
## 0. Read this before you read code: how the report gets home
|
||||
|
||||
Your terminal reaches nobody. The lead sees **only** the text inside your `fleet_reply` call.
|
||||
|
||||
A long report is exactly the case where this goes wrong, so plan for it:
|
||||
|
||||
- **Write the report into the `fleet_reply` argument itself.** Do not compose it in your terminal
|
||||
and then summarise it into the call.
|
||||
- If the report is long, **send it anyway** — one `fleet_reply` with everything.
|
||||
- If you end the turn without replying, the bridge scrapes your pane instead. That scrape carries
|
||||
at most the last 4000 characters, and on a hunt it usually captures the tail of the lead's own
|
||||
brief rather than your findings. The lead then has nothing and has to ask you again.
|
||||
|
||||
## 1. Change nothing
|
||||
|
||||
A hunt is read-only. Do not edit a production file, do not "quickly fix" what you find, and do
|
||||
not run a formatter. You may run the build and tests to *check* a claim, and you should say so
|
||||
when you did.
|
||||
|
||||
## 2. Read the whole scope first
|
||||
|
||||
Read every file in the assigned package before you judge any of it. A defect that a caller
|
||||
elsewhere in the same package makes unreachable is not a defect, and you cannot know that from
|
||||
one file.
|
||||
|
||||
Stay inside the scope. If a defect there depends on a class outside it, read that class to
|
||||
confirm — but the defect itself must live in the scope you were given.
|
||||
|
||||
## 3. The bar — this matters more than the count
|
||||
|
||||
**Name the path into the bad state.** Say which caller, in which state, reaches it. A defect on
|
||||
paper is not a reachable defect. If you cannot name that path, keep the finding but mark it
|
||||
`unproven` and say exactly what you could not check. Do not drop it, and do not dress it up.
|
||||
|
||||
**Say which direction the harm goes.** Data loss, privilege escalation and silent wrong answers
|
||||
are worth reporting even when the window is narrow. A finding whose worst outcome is a worse log
|
||||
line is not worth a block.
|
||||
|
||||
Two workers once ran the same scope: the one that applied the direction-of-harm filter found ten
|
||||
real defects, the one that did not found none. Fewer findings the lead can act on beat many the
|
||||
lead has to triage.
|
||||
|
||||
## 4. Shapes that have produced real merged fixes here
|
||||
|
||||
Read for these first:
|
||||
|
||||
1. **A one-way gate.** A guard added after an incident closes only the direction that incident
|
||||
came from. Do not only ask what closes the gate — ask **which states still open it**.
|
||||
2. **A value read once, then used later to authorise something destructive**, after something
|
||||
else has had a chance to change it.
|
||||
3. **A failure downgraded to a value that looks like a legitimate result** — `-1`, `null`, an
|
||||
empty list, `false` — which a caller then trusts.
|
||||
4. **A lock held for one half of a read-modify-write and not the other**, or two collections
|
||||
updated under different locks.
|
||||
5. **A comment or javadoc stating an invariant the code no longer keeps.** Comments are
|
||||
load-bearing in this repo; a stale one has already caused a bug.
|
||||
|
||||
## 5. What you cannot check, and must not claim you did
|
||||
|
||||
- `fleetd/fleetd.yaml` is gitignored and **absent from your worktree**. You cannot read it. If a
|
||||
finding depends on live configuration, name the key and say you could not check it.
|
||||
- `.mcp.json`, `opencode.json` and `.autoenv` in your worktree are neutralised stubs, not the
|
||||
repo's real files.
|
||||
- The `wiki/` submodule pointer is months old. Do not cite it.
|
||||
|
||||
Reporting a fact you took from the lead's brief as something you measured yourself is a false
|
||||
report, even when the fact is correct. Say where each fact came from.
|
||||
|
||||
## 6. The report — what goes in `fleet_reply`
|
||||
|
||||
One block per finding, most severe first:
|
||||
|
||||
```
|
||||
FINDING N — <one line>
|
||||
file:line
|
||||
Path in: <which caller, in which state, reaches this>
|
||||
Direction: <data loss | escalation | silent wrong answer | outage | ...>
|
||||
Window/trigger: <when it actually happens>
|
||||
Confidence: <confirmed by reading | unproven — say what you could not check>
|
||||
Why nothing else catches it: <the guard or test you checked, and why it misses>
|
||||
```
|
||||
|
||||
End with one line naming every file you read, so the lead knows the denominator.
|
||||
|
||||
**Nothing clears the bar?** Reply `NO FINDINGS`, name the files you read, and say what you ruled
|
||||
out. A clean sweep is a valid result; an invented defect is worse than none.
|
||||
@@ -39,6 +39,19 @@ a worker made all 59 of its edits in the primary's tree and never noticed.
|
||||
test "$(git rev-parse --show-toplevel)" = "$PWD" || cd "$(git rev-parse --show-toplevel)"
|
||||
```
|
||||
|
||||
**Never run `git stash` (or `git stash pop`/`apply`/`drop`).** Your worktree is isolated, but the
|
||||
stash is **not**: `refs/stash` is one stack shared by the primary's checkout and every other
|
||||
worker's worktree of this repo. Measured on 2026-09-04 — `git stash list` from a worker's worktree
|
||||
and from the primary's tree returned byte-identical output. So a `git stash` you run can be popped
|
||||
into someone else's tree, and a `git stash pop` you run can drop **another worker's** uncommitted
|
||||
edits on top of yours. This has already happened here: two workers were running in parallel and one
|
||||
of them had its in-progress edit silently overwritten by the other's stash.
|
||||
|
||||
The branch is your isolation, so use it instead. To set work aside, commit it on your own branch
|
||||
(`git commit -m "wip: ..."`) and carry on; to try something and back out, use
|
||||
`git diff > /tmp/<your-branch>.patch` then `git checkout -- <file>`. Both stay inside your worktree.
|
||||
If you find a stash entry you did not create, leave it alone and say so in your report.
|
||||
|
||||
## 2. Implement
|
||||
|
||||
- Implement exactly the scope the lead named. Keep the diff focused; note anything out of scope
|
||||
|
||||
@@ -0,0 +1,72 @@
|
||||
---
|
||||
name: redeploy-fleetd
|
||||
description: Rebuild and restart the live fleetd daemon after a merge (lead / primary only). Load this before redeploying — it holds the script, the drain step, the permission grant, and the five checks that have each gone wrong here before. Workers must never do this.
|
||||
---
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
scripts/redeploy-fleetd.sh --no-build # restart the jar already on disk
|
||||
```
|
||||
|
||||
`--no-build` skips the build and restarts whatever jar is at `fleetd/target/fleetd.jar`. Use it only
|
||||
when you just built and nothing changed since. It gives up the protection in the next paragraph: no
|
||||
build runs, so a stale or missing jar is not caught early. The script still checks the file is there
|
||||
and dies with `no jar at … — run without --no-build` if it is not, but it cannot tell you the jar is
|
||||
old. A `mvn clean` in the tree deletes that jar while the daemon keeps running on it, and nothing
|
||||
degrades until the next restart. Run `--check` first: it prints the jar's hash and its modification
|
||||
time, so you can see for yourself whether the jar is missing or older than the code you mean to ship.
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
|
||||
@@ -10,6 +10,11 @@ never merge) is in **`CLAUDE.md` → Bridge communication → Worker** and alrea
|
||||
skill is only the *review procedure*: how to work the scope, and the exact shape of what you
|
||||
send back.
|
||||
|
||||
**Wrong skill for a sweep.** This one reviews *one* diff or scope and reports the *single* most
|
||||
important issue. If the lead asked you to hunt or audit a whole package for several defects, load
|
||||
`hunter` instead and ignore this file — the two want different output, and following both is how a
|
||||
worker ends its turn with a good report that never gets sent.
|
||||
|
||||
## 1. Read the whole scope before you judge
|
||||
|
||||
The delegation names your scope — a file, a diff, a PR, a function. **Read all of it first.**
|
||||
|
||||
+10
-4
@@ -87,12 +87,18 @@ jobs:
|
||||
apt-get update && apt-get install -y --no-install-recommends maven
|
||||
mvn -version
|
||||
|
||||
# The `contract` profile clears the default-excludes group, so the @Tag("contract") AMQP test
|
||||
# runs against the RabbitMQ service container (AMQP_URI). Pinned to the one contract test to
|
||||
# avoid re-running the unit suite already covered by the `build` job.
|
||||
# The `contract` profile clears the default-excludes group, so `-Dgroups=contract` runs every
|
||||
# @Tag("contract") test and nothing from the unit suite the `build` job already covered — a
|
||||
# tag selects the whole group, so a test added to it later runs here automatically. A prior
|
||||
# version of this step pinned `-Dtest=AmqpReplyInboxContractTest` by class name instead: that
|
||||
# silently excluded every other contract test (including the herdr ones) from CI, and nobody
|
||||
# noticed until the herdr protocol drifted out from under a test that never ran here
|
||||
# (fleetd #449). If this runner has no herdr socket, the herdr-backed tests in the group
|
||||
# skip on their own `assumeTrue` and only the broker-backed ones actually run — check the
|
||||
# step output rather than assuming which.
|
||||
- name: Contract tests
|
||||
working-directory: fleetd
|
||||
run: mvn -B -Pcontract test -Dtest=AmqpReplyInboxContractTest
|
||||
run: mvn -B -Pcontract test -Dgroups=contract
|
||||
|
||||
- name: Failing test output
|
||||
if: failure()
|
||||
|
||||
@@ -19,3 +19,9 @@
|
||||
fleetd.out
|
||||
fleetd/fleetd.out
|
||||
logs/
|
||||
|
||||
# fleetd #480: the lead rollover handover file. `leadRollover.handoverPath` points here, and the
|
||||
# outgoing lead rewrites it on every rollover. It is a snapshot of one moment's live state —
|
||||
# unpushed branches, running builds, open questions — so it is stale the moment it is written and
|
||||
# has no business in git history.
|
||||
.handover/
|
||||
|
||||
@@ -7,6 +7,14 @@
|
||||
> wiki ([Use Cases](https://git.ltms.dev/fleet/fleetd/wiki/7-Use-Cases) → *The portable
|
||||
> CLAUDE.md block*); improvements go to the template first, then out to each project. Anything
|
||||
> specific to *this* repo lives under §Project addendum below, never inline above it.
|
||||
>
|
||||
> **Anything you measure in an addendum is perishable.** Date it, give the command that
|
||||
> re-measures it and what each outcome means, and tell the reader to delete the section once
|
||||
> it stops reproducing. The four parts work together: deciding what would falsify a claim is
|
||||
> the expensive step, and a reader in the middle of another task will not pay it, so a bare
|
||||
> "verify before relying on this" costs the same space and does nothing. The case this is for
|
||||
> is a note that goes stale as a live restriction — it will tell a future session it cannot do
|
||||
> the thing at the moment doing it becomes the job.
|
||||
|
||||
If no `fleet_*` MCP tools are mounted in this session, this section does not apply — skip it.
|
||||
|
||||
@@ -50,8 +58,9 @@ and the sender silently receives nothing. Fail toward the recoverable error.
|
||||
re-send because a call looks slow — the bridge delivers when the peer is `idle`, `blocked` or
|
||||
`done`. A spawned member must **also** have mounted the bridge MCP: until it has, it is not
|
||||
deliverable, and a send waits on that gate for ~60s and then fails without ever reaching its pane.
|
||||
5. **Never drive the terminal multiplexer directly** (no `herdr` CLI, no socket). The bridge owns
|
||||
policy; the multiplexer owns PTYs. Going around the bridge bypasses every rule above.
|
||||
5. **Never move a fleet session, pane or peer except through the bridge.** The bridge owns policy;
|
||||
the multiplexer owns PTYs. Any route that changes fleet state without the bridge's checks
|
||||
bypasses every rule above — the `herdr` CLI and its socket are the usual example.
|
||||
|
||||
### Primary (lead) — run this on every task, in order
|
||||
|
||||
@@ -71,8 +80,11 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
the final judgment call, verification, merges, and anything that depends on context only you
|
||||
hold. Nothing else is yours by default.
|
||||
3. **Spawn every delegated unit first** — `fleet_spawn{profile, worktree:true, ticket}`, one per
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model and cost, not in
|
||||
tier, so the default is rarely what you want.
|
||||
unit, *before* sending any. Pass `profile` explicitly: profiles differ in model, cost and
|
||||
LIVENESS, not in tier, so the default is rarely what you want. The default is whatever the
|
||||
daemon reports, and on a host where it sits on an exhausted or withdrawn credential every
|
||||
unqualified spawn fails — sometimes loudly, sometimes as a member that spawns fine and then
|
||||
produces nothing. `fleet_profiles` reports the default; check it once per session.
|
||||
4. **Then send them all** — `fleet_send{sessionId, content, wait:false}`. Line 1 of every brief is
|
||||
`Load the <name> skill.` naming the worker's playbook; those skills are opt-in and that line is
|
||||
what makes them reliable. Where the project ships no such skill, spell the procedure out in the
|
||||
@@ -94,6 +106,18 @@ below are the procedure — run them in order, every task, not only the big ones
|
||||
8. **Adjudicate, merge, tear down — yours alone.** Read the diff yourself: fully if it is small,
|
||||
targeted at the reported findings and the risky paths if it is large. Reviewer findings direct
|
||||
your attention; they never substitute for it. Then merge, then `fleet_stop{paneId}`.
|
||||
**If the forge refuses you the merge** — a protected branch, a token without the grant — the
|
||||
adjudication is still yours. Read the diff, decide, and hand the operator a merge-ready queue
|
||||
with the refusal quoted. Never report a PR as merged, and never call one "ready to merge"
|
||||
without having read the diff yourself. A refusal is exactly when that shortcut is tempting,
|
||||
because no action is left that forces you to look, and taking it turns this step into
|
||||
forwarding a reviewer's verdict — which is delegating the merge by proxy, two lines above.
|
||||
**Test a refusal; do not read it off a permissions field.** A protected branch holds its merge
|
||||
rights separately from the repository permissions, so that field can say yes while the merge is
|
||||
refused, and still say no after a grant makes it work. Probe instead, with a request that cannot
|
||||
succeed on its merits, so a rejection can only mean the refusal. Treat a transport failure as a
|
||||
third answer that proves nothing: a timeout, a DNS error or a bad URL is not a refusal, and
|
||||
counting it as one makes you sure of something you never measured.
|
||||
|
||||
**Steps 3 and 4 are separate on purpose** — spawning and sending in one loop is how parallel work
|
||||
silently becomes serial, and it is the most common way this layer is wasted. For the same reason,
|
||||
@@ -114,9 +138,11 @@ the merge — and merging on a reviewer's word is delegating it by proxy.
|
||||
| Answer a member's `fleet_ask` | `fleet_send{turnId, content}` — **not** `sessionId` |
|
||||
| Message a **peer lead** on this host | `fleet_send{sessionId: <their terminal>, content}` — `fleet_list` → `leads` reports it. Coordination only, **never** a task |
|
||||
| Message a **peer lead** on another daemon or host | `fleet_send{coordId: <their coord-id>, content}` — needs a `coordinator:` block; your own coord-id is in `fleet_list`. Coordination only, **never** a task |
|
||||
| Answer a peer lead that messaged you | `fleet_reply{content}` — the one case a lead replies |
|
||||
| Answer a peer lead that messaged you | `fleet_send{coordId}` — or `{sessionId}` if they are on this host. **Not** `fleet_reply`: it has no peer route and the publish is refused |
|
||||
| Read your own held lead-to-lead mail (no ack) | `fleet_poll{coordId: <your own coord-id, from fleet_list's coordinator.selfId>}` — primary-only; never acks, so `fleet_list`'s `held[]` still shows it after. `fleet_list`'s `held[]` gives only a truncated preview — this is the only way to read the full body |
|
||||
| Collect a held reply | `fleet_poll{target}` · then `fleet_ack{target, msgId}` |
|
||||
| Tear down a member | `fleet_stop{paneId}` |
|
||||
| Replace your OWN lead session when its context is full | `fleet_handover{action:"open", reason?}` → write the handover file it names → `fleet_handover{action:"confirm", token, operatorConfirmed}`. Primary-only. **In that order**: the file must be modified *after* `open`, or `confirm` refuses it as stale. There is no terminal parameter — the pane is always your own, so you can never roll another lead. `{action:"cancel", token}` drops a pending request |
|
||||
|
||||
### Lead ↔ lead — coordinate, never delegate
|
||||
|
||||
@@ -140,10 +166,21 @@ The traffic between leads is coordination and nothing else:
|
||||
3. **Verify a peer exactly as you verify yourself.** Peer status buys nothing: check the claim
|
||||
against the code, and re-run the build. A peer's correction gets the same treatment — right or
|
||||
wrong on the evidence, not on who said it. Neither of you merges the other's work unreviewed.
|
||||
**N observations are N data points only if they differ in the axis you are trusting.** This cuts
|
||||
both ways. N *failures* blamed on one cause are one data point when the cases share what you are
|
||||
not varying. N *agreeing measurements* are also one data point when they share an instrument —
|
||||
two hosts, two operators and the same formula is one formula, not two confirmations.
|
||||
4. **Ask a peer to read your project addendum.** Your addendum is instruction surface: every future
|
||||
session on your host obeys it, and a wrong one is obeyed just as faithfully as a right one. The
|
||||
author is the worst reader of their own qualifier placement — measured here, one addendum carried
|
||||
two defects and a non-author found both. If you have no peer, at least re-read it asking "which
|
||||
sentence goes false first, and would a reader reach the caveat before acting?"
|
||||
|
||||
Being messaged by a peer does not make you its worker: answer with `fleet_reply`, and push back on
|
||||
the substance if it is wrong. A peer that simply complies has thrown away the reason there are two of
|
||||
you.
|
||||
Being messaged by a peer does not make you its worker: answer the way you would open —
|
||||
`fleet_send{coordId}` for another daemon, `fleet_send{sessionId}` on this host — and push back on
|
||||
the substance if it is wrong. `fleet_reply` resolves a member's blocked `fleet_send`; a peer's
|
||||
coord-id message is durable and non-blocking, so there is nothing for it to resolve. A peer that
|
||||
simply complies has thrown away the reason there are two of you.
|
||||
|
||||
### Member (worker or architect) — the turn contract
|
||||
|
||||
@@ -188,76 +225,85 @@ must obey belongs in the charter, not here.
|
||||
- **This repo is the bridge.** The daemon is `fleetd`, its MCP mount is `http://127.0.0.1:8765/mcp`,
|
||||
and the code behind the rules above is `mcp/FleetMcp` (tools), `auth/Authz` (the role table),
|
||||
`mcp/ConnectionIdentity` (connection→role), and `worker/*Launcher` (`REPLY_CHARTER`).
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR) and
|
||||
`reviewer` (scoped review → one structured finding). Name one in every delegation.
|
||||
- **Herdr socket tests (measured 2026-09-10).** In this repo, herdr is a subject under test. A
|
||||
worker assigned to herdr code, and the lead, may let a test open the herdr socket directly in a
|
||||
throwaway workspace that the test tears down. This only covers
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/AgentControlContractTest.java`,
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/HerdrContractTest.java`,
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/PaneLocatorContractTest.java`, and
|
||||
`fleetd/src/test/java/dev/ltms/fleet/herdr/WorkspacePlacementContractTest.java`. It is not a
|
||||
general licence. Using the herdr CLI or socket to move a real fleet session, pane, or peer stays
|
||||
banned. That is the control plane that invariant 5 protects. Re-measure with
|
||||
`grep -rl 'UnixSocketHerdrClient.connect()' fleetd/src/test/java --include='*.java'`. A non-empty
|
||||
result means tests still open the socket and this note still applies. An empty result means nobody
|
||||
does this any more; delete this section. Canonical invariant 5 restatement is tracked in #458 and
|
||||
is not part of this change.
|
||||
- **`fleet_profiles`/`fleet_list` report two separate outage states, and they are not the same
|
||||
thing.** *Quarantined* (CB-578) means the backend told us it is out of capacity — a long,
|
||||
1800s-default cooldown. *Cooling off* (fleetd #201/#227) means a profile's credential threw two
|
||||
distinct backend errors (a non-exhaustion failure such as an HTTP 5xx) within 60 seconds — a
|
||||
short, fixed 60s cooldown, not configurable per profile. Each check runs independently, so a
|
||||
profile can show both at once. In the JSON: a cooling profile carries `credentialId` and
|
||||
`coolingOffForSeconds`; a quarantined profile carries `quarantinedForSeconds`; a profile hit by
|
||||
both carries all three fields, and either state alone already sets that profile's `free` to `0`.
|
||||
A `fleet_spawn` naming a cooling-off profile is refused before it ever reaches the backend
|
||||
adapter, with a message naming the credential and the remaining seconds ("cooling off after
|
||||
repeated backend errors") — distinct wording from a quarantine refusal, so don't conflate the
|
||||
two when reading a spawn failure.
|
||||
- **Skills available to delegate:** `implementer` (worktree → commit → push → own PR),
|
||||
`reviewer` (one diff → one structured finding) and `hunter` (sweep a package → several ranked
|
||||
findings, change nothing). Name exactly one in every delegation. **`reviewer` and `hunter` are
|
||||
not interchangeable** — `reviewer` caps the answer at one finding in about 90 words, so naming
|
||||
it for a multi-finding sweep hands the worker two contradictory output contracts. That has
|
||||
already cost three workers' turns: each wrote a good report to its terminal and ended the turn
|
||||
with no `fleet_reply`, and the scrape returned the tail of the brief instead.
|
||||
- **Primary-side skills** (not delegation playbooks — a worker cannot use them):
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace) and
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance).
|
||||
`port-to-opencode` (make an OpenCode session a participant in this workspace),
|
||||
`fleets-status` (report every fleet that shares one LavinMQ instance),
|
||||
`redeploy-fleetd` (rebuild and restart the live daemon after a merge) and
|
||||
`handover` (write the file a fresh lead session inherits when the outgoing one hands off,
|
||||
fleetd #480).
|
||||
- **This repo is also a Claude Code marketplace, and ships a plugin.** `.claude-plugin/marketplace.json`
|
||||
points at `plugin/`, which carries the MCP mount and the `setup` skill
|
||||
(`/claude-bridge:setup` — make any project bridge-ready). It was added in CB-527 and then went
|
||||
unmentioned by every instruction file, so it drifted and a later session planned it from scratch
|
||||
(#362). **Read `plugin/` before designing anything about onboarding a project.** Two limits are
|
||||
structural, not bugs: a plugin cannot carry the role agent files, because
|
||||
`ClaudeCodeLauncher.java:371` requires `<cwd>/.claude/agents/<role>.md` in the member's own
|
||||
worktree; and a plugin cannot deliver anything to members at all, because
|
||||
`ClaudeCodeLauncher.java:285` exports `CLAUDE_CONFIG_DIR` and every Claude profile here sets it,
|
||||
so a member never reads the operator's plugin store. **The plugin is the lead-side surface;
|
||||
member-facing assets travel in the worktree.**
|
||||
- **Never commit** `.mcp.json` (the primary's local copy, flagged `--skip-worktree`) or `wiki/`
|
||||
(a submodule with its own remote).
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, turn-done fallback —
|
||||
are diagrammed in `docs/MCP-Contract.md` **§6 only**. The rest of that page is a pre-build design
|
||||
doc whose tool names, parameter names and REST paths never caught up with the code, so do not use
|
||||
it as the tool reference (CB-609). Section 6 is kept out of this file because this file loads into
|
||||
every session's context.
|
||||
- **A provisioned worktree neutralizes `.mcp.json`, `opencode.json` and `.autoenv`** — the repo's
|
||||
committed copies would otherwise mount the primary's IDE and forge servers (fleetd #134). The
|
||||
worktree's copy of each is a stub, **not** the repo's real file, so a worker that reads one and
|
||||
reports what it found is reporting on the stub. The daemon logs a per-spawn summary, but the
|
||||
worker cannot see that log. From inside its own worktree a worker — or a lead debugging one —
|
||||
reads the list with `git config --worktree --get-all fleet.neutralizedConfig`, and the
|
||||
consequence with `git config --worktree --get fleet.neutralizedConfigNote`. Never brief a worker
|
||||
to edit one of these files: the edit cannot be committed, and it will not tell you so.
|
||||
- **Flows and the error model** — rendezvous, `fleet_ask`, detached delivery, the turn-done
|
||||
fallback and status gating — are diagrammed in `docs/MCP-Contract.md`. That page is now flows
|
||||
only: its pre-build tool catalogue, parameter tables and REST paths were deleted rather than
|
||||
corrected, because a hand-maintained second copy of the tool surface is what drifted for a month
|
||||
while this line pointed every session at it (CB-609 / #114). **The live MCP schema is the tool
|
||||
reference**, with the intent→tool table above as the short form. `McpContractDocTest` fails if
|
||||
that page names a `fleet_*` tool the server does not register. The flows are kept out of this
|
||||
file because this file loads into every session's context.
|
||||
|
||||
### Redeploying the daemon — the lead may do this (primary only)
|
||||
|
||||
**A merge is not a deployment.** The running `fleetd` holds the jar it was started with, so a
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. Saying "shipped"
|
||||
about code the live daemon has never loaded is a false report. The lead **may and should** redeploy
|
||||
rather than hand the job back to the operator.
|
||||
feature merged to `main` does nothing until the daemon is rebuilt and restarted. The lead **may and
|
||||
should** redeploy rather than hand the job back to the operator. **Workers must never do this** — a
|
||||
worker has no business restarting the daemon it is talking through, and stopping it kills the
|
||||
worker's own channel mid-turn.
|
||||
|
||||
Workers must never do this. A worker has no business restarting the daemon it is talking through,
|
||||
and stopping it kills the worker's own channel mid-turn.
|
||||
|
||||
**Use the script — do not hand-roll the steps.**
|
||||
|
||||
```bash
|
||||
scripts/redeploy-fleetd.sh --check # report state, change nothing
|
||||
scripts/redeploy-fleetd.sh # build, confirm drain, restart, verify
|
||||
scripts/redeploy-fleetd.sh --yes # skip the drain prompt (fleet already checked)
|
||||
```
|
||||
|
||||
It builds before it stops anything, so a failed build never leaves the fleet down; it waits for the
|
||||
old process to exit rather than assuming; it polls `/healthz`; and it anchors its log checks to a
|
||||
line marker taken before the restart, so old errors cannot be misread as new ones. Run `--check`
|
||||
first — it is read-only and reports whether the forge token resolves, which nothing else tells you.
|
||||
|
||||
The script encodes the five things below, each of which has gone wrong here before. Read them anyway:
|
||||
if the script is unavailable or a step fails, this is what it was protecting you from.
|
||||
|
||||
1. **Login shell, or workers silently lose their forge token.** The daemon inherits
|
||||
`WORKER_GITEA_TOKEN` from the shell that starts it, and that comes from
|
||||
`${SHARED_ENV}/tools/secrets.sh`. Start it from a non-login shell and the variable is empty, the
|
||||
daemon starts fine, and the failure appears much later as workers that cannot open a PR. Nothing
|
||||
logs this at startup — the script's `--check` is the only thing that reports it, and it checks
|
||||
whether the name resolves without ever printing the value.
|
||||
2. **Drain live members first.** `fleet_list`, then `fleet_stop` each member, and collect anything
|
||||
you still want with `fleet_poll` before you kill anything. A restart drops in-flight tickets and
|
||||
rendezvous, and a member's report is not recoverable once its ticket is gone.
|
||||
3. **A restart is the only way deferred config keys take effect.** That is usually the reason to do
|
||||
it. The startup log names which keys it accepted and which it deferred — read those lines rather
|
||||
than assuming.
|
||||
4. **Re-check identity afterwards.** Call `fleet_whoami` and confirm it still answers `primary`. The
|
||||
lead is found by its tab label (`fleet.leaders.*.tab`), and a lead whose tab no longer matches is
|
||||
demoted to worker, which refuses every orchestration call.
|
||||
5. **Prove the new jar is the one running.** Confirm a *fresh* `fleetd listening` line at the end of
|
||||
`fleetd/fleetd.out`, dated after the restart. An old daemon that never died looks identical from
|
||||
the outside.
|
||||
|
||||
**Permission.** A `CLAUDE.md` rule grants intent, not tool permission — the command classifier
|
||||
refuses a bare `kill` on the daemon whatever this file says. The script is the seam that fixes that:
|
||||
it is one auditable command, so the operator allow-lists it once instead of approving a stop and a
|
||||
start every time. The rule lives in the operator's Claude Code settings:
|
||||
|
||||
```json
|
||||
{ "permissions": { "allow": ["Bash(scripts/redeploy-fleetd.sh:*)"] } }
|
||||
```
|
||||
|
||||
Granted by the operator on 2026-08-15. If a call is still refused, do **not** route around it by
|
||||
running the stop and start as separate commands — that is exactly the approval the script replaced.
|
||||
Say what you were going to run and why, and let the operator decide.
|
||||
**Load the `redeploy-fleetd` skill before you redeploy.** It holds `scripts/redeploy-fleetd.sh`
|
||||
and its flags, the drain step, the operator's permission grant, and the five checks that have each
|
||||
gone wrong here before. Do not hand-roll the steps from memory.
|
||||
|
||||
### The prompt is part of the product — update it with the code (mandatory)
|
||||
|
||||
@@ -305,6 +351,16 @@ print("in sync:", w[i:w.index("\n```\n", i) + 1] == block)
|
||||
PY
|
||||
```
|
||||
|
||||
**Only the lead can run that check (measured 2026-09-10).** A member's provisioned worktree has
|
||||
`wiki/` uninitialized, so the script dies with `FileNotFoundError: wiki/7-Use-Cases.md`. Measured
|
||||
in three worker worktrees: `git submodule status` printed a leading `-` and `wiki/` held 0
|
||||
entries; the primary's own clone printed a leading `+` and the file was there. So never make this
|
||||
check a member's acceptance criterion — it is unsatisfiable for them, and a brief that asks for it
|
||||
is asking a worker to invent a pass. A member told to check it must say it could not run it, and
|
||||
must never report it as passed. The lead runs it in the main clone before merging. Re-measure with
|
||||
`git submodule status` in a member's worktree: a leading `-` means this still applies; once it
|
||||
prints a commit with no `-`, delete this paragraph.
|
||||
|
||||
## IDE MCP tools & validation workflow (enforced)
|
||||
|
||||
> **Primary only.** Workers have no IDE MCP mount — if you are a worker, skip this section and
|
||||
|
||||
+54
-30
@@ -1,56 +1,80 @@
|
||||
# CB-504 — systemd unit for fleetd (Linux).
|
||||
#
|
||||
# The macOS launchd agent (deploy/dev.ltms.fleetd.plist) is the supervision target for the
|
||||
# current single-host deployment. This unit exists for the per-host gateways CB-308 introduces,
|
||||
# which will run on Linux.
|
||||
# fleetd #360: the previous version of this file started clean and broke the daemon in three ways
|
||||
# that nothing logs (see the DO NOT block and the ExecStart/PrivateTmp comments below for what and
|
||||
# why). The unit below, plus its companion deploy/herdr.service, is the version that has actually
|
||||
# run on fleet01 without those failures. Do not "improve" it back toward the old shape without
|
||||
# re-reading why each line is the way it is.
|
||||
#
|
||||
# Install (user service — fleetd drives the user's herdr, not a system daemon):
|
||||
# mkdir -p ~/.config/systemd/user
|
||||
# cp deploy/fleetd.service ~/.config/systemd/user/
|
||||
# # edit ExecStart / WorkingDirectory / Environment below, then:
|
||||
# cp deploy/fleetd.service deploy/herdr.service ~/.config/systemd/user/
|
||||
# # edit WorkingDirectory / ExecStart below for your host's paths and java location
|
||||
# systemctl --user daemon-reload
|
||||
# systemctl --user enable --now fleetd
|
||||
# systemctl --user enable --now herdr fleetd
|
||||
# loginctl enable-linger $USER # REQUIRED -- see below
|
||||
# journalctl --user -u fleetd -f
|
||||
#
|
||||
# `loginctl enable-linger` is not optional and is easy to miss, because leaving it out looks like
|
||||
# success: `systemctl --user enable` reports "enabled" and both units run for as long as you stay
|
||||
# logged in. A user manager without lingering starts at your first login and stops at your last
|
||||
# logout, so the fleet simply does not come back after a reboot -- which is the whole reason to
|
||||
# use systemd here rather than the setsid scripts these units replaced. Check it with
|
||||
# `loginctl show-user $USER -p Linger`; the answer must be `Linger=yes`.
|
||||
#
|
||||
# Secrets (AI_GATEWAY_TOKEN, WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI, ...) are not set
|
||||
# here and need no systemd drop-in: ExecStart runs a login shell, so they come from wherever your
|
||||
# login shell already sources them (this host: ~/.fleet/secrets.sh via ~/.zprofile). If a token is
|
||||
# missing there, fleetd still starts — the daemon reports every secret a configured profile
|
||||
# references, by name, never by value:
|
||||
# journalctl --user -u fleetd | grep 'startup secret'
|
||||
# A resolved one logs "startup secret NAME: set (profile 'x' tokenEnv)"; a missing one logs
|
||||
# "startup secret NAME: MISSING" at WARN and the daemon starts anyway — the first visible symptom
|
||||
# is a member that cannot open a pull request, hours later and in a different component.
|
||||
|
||||
[Unit]
|
||||
Description=fleetd — claude-bridge message server
|
||||
Description=fleetd — fleet message server
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
# Ordering only: herdr is a user process and its socket may appear after us. This is advisory —
|
||||
# fleetd retries the herdr socket rather than exiting, which is what actually makes a late
|
||||
# socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down with it.
|
||||
# Ordering only. fleetd retries the herdr socket rather than exiting, which is what actually makes
|
||||
# a late socket survivable. Do NOT add Requires=: a herdr restart must not take fleetd down too.
|
||||
After=herdr.service
|
||||
Wants=herdr.service
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
WorkingDirectory=%h/src/claude-bridge/fleetd
|
||||
ExecStart=/usr/lib/jvm/temurin-25-jdk/bin/java -jar target/fleetd.jar fleetd.yaml
|
||||
WorkingDirectory=%h/LTMS/fleetd/fleetd
|
||||
|
||||
Environment=HERDR_SOCKET_PATH=%h/.config/herdr/herdr.sock
|
||||
# PATH matters more than it looks (CB-511): fleetd propagates its own PATH to every worker it
|
||||
# spawns, so this line decides whether the fleet can run a build at all. systemd does not source a
|
||||
# login shell, so without it the daemon — and every worker — gets a bare default with no JDK/Maven.
|
||||
Environment=PATH=/usr/lib/jvm/temurin-25-jdk/bin:/usr/share/maven/bin:/usr/local/bin:/usr/bin:/bin
|
||||
# Secrets are NOT set here — this file is committed. Put the API/worker tokens in a private
|
||||
# drop-in that systemd reads with restrictive permissions:
|
||||
# systemctl --user edit fleetd → [Service] / Environment=FLEETD_API_TOKEN=...
|
||||
# or point EnvironmentFile at a 0600 file:
|
||||
# EnvironmentFile=%h/.config/fleetd/env
|
||||
# A LOGIN shell, not java directly. Every secret this daemon needs (AI_GATEWAY_TOKEN,
|
||||
# WORKER_GITEA_TOKEN, LAVINMQ_URI, COORD_AMQP_URI) lives in ~/.fleet/secrets.sh, which only
|
||||
# ~/.zprofile sources. systemd runs no login shell. Started any other way the daemon boots fine
|
||||
# and looks healthy, and the failure appears hours later as a member that cannot open a pull
|
||||
# request. exec keeps it one process, so systemd tracks the right PID.
|
||||
# This also avoids a SECOND copy of the secrets in a systemd drop-in: one source of truth.
|
||||
ExecStart=/bin/zsh -lc "exec java -jar target/fleetd.jar fleetd.yaml"
|
||||
|
||||
# PrivateTmp MUST stay false -- see herdr.service. fleetd creates the member ZDOTDIR scrub dir and
|
||||
# the opencode config dir under java.io.tmpdir, and the member pane (a herdr child, a different
|
||||
# unit) has to read them. A private /tmp turns the credential scrub into a silent no-op.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=10s
|
||||
# A bad config (e.g. a non-loopback bind without token auth) makes fleetd fail fast by design.
|
||||
# Give up rather than restart-loop on a permanent error.
|
||||
# A bad config makes fleetd fail fast by design. Give up rather than restart-loop forever.
|
||||
StartLimitBurst=5
|
||||
StartLimitIntervalSec=120
|
||||
|
||||
# The daemon reads the repo, writes worktrees, and talks to a Unix socket — it needs no more.
|
||||
# DO NOT add ProtectSystem=, ProtectHome=, ProtectKernelTunables= or ProtectControlGroups=.
|
||||
# Measured on fleet01 2026-09-05: each of those gives the unit its own mount namespace, and
|
||||
# fleetd resolves a caller role by running lsof to find the loopback peer PID
|
||||
# (mcp/LsofPeerPidLookup). Inside such a namespace lsof returns nothing, every caller falls back
|
||||
# to ANONYMOUS, and the primary is refused every orchestration call with
|
||||
# "unauthenticated: anonymous may not SPAWN".
|
||||
# The daemon still starts, healthz still returns ok and the secrets still resolve - the only
|
||||
# symptom is that the fleet cannot be driven at all. Verified by bisecting the directives:
|
||||
# no sandbox 3 lsof lines | ProtectSystem=strict 0 | ProtectHome=read-only 0
|
||||
# ProtectKernelTunables 0 | ProtectControlGroups 0 | RestrictSUIDSGID 3 | NoNewPrivileges 3
|
||||
# The two below add no mount namespace and are safe.
|
||||
NoNewPrivileges=true
|
||||
PrivateTmp=true
|
||||
ProtectSystem=strict
|
||||
ProtectHome=read-write
|
||||
ProtectKernelTunables=true
|
||||
ProtectControlGroups=true
|
||||
RestrictSUIDSGID=true
|
||||
|
||||
StandardOutput=journal
|
||||
|
||||
Executable
+19
@@ -0,0 +1,19 @@
|
||||
#!/bin/zsh
|
||||
# fleetd #360 — template for the script deploy/herdr.service's ExecStart wraps in a pty.
|
||||
#
|
||||
# `script -qfec <this> /dev/null` needs a real command to run, and that command has to be a LOGIN
|
||||
# shell script: herdr itself needs the same secrets fleetd.service's login shell picks up (this
|
||||
# host: ~/.fleet/secrets.sh via ~/.zprofile), because members it spawns inherit its environment.
|
||||
# systemd's own Environment= lines in herdr.service are not enough for that -- they set TERM and a
|
||||
# bare PATH so the pty starts at all, nothing more.
|
||||
#
|
||||
# Copy this file to the path deploy/herdr.service's ExecStart names
|
||||
# (%h/LTMS/fleetd/fleetd-run/herdr-inner.sh by default) and `chmod +x` it. Not committed under
|
||||
# that path itself because the session name below is host-specific.
|
||||
|
||||
# A 0x0 pty makes every pane spawn fail with "ghostty error -2" (see herdr-multi-instance-facts /
|
||||
# fleet01-headless-herdr-standup) -- give it a real size before herdr ever touches it.
|
||||
stty rows 50 cols 200
|
||||
|
||||
# -l: login shell, so herdr and everything it spawns gets the real secrets and PATH.
|
||||
exec zsh -lc 'exec herdr --session <name>'
|
||||
@@ -0,0 +1,40 @@
|
||||
# fleetd #360 — systemd unit for herdr (Linux), the terminal multiplexer fleetd drives.
|
||||
#
|
||||
# This is fleetd.service's companion: fleetd.service's After=/Wants=herdr.service assumes this
|
||||
# unit exists. Before this ticket it did not, so on a fresh host fleetd started against a herdr
|
||||
# that systemd never supervised at all.
|
||||
#
|
||||
# Install: see deploy/fleetd.service's header comment (both units install the same way).
|
||||
#
|
||||
# ExecStart below runs deploy/herdr-inner.sh (copy the template of that name from this directory
|
||||
# to the path in ExecStart, or point ExecStart at wherever you keep it, and make it executable).
|
||||
# It is a separate file rather than an inline command because it must itself be a login shell (see
|
||||
# its own header for why) and systemd's ExecStart does not run one.
|
||||
|
||||
[Unit]
|
||||
Description=herdr terminal multiplexer (fleet session)
|
||||
Documentation=https://git.ltms.dev/fleet/fleetd/wiki
|
||||
|
||||
[Service]
|
||||
Type=simple
|
||||
# script(1) gives herdr a real pty. Without it the client reports a 0x0 window and every pane
|
||||
# spawn fails with "ghostty error -2" -- which surfaces as a fleetd spawn failure, not a herdr one.
|
||||
ExecStart=/usr/bin/script -qfec %h/LTMS/fleetd/fleetd-run/herdr-inner.sh /dev/null
|
||||
StandardInput=null
|
||||
Environment=TERM=xterm-256color
|
||||
Environment=PATH=%h/.local/bin:/usr/local/bin:/usr/bin:/bin
|
||||
|
||||
# PrivateTmp MUST stay false. fleetd writes the member ZDOTDIR scrub dir and the opencode config
|
||||
# dir under its own java.io.tmpdir, and the member pane -- a child of THIS process -- has to read
|
||||
# them. A private /tmp here silently breaks the credential scrub instead of failing loudly.
|
||||
PrivateTmp=false
|
||||
|
||||
Restart=on-failure
|
||||
RestartSec=5s
|
||||
|
||||
StandardOutput=journal
|
||||
StandardError=journal
|
||||
SyslogIdentifier=herdr
|
||||
|
||||
[Install]
|
||||
WantedBy=default.target
|
||||
@@ -0,0 +1,531 @@
|
||||
# CB-201 and CB-227 refinement
|
||||
|
||||
Date: 2026-09-03
|
||||
|
||||
## Decision
|
||||
|
||||
#201 and #227 are one delivery program, but they are not one implementation unit.
|
||||
|
||||
#201 has a real seam: `CompletionResolver` can publish a typed backend-error event only after its
|
||||
waiter resolution wins. #227 can consume that event without knowing any pane text. The classifier
|
||||
must land before the final #227 wiring. However, the policy engine, roster state, and lead nudge can
|
||||
be built in parallel with the classifier.
|
||||
|
||||
I propose five units. Units 1 to 4 own separate files and can run in parallel. Unit 5 owns all
|
||||
composition files and lands after them. It also depends on the #234 defect 2 fix named in the task.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
U1["Unit 1: typed backend-error classification"] --> U5["Unit 5: wire policy, spawn gate, and fleet views"]
|
||||
U2["Unit 2: credential outage policy"] --> U5
|
||||
U3["Unit 3: lead outage nudge"] --> U5
|
||||
U4["Unit 4: durable member outcome"] --> U5
|
||||
D234["#234 defect 2: fail-loud target resolution"] --> U5
|
||||
```
|
||||
|
||||
*Figure 1. Four file-disjoint foundations feed one composition unit.*
|
||||
|
||||
This split keeps `Fleetd.java` under one owner. It also keeps every other production file under one
|
||||
unit in this plan.
|
||||
|
||||
## Evidence checked in the current branch
|
||||
|
||||
I read both issue pages in full. Each page reports zero comments.
|
||||
|
||||
| Evidence | What the code says now |
|
||||
|---|---|
|
||||
| `inject/CompletionResolver.java:229-237` | A turn below two seconds fails before normal scrape classification. A matching fast backend error is therefore only a generic failure today. |
|
||||
| `inject/CompletionResolver.java:260-276` and `:332-367` | #211 already added raw-screen classification when `lastAssistantBlock` is empty. The “dead-code question” in #201 is stale on this branch. |
|
||||
| `inject/CompletionResolver.java:288-317` | Exhaustion wins before the hard-coded `API Error:` match. A backend error then goes through generic `fail(...)`. |
|
||||
| `inject/CompletionResolver.java:311-316` | The code admits that the pattern is a heuristic. A member report which quotes an API error may match it. |
|
||||
| `inject/CompletionResolver.java:449-467` | Startup coverage exists only for `exhaustedPattern`. |
|
||||
| `inject/ExhaustedPatternLookup.java:13-25` | The current lookup and explicit `none()` value are a good shape for the new classifier seam. |
|
||||
| `Fleetd.java:196-207` | One `BackendQuarantine` is shared by placement and the exhaustion sink. Its cooldown comes from `quarantineCooldownSeconds`. |
|
||||
| `Fleetd.java:322-363` | Pattern compilation, target-to-profile lookup, and the live `ExhaustionSink` are composed in `Fleetd.main`. The sink on this branch still ends in `.ifPresent(...)`. This plan assumes #234 replaces that silent path. |
|
||||
| `placement/BackendQuarantine.java:60-87` | A repeated exhaustion restarts one long quarantine. The store is credential-keyed and uses an injected monotonic clock. |
|
||||
| `member/CompositePeerLauncher.java:260-317` | Explicit and policy-selected spawns have separate gates. Both paths must learn about outage cool-off. |
|
||||
| `member/CompositePeerLauncher.java:347-379` | Exhaustion refusal already checks a credential for explicit spawns and filters policy candidates. Its error text says “exhausted”. |
|
||||
| `placement/PlacementContext.java:10-22` and `PlacementPolicyUtil.java:14-83` | Automatic placement has only one transient exclusion set named `quarantined`. Reusing it would make outage errors say “backend exhausted”. |
|
||||
| `mcp/FleetMcp.java:913-1025` | `fleet_list` sets `free: 0` and adds `credentialId` plus `quarantinedForSeconds` when quarantine is active. |
|
||||
| `session/MemberSession.java:51-59` | The roster has `DONE` and generic `FAILED`, but no backend-error state or stored reason. |
|
||||
| `session/SessionManager.java:695-773` | A normal boundary moves `BUSY` to `DONE`. A failure moves any non-released session to `FAILED`. The async completion resolver can race the `DONE` update. |
|
||||
| `session/SessionManager.java:648-687` | `rosterView` reports the session state, but it reports no terminal reason. |
|
||||
| `msg/MessageService.java:922-940` | CB-588 already nudges for every terminal async ticket, including failures. Current code would report failed tickets, but it would not report one correlated outage. |
|
||||
| `msg/ReplyPushLoop.java:20-48` | Replies, terminal tickets, and questions share one per-lead schedule. This prevents two push sources from injecting competing turns. |
|
||||
| `msg/ReplyPushLoop.java:305-395` | Each push entry point resolves the owning lead through `PrimaryRegistry`. Missing ownership is logged and the durable or pending item remains the backstop. |
|
||||
| `msg/ReplyPushLoop.java:496-547` | One tick builds one combined nudge. Pending items have separate reminder counts. |
|
||||
| `health/FleetHealthMonitor.java:91-143` | Health is a slow periodic observer of members and message-layer facts. It does not receive completion classifications. |
|
||||
| `health/FleetHealthMonitor.java:206-208` | `healthCoverage` means health enabled plus webhook configured. It does not describe lead-pane alerts. |
|
||||
| `Fleetd.java:465-486` | Health stays `detection-only` without the webhook notification setting. |
|
||||
|
||||
I also read the related unit tests for `CompletionResolver`, `ReplyPushLoop`, `BackendQuarantine`,
|
||||
`CompositePeerLauncher`, `PlacementPolicyUtil`, `SessionManager`, `MessageService`, and `FleetMcp`.
|
||||
|
||||
I did not inspect the in-progress #234 branch. I only used the two measured facts in the task. No
|
||||
peer architect was named, so I did not exchange a design with one.
|
||||
|
||||
## Required behaviour
|
||||
|
||||
The policy should use these first values:
|
||||
|
||||
- Threshold: **2** classified backend errors.
|
||||
- Window: **60 seconds**, measured from the first error to the second.
|
||||
- Cool-off: **60 seconds**, starting when the threshold is reached.
|
||||
- Correlation key: `credentialId`, never profile name and never error text.
|
||||
- Incident rule: one active incident per credential. Errors during its cool-off do not extend it and
|
||||
do not create more lead notices.
|
||||
- Rearm rule: after cool-off ends, two fresh errors are needed for another incident.
|
||||
|
||||
Two errors are the smallest threshold which protects the honest one-turn failure. A 60-second window
|
||||
fits the measured two-member outage. A 60-second cool-off blocks immediate repeat spawns without
|
||||
turning a short backend fault into the default 1,800-second exhaustion quarantine.
|
||||
|
||||
A single classified error still fails its send and marks its member `backend_error`. It does not
|
||||
cool a credential and does not send an outage notice. This is what “a single error changes nothing”
|
||||
must mean at the credential level. It cannot mean that the failed member still looks successful.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant R1 as Resolver for member A
|
||||
participant R2 as Resolver for member B
|
||||
participant P as Outage policy
|
||||
participant S as Spawn gate
|
||||
participant N as Lead push loop
|
||||
participant L as Lead pane
|
||||
|
||||
R1->>P: backend error for credential C
|
||||
Note over P: Count 1, no cool-off
|
||||
R2->>P: backend error for credential C within 60s
|
||||
P->>P: Start one 60s incident
|
||||
P->>S: Credential C is cooling off
|
||||
P->>N: Queue one incident notice
|
||||
N->>L: Inject when lead is idle, blocked, or done
|
||||
L->>S: Request another spawn on credential C
|
||||
S-->>L: Refuse and report remaining cool-off
|
||||
```
|
||||
|
||||
*Figure 2. The second independent classification creates the fleet-level event.*
|
||||
|
||||
Against the 2026-09-01 case, the second failed member would start cool-off. `fleet_list` would show
|
||||
zero free capacity and both members as `backend_error`. The push loop would inject one outage notice
|
||||
even if the lead had not polled either ticket yet. The design reports the outage. It does not recover
|
||||
uncommitted work from the members.
|
||||
|
||||
## Unit 1 — Typed backend-error classification
|
||||
|
||||
### Scope
|
||||
|
||||
Replace the direct hard-coded check inside `CompletionResolver` with a lookup and a sink. Keep the
|
||||
public send result as a failed send. The typed internal event is the seam #227 consumes.
|
||||
|
||||
The lookup returns the pattern for a target. The sink receives the target, matched line, and full
|
||||
failure reason. It fires only after `Rendezvous.resolveFailure(...)` wins for that exact captured
|
||||
waiter. This copies the race rule already used by `ExhaustionSink`.
|
||||
|
||||
The classifier must run in all three current paths:
|
||||
|
||||
1. a normal non-empty assistant block;
|
||||
2. the #211 raw scrape fallback;
|
||||
3. a turn inside `MIN_TURN_NANOS`, before it becomes a generic too-fast failure.
|
||||
|
||||
In every path, the order stays: stale-baseline guard, exhaustion, backend error, then generic
|
||||
failure or completion. A fast turn still fails when no configured pattern matches.
|
||||
|
||||
Keep `(?i)\bAPI Error\s*:` as a compatibility pattern for profiles without `errorPattern` until the
|
||||
operator config is updated. Do not call this full coverage. Startup reporting in Unit 5 must name
|
||||
profiles using this weaker legacy default.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorPatternLookup.java`.
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/inject/BackendErrorSink.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/inject/CompletionResolver.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/inject/CompletionResolverTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. A target-specific error pattern matches a normal assistant block and resolves the send as failed.
|
||||
2. The same match calls `BackendErrorSink` exactly once after the waiter resolution wins.
|
||||
3. A late classification which loses to `fleet_reply` does not call the sink.
|
||||
4. An exhausted line that also matches the generic error pattern stays `BACKEND_EXHAUSTED`. It calls
|
||||
only `ExhaustionSink`.
|
||||
5. A raw pane with leading Terminal User Interface (TUI) chrome and no assistant marker still uses
|
||||
the #211 fallback and calls the backend-error sink.
|
||||
6. A matching error inside the two-second floor is typed and sent to the sink. A non-matching fast
|
||||
turn stays a generic failure.
|
||||
7. An unchanged delivery baseline which contains old backend-error text is suppressed. It never
|
||||
increments outage evidence.
|
||||
8. A non-match keeps the existing completion result and text.
|
||||
9. Constructors used by current callers keep compiling. They use the legacy default lookup and an
|
||||
explicit inert sink until Unit 5 supplies the production objects.
|
||||
10. Unit tests pass. The developer runs the focused test first, then `mvn clean install` from
|
||||
`fleetd/`.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 1 can run with Units 2, 3, and 4.
|
||||
|
||||
Unit 5 depends on its new lookup, sink, and constructor.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact classifier order in all three paths.
|
||||
- The focused test command and result.
|
||||
- The test which proves a losing waiter race does not publish an event.
|
||||
- The test which proves a fast matching failure is typed.
|
||||
- The final `mvn clean install` result.
|
||||
- Any constructor kept only for transition and where Unit 5 replaces it.
|
||||
|
||||
## Unit 2 — Credential outage policy
|
||||
|
||||
### Scope
|
||||
|
||||
Build a small credential-keyed state machine. It accepts already-classified backend-error events.
|
||||
It does not read pane text, profiles, sessions, or lead state.
|
||||
|
||||
Use an injected monotonic clock. A call records `credentialId`, target, and reason. It returns a new
|
||||
incident only on the threshold crossing. The incident contains a stable event id, credential id,
|
||||
the distinct affected targets, evidence count, window, and remaining cool-off.
|
||||
|
||||
This class owns both correlation and short cool-off. Keeping them together makes threshold crossing
|
||||
and the cool-off deadline one atomic state change.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Add `fleetd/src/main/java/dev/ltms/fleet/placement/BackendOutagePolicy.java`.
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/placement/BackendOutagePolicyTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One error creates no incident and no cool-off.
|
||||
2. Two errors for one credential within 60 seconds create exactly one incident and a 60-second
|
||||
cool-off.
|
||||
3. Two errors more than 60 seconds apart do not create an incident.
|
||||
4. The exact 60-second boundary has a pinned result. Use inclusive `<= 60s` so scheduler delay does
|
||||
not discard evidence at the boundary.
|
||||
5. Different credentials never share evidence.
|
||||
6. Different profiles which supply the same credential id do share evidence. The policy itself only
|
||||
sees the credential id.
|
||||
7. More errors during active cool-off do not extend its deadline and do not return another incident.
|
||||
8. After expiry, old evidence is cleared. Two fresh errors are needed to create the next incident.
|
||||
9. Remaining seconds round up, matching `BackendQuarantine` reporting.
|
||||
10. Concurrent second and third errors cannot return two incidents.
|
||||
11. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 2 can run with Units 1, 3, and 4.
|
||||
|
||||
Unit 5 depends on the policy API.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The state transition table and locking method.
|
||||
- The exact threshold, window, cool-off, and boundary rule.
|
||||
- The test which proves one incident under concurrent calls.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 3 — Lead outage nudge
|
||||
|
||||
### Scope
|
||||
|
||||
Add backend incidents as a fourth pending source in `ReplyPushLoop`. Do not create another scheduler
|
||||
or call `AgentControl.send` from `Fleetd`. The existing combined per-lead schedule is the control
|
||||
which prevents competing injected turns.
|
||||
|
||||
The entry point takes an incident id, affected worker targets, credential id, affected profile
|
||||
names, and remaining cool-off. It resolves distinct owning leads through `PrimaryRegistry`.
|
||||
|
||||
Each `(incidentId, lead)` item is one-shot. It waits while the lead is not injectable. After one
|
||||
successful `agents.send`, remove it. A send exception keeps it pending for a bounded retry. It never
|
||||
uses the repeated reminder behaviour of an uncollected ticket.
|
||||
|
||||
Also add a fail-loud entry point for a classified target that Unit 5 cannot map to a credential. It
|
||||
uses `PrimaryRegistry.nudgeTargetFor(target)` and says that correlation could not run. If no lead is
|
||||
known, log at `WARN`, not `DEBUG`.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/msg/ReplyPushLoop.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/msg/ReplyPushLoopTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. One incident affecting two workers owned by one lead causes one successful pane injection.
|
||||
2. Two affected workers owned by two leads cause one successful injection per affected lead. This
|
||||
is one notice per event per lead, not one notice per member.
|
||||
3. Repeating the same incident id is idempotent.
|
||||
4. A busy or unknown lead is not injected. The item stays pending until the lead becomes injectable
|
||||
or its attempt cap is reached.
|
||||
5. After one successful injection, later ticks do not mention that incident again.
|
||||
6. A failed `agents.send` is retried within the existing bound. A successful retry still gives only
|
||||
one successful send.
|
||||
7. A pending ticket and an outage incident for one lead appear in one combined nudge, not two
|
||||
competing turns.
|
||||
8. The text names the credential, profiles, affected workers, and remaining cool-off. It tells the
|
||||
lead to run `fleet_list`.
|
||||
9. An unmapped target produces a direct warning notice when a lead is known. If no lead is known,
|
||||
the code logs a `WARN` naming the target and reason.
|
||||
10. `stop()` clears incident state as it clears other push state.
|
||||
11. Existing reply, ticket, and question tests stay green. The focused tests and
|
||||
`mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. The API uses plain values, not the Unit 2 incident class. This lets Unit 3 run in parallel.
|
||||
|
||||
Unit 5 adapts the Unit 2 incident into this entry point.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact one-shot and retry rules.
|
||||
- The test showing one combined nudge with a failed ticket.
|
||||
- The test showing one successful send for two affected workers.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 4 — Durable member backend-error outcome
|
||||
|
||||
### Scope
|
||||
|
||||
Make a classified backend failure remain visible after its ticket is collected or expires.
|
||||
|
||||
Add `BACKEND_ERROR` to `MemberSession.State`. Add a nullable failure detail to `MemberSession` and
|
||||
render it as `failureReason` in `SessionManager.rosterView`. Add
|
||||
`SessionManager.onBackendError(target, reason)`.
|
||||
|
||||
The transition must handle both completion orderings:
|
||||
|
||||
- `BUSY -> BACKEND_ERROR` when classification wins before the normal completion state update;
|
||||
- `DONE -> BACKEND_ERROR` when the async resolver runs after `SessionManager.onTurnComplete`.
|
||||
|
||||
It must use a compare-and-set retry or another atomic update. `RELEASED` must never return to the
|
||||
roster. A backend-error member is terminal and cannot accept another delivery.
|
||||
|
||||
### Files owned
|
||||
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/MemberSession.java`.
|
||||
- Change `fleetd/src/main/java/dev/ltms/fleet/session/SessionManager.java`.
|
||||
- Change `fleetd/src/test/java/dev/ltms/fleet/session/SessionManagerTest.java`.
|
||||
|
||||
No other unit may edit these files.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `onBackendError` moves a `BUSY` member to `BACKEND_ERROR` and stores the reason.
|
||||
2. It also moves `DONE` to `BACKEND_ERROR`, covering the resolver race.
|
||||
3. A later normal `onTurnComplete` cannot change `BACKEND_ERROR` back to `DONE`.
|
||||
4. A released or unknown member is not recreated. The unknown case logs at `WARN` and returns an
|
||||
explicit false result to its caller.
|
||||
5. `onDelivered` refuses a `BACKEND_ERROR` member, just as it refuses generic `FAILED`.
|
||||
6. `rosterView` reports `state: backend_error` and `failureReason` after the send ticket is gone.
|
||||
7. Ordinary members do not gain a blank or invented `failureReason` field.
|
||||
8. Existing constructors keep source compatibility for tests and adapters.
|
||||
9. The focused tests and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
None. Unit 4 can run with Units 1, 2, and 3.
|
||||
|
||||
Unit 5 calls the new session method from the production sink.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The two race orderings and the tests for both.
|
||||
- The exact roster JSON shape.
|
||||
- The unknown-target result and log level.
|
||||
- The focused test and `mvn clean install` results.
|
||||
|
||||
## Unit 5 — Production wiring, spawn gate, and fleet views
|
||||
|
||||
### Scope
|
||||
|
||||
Compose Units 1 to 4 in production. This is the only unit which edits `Fleetd.java`.
|
||||
|
||||
Add per-profile `errorPattern` config beside `exhaustedPattern`. Compile both once at startup. A
|
||||
configured pattern wins over the legacy default. Report configured profiles and legacy-default
|
||||
profiles separately at startup. A bad regex must stop startup with the profile and key in the
|
||||
message.
|
||||
|
||||
Wire one production `BackendErrorSink` with this order:
|
||||
|
||||
1. mark the member `backend_error` with its reason;
|
||||
2. resolve the profile and its current `effectiveCredentialId()` through the fail-loud #234 seam;
|
||||
3. record the error in `BackendOutagePolicy`;
|
||||
4. on a new incident, submit one event to `ReplyPushLoop`.
|
||||
|
||||
If target metadata cannot be resolved, do not end in `Optional.ifPresent`. Log an error and call the
|
||||
Unit 3 unmapped-target notice. The failed send still reaches its ticket through CB-588.
|
||||
|
||||
Teach both spawn paths about a separate cool-off source. Exhaustion quarantine has priority when
|
||||
both states are active. Automatic placement needs a distinct `coolingOff` set so its refusal does
|
||||
not say “exhausted”.
|
||||
|
||||
Extend the MCP (Model Context Protocol) views:
|
||||
|
||||
- A cooling profile has `free: 0`, `credentialId`, and `coolingOffForSeconds` in `fleet_list`.
|
||||
- It does not have `quarantinedForSeconds` unless exhaustion quarantine is also active.
|
||||
- `fleet_profiles` has a separate `coolingOff` map, not an entry in `quarantined`.
|
||||
- A direct spawn refusal says the credential is cooling off after repeated backend errors and gives
|
||||
the remaining seconds.
|
||||
|
||||
Do not change `FleetHealthMonitor.coverage`. It still describes the periodic health webhook path.
|
||||
Lead-pane outage delivery is a separate capability.
|
||||
|
||||
### Files owned
|
||||
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/Fleetd.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/config/ConfigRef.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/member/CompositePeerLauncher.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementContext.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/placement/PlacementPolicyUtil.java`
|
||||
- `fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/FleetConfigTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/config/ConfigRefTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/member/CompositePeerLauncherTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/placement/PlacementPolicyTest.java`
|
||||
- `fleetd/src/test/java/dev/ltms/fleet/mcp/FleetMcpTest.java`
|
||||
- Add `fleetd/src/test/java/dev/ltms/fleet/BackendOutageFlowTest.java`.
|
||||
- `fleetd/fleetd.example.yaml`
|
||||
- `CLAUDE.md`
|
||||
|
||||
No earlier unit edits these files.
|
||||
|
||||
The lead, not a worker, must update `wiki/7-Use-Cases.md`, `wiki/9-Implementation.md`, and
|
||||
`wiki/11-Features.md`. Project rules forbid workers from committing `wiki/`. The portable block in
|
||||
`CLAUDE.md` and `wiki/7-Use-Cases.md` must remain byte-identical.
|
||||
|
||||
### Acceptance criteria
|
||||
|
||||
1. `errorPattern` binds per profile. Blank uses the legacy default and is reported as degraded
|
||||
coverage. A config reload which changes it is reported as deferred because patterns are compiled
|
||||
at startup.
|
||||
2. A malformed `errorPattern` stops startup and names `profiles.<name>.errorPattern`.
|
||||
3. The real production sink never silently drops an unknown target. A test captures its error log
|
||||
and the fallback notice call.
|
||||
4. One real backend-error classification through `CompletionResolver` marks only its member. It does
|
||||
not cool the credential and does not send an outage notice.
|
||||
5. Two real classifications for one credential within 60 seconds start one incident.
|
||||
6. The integration test then calls the real explicit-profile spawn gate. It is refused before any
|
||||
adapter spawn call, with “cooling off” and remaining seconds in the message.
|
||||
7. The same test calls an automatic placement path. A cooling candidate is skipped. If every
|
||||
candidate is cooling, the error names cool-off rather than exhaustion.
|
||||
8. Profiles sharing the credential are all blocked. A profile on another credential stays usable.
|
||||
9. `fleet_list` from the same fixture shows both members as `backend_error`, preserves each failure
|
||||
reason, and reports `free: 0`, the credential, and `coolingOffForSeconds`.
|
||||
10. `fleet_profiles` reports cool-off separately from quarantine.
|
||||
11. The real `ReplyPushLoop` receives one incident and makes one successful lead-pane send. Existing
|
||||
failed-ticket notice content may share that same combined send.
|
||||
12. Exhaustion still wins when a line matches both patterns. A simultaneous exhaustion quarantine
|
||||
also wins in spawn errors and fleet views.
|
||||
13. After the 60-second cool-off, spawn is allowed again. A new incident needs two fresh errors.
|
||||
14. `healthCoverage` has the same value before and after this change for the same health config.
|
||||
15. `fleetd.example.yaml` explains `errorPattern`, the legacy fallback, 2/60/60 policy, and the
|
||||
difference between cool-off and exhaustion quarantine.
|
||||
16. `CLAUDE.md` tells leads how `fleet_profiles` and `fleet_list` report cool-off. The lead later
|
||||
applies the matching wiki updates and runs the documented byte-sync check.
|
||||
17. The developer records the new end-to-end test failing before implementation, then passing. The
|
||||
focused suites and `mvn clean install` pass.
|
||||
|
||||
### Dependencies
|
||||
|
||||
Unit 5 starts only after Units 1 to 4 are merged or rebased into its branch. It also starts after the
|
||||
#234 defect 2 fix lands, because both areas touch the same target-resolution control path.
|
||||
|
||||
### What to report back
|
||||
|
||||
- The exact commits used for Units 1 to 4 and #234.
|
||||
- The startup coverage line with one configured and one legacy-default profile.
|
||||
- The red test output before the implementation and its green result after.
|
||||
- The explicit and automatic spawn refusal text.
|
||||
- Sample `fleet_list` and `fleet_profiles` JSON for cool-off and exhaustion.
|
||||
- The number and text of lead-pane sends in the real-path test.
|
||||
- The focused test commands and final `mvn clean install` result.
|
||||
- The exact `CLAUDE.md` change and the wiki edits the lead must apply.
|
||||
|
||||
## File ownership summary
|
||||
|
||||
| Area | Unit | Shared edit risk |
|
||||
|---|---:|---|
|
||||
| Completion classification | 1 | Only Unit 1 edits `CompletionResolver` and its test. |
|
||||
| Correlation and cool-off state | 2 | New files only. |
|
||||
| Lead push scheduling | 3 | Only Unit 3 edits `ReplyPushLoop` and its test. |
|
||||
| Member terminal state | 4 | Only Unit 4 edits `MemberSession`, `SessionManager`, and their test. |
|
||||
| Main composition, config, placement, MCP views, shipped prompt | 5 | Only Unit 5 edits `Fleetd`, `FleetConfig`, `CompositePeerLauncher`, placement context, `FleetMcp`, and `CLAUDE.md`. |
|
||||
| Wiki propagation | Lead after Unit 5 | Workers do not commit the wiki submodule. |
|
||||
|
||||
## What I would not build
|
||||
|
||||
1. **Do not reuse `BackendQuarantine` for outages.** Its repeat call restarts a long credential
|
||||
quarantine. Its fields and errors say “exhausted”. That is wrong for a short outage.
|
||||
2. **Do not merge the exhaustion and generic error patterns.** Exhaustion must win because it has a
|
||||
different policy and duration.
|
||||
3. **Do not group by error string.** One outage can produce different text. The shared operational
|
||||
limit is the credential.
|
||||
4. **Do not mark a profile unusable until config changes.** The current classifier cannot safely
|
||||
tell a permanent malformed request from a transient service fault. A permanent state would need
|
||||
a stronger error taxonomy first.
|
||||
5. **Do not quarantine on the first generic backend error.** That would turn one bad request or one
|
||||
false pattern match into a fleet-wide capacity loss.
|
||||
6. **Do not add this to `FleetHealthMonitor`.** The monitor samples slow member health. The exact
|
||||
backend event already exists at completion resolution, and moving it to polling would lose type
|
||||
and time.
|
||||
7. **Do not add another direct lead injector.** `ReplyPushLoop` already owns status gating,
|
||||
per-lead coalescing, retry bounds, and heartbeat stand-down.
|
||||
8. **Do not change `healthCoverage` to `full`.** That field still means a webhook notification sink
|
||||
exists for periodic health. A backend outage nudge does not make every health event visible.
|
||||
9. **Do not persist incident history across daemon restart in this work.** Existing exhaustion
|
||||
quarantine is also in memory. A 60-second state does not justify a new durable store.
|
||||
10. **Do not build work recovery.** The PR-body survival story proves why checkpoint-first work is
|
||||
useful, but these tickets are about detection, capacity, and signalling.
|
||||
11. **Do not remove the legacy `API Error:` fallback in the first release.** Doing so would turn an
|
||||
unedited config back into a false successful completion. Report it as degraded coverage instead.
|
||||
12. **Do not reorder or add the old `visibleTurn` fallback from #201.** #211 already implemented the
|
||||
narrow raw-scrape fallback at `CompletionResolver.classifyRawScrapeFallback`.
|
||||
|
||||
## Riskiest assumption and cheapest experiment
|
||||
|
||||
The riskiest assumption is that a configured error regex means “the backend failed this turn”. The
|
||||
current code and test already show the counterexample: a worker may quote `API Error:` while writing
|
||||
a valid report. Two such false matches on one credential would now remove capacity for 60 seconds.
|
||||
|
||||
The cheapest experiment is a replay corpus before Unit 5 ships:
|
||||
|
||||
1. Save the full pane text from the measured 2026-09-01 outage.
|
||||
2. Produce one safe failure per backend with a disposable invalid endpoint or request.
|
||||
3. Save one valid member report which quotes each error line.
|
||||
4. Replay all samples through the real `CompletionResolver` test fixture.
|
||||
5. Require outage samples to match and quoted-report samples not to match after assistant-block
|
||||
extraction and baseline checks.
|
||||
|
||||
This costs no outage deployment and no real sleep. If quoted reports still match, narrow the profile
|
||||
patterns before enabling correlation. Do not raise the threshold to hide a bad classifier.
|
||||
|
||||
## Sequencing with three developers
|
||||
|
||||
First wave:
|
||||
|
||||
1. Developer A: Unit 1, typed classification.
|
||||
2. Developer B: Unit 2, credential outage policy.
|
||||
3. Developer C: Unit 3, lead outage nudge.
|
||||
|
||||
As soon as one slot is free, start Unit 4. It is file-disjoint from every first-wave unit. Merge and
|
||||
review Units 1 to 4 independently.
|
||||
|
||||
Start Unit 5 only after all four foundations and #234 are available. Unit 5 is the only high-conflict
|
||||
integration branch, so no other active unit should touch its file list.
|
||||
|
||||
## Checks performed for this refinement
|
||||
|
||||
- Read issue #201 and issue #227 through their Gitea pages. Both showed zero comments.
|
||||
- Read the source and tests named in the evidence section.
|
||||
- Ran `git status --short --branch`; the branch was clean before this document was added.
|
||||
- Ran `git log --oneline -12` to identify the branch base.
|
||||
- I did not run Maven because this change adds only a design document.
|
||||
- Rendered both Mermaid blocks with `npx @mermaid-js/mermaid-cli`; both commands succeeded.
|
||||
@@ -192,7 +192,7 @@ Deliberately small; every one maps to a failure mode we have actually hit.
|
||||
| `fleet_send_duration_seconds` | histogram | delegated turn latency |
|
||||
| `fleet_replies_total{path}` | counter | path ∈ rendezvous\|inbox — how often a reply strands (CB-307's whole reason to exist) |
|
||||
| `fleet_inbox_depth{target}` | gauge | undrained replies; steady-state should be 0 |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ delivered\|exhausted — a rising `exhausted` means the primary is not draining |
|
||||
| `fleet_push_nudges_total{outcome}` | counter | outcome ∈ sent\|exhausted — a rising `exhausted` means the primary is not draining. `sent` was called `delivered` until fleetd #365; it counts the herdr paste-and-submit call returning, never a confirmation the pane read it |
|
||||
| `fleet_spawns_total{kind,outcome}` | counter | outcome ∈ ready\|timeout\|guard_rejected; per peer kind (CB-402) |
|
||||
| `fleet_sessions{state}` | gauge | SPAWNING/READY/BUSY/DONE census |
|
||||
| `fleet_herdr_calls_total{method,outcome}` | counter | socket health — the dependency everything rests on |
|
||||
|
||||
+123
-324
@@ -1,311 +1,162 @@
|
||||
# MCP Contract — `fleetd`'s unified gateway
|
||||
# MCP flows and error model — `fleetd`
|
||||
|
||||
> **Status: 🔴 HISTORICAL DESIGN — do NOT use as the tool reference.** Written 2026-07-14, before
|
||||
> any MCP code existed. The system shipped and this page never caught up, so **its tool names,
|
||||
> parameter names and REST paths are wrong today**. Audited 2026-08-17; the specific drift:
|
||||
> **What this page is.** The **flows**: how a delegation, a clarification, a detached task and a
|
||||
> silent member each travel through `fleetd`. These shapes are what shipped, and they are hard to
|
||||
> read off the code because they span the MCP face, the rendezvous registry, the `Injector` and
|
||||
> herdr.
|
||||
>
|
||||
> - **Tools it names that do not exist:** `fleet_read`, `fleet_cancel`.
|
||||
> - **Shipped tools it omits:** `fleet_poll`, `fleet_ack`, `fleet_profiles`, `fleet_whoami`.
|
||||
> - **Parameter names are wrong nearly everywhere** — it says `message`/`target`/`timeout_seconds`/
|
||||
> `block` where the code takes `content`/`sessionId`/`timeoutMs`/`wait`; `text` where
|
||||
> `fleet_reply` takes `content`; `target` where `fleet_stop` takes `paneId`.
|
||||
> - **REST paths are wrong:** it says `POST /workers` and `DELETE /workers/{paneId}`; the daemon
|
||||
> serves `POST /members` and `DELETE /members/{paneId}`.
|
||||
> **What this page is NOT: a tool reference.** It deliberately holds no tool catalogue, no
|
||||
> parameter tables and no REST paths. **The live MCP schema is the authority** — each tool's own
|
||||
> description and parameters, as mounted — with the intent→tool table in `CLAUDE.md` as the short
|
||||
> form.
|
||||
>
|
||||
> **The authoritative tool surface is the live MCP schema** (each tool's own description and
|
||||
> parameters, as mounted), with the intent→tool table in `CLAUDE.md` as the short form. Both were
|
||||
> checked against `mcp/FleetMcp.java` on 2026-08-17 and are accurate.
|
||||
> That absence is the fix for fleetd #114 (CB-609), and it is worth stating why. This page used to
|
||||
> carry a full tool catalogue written in July 2026, before any MCP code existed. The code shipped;
|
||||
> the page did not follow. By August it named two tools that do not exist, omitted five that do,
|
||||
> had the wrong name for nearly every parameter, pointed at REST paths the daemon does not serve,
|
||||
> and — worst — still described an identity model (*"any connection that does not map to a known
|
||||
> worker is treated as a primary"*) that was a real privilege bug, fixed since by the ancestry
|
||||
> walk in fleetd #161. Every one of those errors is the same error: **a second, hand-maintained
|
||||
> copy of something the code already states**. So the second copy is gone rather than corrected.
|
||||
> Only the flows remain, because a flow is a shape rather than a name, and shapes are what this
|
||||
> page was ever good for.
|
||||
>
|
||||
> What is still worth reading here is **§6 — the flows and the error model** (rendezvous,
|
||||
> `fleet_ask`, detached delivery, the turn-done fallback). The shapes it describes are the ones
|
||||
> that shipped; only the names around them drifted. Rewriting this page is tracked as **CB-609**.
|
||||
|
||||
`fleetd` is the **sole communication gateway** for every Claude session in the bridge. Both
|
||||
the **primary** (Opus, on subscription) and every **worker** (off-subscription Claude Code)
|
||||
mount the *same* MCP server with a single `claude mcp add` line, and talk only through its
|
||||
tools. No Claude session ever addresses a broker, a peer, or the network directly.
|
||||
|
||||
This document defines every MCP tool that face must expose, who may call it, its blocking
|
||||
semantics, and how it maps onto the code already in the tree.
|
||||
> The names that do appear below are checked by `McpContractDocTest`, which fails if this page
|
||||
> names a `fleet_*` tool the server does not register. That test is the whole reason it is safe to
|
||||
> write a tool name here at all.
|
||||
|
||||
---
|
||||
|
||||
## 1. Design constraints (non-negotiable)
|
||||
## 1. Rendezvous flows
|
||||
|
||||
These come from the project's core invariants and bound every decision below.
|
||||
### 1.1 Delegation — happy path
|
||||
|
||||
1. **One server, both roles.** The primary and all workers mount an identical server. The
|
||||
catalog must serve both, and `fleetd` must decide *who is calling* from the connection —
|
||||
never from a caller-supplied argument that could be spoofed.
|
||||
2. **Subscription-safe by construction.** No MCP tool ever reads, sets, or forwards
|
||||
`ANTHROPIC_BASE_URL`. Mounting the bridge cannot move a session off subscription.
|
||||
Enforced today by [`SubscriptionGuard`](1-Architecture).
|
||||
3. **Blocking rendezvous, no busy-poll.** The primary consumes a worker's reply through a
|
||||
*single* MCP call that `fleetd` holds open — never a cross-turn poll loop that would burn
|
||||
subscription quota.
|
||||
4. **Status-gated delivery.** Anything that puts text into a worker flows through the existing
|
||||
[`Injector`](1-Architecture): delivered only when the worker is `idle`/`blocked`, at most
|
||||
one message per turn.
|
||||
5. **`fleetd` owns policy; herdr owns PTYs.** MCP tools express *intent*; `fleetd`
|
||||
translates it into guard checks, rendezvous bookkeeping, and herdr `agent.*` calls.
|
||||
|
||||
---
|
||||
|
||||
## 2. Topology
|
||||
|
||||
Both faces live in the one daemon. The **north face** is MCP (this document); the **south
|
||||
face** is the herdr Unix socket. REST/SSE remains only for non-Claude clients and dashboards.
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
OPUS["Opus — primary<br/>(Claude Code, env CLEAN)<br/>MCP client"]
|
||||
subgraph BD["fleetd — standalone daemon"]
|
||||
MCP["MCP server (north face)<br/>fleet_send · fleet_reply<br/>fleet_ask · fleet_status · lifecycle"]
|
||||
RDV["rendezvous registry<br/>(blocking-call waiters)"]
|
||||
INJ["Injector + StatusPoller<br/>(status-gated writer)"]
|
||||
SOCK["herdr socket client (south face)"]
|
||||
MCP --> RDV
|
||||
RDV --> INJ
|
||||
INJ --> SOCK
|
||||
MCP --> SOCK
|
||||
end
|
||||
HERDR["herdr<br/>panes · agent-status"]
|
||||
W["worker claude pane<br/>ANTHROPIC_BASE_URL set<br/>MCP client"]
|
||||
|
||||
OPUS -->|"fleet_send (blocks)"| MCP
|
||||
W -.->|"fleet_reply / fleet_ask"| MCP
|
||||
SOCK -->|"agent.start · agent.send<br/>agent.get · pane.close"| HERDR
|
||||
HERDR -->|"drives PTY"| W
|
||||
|
||||
classDef ext fill:#2b6cb0,stroke:#1a365d,color:#ffffff;
|
||||
classDef core fill:#2f855a,stroke:#22543d,color:#ffffff;
|
||||
class OPUS,W ext
|
||||
class MCP,RDV,INJ,SOCK core
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
## 3. Identity & addressing
|
||||
|
||||
Because the same server is mounted by everyone, `fleetd` resolves the caller's role on every
|
||||
request — this is the linchpin of the whole contract and has no code yet.
|
||||
|
||||
- **Workers are known.** `fleetd` spawns every worker
|
||||
([`WorkerService`](1-Architecture)) and records its herdr session UUID / `terminal_id` on
|
||||
the returned [`Agent`]. When a call arrives on a connection that maps to a known worker,
|
||||
the caller is *that* worker — so **workers never pass a target**; routing is implicit.
|
||||
- **The primary is "not a worker".** Any connection that does not map to a known worker is
|
||||
treated as a primary. It addresses workers **explicitly** by `target` — a session UUID,
|
||||
a `terminal_id`, or a friendly `profile` name.
|
||||
- **Turn correlation.** A blocking `fleet_send` registers a *waiter* keyed by worker
|
||||
identity. A worker's later `fleet_reply` / `fleet_ask` on the same identity resolves that
|
||||
waiter. A `turn_id` is minted per exchange so a clarification round-trip
|
||||
(§6.2) rejoins the right turn.
|
||||
|
||||
---
|
||||
|
||||
## 4. Transport
|
||||
|
||||
`fleetd` is a long-lived daemon serving **multiple** concurrent clients (one primary + N
|
||||
workers), so a per-client stdio child is the wrong shape. The recommended transport is
|
||||
**streamable-HTTP / SSE** on the same bind as the REST face:
|
||||
|
||||
```bash
|
||||
# identical on primary and every worker
|
||||
claude mcp add --transport http fleetd http://127.0.0.1:8080/mcp
|
||||
```
|
||||
|
||||
This adds an MCP-server dependency the pom does not yet carry. See [Open decisions](#10-open-decisions).
|
||||
|
||||
---
|
||||
|
||||
## 5. Tool catalog
|
||||
|
||||
| Tool | Caller | Blocks? | Backing (exists today?) |
|
||||
|---|---|---|---|
|
||||
| [`fleet_send`](#fleet_send) | primary | yes (default) | `Injector.enqueue` ✅ · rendezvous registry ❌ (CB-104) |
|
||||
| [`fleet_reply`](#fleet_reply) | worker | no | rendezvous ❌ · pane injection via `Injector` ✅ |
|
||||
| [`fleet_ask`](#fleet_ask) | worker | yes | reverse rendezvous ❌ |
|
||||
| [`fleet_status`](#fleet_status) | either | no | `AgentControl.status` ✅ · `Injector.activeTargets` ✅ |
|
||||
| [`fleet_spawn`](#lifecycle) | primary | no | `WorkerService.spawn` ✅ (`POST /workers`) |
|
||||
| [`fleet_list`](#lifecycle) | either | no | `WorkerService.list` ✅ (`/agents`) |
|
||||
| [`fleet_stop`](#lifecycle) | primary | no | `WorkerService.stop` ✅ (`DELETE /workers/{paneId}`) |
|
||||
| [`fleet_read`](#fleet_read) | primary | no | `AgentControl.read` ✅ |
|
||||
| [`fleet_cancel`](#fleet_cancel) | primary | no | — ❌ (future) |
|
||||
|
||||
### Core: delegation & rendezvous
|
||||
|
||||
#### `fleet_send`
|
||||
*(primary → worker — the headline tool, CB-104)*
|
||||
|
||||
- **Params:** `message` (required); `target` (optional — defaults to the sole worker / default
|
||||
profile); `timeout_seconds` (default 600); `block` (default `true`); `auto_spawn`
|
||||
(default `true`); `turn_id` (optional — supplied when answering a worker's `fleet_ask`).
|
||||
- **Blocking (`block:true`):** enqueue `message` via the `Injector`, then hold the call open
|
||||
until exactly one of:
|
||||
- worker calls `fleet_reply` → `{ outcome:"reply", text }`
|
||||
- worker calls `fleet_ask` → `{ outcome:"question", text, turn_id }`
|
||||
- worker's `agent_status` reaches done/idle with no reply → `{ outcome:"turn_done", text:<terminal tail> }`
|
||||
- deadline elapses → `{ outcome:"timeout" }`
|
||||
- worker gone → error `worker_gone`
|
||||
- **Detached (`block:false`):** enqueue and return `{ outcome:"dispatched", dispatch_id }`
|
||||
immediately. The eventual reply is injected into the primary's idle pane (§6.3), or drained
|
||||
via `fleet_status` on a split-host primary.
|
||||
|
||||
#### `fleet_reply`
|
||||
*(worker → primary)*
|
||||
|
||||
- **Params:** `text` (required); `final` (default `true`).
|
||||
- **Behavior:** resolve the primary waiter registered against this worker with `text`. If no
|
||||
waiter exists (detached delegation), `fleetd` **injects the primary's idle pane** instead.
|
||||
Returns `{ delivered:true, mode:"resolved"|"injected" }`. No `target` — identity is implicit.
|
||||
|
||||
#### `fleet_ask`
|
||||
*(worker → primary — the reverse rendezvous)*
|
||||
|
||||
- **Params:** `question` (required); `timeout_seconds`.
|
||||
- **Behavior:** blocks the *worker's* call. Surfaces the question to the primary (resolving its
|
||||
open `fleet_send` with `outcome:"question"`, or injecting its pane). When the primary
|
||||
answers — a `fleet_send` carrying the matching `turn_id` — that unblocks this call and
|
||||
returns `{ answer }` to the worker, which continues **in the same turn**.
|
||||
|
||||
### Worker lifecycle
|
||||
<a id="lifecycle"></a>
|
||||
Thin adapters over [`WorkerService`](1-Architecture) — parity with the existing REST routes.
|
||||
|
||||
- **`fleet_spawn`** — `{ profile? }` → worker view (`sessionId`, `terminalId`, `paneId`,
|
||||
`status`). Guard-checked; a boundary breach returns error `subscription_boundary` (the
|
||||
REST `403`).
|
||||
- **`fleet_list`** — no params → all workers + `agent_status`. Read-only, either role.
|
||||
- **`fleet_stop`** — `{ target }` → tears down the pane and its dedicated tab. Idempotent.
|
||||
|
||||
### Observability
|
||||
|
||||
#### `fleet_status`
|
||||
*(either role — the README's 4th named tool)*
|
||||
|
||||
- **Params:** `target?`.
|
||||
- **Behavior:** per-worker `agent_status`, queue depth (`Injector.activeTargets`), whether a
|
||||
rendezvous is open, and ids. For the *calling* session it also reports/drains **pending
|
||||
messages addressed to me** — the path a split-host primary's `Stop`-hook uses to wake and
|
||||
collect replies without being injectable. Read-only, non-blocking.
|
||||
|
||||
#### `fleet_read`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`; `source` ∈ `visible | recent | recent_unwrapped | detection`.
|
||||
- **Behavior:** returns the worker's terminal text so the primary can peek at a *detached*
|
||||
worker's progress. Adapter over `AgentControl.read`.
|
||||
|
||||
### Control (future)
|
||||
|
||||
#### `fleet_cancel`
|
||||
*(primary)*
|
||||
|
||||
- **Params:** `target`. Interrupt the worker's current turn / abandon the rendezvous. No
|
||||
backing code yet.
|
||||
|
||||
---
|
||||
|
||||
## 6. Rendezvous flows
|
||||
|
||||
### 6.1 Delegation — happy path
|
||||
|
||||
One blocking call, zero polls.
|
||||
One blocking call, zero polls. The lead's call is held open by `fleetd` until the member answers.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary (Opus)
|
||||
participant B as fleetd (MCP + Injector)
|
||||
participant P as "Lead (primary)"
|
||||
participant B as "fleetd (MCP + Injector)"
|
||||
participant H as herdr
|
||||
participant W as Worker (Claude)
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B->>B: register waiter(w)
|
||||
B->>H: agent.send(w, "do X") (idle window)
|
||||
H-->>W: prompt injected
|
||||
W->>W: works the turn
|
||||
W->>B: fleet_reply("result")
|
||||
B->>B: resolve waiter(w)
|
||||
B-->>P: { outcome:"reply", text:"result" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B->>B: "register waiter(sessionId)"
|
||||
B->>H: "agent.send — only in an injectable window"
|
||||
H-->>W: "prompt injected"
|
||||
W->>W: "works the turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B->>B: "resolve waiter"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.2 Clarification — reverse rendezvous (`fleet_ask`)
|
||||
**The cap that matters:** a blocking `fleet_send` is bounded by the *caller's own* MCP client
|
||||
timeout, about 60 seconds — not by the task. Anything slower than that must use the detached flow
|
||||
in §1.3, or the lead's call returns while the member is still working.
|
||||
|
||||
The worker pauses mid-turn to ask; the primary answers; the worker resumes in the same turn.
|
||||
### 1.2 Clarification — reverse rendezvous
|
||||
|
||||
The member pauses mid-turn to ask, the lead answers, and the member resumes **the same turn** with
|
||||
its context intact.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>B: fleet_ask("which config?") — worker blocks
|
||||
B-->>P: { outcome:"question", text:"which config?", turn_id }
|
||||
P->>B: fleet_send("config.yaml", target=w, turn_id) — blocks again
|
||||
B-->>W: resolve fleet_ask → { answer:"config.yaml" }
|
||||
W->>W: resumes same turn
|
||||
W->>B: fleet_reply("done")
|
||||
B-->>P: { outcome:"reply", text:"done" }
|
||||
P->>B: "fleet_send{sessionId, content} — blocks"
|
||||
B-->>W: "content injected"
|
||||
W->>B: "fleet_ask{question} — member blocks"
|
||||
B-->>P: "{ outcome: question, turnId }"
|
||||
P->>B: "fleet_send{turnId, content} — answers THIS turn"
|
||||
B-->>W: "fleet_ask returns the answer"
|
||||
W->>W: "resumes the same turn"
|
||||
W->>B: "fleet_reply{content}"
|
||||
B-->>P: "{ outcome: reply }"
|
||||
```
|
||||
|
||||
### 6.3 Detached delegation — pane injection
|
||||
**Answer with `turnId`, never `sessionId`.** A `sessionId` send starts a new turn; it does not
|
||||
resolve the waiting `fleet_ask`.
|
||||
|
||||
The primary does not block; the reply arrives later in its idle pane.
|
||||
**The window is about 55 seconds and no nudge extends it.** So never brief a member to "ask me":
|
||||
decide before delegating, or give the member an explicit default to fall back on.
|
||||
|
||||
### 1.3 Detached delegation — the lead does not block
|
||||
|
||||
The lead gets a ticket immediately and collects the answer later. This is the flow for any real
|
||||
task, because of the ~60s cap in §1.1.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w, block=false)
|
||||
B-->>P: { outcome:"dispatched", dispatch_id }
|
||||
P->>P: continues its own work
|
||||
W->>B: fleet_reply("result")
|
||||
Note over B: no waiter → detached path
|
||||
B->>B: Injector.enqueue(primary_pane, "result")
|
||||
B-->>P: injected into idle pane (status-gated)
|
||||
P->>B: "fleet_send{sessionId, content, wait:false}"
|
||||
B-->>P: "accepted — ticket"
|
||||
P->>P: "continues its own work"
|
||||
W->>B: "fleet_reply{content}"
|
||||
Note over B: "no waiter is blocked — the reply is held"
|
||||
B->>B: "nudge the lead's own pane (status-gated)"
|
||||
P->>B: "fleet_poll{ticket}"
|
||||
B-->>P: "the member's report"
|
||||
P->>B: "fleet_ack{target, msgId}"
|
||||
```
|
||||
|
||||
### 6.4 Uncooperative worker — turn-done fallback
|
||||
A terminal ticket nudges the lead's pane by itself, so a detached task does not need watching. The
|
||||
nudge needs an injectable lead pane and is capped, so it is a convenience rather than a guarantee.
|
||||
|
||||
A worker that never calls `fleet_reply` still returns a result: `fleetd` reads its terminal
|
||||
tail when the turn completes.
|
||||
### 1.4 The member never replies — turn-done fallback
|
||||
|
||||
A member that ends its turn without `fleet_reply` still produces something: `fleetd` reads its
|
||||
pane tail. This is a **fallback, not a channel** — it is lossy in three separate ways, and every
|
||||
one of them has produced a wrong answer in practice.
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant P as Primary
|
||||
participant P as "Lead"
|
||||
participant B as fleetd
|
||||
participant W as Worker
|
||||
participant W as "Member"
|
||||
|
||||
P->>B: fleet_send("do X", target=w) — blocks
|
||||
B-->>W: "do X" (injected)
|
||||
W->>W: works, never calls fleet_reply
|
||||
B->>B: StatusPoller sees agent_status → idle/done
|
||||
B->>B: AgentControl.read(w, "recent")
|
||||
B-->>P: { outcome:"turn_done", text:<terminal tail> }
|
||||
P->>B: "fleet_send — blocks or detaches"
|
||||
B-->>W: "content injected"
|
||||
W->>W: "works, never calls fleet_reply"
|
||||
B->>B: "StatusPoller sees the turn end"
|
||||
B->>B: "read the pane tail"
|
||||
B->>B: "classify: exhausted? echoed brief? real report?"
|
||||
B-->>P: "{ outcome: turn_done } or a named failure"
|
||||
```
|
||||
|
||||
The three ways it goes wrong, and what each looks like now:
|
||||
|
||||
| What happened | What the lead used to get | What it gets today |
|
||||
|---|---|---|
|
||||
| The report is longer than the scrape window | The **end** silently cut off | Still clipped, but marked partial |
|
||||
| The member never started — spent credential | The lead's **own brief** echoed back as a report | A named failure: backend exhausted |
|
||||
| The member is simply slow | A tail of work in progress | Unchanged — read it as a hint, not a result |
|
||||
|
||||
The echoed-brief case is the one to remember: it reads as a long, on-topic report with nothing in
|
||||
it from the member. It is suppressed now, but the general rule stands — **check the member's
|
||||
worktree with `git log` before believing a report you did not watch arrive.**
|
||||
|
||||
---
|
||||
|
||||
## 7. Status gating
|
||||
## 2. Status gating
|
||||
|
||||
Delivery only happens in a safe window. This is the state machine the `Injector` already
|
||||
enforces via `AgentStatus.injectable()`; MCP `fleet_send` is simply its producer.
|
||||
Delivery only happens in a safe window. `fleet_send` is a producer for the `Injector`, which
|
||||
already enforces this through `AgentStatus.injectable()`.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> IDLE
|
||||
IDLE --> WORKING: message delivered / picks up
|
||||
WORKING --> IDLE: turn done
|
||||
WORKING --> BLOCKED: awaits input
|
||||
BLOCKED --> WORKING: input delivered
|
||||
IDLE --> UNKNOWN: detection glitch
|
||||
BLOCKED --> UNKNOWN: detection glitch
|
||||
UNKNOWN --> IDLE: re-detected
|
||||
IDLE --> WORKING: "message delivered, picked up"
|
||||
WORKING --> IDLE: "turn done"
|
||||
WORKING --> BLOCKED: "awaits input"
|
||||
BLOCKED --> WORKING: "input delivered"
|
||||
IDLE --> UNKNOWN: "detection glitch"
|
||||
BLOCKED --> UNKNOWN: "detection glitch"
|
||||
UNKNOWN --> IDLE: "re-detected"
|
||||
|
||||
note right of IDLE
|
||||
injectable — deliver head of FIFO
|
||||
@@ -321,69 +172,17 @@ stateDiagram-v2
|
||||
end note
|
||||
```
|
||||
|
||||
At most one message is delivered per turn: after a send the `Injector` waits for a `WORKING`
|
||||
pickup before delivering the next, with a `PICKUP_GRACE_POLLS` fallback for turns faster than
|
||||
the poll interval. A herdr `events.subscribe` stream can later replace the sampling without
|
||||
touching this state machine.
|
||||
**At most one message per turn.** After a send, the `Injector` waits for a `WORKING` pickup before
|
||||
delivering the next, with a grace-poll fallback for turns that finish faster than the poll
|
||||
interval.
|
||||
|
||||
---
|
||||
Two consequences a lead feels directly:
|
||||
|
||||
## 8. Error model
|
||||
- **A second send to a busy member never lands.** It reports as queued and times out. The member
|
||||
is fine; the message simply waits, and then restarts the member when it next goes idle.
|
||||
- **A spawned member is not deliverable until it has mounted the MCP.** Until then a send waits on
|
||||
that gate for about 60 seconds and then fails without ever reaching the pane.
|
||||
|
||||
| Condition | `fleet_send` result | Notes |
|
||||
|---|---|---|
|
||||
| Worker replies | `{ outcome:"reply" }` | normal |
|
||||
| Worker asks | `{ outcome:"question", turn_id }` | answer with `fleet_send(turn_id)` |
|
||||
| Turn ends, no reply | `{ outcome:"turn_done" }` | terminal tail as text |
|
||||
| Deadline elapsed | `{ outcome:"timeout" }` | message may still be queued/delivered |
|
||||
| Worker vanished | error `worker_gone` | `Injector.drop` fails the queued future |
|
||||
| Guard breach on spawn | error `subscription_boundary` | REST `403` parity |
|
||||
| Delivery failed at herdr | error, message dropped | poisoned message not left blocking the FIFO |
|
||||
|
||||
`fleet_reply` from a worker with no open waiter is **not** an error — it falls through to
|
||||
detached pane injection (§6.3).
|
||||
|
||||
---
|
||||
|
||||
## 9. Mapping to existing code
|
||||
|
||||
The MCP face is a thin adapter layer; nearly every capability already exists behind the REST
|
||||
seam. Only the **rendezvous registry** and the **caller-identity resolver** are new.
|
||||
|
||||
| MCP tool | Existing collaborator | New work |
|
||||
|---|---|---|
|
||||
| `fleet_send` | `Injector.enqueue`, `AgentControl.send` | waiter registry, timeout, outcome mux (CB-104) |
|
||||
| `fleet_reply` / `fleet_ask` | `Injector` (pane injection) | reverse rendezvous, identity resolver |
|
||||
| `fleet_status` | `AgentControl.status`, `Injector.activeTargets` | pending-drain projection |
|
||||
| `fleet_spawn` / `list` / `stop` | `WorkerService.{spawn,list,stop}` | MCP adapter only |
|
||||
| `fleet_read` | `AgentControl.read` | MCP adapter only |
|
||||
|
||||
Because the REST routes in `FleetApp` already exercise the collaborators, MCP tools are
|
||||
validated by **parity** against those routes, not by re-testing behavior.
|
||||
|
||||
---
|
||||
|
||||
## 10. Open decisions
|
||||
|
||||
1. **`fleet_ask` direction.** This page defines it as *worker-asks-primary* (a genuine reverse
|
||||
channel, matching the "inject the primary's pane" language). The alternative — a synonym for
|
||||
a blocking primary→worker send — is weaker and produces different plumbing. **Recommend
|
||||
worker-asks-primary.**
|
||||
2. **Detached delivery shape.** A `block:false` param on `fleet_send` (keeps the catalog
|
||||
small) vs. a separate `fleet_dispatch` tool. **Recommend the param.**
|
||||
3. **Auto-spawn on send.** `fleet_send` provisions a worker per profile when none exists
|
||||
(simplest primary UX) vs. requiring an explicit `fleet_spawn` first. **Recommend
|
||||
auto-spawn, defaulting on.**
|
||||
4. **Transport & SDK.** Streamable-HTTP/SSE co-located with the REST bind (recommended) vs.
|
||||
stdio. Requires choosing a Java MCP server SDK and adding it to the pom.
|
||||
|
||||
---
|
||||
|
||||
## 11. Implementation staging
|
||||
|
||||
- **CB-104** — blocking `fleet_send` + rendezvous registry + caller-identity resolver
|
||||
(the producer that finally drives the inert `StatusPoller`).
|
||||
- **CB-1xx** — `fleet_reply` / `fleet_ask` reverse rendezvous + detached pane injection.
|
||||
- **CB-1xx** — lifecycle + observability adapters (`fleet_spawn/list/stop/status/read`).
|
||||
- **CB-1xx** — transport wiring + `claude mcp add` docs; parity tests vs. REST.
|
||||
- **Later** — `fleet_cancel`; swap `StatusPoller` for herdr `events.subscribe`.
|
||||
`UNKNOWN` is deliberately neither injectable nor a pickup. A pane whose status cannot be read is
|
||||
not a pane that is safe to write to — see fleetd #176 for what happens when a gate treats an
|
||||
unreadable pane as a ready one.
|
||||
|
||||
@@ -0,0 +1,208 @@
|
||||
# Wiki audit for #168
|
||||
|
||||
**Source checked:** `.wiki-snapshot/` at `68e32c6` (2026-08-31). I did not use
|
||||
`wiki/`. Code references below are from the current `fleetd` source tree. A quoted
|
||||
line is a concrete claim that needs correction, unless the table says `KEEP`.
|
||||
|
||||
| Page | Verdict | One-line reason |
|
||||
|---|---|---|
|
||||
| `Home.md` | REVISE | Good overview, but it still names the retired product. |
|
||||
| `_Sidebar.md` | REVISE | The heading still says `claude-bridge`. |
|
||||
| `1-Architecture.md` | REBUILD | Its component contract mixes current names with removed tools, routes, and planned backends. |
|
||||
| `2-Message-Server.md` | REBUILD | The claimed MCP schema, mount command, REST/SSE surface, and fallback paths are pre-build design. |
|
||||
| `3-Approaches.md` | REVISE | Useful research history, but it presents unbuilt AgentAPI as a selectable fallback. |
|
||||
| `4-Setup.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `5-Operations.md` | RETIRE | It is an intentional stub that only redirects to chapter 13. |
|
||||
| `6-Team.md` | REBUILD | It teaches role-addressed sends and a Claude-only team model that the shipped API does not have. |
|
||||
| `7-Use-Cases.md` | REBUILD | Its flagship flow depends on removed `ccs` profiles and removed send parameters. |
|
||||
| `8-Roadmap.md` | REBUILD | It is a historical plan, but it presents old implementation choices and planned work as the current stack. |
|
||||
| `9-Implementation.md` | REBUILD | Its package, class, endpoint, and outcome map has drifted from the source. |
|
||||
| `10-Cross-Host-Messaging.md` | REVISE | It labels most federation work proposed, but misses the shipped `coordinator:` lead channel. |
|
||||
| `11-Features.md` | REVISE | It is the right catalogue, but code-path names are old and it misses the second-herdr-daemon capability. |
|
||||
| `12-Claude-to-OpenCode.md` | REVISE | The porting guide is mostly current, but calls the product and spawned-member path a bridge. |
|
||||
| `13-User-Guide.md` | REVISE | It is the best operator page, but needs the product rename and the second-herdr-daemon setup. |
|
||||
|
||||
## Pages needing work
|
||||
|
||||
### `Home.md` — REVISE
|
||||
|
||||
- Quote: `# claude-bridge` (line 1) and `` `claude-bridge` keeps`` (line 11).
|
||||
The product is `fleet` / `fleetd`. The MCP server identifies itself as `fleet` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:313-315`.
|
||||
- Quote: `AgentAPI ... swappable fallback injector` (lines 73-76).
|
||||
There is no AgentAPI implementation under `fleetd/src/main/java`; the actual
|
||||
launchers are selected by `Profile.kind` in
|
||||
`fleetd/src/main/java/dev/ltms/fleet/config/FleetConfig.java:265-270`.
|
||||
|
||||
### `_Sidebar.md` — REVISE
|
||||
|
||||
- Quote: `### 📖 claude-bridge` (line 1).
|
||||
Rename it to `fleet`. `FleetMcp` registers the current product-facing tool set at
|
||||
`fleetd/src/main/java/dev/ltms/fleet/mcp/FleetMcp.java:301-326`.
|
||||
|
||||
### `1-Architecture.md` — REBUILD
|
||||
|
||||
- Quote: `` `claude-bridge` lets`` (line 3). The product was renamed; the MCP
|
||||
server name is `fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: ``fleet_read`` in the tool list (line 102). No such tool is registered.
|
||||
The complete registered list is `fleet_send` through `fleet_whoami` at
|
||||
`FleetMcp.java:301-326`; `fleet_read` is absent.
|
||||
- Quote: `SSE (GET /events)` (line 143). `FleetApp.build()` registers no `/events`
|
||||
route; its routes are listed at `FleetApp.java:143-159`.
|
||||
- Quote: `Redis Streams / NATS JetStream, or an embedded queue` (line 106).
|
||||
The shipped durable inbox is AMQP, configured by `broker`, at
|
||||
`FleetConfig.java:49-50` and `FleetConfig.java:655-714`.
|
||||
- Quote: `AgentAPI (fallback)` (line 107). No AgentAPI adapter exists; shipped
|
||||
launcher kinds are `claude-code` and `opencode` (`FleetConfig.java:265-270`).
|
||||
|
||||
### `2-Message-Server.md` — REBUILD
|
||||
|
||||
- Quote: `claude mcp add --transport http bridge http://127.0.0.1:8080/mcp`
|
||||
(line 67). The daemon defaults to port `8765` in `FleetConfig.java:183-187`,
|
||||
and identifies its server as `fleet` at `FleetMcp.java:313-315`.
|
||||
- Quote: ``fleet_send(message, target?, {block, timeout_seconds, auto_spawn,
|
||||
turn_id})`` (line 80). The real parameters are `sessionId`, `content`,
|
||||
`timeoutMs`, `wait`, `turnId`, and `coordId` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: ``fleet_read(target, source)`` (line 85). It is not registered; see the
|
||||
complete registration at `FleetMcp.java:301-326`.
|
||||
- Quote: `docs/MCP-Contract.md ... normative` (lines 87-88). That is not a valid
|
||||
reference: only §6 is current, as the current operator guide itself says at
|
||||
`.wiki-snapshot/13-User-Guide.md:466`.
|
||||
- Quote: `SSE (GET /events)` (line 45). No route exists in the built REST surface,
|
||||
`FleetApp.java:143-159`.
|
||||
|
||||
### `3-Approaches.md` — REVISE
|
||||
|
||||
- Quote: `AgentAPI ... remains a swappable fallback injector` (lines 78-84).
|
||||
It was never built. The shipped adapter selection is only `claude-code` or
|
||||
`opencode` (`FleetConfig.java:265-270`). Keep it as discarded research, not an
|
||||
operational fallback.
|
||||
- Quote: `claude-bridge` (line 109). Rename the product to `fleet`; the runtime
|
||||
package is `dev.ltms.fleet`, for example `FleetMcp.java:1`.
|
||||
|
||||
### `4-Setup.md` — RETIRE
|
||||
|
||||
It is a 25-line redirect and says its procedure was never written (lines 3-9).
|
||||
Chapter 13 is the maintained install procedure. Keeping a second navigation page
|
||||
adds no working documentation.
|
||||
|
||||
### `5-Operations.md` — RETIRE
|
||||
|
||||
It is a 35-line redirect and says its runbook was never written (lines 3-14).
|
||||
Chapter 13 now owns run and recovery instructions.
|
||||
|
||||
### `6-Team.md` — REBUILD
|
||||
|
||||
- Quote: `fleet_send {role: w-claude, prompt: A}` (line 98). `fleet_send` accepts
|
||||
`sessionId` and `content`, not `role` or `prompt` (`FleetMcp.java:1096-1108`).
|
||||
- Quote: `some on Claude, some on the remote local LLM` (lines 3-5) and `Every
|
||||
worker is ... Claude Code` (line 25). `opencode` is a first-class launcher kind,
|
||||
not a Claude worker (`FleetConfig.java:265-270`).
|
||||
- Quote: `fleetd's concurrency policy` (line 121). The configured capacity control
|
||||
is per-profile `maxLoad` (`FleetConfig.java:251-264`), not the role routing model
|
||||
described here.
|
||||
|
||||
### `7-Use-Cases.md` — REBUILD
|
||||
|
||||
- Quote: `ccs profile` (line 10), `ccs + herdr` (line 22), and `ccs-spawn`
|
||||
(line 45). The configuration has `profiles` and `fleet`, not `ccs`:
|
||||
`FleetConfig.java:34-58` and `FleetConfig.java:81-101`.
|
||||
- Quote: `fleet_send({"to", "kind", "body", "block"})` (lines 55-62).
|
||||
None of those are the shipped send parameters. The schema is
|
||||
`FleetMcp.java:1096-1108`.
|
||||
- Quote: `fleet_list() → { "profiles": ... }` (lines 74-80). `fleet_list` is a
|
||||
roster view; `fleet_profiles` is the configured-backend view, as registered at
|
||||
`FleetMcp.java:307-311` and described at `FleetMcp.java:1176-1182`.
|
||||
|
||||
### `8-Roadmap.md` — REBUILD
|
||||
|
||||
- Quote: `Java 21+` (line 43). The current project guidance and source use Java 25;
|
||||
the `FleetConfig` source itself uses Java 25 unnamed lambda parameters, for
|
||||
example `FleetConfig.java:102`.
|
||||
- Quote: `herdr 0.7.0 / protocol 14` (line 46). The current REST health endpoint
|
||||
reports the live protocol returned by herdr (`FleetApp.java:240-244`), while the
|
||||
current operator guide records protocol 19 at
|
||||
`.wiki-snapshot/13-User-Guide.md:76-85`.
|
||||
- Quote: `ccs <profile> claude` and `ccs env <profile>` (lines 47-48). Shipped
|
||||
configuration uses `Profile` records and launcher `kind`,
|
||||
`FleetConfig.java:313-330` and `FleetConfig.java:265-270`.
|
||||
- Quote: `Redis Streams via Lettuce` (line 50). The actual durable inbox is AMQP
|
||||
`broker`, `FleetConfig.java:655-714`.
|
||||
|
||||
### `9-Implementation.md` — REBUILD
|
||||
|
||||
- Quote: `rest.FleetdApp` and `mcp.BridgeMcp` (lines 29-30). The classes are
|
||||
`rest.FleetApp` and `mcp.FleetMcp` (`FleetApp.java:46`; `FleetMcp.java:67`).
|
||||
- Quote: `dev.ltms.fleetd` (line 67). The source package is `dev.ltms.fleet`
|
||||
(`FleetMcp.java:1`).
|
||||
- Quote: `WorkerPresence` (line 110). The current class is `MemberPresence`, as
|
||||
imported and used by `FleetMcp` at `FleetMcp.java:12` and `465-469`.
|
||||
- Quote: the outcome list ending in `STALE_TURN` (lines 128-131). The code also
|
||||
has `BACKEND_EXHAUSTED` (`FleetMcp.java:550-554`) and async `ASKING` handling
|
||||
(`FleetMcp.java:664-668`).
|
||||
- Quote: `FleetdApp` (line 207) and `FleetdConfig` (line 211). These names do not
|
||||
resolve; current classes are `FleetApp` and `FleetConfig`.
|
||||
|
||||
### `10-Cross-Host-Messaging.md` — REVISE
|
||||
|
||||
- Quote: the chapter says the cross-host fabric is proposed except for the
|
||||
single-host inbox (lines 3-8). Cross-host **lead-to-lead** delivery shipped:
|
||||
`fleet_send` accepts `coordId` (`FleetMcp.java:1094-1107`) and publishes it at
|
||||
`FleetMcp.java:616-641`; configuration has `coordinator` at
|
||||
`FleetConfig.java:74-78` and `99-101`.
|
||||
- Quote: `bridge.dlx` (line 90). This product name is stale. The shipped lead path
|
||||
uses `LeadChannel`, not the proposed exchange flow (`FleetMcp.java:95-96` and
|
||||
`616-641`). Keep the proposed federation design, but add a clear shipped/proposed
|
||||
boundary for CB-637.
|
||||
|
||||
### `11-Features.md` — REVISE
|
||||
|
||||
- Quote: `mcp/BridgeMcp` (line 22), `config/FleetdConfig` (lines 25-27), and other
|
||||
index references. These paths no longer resolve; the source classes are
|
||||
`mcp/FleetMcp` (`FleetMcp.java:67`) and `config/FleetConfig`
|
||||
(`FleetConfig.java:81`).
|
||||
- Quote: `fleet_whoami` returns only `primary` or `worker` (lines 99-100).
|
||||
It also returns `architect` (`FleetMcp.java:1235-1244`).
|
||||
- The page needs the missing separate member-herdr-daemon feature listed below.
|
||||
|
||||
### `12-Claude-to-OpenCode.md` — REVISE
|
||||
|
||||
- Quote: `same bridge mount` (line 5) and `a bridge-spawned worker` (line 94).
|
||||
Rename the product path to `fleet`. The daemon exposes the MCP server as `fleet`
|
||||
(`FleetMcp.java:313-315`), and profiles select OpenCode with `kind: opencode`
|
||||
(`FleetConfig.java:332-335`).
|
||||
- Quote: the sample mount name is `fleetd` (line 67). The server name is `fleet`;
|
||||
update the sample to avoid teaching a second product name.
|
||||
|
||||
### `13-User-Guide.md` — REVISE
|
||||
|
||||
- Quote: `The bridge is the only channel` (line 63). The invariant is correct, but
|
||||
the product term needs the `fleet` rename. The daemon's MCP server name is
|
||||
`fleet` (`FleetMcp.java:313-315`).
|
||||
- Quote: it describes one herdr socket (lines 72-85). It needs the optional
|
||||
`memberHerdrSocket` setup and two-daemon health meaning. The config key is in
|
||||
`FleetConfig.java:34-37`, and `/healthz` checks both daemons when configured at
|
||||
`FleetApp.java:210-245`.
|
||||
|
||||
## MISSING
|
||||
|
||||
`11-Features.md` has a body section for **routing members through a separate herdr daemon**
|
||||
(`## memberHerdrSocket`, line 2174), but **no row in the index table** at the top of the page
|
||||
(lines 20-95). That table is how the page is meant to be read, so a capability absent from it is
|
||||
effectively undiscoverable. Lead note: this is my own omission — I added the section on 2026-08-31
|
||||
and did not add the matching row. Fixed in the wiki at `68e32c6`'s successor.
|
||||
|
||||
The original audit stated the feature had no entry at all. That was wrong: the section exists. The
|
||||
gap is the index row. Recorded here rather than silently corrected, because the difference matters —
|
||||
"undocumented" and "documented but unindexed" are different jobs.
|
||||
|
||||
Evidence for the feature itself: `FleetConfig.java:34-37` and `FleetApp.java:103-115`, `210-245`,
|
||||
and `247-263`.
|
||||
|
||||
## Audit method and coverage
|
||||
|
||||
I checked all 15 pages. I checked concrete tool, route, config, class, file, and
|
||||
product-name claims claim-by-claim on 11 pages: Home, Sidebar, 1, 2, 4, 5, 6, 7, 9,
|
||||
11, and 13. I skimmed the remaining four long historical or research pages (3, 8, 10,
|
||||
12), then checked their concrete claims that affect the verdict. This is an audit of
|
||||
the supplied snapshot, not a wiki rewrite.
|
||||
+282
-17
@@ -88,6 +88,47 @@ bind:
|
||||
# backoffMs: 60000
|
||||
# quietNudgeCap: 3
|
||||
|
||||
# Lead rollover (fleetd #480): replace a lead session that has decided it is ready to be replaced,
|
||||
# without an operator doing it by hand. A lead writes a handover file, then asks fleetd to clear its
|
||||
# own pane and bootstrap a fresh session against that file.
|
||||
#
|
||||
# Opt-in on purpose — it clears the lead's own pane on request, so upgrading the daemon must never
|
||||
# acquire that ability for you. Absent block = feature off, and nothing is constructed at all. Even
|
||||
# once present, nothing but an explicit confirm() call — one that passes every check — can ever
|
||||
# cause a /clear: there is no recurring timer, heartbeat or scheduler anywhere in this feature that
|
||||
# fires one on its own initiative. confirm() itself is called FROM the calling lead's own turn, so
|
||||
# it cannot clear the pane inline (that pane is still WORKING); instead it schedules a one-shot
|
||||
# continuation that waits for the SAME confirm() call's turn to end, then does the actual work. See
|
||||
# dev.ltms.fleet.lead.LeadRollover's class javadoc for the exact order (fleetd #480 correction).
|
||||
#
|
||||
# handoverPath: REQUIRED when this block is present — where the handover file a fresh lead session
|
||||
# reads must live. No default (an operator-specific path); a present block with no
|
||||
# handoverPath refuses to start. May be relative: it then resolves against the
|
||||
# CALLING lead's own fleet.leaders.<name>.cwd (falling back to the daemon's own
|
||||
# working directory when that lead has none configured) — never against whatever
|
||||
# directory the daemon process happens to have been started in. An absolute path is
|
||||
# used unchanged. Prefer an absolute path if the daemon and the lead's pane might not
|
||||
# share a working directory (fleetd #480 follow-up).
|
||||
# requireOperatorConfirm: true # default true — confirm() refuses unless the caller also passes
|
||||
# # operatorConfirmed: true
|
||||
# maxDocAgeSeconds: 3600 # default 3600 — refuse a handover file older than this
|
||||
# turnSettleSeconds: 20 # default 20 — how long the deferred roll waits for the CALLING
|
||||
# # lead's own turn to end (its pane to report injectable again)
|
||||
# # before sending /clear at all. If this elapses, /clear is NEVER
|
||||
# # sent — a lead that never goes idle is still doing real work.
|
||||
# clearSettleSeconds: 20 # default 20 — how long to wait for the pane to become injectable
|
||||
# # again AFTER /clear before giving up (never sends bootstrapText
|
||||
# # if this elapses). A separate, second wait from turnSettleSeconds.
|
||||
# bootstrapText: "..." # default names the RESOLVED (absolute) handoverPath — sent to
|
||||
# # the lead once its pane settles after /clear
|
||||
# leadRollover:
|
||||
# handoverPath: /path/to/handover.md
|
||||
# requireOperatorConfirm: true
|
||||
# maxDocAgeSeconds: 3600
|
||||
# turnSettleSeconds: 20
|
||||
# clearSettleSeconds: 20
|
||||
# bootstrapText: "Fresh lead session: read the handover file and carry on."
|
||||
|
||||
# Fleet health detection is dormant unless enabled (CB-573). It reads one whole-fleet agent list
|
||||
# per tick.
|
||||
# intervalSeconds → how often a tick runs (default 30). ENFORCED floor of 15: the code computes
|
||||
@@ -110,10 +151,30 @@ bind:
|
||||
# notifications:
|
||||
# mode: disabled
|
||||
|
||||
# Idle-sleep guard: while at least one member is live, hold an OS-level assertion against idle
|
||||
# sleep (macOS only — a `caffeinate -i` child; a no-op elsewhere or if caffeinate is missing), so
|
||||
# an unattended host does not idle-sleep out from under a member's long turn. Unlike health/
|
||||
# configReload above, this is ON BY DEFAULT — omitting the block entirely leaves it enabled, the
|
||||
# same as `enabled: true`. Uncomment only to turn it off:
|
||||
# idleSleepGuard:
|
||||
# enabled: false
|
||||
|
||||
# herdr Unix socket. Omit to use the client default
|
||||
# (${HERDR_SOCKET_PATH:-~/.config/herdr/herdr.sock}).
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
|
||||
# Optional socket for member panes. Omit this to use herdrSocket for both leads and members.
|
||||
# memberHerdrSocket: /Users/member/.config/herdr/herdr.sock
|
||||
|
||||
# fleetd #213: the login shell the member OS user (memberHerdrSocket above) actually runs. ONLY
|
||||
# read when memberHerdrSocket is set — fleetd's own $SHELL says nothing about a pane running
|
||||
# under a different OS user, and there is no channel to ask herdr for that user's shell, so this
|
||||
# must be told rather than guessed. Absent, blank, or anything not ending in "zsh" is treated the
|
||||
# same as "not zsh": the memberCredentials.policy: allow-list ZDOTDIR scrub (see worktreeGroup
|
||||
# below) is skipped in favour of the weaker CB-596 sentinel overlay — a degraded control, never a
|
||||
# refusal to spawn. When memberHerdrSocket is absent this key is never consulted at all.
|
||||
# memberLoginShell: /bin/zsh
|
||||
|
||||
# How member sessions are spawned. Define one or more named profiles (backends) under
|
||||
# `profiles`; each key is the profile name (also the ccs profile). A profile says only WHICH
|
||||
# BACKEND — model, CLI adapter, credentials, cost. It says nothing about what a member spawned on
|
||||
@@ -160,7 +221,11 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# skills/MCP/hooks. Omit to leave the worker on the host default.
|
||||
# parityOverlay → repo-relative paths copied primary→worktree so a worker in a provisioned
|
||||
# worktree sees the same local config (CB-301-ext). Omit for the default set:
|
||||
# [.env, .envrc]. (.claude/settings.local.json is NOT in the default — it
|
||||
# [.env] only (CB-148). .envrc is left out of the default on purpose: it is
|
||||
# executable shell that direnv runs on every cd, so copying it carries
|
||||
# behaviour into the worker, not just values, unlike .env. An operator who
|
||||
# wants it copied can still write parityOverlay: [.env, .envrc] explicitly.
|
||||
# (.claude/settings.local.json is NOT in the default — it
|
||||
# pre-approves IDE/tool grants a member must not hold ambiently; CB-525/CB-634.)
|
||||
#
|
||||
# Do NOT add .mcp.json (CB-525). A worker's tools are whatever its launcher
|
||||
@@ -183,8 +248,12 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# omit and this profile's completion fallback behaves exactly as before.
|
||||
# Every backend words its refusal differently, so this is config, never a
|
||||
# vendor string baked into fleetd itself.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart, same as this profile's model/baseUrl/argv.
|
||||
# HOT (fleetd #446): read live, cached by profile name, at every
|
||||
# completion-fallback check AND by fleet_profiles' exhaustionDetectionArmed —
|
||||
# editing it and reloading arms or disarms usage-limit detection for this
|
||||
# profile with no daemon restart. (Before fleetd #446 this was DEFERRED,
|
||||
# compiled once into a startup pattern map like model/baseUrl/argv still are —
|
||||
# see errorPattern below, which is still deferred that way on purpose.)
|
||||
# credentialId → CB-578 stage B: the credential this profile quarantines WITH when a
|
||||
# BACKEND_EXHAUSTED classification fires. Two profiles that set the SAME
|
||||
# credentialId share one quarantine — the case this exists for is two models
|
||||
@@ -194,6 +263,40 @@ herdrSocket: ~/.config/herdr/herdr.sock
|
||||
# profile quarantines alone, under its own name, exactly as if the field did
|
||||
# not exist. Cooldown length is the top-level quarantineCooldownSeconds below.
|
||||
# HOT: read live at every spawn/exhaustion check — no restart needed.
|
||||
# errorPattern → fleetd #201 / #227: regex matched against a completion-fallback scrape to
|
||||
# classify a turn that ended with no fleet_reply as a BACKEND ERROR — a
|
||||
# credential outage or a provider 5xx — rather than a real answer or a
|
||||
# usage-limit exhaustion (exhaustedPattern above always wins when a line
|
||||
# matches both). Opt-in. Omit it and this profile falls back to fleetd's
|
||||
# built-in legacy pattern `(?i)\bAPI Error\s*:` — classification still
|
||||
# happens, just without a profile-specific match; every backend words its
|
||||
# failure differently, so a hardcoded sentence would only ever match one
|
||||
# of them.
|
||||
# DEFERRED: compiled once into a startup pattern map — editing it needs a
|
||||
# daemon restart. Unlike exhaustedPattern above (made hot by fleetd #446),
|
||||
# errorPattern was scoped out of that ticket on purpose and stays deferred.
|
||||
# # errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage
|
||||
#
|
||||
# What happens once a match fires (BackendOutagePolicy, credentialId-keyed,
|
||||
# SEPARATE from the CB-578 stage B quarantine above and never merged with it):
|
||||
# - threshold 2 — TWO DISTINCT TARGETS (never raw events) on the same
|
||||
# effective credential inside a 60-second window start an "incident" and a
|
||||
# 60-second cool-off for that credential. One member repeating the same
|
||||
# classified line twice never cools anything off — a real outage hits
|
||||
# every target on that credential, so requiring a second, independent
|
||||
# target loses nothing against the case this guards against, while
|
||||
# protecting against a heuristic misfire on one flaky member.
|
||||
# - a fresh error while a credential is already cooling off is ignored
|
||||
# outright: it neither extends the 60s deadline nor starts a new incident.
|
||||
# - `fleet_list`/`fleet_profiles` report a cooling credential with
|
||||
# `coolingOffForSeconds` (never `quarantinedForSeconds`, unless CB-578
|
||||
# exhaustion quarantine is ALSO independently active for the same
|
||||
# credential — the two checks can both fire at once). A spawn onto a
|
||||
# cooling profile is refused with a message naming the credential and
|
||||
# remaining seconds — "cooling off", never "exhausted", so an operator can
|
||||
# tell a short transient fault from a spent subscription at a glance.
|
||||
# - the lead gets ONE nudge per incident (not one per affected target), via
|
||||
# the same push loop that already delivers ticket/question reminders.
|
||||
# env → extra environment for this profile's workers, as a literal key/value map
|
||||
# (CB-511). Use it to give workers a toolchain.
|
||||
#
|
||||
@@ -257,13 +360,49 @@ profiles:
|
||||
# GOTCHA 2 — `maxLoad` is the ONLY throttle you have here. There is no metering, no budget
|
||||
# and no refusal on cost; the cap on live members is the single thing standing between a
|
||||
# fan-out and your monthly limit. Set it deliberately and keep it small.
|
||||
#
|
||||
# GOTCHA 3 (fleetd #176, corrected by fleetd #257) — `maxLoad` counts members, never the lead
|
||||
# itself. The lead is a live `claude` session on this SAME account (a lead is never moved
|
||||
# off-subscription, whatever its own profile says), so it already holds one seat before any
|
||||
# member spawns. If a lead's `fleet.leaders.<name>.profile` names THIS profile — or ANY OTHER
|
||||
# `subscription: true` profile that shares this one's account (see THE SENTINEL, just below,
|
||||
# next to `credentialId:`) — `fleet_list` reports that seat count under `leadSeats`; see
|
||||
# `profile:` under THE FLEET below. `free` itself is NEVER reduced by `leadSeats`: `free` means
|
||||
# "what the real placement gate (`CompositePeerLauncher#enforceMaxLoad`) will actually grant a
|
||||
# fresh `fleet_spawn` right now", and that gate only ever compares live members against
|
||||
# `maxLoad` — it has no notion of the lead's own seat. An earlier cut of this feature
|
||||
# subtracted `leadSeats` from `free` on the theory it made `free` describe the true ceiling on
|
||||
# the account, but no backend seat ceiling shared with the lead has ever actually been
|
||||
# measured, and the subtraction just made `free` disagree with the one thing it is supposed to
|
||||
# describe — the fleetd #257 fix. `maxLoad: 3` means 3 member slots, full stop; a lead sharing
|
||||
# the account is a fact you can see in `leadSeats`, not a reason `free` undercounts spawns that
|
||||
# will, in practice, succeed.
|
||||
#
|
||||
# THE SENTINEL (fleetd #176 stage 2, correcting an inert stage 1 fix): every `subscription:
|
||||
# true` profile that leaves `credentialId` unset shares ONE implicit account-wide credential
|
||||
# id with every other such profile on this host — because a subscription profile doesn't
|
||||
# authenticate with a credential of its own, it authenticates as the operator's own Claude
|
||||
# login, and there is exactly one of those. So on a typical host, `opus` (the lead's profile)
|
||||
# and `sonnet` (the members' profile) are linked automatically, with NOTHING to set here — that
|
||||
# is what makes GOTCHA 3 above work without also writing matching `credentialId:` values on
|
||||
# both. This linkage is not just cosmetic: it is the same key `BackendQuarantine`/cool-off use,
|
||||
# so a usage-limit hit on `opus` now quarantines `sonnet` too (and vice versa) — correct, since
|
||||
# they are one Claude account, but worth knowing before you wonder why an unrelated-looking
|
||||
# profile went quarantined.
|
||||
#
|
||||
# WHEN TO OVERRIDE — set explicit, DIFFERENT `credentialId:` values on two `subscription: true`
|
||||
# profiles only when they are genuinely two separate Claude logins on the same host (a real,
|
||||
# if unusual, setup). An explicit `credentialId` always wins over the sentinel, so this is the
|
||||
# one way to keep two subscription profiles from being treated as one account for lead-seat
|
||||
# counting AND for quarantine/cool-off grouping alike.
|
||||
# gitTokenEnv: GITEA_TOKEN # opt-in: let this profile's workers open their own PR (CB-302)
|
||||
# gitHostEnv: GITEA_HOST # defaults to GITEA_HOST; injected only with gitTokenEnv
|
||||
# exhaustedPattern: "usage limit has been reached" # opt-in: classify a usage-limit refusal (CB-578)
|
||||
# credentialId: shared-openai # opt-in: quarantine together with every other profile sharing this id (CB-578)
|
||||
# errorPattern: "503 Service Unavailable" # opt-in: classify a backend outage (fleetd #201/#227) — see the key doc above
|
||||
# configDir: /Users/me/.ccs/instances/gx10 # CLAUDE_CONFIG_DIR — inherit that profile's skills/MCP
|
||||
# cwd: /Users/me/src/myrepo # pin the working dir; omit to inherit the primary's
|
||||
# parityOverlay: [".env", ".envrc"] # the default; never add .mcp.json or .claude/settings.local.json — see above
|
||||
# parityOverlay: [".env"] # the default; add ".envrc" explicitly if you want it copied too (CB-148) — never add .mcp.json or .claude/settings.local.json — see above
|
||||
# ideMcpUrl: http://127.0.0.1:29170/index-mcp/streamable-http # opt-in (CB-634): IDE code intelligence, pinned to the worktree
|
||||
# ideProjectDir: fleetd # CB-634: module dir the IDE opens + the overlay pins (this repo's pom is in fleetd/)
|
||||
# ideOpenCommand: env DISPLAY=:10.0 idea {dir} # CB-634 auto-open: opens {dir} in the IDE at spawn; omit to open by hand
|
||||
@@ -354,9 +493,24 @@ placement: weighted
|
||||
# seconds, before a spawn may land on it again. Applies to every profile's effective credential
|
||||
# (its own name, or its credentialId if set above) — there is no per-profile override. Default
|
||||
# 1800 (30 minutes) when omitted or non-positive.
|
||||
#
|
||||
# fleetd #466: this is now only the BASE of an escalating backoff, not a flat retry rate. A
|
||||
# credential quarantined again within one base cooldown of the previous quarantine ending (still
|
||||
# reporting exhausted — e.g. a weekly subscription limit that hasn't reset) backs off further:
|
||||
# cooldown doubles each such time, capped at 12x this value (~6 hours at the 1800s default). A
|
||||
# quarantine that starts after a base-cooldown's worth of quiet resets back to this value. Not
|
||||
# configurable per se — the multiplier and ceiling are constants in BackendQuarantine, not new
|
||||
# YAML keys; see its class doc for the exact formula and why there is no automatic probe to clear
|
||||
# it early (the operator's own design constraint — a probe spends the quota it's measuring).
|
||||
# DEFERRED: baked once into the BackendQuarantine built at startup — a running quarantine keeps
|
||||
# its original cooldown regardless; a new value only applies to a quarantine that starts after a
|
||||
# restart. Editing this needs a daemon restart to take effect.
|
||||
#
|
||||
# This does NOT govern the fleetd #201 / #227 backend-error cool-off documented under errorPattern
|
||||
# above — that mechanism is a separate, shorter-lived, NOT-configurable policy (threshold 2 distinct
|
||||
# targets, 60-second window, 60-second cool-off), on purpose: it exists to survive a brief transient
|
||||
# fault, not to replace this 30-minute exhaustion quarantine. Do not conflate the two when reading
|
||||
# fleet_list/fleet_profiles — coolingOffForSeconds and quarantinedForSeconds are independent facts.
|
||||
# quarantineCooldownSeconds: 1800
|
||||
|
||||
# Re-read this file without restarting the daemon (CB-559). Off unless you add this block, so an
|
||||
@@ -369,9 +523,10 @@ placement: weighted
|
||||
# when the reload happens — not about how important the key is:
|
||||
# HOT → takes effect on the next spawn: the whole `fleet:` block (every role pool,
|
||||
# `charters`, and `tabLabel`), `placement:`, and an existing profile's weight / maxLoad
|
||||
# / credentialId. Those are hot because the placement policy (and, for credentialId,
|
||||
# the CB-578 stage B quarantine check) reads them through a supplier — being config is
|
||||
# not by itself enough to make a key hot.
|
||||
# / credentialId / exhaustedPattern. Those are hot because the placement policy (and,
|
||||
# for credentialId, the CB-578 stage B quarantine check; for exhaustedPattern, fleetd
|
||||
# #446's LiveExhaustedPatterns) reads them through a supplier — being config is not by
|
||||
# itself enough to make a key hot.
|
||||
# EXCEPT `fleet.leaders`: Fleetd.main reads it once at startup to build the lead tab
|
||||
# scanner and launcher, and neither is rebuilt on reload. A changed/added/removed
|
||||
# `fleet.leaders` entry is silently accepted — the reload reports "config reloaded"
|
||||
@@ -383,8 +538,10 @@ placement: weighted
|
||||
# stage B — baked once into the quarantine tracker built at startup), ADDING or
|
||||
# REMOVING a profile (a new backend needs its own launcher, and launchers are built
|
||||
# once), AND an existing profile's launch settings — model, baseUrl, argv, env,
|
||||
# configDir, mcpUrl, tabLabel, exhaustedPattern. The launcher takes a copy of
|
||||
# `profiles:` at startup and resolves every spawn out of that copy, so those never
|
||||
# configDir, mcpUrl, tabLabel, errorPattern (fleetd #201 / #227 — compiled once into
|
||||
# a startup pattern map; exhaustedPattern used to be compiled the same way until
|
||||
# fleetd #446 made it hot — see above). The launcher takes a copy of `profiles:` at
|
||||
# startup and resolves every spawn out of that copy, so those never
|
||||
# reach a launch until you restart. The reload logs them by name rather than
|
||||
# pretending they applied.
|
||||
# COLD → cannot change at all: `bind:`, `herdrSocket:`, `broker:` and `auth:`. The socket is
|
||||
@@ -445,6 +602,23 @@ fleet:
|
||||
# recognised: give it a `profile:` and the daemon launches the shortfall when fewer than
|
||||
# `instances` are live. Omit `profile:` and it is recognise-only, as before.
|
||||
#
|
||||
# `profile:` has a SECOND job as of fleetd #176, even for a recognise-only lead you never want
|
||||
# auto-launched: it is also how fleetd learns which account this lead's own session shares. A
|
||||
# `subscription: true` profile bills the operator's Claude account, and the lead itself is always
|
||||
# a live `claude` session on that same account — `maxLoad` never counted that seat. If a lead
|
||||
# entry here names a profile that shares a worker profile's account, `fleet_list` reports the
|
||||
# lead's live seat(s) on that worker profile under `leadSeats` — informational only, as of fleetd
|
||||
# #257 it is NEVER subtracted from `free` (see GOTCHA 3, next to `maxLoad:`, in THE WORKERS above,
|
||||
# for why). "Shares the account" is decided by matching `effectiveCredentialId()`, which (fleetd
|
||||
# #176 stage 2 — see THE SENTINEL, next to `credentialId:`, in THE WORKERS above) means: an
|
||||
# explicit, matching `credentialId:` on both, OR — the common case, needing NO extra config — both
|
||||
# being `subscription: true` with `credentialId` left unset, since those all share one implicit
|
||||
# account-wide id. A lead on `opus` and workers on `sonnet` link automatically this way; they do
|
||||
# NOT need the same profile name. Setting `profile:` on an already-running, recognise-only lead is
|
||||
# safe — the daemon only launches the SHORTFALL below `instances`, so naming a profile here does
|
||||
# not, by itself, start anything. Omit it and fleetd has no way to derive the sharing — there is
|
||||
# no other reliable signal on the daemon's side — so that lead's seat never appears in `leadSeats`.
|
||||
#
|
||||
# `tab:` (CB-579) is REQUIRED and is the only field identity depends on — the exact label of the
|
||||
# tab hosting the lead, matched case-insensitively. Label the tab yourself and put that same
|
||||
# string here, and the pane is recognised on the next rescan. Reopen the tab later, or the session
|
||||
@@ -566,20 +740,35 @@ guard:
|
||||
# every name here NOT also in `allow` is overlaid with a non-secret sentinel value before
|
||||
# the pane's login shell runs — real protection only for names that shell does not itself
|
||||
# re-export (see the ROUND-2 CORRECTION note above). Under allow-list: reporting only.
|
||||
# sshAuthSock → whether SSH_AUTH_SOCK may pass through under allow-list ("allow") or must be
|
||||
# blanked like any other non-derived name ("block", the default). This is a decision you
|
||||
# have to make explicitly: SSH_AUTH_SOCK is a handle to YOUR ssh-agent, and a member
|
||||
# holding it can sign with your keys — it sits in no secret file and looks like no
|
||||
# credential, which is why it slipped past three earlier tickets (gitea #110). Blocking
|
||||
# it breaks git over SSH inside members (push/fetch authenticate as you); use HTTPS
|
||||
# remotes or scoped deploy keys instead of allowing it lightly.
|
||||
# sshAgentEnv → whether SSH_AUTH_SOCK may pass through under allow-list ("inherit") or is omitted
|
||||
# from the member environment ("omit", the default). Omitting it only omits the
|
||||
# inherited ssh-agent path. It discourages automatic use of the operator's agent.
|
||||
# It does not deny same-user access to that socket. It also does not block SSH keys that
|
||||
# are readable on disk. Git over SSH may still work from inside a member. Keep the block:
|
||||
# it is correct and costs nothing, but it is not a control. A member runs as the same OS
|
||||
# user as the lead. Inside one uid, ordinary Unix permissions provide no meaningful
|
||||
# confidentiality boundary. A real boundary needs a different OS user or OS-level
|
||||
# confinement, such as a container or VM. That is the open question in fleetd #184.
|
||||
#
|
||||
# Still do not set this to "inherit" casually. SSH_AUTH_SOCK is a live handle to YOUR
|
||||
# ssh-agent, so a member holding it can sign with EVERY key the agent holds. It sits in
|
||||
# no secret file and looks like no credential, which is why it slipped past three
|
||||
# earlier tickets (gitea #110). Blocking it does not contain a member, but allowing it
|
||||
# hands one a signing capability for no gain — the block costs nothing, so keep it.
|
||||
#
|
||||
# Both halves of this are measured, not argued. 2026-08-28: a member with
|
||||
# SSH_AUTH_SOCK blanked pushed to the forge over SSH successfully, because `ssh -G`
|
||||
# resolves an IdentityFile outside ~/.ssh that is readable and has no passphrase. An
|
||||
# earlier version of this comment claimed blocking the socket BREAKS git over SSH. It
|
||||
# does not. That claim came from looking only in ~/.ssh, which holds nothing but four
|
||||
# `Include` lines — looking in one place and concluding about the whole host.
|
||||
#
|
||||
# HOT-RELOADABLE the same way `fleet:` is (CB-559): read fresh on every spawn, so editing this list
|
||||
# and reloading config (or restarting) changes what the NEXT spawn inherits; already-running members
|
||||
# are unaffected either way.
|
||||
# memberCredentials:
|
||||
# policy: deny-by-default # or "deny-list", or "allow-list" (CB-633) — see above
|
||||
# sshAuthSock: block # allow-list only; see the sshAuthSock note above
|
||||
# sshAgentEnv: omit # allow-list only; see the sshAgentEnv note above
|
||||
# allow:
|
||||
# - AI_GATEWAY_TOKEN # named in a profile's tokenEnv (local/gx) — a member reaching the
|
||||
# # gateway is by design, not a leak
|
||||
@@ -631,6 +820,46 @@ guard:
|
||||
# to a sibling directory of the repo root.
|
||||
# worktreeRoot: /Users/me/src/.bridged-worktrees
|
||||
|
||||
# Worktree group sharing (fleetd #185 stage 3). OPTIONAL, off by default. Names an OS group
|
||||
# that a provisioned worktree's repo is made group-writable for (git config
|
||||
# core.sharedRepository group, plus a one-time chgrp/chmod/setgid fix-up), so a member spawned
|
||||
# under a DIFFERENT OS user (see memberHerdrSocket) can write its own worktree, its
|
||||
# per-worktree git metadata, and its own commit objects — without it, every file GitWorktrees
|
||||
# creates is owned by fleetd's own uid and unwritable by another user.
|
||||
# CAUTION: this isolates credentials, not the repository — a member in the group can still
|
||||
# write the operator's git objects and refs in the shared repo. The operator running fleetd
|
||||
# must already be a member of the named group, or every provisioning spawn fails loudly.
|
||||
#
|
||||
# fleetd #213: this is also the ONE group the memberCredentials.policy: allow-list ZDOTDIR scrub
|
||||
# reuses when memberHerdrSocket is set — deliberately not a second config key. Under
|
||||
# memberHerdrSocket, the scrub directory is generated under worktreeRoot (never java.io.tmpdir,
|
||||
# which the member OS user cannot reach) and shared read-only with this group. If worktreeGroup
|
||||
# is unset while memberHerdrSocket is set, the scrub cannot be guaranteed reachable by the member,
|
||||
# so fleetd falls back to the weaker CB-596 sentinel overlay instead (a WARN names the gap).
|
||||
# worktreeGroup: fleet-workers
|
||||
|
||||
# fleetd #362: a directory of skill folders (each a subdirectory holding a SKILL.md, the same
|
||||
# shape as this repo's own .claude/skills/) copied into every PROVISIONED worktree's
|
||||
# .claude/skills/, so a member spawned against ANY repo — not only one that already ships its own
|
||||
# copy — can load a bridge skill (e.g. implementer). Unset (the default): no worktree is touched
|
||||
# beyond today's behaviour. A skill folder the target repo already carries under
|
||||
# .claude/skills/<name> is never overwritten — the repo's own copy always wins. Best-effort like
|
||||
# worktreeGroup above: a missing/unreadable directory here is logged and skipped, never a failed
|
||||
# spawn. Every non-hidden subdirectory of this directory is copied wholesale, with no per-file
|
||||
# allowlist — don't park scratch files or drafts alongside the real skill folders, they will be
|
||||
# copied into every provisioned worktree too.
|
||||
#
|
||||
# fleetd #393: which member KINDS actually consume this once it is copied. kind: claude-code —
|
||||
# the Claude Code CLI discovers .claude/skills/ on its own; nothing else is needed. kind: opencode
|
||||
# — opencode has no such discovery, so OpenCodeLauncher reads whatever landed under
|
||||
# .claude/skills/ and appends each seeded skill's SKILL.md to the generated instructions[] file
|
||||
# (opencode's only channel for static guidance text; unlike Claude Code's Skill tool, the content
|
||||
# is always part of the system prompt, not loaded on demand). Both kinds are covered as of #393 —
|
||||
# earlier builds copied the files for every kind but only claude-code could read them, and the
|
||||
# seeding log said "N of M" regardless. Check the per-spawn launcher log (not just the seeding
|
||||
# log) to see what a given member actually got.
|
||||
# memberSkills: /path/to/fleetd/checkout/.claude/skills
|
||||
|
||||
# Session lifecycle limits (CB-303). All knobs are opt-in; omit or set to null to keep
|
||||
# the feature disabled. By default the daemon never reaps, caps, or drains sessions.
|
||||
# idleTtlSeconds → reap READY/DONE sessions idle longer than this (never BUSY/SPAWNING)
|
||||
@@ -678,10 +907,17 @@ guard:
|
||||
# across every daemon sharing this vhost.
|
||||
# prefetch → consumer basicQos, capping how many unacked messages the mailbox holds in-heap.
|
||||
# Default 32 when omitted.
|
||||
# peers → fleetd #361: the coord-ids of the OTHER daemons on this vhost, declared by the
|
||||
# operator (the daemon never guesses). fleet_list reports each one's live reachability
|
||||
# (a passive queue check, never a presence protocol) alongside this daemon's own
|
||||
# mailbox state. Omit, or leave empty, for a daemon with no known peers yet — an
|
||||
# undeclared peer can still reach you and be reached by fleet_send, it just will not
|
||||
# show up as a row in fleet_list.
|
||||
# coordinator:
|
||||
# uriEnv: LEAD_COORD_URI
|
||||
# selfId: mac-opus
|
||||
# prefetch: 32
|
||||
# peers: [fleet01-lead]
|
||||
|
||||
# Active push-to-primary (CB-307 Stage 3). When a worker reply lands with no open fleet_send,
|
||||
# the ReplyPushLoop injects a *drain nudge* (never the payload) into the primary's own herdr
|
||||
@@ -706,3 +942,32 @@ guard:
|
||||
# terminal: term_65619bd6174568
|
||||
# pushReminders: 5
|
||||
# pushBackoffMs: 15000
|
||||
|
||||
# Central allow-list of models any profiles: entry may name. Nothing checked a profile's model:
|
||||
# value before this block existed — it was a free-form string handed straight to the backend
|
||||
# adapter, and a withdrawn or misspelled name failed silently instead of at config load (opencode
|
||||
# falls back to a default model rather than erroring on an unknown -m).
|
||||
#
|
||||
# Absent, or present with an empty allow:, is OFF: no profile's model: is checked, exactly like
|
||||
# before this block existed. fleetd.yaml is gitignored on every host, so an upgrade must not force
|
||||
# every operator to enumerate their models before the daemon will start.
|
||||
#
|
||||
# The list is the authority; profiles: is checked against it, never the reverse — adding or
|
||||
# editing a profiles: entry cannot, by itself, widen what is permitted here.
|
||||
#
|
||||
# Enforcement is at CONFIG LOAD only (a bad model: fails the daemon at startup, naming both the
|
||||
# model and the profile). There is no spawn-time enforcement, no runtime on/off switch, and no
|
||||
# interaction with BackendQuarantine — those are separate, later units.
|
||||
#
|
||||
# allow → the permitted models. Each entry is its own block (not a bare string) so a later unit
|
||||
# can add an on/off state or a load limit per model without changing this shape.
|
||||
# model → the model id exactly as a profiles: entry's model: field would write it. One flat,
|
||||
# opaque-string namespace: a bare Claude id (claude-sonnet-5) and an opencode
|
||||
# provider-prefixed id (openai/gpt-5.6-terra) both fit here unchanged — the check is a
|
||||
# plain string match, never a parse of the provider prefix or a branch on kind:.
|
||||
# models:
|
||||
# allow:
|
||||
# - model: claude-sonnet-5
|
||||
# - model: claude-opus-5
|
||||
# - model: openai/gpt-5.6-terra
|
||||
# - model: amazon.nova-pro-v1:0
|
||||
|
||||
@@ -28,6 +28,8 @@
|
||||
<testcontainers.version>1.20.4</testcontainers.version>
|
||||
<commons-compress.version>1.27.1</commons-compress.version>
|
||||
<commons-lang3.version>3.18.0</commons-lang3.version>
|
||||
<sqlite-jdbc.version>3.53.4.0</sqlite-jdbc.version>
|
||||
<archunit.version>1.5.0</archunit.version>
|
||||
</properties>
|
||||
|
||||
<!--
|
||||
@@ -44,6 +46,12 @@
|
||||
3.0-rc5; bumping Jackson 3 to the patched 3.2.x breaks the SDK (annotation mismatch).
|
||||
Only the loopback /mcp endpoint parses this JSON, from trusted local Claude clients.
|
||||
The 11.0.23 -> 11.0.25 bump did clear jetty CVE-2024-8184 (5.9) and CVE-2024-6763.
|
||||
|
||||
fleetd #206: org.xerial:sqlite-jdbc 3.53.4.0 (added for OpenCodeSessionDiscovery) — the
|
||||
only known advisory against this artifact is CVE-2023-32697 (RCE via an attacker-controlled
|
||||
JDBC URL), fixed in 3.41.2.2; 3.53.4.0 is well past that fix and OSV.dev reports no open
|
||||
advisory against it. Checked via the OSV.dev API (no Mend.io/JetBrains IDE MCP mount
|
||||
available from this worktree) on 2026-08-31.
|
||||
-->
|
||||
|
||||
<!-- Force the latest patched Jetty 11.x across all Javalin-pulled Jetty modules (no version
|
||||
@@ -123,6 +131,17 @@
|
||||
<version>${amqp.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #206: opencode moved its session store from a JSON tree to SQLite
|
||||
(opencode.db). This is the JDBC driver OpenCodeSessionDiscovery uses to read it
|
||||
read-only. Ships bundled native libraries (linux/mac/windows, several archs), so it
|
||||
is a heavier jar than most deps here — see the pom's dependency-security note below
|
||||
for the size/CVE tradeoff actually measured. -->
|
||||
<dependency>
|
||||
<groupId>org.xerial</groupId>
|
||||
<artifactId>sqlite-jdbc</artifactId>
|
||||
<version>${sqlite-jdbc.version}</version>
|
||||
</dependency>
|
||||
|
||||
<!-- Logging -->
|
||||
<dependency>
|
||||
<groupId>org.slf4j</groupId>
|
||||
@@ -158,6 +177,14 @@
|
||||
<version>${testcontainers.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
|
||||
<!-- fleetd #131: package-boundary and cycle enforcement (PackageCyclesTest). -->
|
||||
<dependency>
|
||||
<groupId>com.tngtech.archunit</groupId>
|
||||
<artifactId>archunit-junit5</artifactId>
|
||||
<version>${archunit.version}</version>
|
||||
<scope>test</scope>
|
||||
</dependency>
|
||||
</dependencies>
|
||||
|
||||
<build>
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -30,8 +30,27 @@ public final class Authz {
|
||||
DRAIN,
|
||||
/** Read-only observation: status, roster, profiles, task polling. */
|
||||
READ,
|
||||
/**
|
||||
* Read (never ack) this daemon's own held lead-to-lead coordination mail (fleetd #421).
|
||||
*
|
||||
* <p>Deliberately <strong>not</strong> folded into {@link #READ}. {@code READ}'s grant
|
||||
* rests on "the roster carries no secrets" (see its case below) — a lead-to-lead body is
|
||||
* not the roster; it is where leads discuss host shapes, credentials and unmerged work.
|
||||
* Mapping this to {@code READ} would let any worker read every peer lead's mail in full
|
||||
* and would silently falsify that comment for every other {@code READ} caller.
|
||||
*/
|
||||
COORD_READ,
|
||||
/** Scrape the metrics endpoint. */
|
||||
METRICS
|
||||
METRICS,
|
||||
/**
|
||||
* Drive the lead-rollover executor ({@code fleet_handover}: open/confirm/cancel a
|
||||
* self-replace, fleetd #480 Unit C). Primary-only, same as {@link #SPAWN}/{@link #STOP}/
|
||||
* {@link #DRAIN} — and, unlike those, the terminal it acts on is never even an argument:
|
||||
* {@code LeadRollover#open}/{@code #confirm} are always called with the CALLER's own
|
||||
* connection-resolved terminal (see {@code dev.ltms.fleet.lead.LeadRollover}'s class
|
||||
* javadoc, fleetd #480 correction 2), so a primary can only ever roll itself.
|
||||
*/
|
||||
HANDOVER
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -50,7 +69,7 @@ public final class Authz {
|
||||
// deliberately does NOT get these (CB-548), so it cannot tear down or stand up workers
|
||||
// even though it coordinates them; and a worker driving any of these would be a worker
|
||||
// escalating into the orchestrator role.
|
||||
case SPAWN, STOP, DRAIN -> caller.isPrimary();
|
||||
case SPAWN, STOP, DRAIN, HANDOVER -> caller.isPrimary();
|
||||
|
||||
// Delivering a turn is open to the primary and the architect: an architect delegates
|
||||
// to workers (that is the role's point) but still has no lifecycle rights. A worker is
|
||||
@@ -68,6 +87,11 @@ public final class Authz {
|
||||
// Observation is open to every authenticated role: a worker legitimately polls its own
|
||||
// status, and the roster carries no secrets.
|
||||
case READ, METRICS -> caller.isPrimary() || caller.isWorker() || caller.isArchitect();
|
||||
|
||||
// fleetd #421: reading held lead-to-lead mail is the primary's alone. An architect
|
||||
// holds READ today (CB-548), so "not primary" must mean not-architect here too — this
|
||||
// is coordination between leads, not observation of the roster.
|
||||
case COORD_READ -> caller.isPrimary();
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -236,7 +236,15 @@ public final class CallerResolver {
|
||||
// loopback-trust: same-host callers that are not workers are the primary. A non-loopback
|
||||
// caller is anonymous even here — and startup refuses that combination anyway
|
||||
// (FleetConfig.validateAuthExposure), so this is defence in depth, not the control.
|
||||
return isLoopback(remoteAddr) ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
//
|
||||
// fleetd #317: "not a worker" must not be conflated with "identity unresolved". The real
|
||||
// primary is a real process — its pid resolves (c.resolved()), it just owns no herdr pane.
|
||||
// A caller whose peer-PID lookup failed (LsofPeerPidLookup's -1 sentinel — on any failure,
|
||||
// silently including "lsof found no match") has no such pid, and PaneLocator's own javadoc
|
||||
// already names what happens if that case is handed the primary role: a worker→primary
|
||||
// escalation. So an unresolved caller is refused (ANONYMOUS — the same clean, already-tested
|
||||
// "authenticated as nothing" outcome used everywhere else in this method), never promoted.
|
||||
return isLoopback(remoteAddr) && c.resolved() ? Principal.primary(c.pid()) : Principal.anonymous();
|
||||
}
|
||||
|
||||
private boolean presentedTokenMatches(String authorizationHeader) {
|
||||
@@ -262,11 +270,15 @@ public final class CallerResolver {
|
||||
return token.isEmpty() ? null : token;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #305: delegates to {@link ConnectionIdentity#isLoopback}. This used to be a second,
|
||||
* independent copy of the same rule, and the two drifted: this one accepted all of
|
||||
* {@code 127.0.0.0/8}, {@code ConnectionIdentity}'s accepted only {@code 127.0.0.1}. A caller
|
||||
* from {@code 127.0.0.2} therefore had its identity skipped (so it had no terminal) and was
|
||||
* then read as loopback here — which under loopback-trust is the primary. Sharing the inputs
|
||||
* would not have prevented that; only sharing the computation does.
|
||||
*/
|
||||
private static boolean isLoopback(String remoteAddr) {
|
||||
if (remoteAddr == null) {
|
||||
return false;
|
||||
}
|
||||
return remoteAddr.equals("127.0.0.1") || remoteAddr.equals("::1")
|
||||
|| remoteAddr.equals("0:0:0:0:0:0:0:1") || remoteAddr.startsWith("127.");
|
||||
return ConnectionIdentity.isLoopback(remoteAddr);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,17 +5,70 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
/** Optional session lifecycle hook for live member-slot bindings. */
|
||||
public interface MemberLifecycle {
|
||||
|
||||
/** A slot held before a member process starts. */
|
||||
record SlotReservation(String slot, String profile) { }
|
||||
|
||||
MemberLifecycle NONE = new MemberLifecycle() {
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
return role; // no registry configured — nothing to bind against, so the request stands
|
||||
}
|
||||
|
||||
@Override
|
||||
public void released(String terminal) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
// no registry configured — nothing to validate against, so nothing is refused
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
}
|
||||
};
|
||||
|
||||
void acquired(MemberRole role, String profile, String terminal);
|
||||
/**
|
||||
* Try to bind a newly spawned {@code terminal} into the role it was granted.
|
||||
*
|
||||
* @return the role this session actually holds: {@code role} unchanged for a role with no
|
||||
* live slot-binding semantics (dev, reviewer), or when the bind succeeded; a fallback
|
||||
* role — never {@code role} — when a slot-bound role (architect) could not be bound.
|
||||
* Callers must record THIS value on the session, never the requested {@code role}, so
|
||||
* a later roster read never reports a role the session does not hold (CB-619). In
|
||||
* normal operation this fallback should not happen once a reservation has been bound.
|
||||
* It remains the honest answer if a caller has no reservation, or if binding a
|
||||
* reservation unexpectedly fails.
|
||||
*/
|
||||
MemberRole acquired(MemberRole role, String profile, String terminal);
|
||||
|
||||
void released(String terminal);
|
||||
|
||||
/**
|
||||
* Refuse an acquire before anything spawns when {@code role} requires a live slot binding and
|
||||
* no configured slot carries {@code profile} (CB-619 / fleetd #123). A no-op for a role with
|
||||
* no slot-binding semantics.
|
||||
*
|
||||
* @throws IllegalArgumentException naming the role, the profile, and the pools that do carry it
|
||||
*/
|
||||
void requireSlotFor(MemberRole role, String profile);
|
||||
|
||||
/** Reserve a matching slot before launch, or refuse before a charter can be delivered. */
|
||||
SlotReservation reserve(MemberRole role, String profile);
|
||||
|
||||
/** Convert a reservation into a live terminal binding. */
|
||||
boolean bind(SlotReservation reservation, String terminal);
|
||||
|
||||
/** Return an unbound reservation after a failed launch. */
|
||||
void release(SlotReservation reservation);
|
||||
}
|
||||
|
||||
@@ -8,8 +8,10 @@ import org.slf4j.LoggerFactory;
|
||||
import java.util.Collections;
|
||||
import java.util.HashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* The architect-slot registry (CB-548): every gateway-local architect name and the strong-model
|
||||
@@ -18,16 +20,43 @@ import java.util.Objects;
|
||||
*
|
||||
* <p>Two halves, split by who owns each:
|
||||
* <ul>
|
||||
* <li><b>slots</b> — configured once, keyed by the gateway-local unique name; each carries the
|
||||
* {@code profile} reference the spawn lifecycle reads when it stands the slot up. A read-only
|
||||
* snapshot taken at construction.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry and initially <em>empty</em>. Config
|
||||
* declares no architect terminal, so at startup every slot is idle and nothing resolves to an
|
||||
* architect; a session only becomes one when the spawn lifecycle {@linkplain #bind(String,
|
||||
* String) binds} its terminal to a slot. {@link CallerResolver} reads this through
|
||||
* {@link #snapshot()} to turn a pane into an {@link Role#ARCHITECT}.</li>
|
||||
* <li><b>slots</b> — read from {@code fleet.architects}/{@code developers}/{@code reviewers}
|
||||
* (see {@link #slots()}), each carrying the {@code profile} reference the spawn lifecycle
|
||||
* reads when it stands the slot up. <strong>Live, since fleetd #424</strong>: {@link #live}
|
||||
* re-reads {@code fleet:} on every call, through a supplier the same shape as
|
||||
* {@code CompositePeerLauncher}'s (see {@code ConfigRef}'s class doc) — so a config reload
|
||||
* that removes or adds an architect slot governs the <em>next</em> spawn with no restart.
|
||||
* Only {@link #MemberRegistry(FleetConfig.Fleet)} freezes the pool at construction, and that
|
||||
* constructor exists for tests and for the (rare) case of wiring a fixed, code-built config.</li>
|
||||
* <li><b>terminal bindings</b> — owned by this registry, initially <em>empty</em>, and
|
||||
* <strong>never</strong> touched by a reload. Config declares no architect terminal, so at
|
||||
* startup every slot is idle and nothing resolves to an architect; a session only becomes one
|
||||
* when the spawn lifecycle {@linkplain #bind(String, String) binds} its terminal to a slot.
|
||||
* {@link CallerResolver} reads this through {@link #snapshot()} to turn a pane into an
|
||||
* {@link Role#ARCHITECT}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The binding rule (fleetd #424): config governs what a bound slot still grants, as
|
||||
* well as what may be bound next.</strong> Removing a slot from config revokes it — that is the
|
||||
* ticket's entire point ("Revoking an architect slot does not revoke it"). Revoking it means an
|
||||
* architect already bound to that slot loses the ARCHITECT privilege on its very next request:
|
||||
* {@link #roleForSlot} and {@link #nameForSlot} read {@link #slots()} directly, with no cache, so
|
||||
* the moment a slot drops out of config, {@link CallerResolver#resolve} (which calls both on every
|
||||
* request from a bound pane, {@code CallerResolver.java:220}) can no longer confirm the pane's slot
|
||||
* is an architect slot, and the pane falls through to {@code Principal.worker(...)}. What does
|
||||
* <em>not</em> change is the {@code terminalToSlot} <em>occupancy</em> — the binding created by
|
||||
* {@link #bind} is untouched by a reload, on purpose: unbinding it here would double-book the slot
|
||||
* key (a second terminal could then bind to the "freed" key while the first is still the terminal
|
||||
* the operator actually meant to demote) and would silently break {@link #unbind}'s compare-safe
|
||||
* contract, which needs the original {@code terminal → slot} pair intact to remove it cleanly. So
|
||||
* the demoted session keeps occupying its slot — {@link #slotForTerminal} and {@link #snapshot()}
|
||||
* still name it — it just no longer resolves as an architect through that occupancy, and a fresh
|
||||
* spawn still cannot bind to the same key while it is occupied ({@link #reserve}/
|
||||
* {@link #requireSlotFor} refuse it anyway, since it is gone from {@link #slots()}). The demoted
|
||||
* session's own turn is unaffected: {@code fleet_reply}'s authorization
|
||||
* ({@code Authz.Action.REPLY}) is {@code caller.ownsSession(targetSession)} — identity by terminal,
|
||||
* not by role — so a demoted architect can still end its own turn normally.
|
||||
*
|
||||
* <p>Spawning/lifecycle is deliberately a separate unit: this class only owns the bindings and
|
||||
* exposes the map the resolver resolves against plus the profile lookup lifecycle will call.
|
||||
* Nothing here creates or manages an architect session.
|
||||
@@ -54,12 +83,42 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
}
|
||||
|
||||
private final Map<String, Entry> slots;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code this}. */
|
||||
private final Supplier<FleetConfig.Fleet> fleet;
|
||||
/** Live {@code terminal_id → qualified slot key}; guarded by {@code terminalToSlot}. */
|
||||
private final Map<String, String> terminalToSlot = new HashMap<>();
|
||||
/** Slot keys held between reservation and the terminal binding. Guarded by terminalToSlot. */
|
||||
private final java.util.Set<String> reservedSlots = new java.util.HashSet<>();
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one registry. Leaders are not members. */
|
||||
/**
|
||||
* Freeze the pool at construction — for tests, and for the rare case of wiring a fixed,
|
||||
* code-built config. Production wiring should prefer {@link #live}, which re-reads
|
||||
* {@code fleet:} on every call.
|
||||
*/
|
||||
public MemberRegistry(FleetConfig.Fleet fleet) {
|
||||
this(() -> fleet);
|
||||
}
|
||||
|
||||
private MemberRegistry(Supplier<FleetConfig.Fleet> fleet) {
|
||||
this.fleet = fleet;
|
||||
}
|
||||
|
||||
/**
|
||||
* Live variant (fleetd #424): {@code fleet} is read fresh on every {@link #slots()} call — pass
|
||||
* {@code () -> config.get().fleet()}, the same supplier shape {@code CompositePeerLauncher}
|
||||
* already uses for placement — so a reload that adds or removes an architect slot governs the
|
||||
* next spawn's {@link #reserve}/{@link #requireSlotFor} check with no restart. A separate,
|
||||
* private constructor rather than a same-arity public overload of
|
||||
* {@link #MemberRegistry(FleetConfig.Fleet)}: a {@code FleetConfig.Fleet} and a
|
||||
* {@code Supplier<FleetConfig.Fleet>} overload are ambiguous for a literal {@code null} — the
|
||||
* same reason {@code CallerResolver.withLeads} is a static factory rather than a fourth
|
||||
* constructor overload.
|
||||
*/
|
||||
public static MemberRegistry live(Supplier<FleetConfig.Fleet> fleet) {
|
||||
return new MemberRegistry(Objects.requireNonNull(fleet, "fleet"));
|
||||
}
|
||||
|
||||
/** Flatten every role pool in {@code fleet} into one map. Leaders are not members. */
|
||||
private static Map<String, Entry> flatten(FleetConfig.Fleet fleet) {
|
||||
Map<String, Entry> flat = new LinkedHashMap<>();
|
||||
if (fleet != null) {
|
||||
for (MemberRole role : MemberRole.values()) {
|
||||
@@ -71,18 +130,22 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
});
|
||||
}
|
||||
}
|
||||
this.slots = Collections.unmodifiableMap(flat);
|
||||
return Collections.unmodifiableMap(flat);
|
||||
}
|
||||
|
||||
/** The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot. */
|
||||
/**
|
||||
* The configured slots, keyed by qualified {@link Entry#key()}. Unmodifiable snapshot of
|
||||
* {@code fleet:} <em>as of this call</em> — see the class doc for which constructor makes that
|
||||
* live versus frozen.
|
||||
*/
|
||||
public Map<String, Entry> slots() {
|
||||
return slots;
|
||||
return flatten(fleet.get());
|
||||
}
|
||||
|
||||
/** The slots belonging to {@code role}, in definition order. */
|
||||
/** The slots belonging to {@code role}, in definition order, as of this call. */
|
||||
public Map<String, Entry> slotsFor(MemberRole role) {
|
||||
Map<String, Entry> out = new LinkedHashMap<>();
|
||||
slots.forEach((key, e) -> {
|
||||
slots().forEach((key, e) -> {
|
||||
if (e.role() == role) {
|
||||
out.put(key, e);
|
||||
}
|
||||
@@ -114,31 +177,49 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
}
|
||||
|
||||
/**
|
||||
* The strong-model profile a slot runs under — what the spawn lifecycle reads.
|
||||
* The strong-model profile a slot runs under, as of this call.
|
||||
*
|
||||
* <p>Nothing in {@code src/main} calls this (fleetd #431 — grepped both the {@code
|
||||
* .profileForSlot(} and the {@code ::profileForSlot} form). This javadoc used to say "what the
|
||||
* spawn lifecycle reads", and that seam does not exist: the spawn lifecycle takes its profile
|
||||
* from the {@link MemberLifecycle.SlotReservation} that {@code reserve} returns, never from
|
||||
* here. Kept and pinned rather than deleted because it is the natural accessor for that seam
|
||||
* if one is added; live for the same reason as {@link #roleForSlot}, so a reload cannot leave
|
||||
* it answering for the old config.
|
||||
*
|
||||
* @return the slot's configured {@code profile}, or {@code null} if the slot is unknown or
|
||||
* declares none
|
||||
*/
|
||||
public String profileForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return (e == null || e.profile() == null) ? null : e.profile();
|
||||
}
|
||||
|
||||
/** The role a qualified slot key belongs to, or {@code null} when the key is unknown. */
|
||||
/**
|
||||
* The role a qualified slot key belongs to, or {@code null} when the key is not currently
|
||||
* configured. Deliberately live, with no cache (fleetd #424, see the class doc's binding rule):
|
||||
* removing a slot from config must make {@link CallerResolver#resolve} stop granting the
|
||||
* ARCHITECT role for it on the very next request from a terminal that was bound to it, which is
|
||||
* the ticket's whole point — revoking a slot must actually revoke it, not just refuse the next
|
||||
* spawn.
|
||||
*/
|
||||
public MemberRole roleForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.role();
|
||||
}
|
||||
|
||||
/** The unqualified configured name for a slot, or {@code null} if it is unknown. */
|
||||
/**
|
||||
* The unqualified configured name for a slot, or {@code null} if it is not currently configured.
|
||||
* Live for the same reason as {@link #roleForSlot} — see the class doc's binding rule.
|
||||
*/
|
||||
public String nameForSlot(String slotName) {
|
||||
Entry e = slots.get(slotName);
|
||||
Entry e = slots().get(slotName);
|
||||
return e == null ? null : e.name();
|
||||
}
|
||||
|
||||
/** True when {@code slotName} is a configured architect slot. */
|
||||
public boolean isSlot(String slotName) {
|
||||
return slots.containsKey(slotName);
|
||||
return slots().containsKey(slotName);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -166,7 +247,7 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
if (existingSlot != null) {
|
||||
return slot.equals(existingSlot); // already this slot (idempotent) or a different one
|
||||
}
|
||||
if (terminalToSlot.containsValue(slot)) {
|
||||
if (terminalToSlot.containsValue(slot) || reservedSlots.contains(slot)) {
|
||||
return false; // slot already hosts a terminal — no second one
|
||||
}
|
||||
terminalToSlot.put(terminal, slot);
|
||||
@@ -206,19 +287,116 @@ public final class MemberRegistry implements MemberLifecycle {
|
||||
*
|
||||
* <p>The role check is lifecycle policy. {@link CallerResolver} repeats it when resolving a
|
||||
* binding, so a later lifecycle regression cannot turn a worker into an architect.
|
||||
*
|
||||
* <p>CB-619 / fleetd #123: the return value is the role this session actually holds, and the
|
||||
* caller is required to record THAT — never the requested {@code role} — on the session. Before
|
||||
* this fix the caller kept the requested role regardless of whether the bind below succeeded, so
|
||||
* a demoted session's {@code GET /members} row still said {@code "architect"} while
|
||||
* {@code fleet_whoami} (which reads the live binding, not the request) correctly said
|
||||
* {@code "worker"} — three sources of truth that disagreed about one live member, silently.
|
||||
*/
|
||||
@Override
|
||||
public void acquired(MemberRole role, String profile, String terminal) {
|
||||
public MemberRole acquired(MemberRole role, String profile, String terminal) {
|
||||
if (role != MemberRole.ARCHITECT || terminal == null || terminal.isBlank()) {
|
||||
return;
|
||||
return role;
|
||||
}
|
||||
// slotsFor preserves definition order, so duplicate-profile slots use the first free one.
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if (Objects.equals(profile, entry.profile()) && bind(entry.key(), terminal)) {
|
||||
return;
|
||||
return MemberRole.ARCHITECT;
|
||||
}
|
||||
}
|
||||
// fleetd #123: at least WARN — a role downgrade that the roster must now also reflect is
|
||||
// not routine bookkeeping. requireSlotFor already refuses the config-gap case (no slot at
|
||||
// all carries this profile) before a process ever spawns; reaching here means the config DID
|
||||
// carry a matching slot but every one of them was already bound to a different terminal — a
|
||||
// race this pre-spawn check cannot close on its own (see requireSlotFor's javadoc).
|
||||
log.warn("member slot: no free architect slot for profile={} terminal={}; holding the session "
|
||||
+ "as {} instead of the architect it asked for — every configured slot for this "
|
||||
+ "profile is already bound to a different terminal", profile, terminal,
|
||||
MemberRole.DEV.wireName());
|
||||
return MemberRole.DEV;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-619 / fleetd #123: refuse an architect acquire before anything spawns when no configured
|
||||
* slot carries {@code profile} — the config-gap case from the original defect report (a spawn
|
||||
* asked for {@code role=architect, profile=sonnet}, and {@code fleet.architects} carried only
|
||||
* {@code opus} and {@code sol}). A dev/reviewer acquire is always a no-op: those pools are
|
||||
* placement candidates only (see {@code CompositePeerLauncher}), never a live identity binding,
|
||||
* so there is nothing here to refuse — an explicit profile outside the pool for those roles is a
|
||||
* documented operator override, not a defect.
|
||||
*
|
||||
* <p>This closes the config-gap case, not the live-capacity case: a profile that DOES carry a
|
||||
* slot can still lose the race to a concurrent spawn between this check and the actual
|
||||
* {@link #bind}, which is why {@link #acquired} must still answer honestly even after this
|
||||
* check has passed.
|
||||
*/
|
||||
@Override
|
||||
public void requireSlotFor(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return;
|
||||
}
|
||||
boolean hasSlot = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.anyMatch(e -> Objects.equals(profile, e.profile()));
|
||||
if (hasSlot) {
|
||||
return;
|
||||
}
|
||||
List<String> pools = slotsFor(MemberRole.ARCHITECT).values().stream()
|
||||
.map(Entry::profile)
|
||||
.distinct()
|
||||
.toList();
|
||||
throw new IllegalArgumentException(
|
||||
"no " + role.wireName() + " slot for profile '" + profile + "' — an architect's "
|
||||
+ "identity IS the slot it is bound to, so there is nothing to bind this "
|
||||
+ "session's identity to. fleet." + role.configKey() + " carries profiles: "
|
||||
+ (pools.isEmpty() ? "(none configured)" : String.join(", ", pools))
|
||||
+ "; add profile '" + profile + "' there, or spawn " + role.wireName()
|
||||
+ " on one of those profiles instead");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SlotReservation reserve(MemberRole role, String profile) {
|
||||
if (role != MemberRole.ARCHITECT) {
|
||||
return null;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
for (Entry entry : slotsFor(MemberRole.ARCHITECT).values()) {
|
||||
if ((profile == null || profile.isBlank() || Objects.equals(profile, entry.profile()))
|
||||
&& !terminalToSlot.containsValue(entry.key()) && reservedSlots.add(entry.key())) {
|
||||
return new SlotReservation(entry.key(), entry.profile());
|
||||
}
|
||||
}
|
||||
}
|
||||
throw new IllegalArgumentException("no free architect slot for profile '" + profile
|
||||
+ "' — every matching slot is already bound or reserved");
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean bind(SlotReservation reservation, String terminal) {
|
||||
if (reservation == null || terminal == null || terminal.isBlank()) {
|
||||
return false;
|
||||
}
|
||||
synchronized (terminalToSlot) {
|
||||
if (!reservedSlots.remove(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
if (!isSlot(reservation.slot()) || terminalToSlot.containsKey(terminal)
|
||||
|| terminalToSlot.containsValue(reservation.slot())) {
|
||||
return false;
|
||||
}
|
||||
terminalToSlot.put(terminal, reservation.slot());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public void release(SlotReservation reservation) {
|
||||
if (reservation != null) {
|
||||
synchronized (terminalToSlot) {
|
||||
reservedSlots.remove(reservation.slot());
|
||||
}
|
||||
}
|
||||
log.info("member slot: no free architect slot for profile={}; session remains a worker", profile);
|
||||
}
|
||||
|
||||
/** Unbind a released terminal using the compare-safe registry operation. */
|
||||
|
||||
@@ -11,6 +11,7 @@ import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
@@ -22,63 +23,296 @@ import java.util.function.Supplier;
|
||||
* choice rather than an accident of where the field was initialised.
|
||||
*
|
||||
* <h2>Not every key can change under a running daemon</h2>
|
||||
* Keys fall into three classes, and the difference is about what already exists when the reload
|
||||
* Keys fall into four classes, and the difference is about what already exists when the reload
|
||||
* happens — not about how important the key is.
|
||||
*
|
||||
* <ul>
|
||||
* <li><strong>Hot</strong> — re-read per use, so a reload takes effect on the next spawn:
|
||||
* {@code fleet:} (every role pool, {@code charters}, and {@code tabLabel}),
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Those
|
||||
* three are read through a supplier on {@code CompositePeerLauncher}, which is what makes
|
||||
* them hot — not the fact that they are config. <strong>This does NOT include
|
||||
* {@code fleet.leaders}</strong>: {@code Fleetd.main} reads {@code cfg.fleet().leaders()}
|
||||
* once at startup to build the {@code LeadTabScanner} and the {@code LeadLauncher}, and
|
||||
* neither is reconstructed on reload — so a lead added, removed, or re-{@code tab}'d under
|
||||
* {@code fleet.leaders} needs a restart, the same as any deferred key below.</li>
|
||||
* {@code placement:}, and an existing profile's {@code weight} / {@code maxLoad}. Both are
|
||||
* read through a supplier on {@code CompositePeerLauncher}, which is what makes them hot —
|
||||
* not the fact that they are config. Most of {@code fleet:} — every role pool
|
||||
* ({@code architects}/{@code developers}/{@code reviewers}), {@code charters}, and
|
||||
* {@code tabLabel} — is read the same live way, through the same supplier
|
||||
* ({@code () -> config.get().fleet()}). {@code architects} in particular is hot for
|
||||
* <strong>two independent consumers</strong> (fleetd #424): {@code CompositePeerLauncher}
|
||||
* reads it live for placement (which profile an unqualified architect spawn may land on), and
|
||||
* {@code MemberRegistry} separately reads it live, through its own instance of the same
|
||||
* supplier shape, for identity — both which slot a spawn may bind to <em>and</em> what a slot
|
||||
* already bound still grants. Removing an architect slot from config therefore revokes the
|
||||
* {@link dev.ltms.fleet.auth.Role#ARCHITECT} role on the bound pane's very next request; only
|
||||
* the slot <em>occupancy</em> survives, so the demoted session still holds its slot key until
|
||||
* it unbinds. See {@code MemberRegistry}'s class doc for that binding rule.
|
||||
* <strong>But {@code fleet:} as a whole is NOT in this class</strong>: {@code fleet.leaders}
|
||||
* inside the same key is frozen, which is exactly what makes {@code fleet:} split rather than
|
||||
* hot — see below. {@code models:} (fleetd #422) joined this class whole: {@link
|
||||
* FleetConfig#validateModels()} re-runs fully against the fresh config on every {@link
|
||||
* #reload()} (via {@link FleetConfig#validateAll()}), refusing a bad edit outright rather than
|
||||
* caching a stale copy anywhere, and the on/off half added by fleetd #422 is read live both by
|
||||
* {@code CompositePeerLauncher}'s spawn gate ({@code enforceModelEnabled} and its candidate
|
||||
* filter) and by {@code fleet_profiles}/{@code GET /profiles} (via
|
||||
* {@code PeerLauncher.disabledModels()}). Nothing about {@code models:} is baked into an
|
||||
* object built at startup, so — unlike the deferred keys below — there is no frozen half left
|
||||
* to report; it moved here from deferred rather than joining split. An existing profile's
|
||||
* {@code exhaustedPattern} (fleetd #446) joined this class the same way: it used to be
|
||||
* compiled once into {@code Fleetd.main}'s startup pattern map (see the Deferred bullet's old
|
||||
* wording, and {@code LiveExhaustedPatterns}'s class doc for the history), and is now read
|
||||
* live, cached by profile name, by both {@code CompletionResolver}'s classification (via
|
||||
* {@code LiveExhaustedPatterns.patternFor}) and {@code fleet_profiles}'s {@code
|
||||
* exhaustionDetectionArmed} (via {@code LiveExhaustedPatterns.armed}) — the one live object
|
||||
* both read, so a reload that arms or disarms a profile's usage-limit detection takes effect
|
||||
* on the next check with no restart. {@code errorPattern}, {@code exhaustedPattern}'s sibling
|
||||
* key for backend-error (not usage-limit) classification, was deliberately left OUT of this
|
||||
* fleetd #446 change and stays deferred below — the ticket scoped it out explicitly.
|
||||
* {@code leadRollover:} (fleetd #480) joined this class whole, the same shape as
|
||||
* {@code models:} above: {@code dev.ltms.fleet.lead.LeadRollover} holds a
|
||||
* {@code Supplier<FleetConfig.LeadRollover>} (the same {@code () -> config.get().x()} shape)
|
||||
* and reads {@code handoverPath}/{@code requireOperatorConfirm}/{@code maxDocAgeSeconds}/
|
||||
* {@code turnSettleSeconds}/{@code clearSettleSeconds}/{@code bootstrapText} fresh on every
|
||||
* {@code open()}/{@code confirm()} call (and on the deferred post-{@code confirm()}
|
||||
* continuation fleetd #480's correction added — see {@code LeadRollover}'s class doc) rather
|
||||
* than capturing them into fields at construction — unlike its closest
|
||||
* structural cousin {@code leadHeartbeat:}, whose {@code LeadHeartbeatLoop} bakes
|
||||
* {@code idleAfterNanos}/{@code backoffMs}/{@code quietNudgeCap} into final fields. The one
|
||||
* restart-only edge is structural, not a stale value: {@code Fleetd.java} decides whether to
|
||||
* construct the {@code LeadRollover} object at all off the startup snapshot (the same
|
||||
* presence gate {@code leadHeartbeat:} uses), so a block ADDED where it was absent at boot
|
||||
* needs a restart before anything exists to call — the same fact already true of adding a
|
||||
* brand-new {@code profiles:} entry.</li>
|
||||
* <li><strong>Deferred</strong> — accepted into the new snapshot, but the wiring built at startup
|
||||
* keeps the old value until a restart: {@code lifecycle:}, {@code leadHeartbeat:},
|
||||
* {@code idleSleepGuard:} ({@code Fleetd.java} reads it once, at startup, to decide whether
|
||||
* to construct an {@code IdleSleepGuard} and wire {@code SessionManager}'s
|
||||
* {@code onAcquire}/{@code onRelease} hooks to it — neither is rebuilt on reload, so a
|
||||
* running daemon keeps whatever this was at startup regardless of a later edit),
|
||||
* {@code spawnReadyTimeoutMs} / {@code spawnReadyPollMs}, {@code quarantineCooldownSeconds}
|
||||
* (CB-578 stage B — baked once into the {@code BackendQuarantine} built at startup),
|
||||
* {@code guard:}, {@code worktreeRoot:}, adding or removing a profile (a new backend needs its own launcher,
|
||||
* {@code guard:}, {@code worktreeRoot:}, {@code worktreeGroup:} and {@code memberSkills:}
|
||||
* (all three of the latter baked once into the {@code GitWorktrees} built at
|
||||
* {@code Fleetd.java:251} and never rebuilt — fleetd #323 instance 2 found
|
||||
* {@code worktreeGroup} missing from this list and from {@link #changedDeferredKeys};
|
||||
* {@code memberSkills} (fleetd #362) followed the same shape), {@code primary:} (fleetd #326 — {@code Fleetd.java:506, 519,
|
||||
* 520} read {@code cfg.primary()} only off the startup snapshot to build {@code
|
||||
* PrimaryRegistry} and size {@code ReplyPushLoop}'s reminder cap/backoff, and neither is
|
||||
* rebuilt on reload. Say the consequence exactly: {@code primary.terminal} is DEPRECATED
|
||||
* (CB-532, and {@code Fleetd.java:511} warns about it at startup) — a lead's identity comes
|
||||
* from {@code leaders:}/{@code leadScan:}, so changing this pin does not demote or promote a
|
||||
* lead that uses those. What a changed pin still does not take effect on until a restart is
|
||||
* the fallback nudge destination the pin remains, the deprecated identity path for an operator
|
||||
* who still relies on it, and {@code pushReminders}/{@code pushBackoffMs}), {@code configReload:} (fleetd #326 — {@code
|
||||
* Fleetd.java:679-680} read it only at startup to decide whether to build a {@code
|
||||
* ConfigWatcher} at all and with what interval; the watcher that would apply a later change is
|
||||
* itself built once, so a running watcher keeps polling on its original enabled flag and
|
||||
* interval regardless of what a reload changes it to, the same shape as {@code lifecycle} —
|
||||
* not cold, because no already-open resource goes inconsistent with the new value, the watcher
|
||||
* (if any) simply keeps its old settings), adding or removing a profile (a new backend needs its own launcher,
|
||||
* which is constructed once), <em>and an existing profile's launch settings</em> —
|
||||
* {@code model}, {@code baseUrl}, {@code argv}, {@code env}, {@code mcpUrl},
|
||||
* {@code exhaustedPattern} (CB-578 stage A — compiled once into {@code Fleetd.main}'s
|
||||
* pattern map at startup), and the rest. {@code credentialId} (CB-578 stage B) is NOT on
|
||||
* this list — it is read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so it is hot instead.
|
||||
* {@code errorPattern} (fleetd #201 Unit 5 — compiled once into
|
||||
* {@code Fleetd.main}'s backend-error pattern map at startup; deliberately NOT made hot
|
||||
* alongside {@code exhaustedPattern} by fleetd #446 — that ticket scoped {@code errorPattern}
|
||||
* and cooling-off out explicitly),
|
||||
* {@code ideProjectDir} / {@code ideOpenCommand} / {@code autoCompactWindow} (fleetd #323
|
||||
* instance 1 — all three are read at spawn off the same frozen profile map and were missing
|
||||
* from {@link #sameLaunchSettings}), and the rest of {@link #sameLaunchSettings}.
|
||||
* {@code credentialId} (CB-578 stage B) and {@code exhaustedPattern} (fleetd #446) are NOT on
|
||||
* this list — both are read live off the config supplier at every quarantine check and
|
||||
* exhaustion event, exactly like {@code weight} / {@code maxLoad}, so both are hot instead
|
||||
* (see the Hot bullet above for {@code exhaustedPattern}'s history).
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} at construction and resolves
|
||||
* each spawn out of that copy, so those never reach a launch until the daemon restarts. A
|
||||
* reload logs these rather than pretending they applied.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon: {@code bind:},
|
||||
* {@code herdrSocket:}, {@code broker:} and {@code auth:}. The socket is bound, the broker
|
||||
* <li><strong>Split</strong> (fleetd #330; extended to a third key by fleetd #333) — read
|
||||
* <em>both</em> ways at different sites, so the key does not fit any class above as a whole:
|
||||
* {@code health:}, {@code coordinator:} and {@code fleet:}. Each is read off the startup
|
||||
* snapshot to build a long-lived object, and read live off {@link #get()} at a different,
|
||||
* unrelated site — so half of a reload's effect already applies while the other half waits
|
||||
* for a restart, and a bare "config reloaded" would under-claim by exactly that half.
|
||||
* <ul>
|
||||
* <li>{@code health:} — the monitor itself ({@code enabled}, {@code intervalSeconds},
|
||||
* {@code workingSuspectAfterSeconds}) is built once at {@code Fleetd.java:556-563}
|
||||
* and never rebuilt, so a changed value needs a restart to actually start, stop, or
|
||||
* retime it. The coverage string {@code fleet_profiles} reports
|
||||
* ({@code Fleetd.java:648-650}) is read live off {@link #get()} on every call, so it
|
||||
* already reflects the new value.</li>
|
||||
* <li>{@code coordinator:} — the {@code LeadMailbox} connection ({@code uri},
|
||||
* {@code uriEnv}, {@code selfId}, {@code prefetch}) is opened once at
|
||||
* {@code Fleetd.java:502} and never reopened, so a changed value needs a restart —
|
||||
* {@code selfId} in particular names this daemon's own AMQP inbox queue, and a peer
|
||||
* lead that learned the old name would not discover a new one on its own. The broker
|
||||
* URI env-var <em>name</em> that {@code MemberEnvAllowList} keeps out of a member's
|
||||
* environment is read live off {@link #get()} on every spawn
|
||||
* ({@code HerdrPeerLauncher.java:1530}), so it already applies.</li>
|
||||
* <li>{@code fleet:} (fleetd #333) — {@code fleet.leaders} is the frozen half:
|
||||
* {@code Fleetd.java:281} reads {@code cfg.fleet().leaders()} off the startup snapshot
|
||||
* for two long-lived objects built right after it and never rebuilt — the
|
||||
* {@code LeadTabScanner}'s {@code tab label → lead name} map ({@code Fleetd.java:301},
|
||||
* wired into {@code CallerResolver.withLeadsAndMembers} at {@code Fleetd.java:620/624},
|
||||
* which is how a caller's pane is recognised as a lead at all) and, when herdr answered
|
||||
* at startup, {@code LeadLauncher(...).ensureLeads()} ({@code Fleetd.java:315}), which
|
||||
* auto-launches each declared lead up to its {@code instances} count. So a lead added,
|
||||
* removed, or given a new {@code tab:} label under {@code fleet.leaders} needs a
|
||||
* restart — until then it is invisible to identity resolution, and this is exactly the
|
||||
* scenario fleetd #333 named: an operator edits a lead's {@code tab:} to match a
|
||||
* renamed pane, sees "config reloaded", and the pane keeps resolving as a worker,
|
||||
* because {@code CallerResolver} is still matching against the old label. The live
|
||||
* half is the rest of {@code fleet:} — {@code architects}/{@code developers}/
|
||||
* {@code reviewers}, {@code charters}, {@code tabLabel} — read live through the same
|
||||
* {@code CompositePeerLauncher} supplier the Hot bullet above names, so a reload that
|
||||
* only touches those already applies with nothing to report. Because the hot and frozen
|
||||
* halves of {@code fleet:} are disjoint sub-fields rather than the same fields read two
|
||||
* ways (contrast {@code coordinator.uriEnv} above), {@link #changedSplitKeys} compares
|
||||
* {@code fleet.leaders} alone, not the whole {@code Fleet} record — comparing the whole
|
||||
* record would report "split" for a {@code tabLabel}-only change that is actually fully
|
||||
* hot, over-claiming in exactly the direction this class exists to avoid under-claiming
|
||||
* in.</li>
|
||||
* </ul>
|
||||
* A split change is still accepted — {@link Outcome#applied()} stays {@code true}, the same
|
||||
* as a deferred change — because the live half genuinely took effect; refusing the whole
|
||||
* reload would leave the operator worse off than today. {@link Outcome#split()} names the
|
||||
* key and says which half is which each time, rather than trying to score "how changed" a
|
||||
* mixed key is or handle "both halves changed in one reload" as a special case.</li>
|
||||
* <li><strong>Cold</strong> — cannot change at all under a running daemon. All five of
|
||||
* {@link #COLD_KEYS}: {@code bind:}, {@code herdrSocket:}, {@code memberHerdrSocket:},
|
||||
* {@code broker:} and {@code auth:}. The sockets are already connected, the broker
|
||||
* connection is open, and the auth mode decides who may reach the port that is already
|
||||
* listening.</li>
|
||||
* listening. This bullet omitted {@code memberHerdrSocket:} until fleetd #333 — say "all
|
||||
* five of COLD_KEYS" rather than re-listing them, so prose and set cannot drift again.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>The denominator, measured on 2026-09-04 (fleetd #330; recounted for fleetd #333);
|
||||
* recounted again for fleetd #362, again after {@code idleSleepGuard:} was added, again after
|
||||
* {@code models:} was added as deferred, again for fleetd #422, which moved {@code models:}
|
||||
* from deferred to hot-excluded once its on/off half was read live everywhere, and again after
|
||||
* {@code leadRollover:} was added (fleetd #480).</strong>
|
||||
* {@code FleetConfig} has 26 top-level record components: 5 cold, 13 deferred, 3 split, 5
|
||||
* hot-excluded. Five of them are named nowhere in this file, and the reason is the same for all
|
||||
* five: {@code placement}, {@code memberCredentials}, {@code memberLoginShell}, {@code models} and
|
||||
* {@code leadRollover} are <strong>hot</strong> and correctly absent — all five are read live off
|
||||
* {@code config.get()} (placement through the {@code CompositePeerLauncher} supplier the Hot bullet
|
||||
* names; {@code memberCredentials}/{@code memberLoginShell} at spawn time, {@code Fleetd.java:198,
|
||||
* 205, 729} and {@code HerdrPeerLauncher#configuredMemberLoginShell}; {@code models} the same way,
|
||||
* through the Hot bullet's {@code models:} paragraph; {@code leadRollover} through the Hot bullet's
|
||||
* {@code leadRollover:} paragraph), so a reload takes effect on the next spawn (or, for
|
||||
* {@code models}, the next reported status; for {@code leadRollover}, the next {@code open()}/
|
||||
* {@code confirm()} call) with no entry needed here.
|
||||
* {@code health} and {@code coordinator} used to be a third kind — <strong>undecided</strong>, not
|
||||
* hot — until fleetd #330 added the <strong>split</strong> class above and gave them a home. A
|
||||
* reload touching either used to report a bare "config reloaded", which under-claimed; now it names
|
||||
* the key and says which half is which. {@code fleet} was the same story in reverse: fleetd #330's
|
||||
* own fact-find named {@code health}/{@code coordinator} as "the complete set of split-shaped keys"
|
||||
* and filed {@code fleet.leaders}'s restart requirement as a documented caveat sitting in the
|
||||
* <strong>hot-excluded</strong> escape hatch instead — correctly documented, but in the one bucket
|
||||
* this file's own coverage test cannot check the truth of (see that test's javadoc). fleetd #333
|
||||
* moved it into <strong>split</strong>, where {@link #changedSplitKeys} actually reports it.
|
||||
* <p>The point of writing the count down: "not mentioned in this file" looks identical for a key
|
||||
* that is correctly hot and for a key nobody triaged. Three times now — {@code worktreeGroup} (#323),
|
||||
* {@code primary}/{@code configReload} (#326), and {@code fleet.leaders} sitting in the escape hatch
|
||||
* (#333) — the second kind hid among the first. A top-level coverage checker in the
|
||||
* {@link ConfigRefProfileCoverageTest} shape (one level up, over {@code FleetConfig} itself rather
|
||||
* than {@code FleetConfig.Profile}) proves this file's four classes exhaust the record's components
|
||||
* — see {@code ConfigRefTopLevelCoverageTest}. That test proves the record's <em>shape</em> is fully
|
||||
* triaged; it does NOT prove a {@code SPLIT_KEYS}/{@code COLD_KEYS}/{@code DEFERRED_KEYS} member has
|
||||
* any reporting code behind it at all — {@code ConfigRefTopLevelReportingCoverageTest} is what
|
||||
* fleetd #333 added for that, after measuring that a {@code SPLIT_KEYS} entry with its reporting
|
||||
* branch deleted passes both this file's own "kept in step" assert and
|
||||
* {@code ConfigRefTopLevelCoverageTest} unchanged. fleetd #337 extended it to {@code DEFERRED_KEYS}
|
||||
* after measuring the same one-way gap there directly: dropping {@code guard}'s branch out of
|
||||
* {@link #changedDeferredKeys} while {@code "guard"} stayed in the set left the whole suite green.
|
||||
*
|
||||
* <p><strong>A cold change refuses the whole reload.</strong> Not the hot half applied and the cold
|
||||
* half warned about: that would leave the running daemon in a state matching no file on disk, which
|
||||
* is the worst thing a reload can do to an operator debugging one. Refusing keeps the invariant that
|
||||
* the live config is always some version of the file, and the message names the keys that must
|
||||
* change through a restart.
|
||||
* change through a restart. A split change does <em>not</em> refuse, for a different reason than a
|
||||
* deferred change does not: its live half genuinely took effect, so refusing would throw that away
|
||||
* and leave the operator worse off than the partial-but-honest report {@link Outcome#split()} gives.
|
||||
*
|
||||
* <p>A reload that fails to parse or fails validation is also refused, and the previous config keeps
|
||||
* running. A config file being edited is normally read once mid-save; degrading a working daemon
|
||||
* because it caught a half-written file would be a bad trade.
|
||||
*
|
||||
* <p><strong>fleetd #474</strong> — {@link FleetConfig#validateAll()} is not the only gate startup
|
||||
* runs before a config takes effect: {@code Fleetd.main} also calls {@code
|
||||
* dev.ltms.fleet.mcp.CharterToolSurface#assertChartersNameOnlyRegisteredTools}, right after {@code
|
||||
* cfg.validateAll()}, to refuse a charter that names an MCP tool the server does not register. That
|
||||
* check cannot live inside {@link FleetConfig} — {@code CharterToolSurface} lives in the {@code mcp}
|
||||
* package because the canonical tool set ({@code FleetTool}) does, and config is loaded before the
|
||||
* MCP server exists, so {@code FleetConfig} must not gain a dependency on {@code mcp}. {@link
|
||||
* #reload} cannot import {@code mcp} either, for the same reason applied one layer up: {@code
|
||||
* dev.ltms.fleet.config} is loaded before {@code dev.ltms.fleet.mcp} exists, same as {@code
|
||||
* FleetConfig}. So this class accepts the check as a {@code Consumer<FleetConfig>} —
|
||||
* {@link #extraValidation} — supplied by whichever caller already sits at the seam that holds both
|
||||
* a loaded {@code FleetConfig} and the {@code mcp} package: {@code Fleetd.main}. It is invoked
|
||||
* inside the same try/catch as {@code fresh.validateAll()}, so a charter that would have refused to
|
||||
* boot refuses a reload too, and keeps the running config exactly like any other {@code
|
||||
* validateAll()} failure. A ref built through the two-argument constructor (every test fixture that
|
||||
* does not care about this check, and {@link #fixed}) gets a no-op consumer, so nothing outside
|
||||
* {@code Fleetd.main} needs to know this hook exists.
|
||||
*/
|
||||
public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ConfigRef.class);
|
||||
|
||||
/** Keys that cannot change under a running daemon — see the class doc. */
|
||||
private static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "broker", "auth");
|
||||
/**
|
||||
* Keys that cannot change under a running daemon — see the class doc.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} can fold it
|
||||
* into the top-level triage it checks, the same way it reads {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> COLD_KEYS =
|
||||
Set.of("bind", "herdrSocket", "memberHerdrSocket", "broker", "auth");
|
||||
|
||||
/**
|
||||
* Keys read BOTH off the startup snapshot and live off {@link #get()} at different sites, so
|
||||
* neither the hot, deferred nor cold class fits them as a whole — see the class doc's Split
|
||||
* bullet (fleetd #330). A changed split key is accepted ({@link Outcome#applied()} stays
|
||||
* {@code true}) and reported by name, with a message naming which half is live and which needs
|
||||
* a restart.
|
||||
*/
|
||||
static final Set<String> SPLIT_KEYS = Set.of("health", "coordinator", "fleet");
|
||||
|
||||
/**
|
||||
* Top-level keys {@link #changedDeferredKeys} compares — see the class doc's Deferred bullet.
|
||||
* Promoted here from a test-side copy in {@code ConfigRefTopLevelCoverageTest} by fleetd #337,
|
||||
* the same reason {@link #COLD_KEYS} and {@link #SPLIT_KEYS} live here rather than in a test: a
|
||||
* second, hand-maintained copy of this set is exactly the kind of thing that silently drifts
|
||||
* from the method it is supposed to describe. {@code spawnReadyTimeoutMs} and
|
||||
* {@code spawnReadyPollMs} are compared together in one branch and reported under the combined
|
||||
* label {@code "spawnReady*"}; {@code profiles} is compared twice over (added/removed names,
|
||||
* then an existing profile's launch settings) — see {@link #changedDeferredKeys}.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelCoverageTest} and
|
||||
* {@code ConfigRefTopLevelReportingCoverageTest} can both read it, the same way they already
|
||||
* read {@link #COLD_KEYS} and {@link #SPLIT_KEYS}.
|
||||
*/
|
||||
static final Set<String> DEFERRED_KEYS = Set.of(
|
||||
"guard", "worktreeRoot", "worktreeGroup", "memberSkills", "primary", "configReload",
|
||||
"leadHeartbeat", "lifecycle", "spawnReadyTimeoutMs", "spawnReadyPollMs",
|
||||
"quarantineCooldownSeconds", "profiles", "idleSleepGuard");
|
||||
|
||||
private final Path path;
|
||||
private final AtomicReference<FleetConfig> current;
|
||||
private final Consumer<FleetConfig> extraValidation;
|
||||
|
||||
/** Equivalent to the three-argument constructor with a no-op {@code extraValidation}. */
|
||||
public ConfigRef(Path path, FleetConfig initial) {
|
||||
this(path, initial, cfg -> { });
|
||||
}
|
||||
|
||||
/**
|
||||
* @param extraValidation run on every {@link #reload} candidate, inside the same try/catch as
|
||||
* {@code fresh.validateAll()} — see the class doc's fleetd #474 note.
|
||||
* {@code Fleetd.main} passes {@code
|
||||
* Fleetd::assertChartersNameOnlyRegisteredTools} (a package-private
|
||||
* {@code FleetConfig -> void} adapter over {@code
|
||||
* CharterToolSurface#assertChartersNameOnlyRegisteredTools}), so a reload
|
||||
* runs the same gate startup does without this class depending on the
|
||||
* {@code mcp} package.
|
||||
*/
|
||||
public ConfigRef(Path path, FleetConfig initial, Consumer<FleetConfig> extraValidation) {
|
||||
this.path = path;
|
||||
this.current = new AtomicReference<>(Objects.requireNonNull(initial, "initial config"));
|
||||
this.extraValidation = Objects.requireNonNull(extraValidation, "extraValidation");
|
||||
}
|
||||
|
||||
/** A fixed reference that never reloads — for tests and for wiring built from a config in code. */
|
||||
@@ -100,25 +334,37 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
/**
|
||||
* What a reload attempt did.
|
||||
*
|
||||
* <p>{@code split} is a separate field from {@code deferred} rather than a differently-worded
|
||||
* entry inside it, because the two carry different guarantees for any caller that branches on
|
||||
* them rather than just printing {@link #summary()}: every {@code deferred} entry means "this
|
||||
* key's whole change waits for a restart", while every {@code split} entry means "part of this
|
||||
* key's change already applied, and the message says which part" — collapsing them would force
|
||||
* a caller to re-parse the message to tell those apart. See the class doc's Split bullet
|
||||
* (fleetd #330) for why the key needs this at all.
|
||||
*
|
||||
* @param applied true when the new config is now live
|
||||
* @param coldKeys cold keys whose value changed, which is why an unapplied reload was refused
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits for a restart
|
||||
* @param deferred keys that changed and were accepted, but whose effect waits entirely on a
|
||||
* restart
|
||||
* @param split split keys that changed and were accepted, each named with which half of it
|
||||
* is already live and which half waits for a restart
|
||||
* @param error the parse or validation failure that refused the reload, else {@code null}
|
||||
*/
|
||||
public record Outcome(boolean applied, List<String> coldKeys, List<String> deferred,
|
||||
String error) {
|
||||
List<String> split, String error) {
|
||||
|
||||
public Outcome {
|
||||
coldKeys = List.copyOf(coldKeys);
|
||||
deferred = List.copyOf(deferred);
|
||||
split = List.copyOf(split);
|
||||
}
|
||||
|
||||
static Outcome refusedCold(List<String> keys) {
|
||||
return new Outcome(false, keys, List.of(), null);
|
||||
return new Outcome(false, keys, List.of(), List.of(), null);
|
||||
}
|
||||
|
||||
static Outcome failed(String error) {
|
||||
return new Outcome(false, List.of(), List.of(), error);
|
||||
return new Outcome(false, List.of(), List.of(), List.of(), error);
|
||||
}
|
||||
|
||||
/** A one-line summary for the operator — the reason, not just the verdict. */
|
||||
@@ -130,11 +376,18 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
return "config reload refused — these keys cannot change under a running daemon: "
|
||||
+ String.join(", ", coldKeys) + ". Restart fleetd to apply them.";
|
||||
}
|
||||
if (!deferred.isEmpty()) {
|
||||
return "config reloaded; these changes need a restart to take effect: "
|
||||
+ String.join(", ", deferred);
|
||||
if (deferred.isEmpty() && split.isEmpty()) {
|
||||
return "config reloaded";
|
||||
}
|
||||
return "config reloaded";
|
||||
StringBuilder out = new StringBuilder("config reloaded");
|
||||
if (!deferred.isEmpty()) {
|
||||
out.append("; these changes need a restart to take effect: ")
|
||||
.append(String.join(", ", deferred));
|
||||
}
|
||||
if (!split.isEmpty()) {
|
||||
out.append("; partially live — ").append(String.join(" | ", split));
|
||||
}
|
||||
return out.toString();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -155,12 +408,18 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
fresh = FleetConfig.load(path);
|
||||
// The same gate startup runs. A config that would have refused to boot must not be able
|
||||
// to slip in through a reload — that is how a daemon ends up in a state it could never
|
||||
// have started in, which is the hardest kind to debug.
|
||||
fresh.validateAuthExposure();
|
||||
fresh.validateLeadTabPrefixes();
|
||||
fresh.validateSubscriptionProfiles();
|
||||
fresh.validateCharters();
|
||||
fresh.validateMembers();
|
||||
// have started in, which is the hardest kind to debug. fleetd ticket "central allow-list
|
||||
// of usable models" follow-up: this used to be six individual validateXxx() calls, and
|
||||
// mutation testing found two of the six unpinned here even though startup pinned nothing
|
||||
// at all — see FleetConfig#validateAll's javadoc for why the fix is one reflective call,
|
||||
// not a longer hand-maintained list.
|
||||
fresh.validateAll();
|
||||
// fleetd #474: validateAll() does not cover everything startup refuses on — the charter
|
||||
// tool-surface check (Fleetd.main, right after cfg.validateAll()) lives outside
|
||||
// FleetConfig on purpose (see this class's doc) and is supplied here as extraValidation.
|
||||
// Same try/catch as validateAll() above, on purpose: either failure must refuse the whole
|
||||
// reload and keep the running config the same way.
|
||||
extraValidation.accept(fresh);
|
||||
} catch (RuntimeException e) {
|
||||
String msg = e.getMessage() == null ? e.toString() : e.getMessage();
|
||||
log.warn("config reload from {} refused, keeping the running config: {}", path, msg);
|
||||
@@ -175,14 +434,21 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
}
|
||||
|
||||
List<String> deferred = changedDeferredKeys(old, fresh);
|
||||
List<String> split = changedSplitKeys(old, fresh);
|
||||
current.set(fresh);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, null);
|
||||
Outcome out = new Outcome(true, List.of(), deferred, split, null);
|
||||
log.info(out.summary());
|
||||
return out;
|
||||
}
|
||||
|
||||
/** Cold keys whose value differs between the running config and the candidate. */
|
||||
private static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
/**
|
||||
* Cold keys whose value differs between the running config and the candidate.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333).
|
||||
*/
|
||||
static List<String> changedColdKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.bind(), fresh.bind())) {
|
||||
changed.add("bind");
|
||||
@@ -190,6 +456,9 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
if (!Objects.equals(old.herdrSocket(), fresh.herdrSocket())) {
|
||||
changed.add("herdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.memberHerdrSocket(), fresh.memberHerdrSocket())) {
|
||||
changed.add("memberHerdrSocket");
|
||||
}
|
||||
if (!Objects.equals(old.broker(), fresh.broker())) {
|
||||
changed.add("broker");
|
||||
}
|
||||
@@ -201,8 +470,16 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** Changed keys that were accepted but whose effect waits for a restart. */
|
||||
private static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
/**
|
||||
* Changed keys that were accepted but whose effect waits for a restart.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #changedColdKeys} and {@link #changedSplitKeys} already are (fleetd #333, extended to
|
||||
* this method by fleetd #337 — membership in {@link #DEFERRED_KEYS} proved nothing about this
|
||||
* method on its own until then; see that test's class doc).
|
||||
*/
|
||||
static List<String> changedDeferredKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.lifecycle(), fresh.lifecycle())) {
|
||||
changed.add("lifecycle");
|
||||
@@ -216,6 +493,44 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
if (!Objects.equals(old.worktreeRoot(), fresh.worktreeRoot())) {
|
||||
changed.add("worktreeRoot");
|
||||
}
|
||||
// Baked into the same GitWorktrees as worktreeRoot (Fleetd.java:251) and never rebuilt
|
||||
// either — see the class doc. Missing this check was fleetd #323 instance 2: a reload
|
||||
// that changed only worktreeGroup reported "config reloaded" with nothing deferred, and
|
||||
// newly provisioned worktrees kept the old sharing behaviour.
|
||||
if (!Objects.equals(old.worktreeGroup(), fresh.worktreeGroup())) {
|
||||
changed.add("worktreeGroup");
|
||||
}
|
||||
// fleetd #362: baked into the same GitWorktrees as worktreeRoot/worktreeGroup
|
||||
// (Fleetd.java:251) and never rebuilt either — a reload that changes only memberSkills
|
||||
// must be reported the same way, or a newly provisioned worktree keeps seeding from (or
|
||||
// skipping) the old source directory with nothing telling the operator why.
|
||||
if (!Objects.equals(old.memberSkills(), fresh.memberSkills())) {
|
||||
changed.add("memberSkills");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:506, 519, 520 read cfg.primary() only off the startup snapshot
|
||||
// (PrimaryRegistry's pinned terminal, ReplyPushLoop's reminder cap and backoff) — neither is
|
||||
// rebuilt on reload, so a changed value needs a restart. Note what it does NOT mean:
|
||||
// primary.terminal is deprecated (CB-532), identity comes from leaders:/leadScan:, so a lead
|
||||
// using those is unaffected by this pin either way. See the class doc for the exact scope.
|
||||
if (!Objects.equals(old.primary(), fresh.primary())) {
|
||||
changed.add("primary");
|
||||
}
|
||||
// fleetd #326: Fleetd.java:679-680 read cfg.configReload() only at startup to decide whether
|
||||
// to build a ConfigWatcher at all and with what interval — the watcher that would apply a
|
||||
// later change is itself built once, so a running watcher keeps its original enabled flag and
|
||||
// interval regardless of what a reload changes it to. Not cold: no already-open resource goes
|
||||
// inconsistent with the new value, a watcher (if any) simply keeps polling on the old settings.
|
||||
if (!Objects.equals(old.configReload(), fresh.configReload())) {
|
||||
changed.add("configReload");
|
||||
}
|
||||
// Fleetd.java reads cfg.idleSleepGuard() once, at startup, to decide whether to construct
|
||||
// an IdleSleepGuard at all and wire SessionManager's onAcquire/onRelease hooks to it —
|
||||
// neither is rebuilt on reload, so a running daemon keeps whatever this was at startup
|
||||
// (armed or not) regardless of a later edit here. Not cold: nothing already-open goes
|
||||
// inconsistent with the new value, an armed-or-not guard just keeps its original answer.
|
||||
if (!Objects.equals(old.idleSleepGuard(), fresh.idleSleepGuard())) {
|
||||
changed.add("idleSleepGuard");
|
||||
}
|
||||
if (!Objects.equals(old.spawnReadyTimeoutMs(), fresh.spawnReadyTimeoutMs())
|
||||
|| !Objects.equals(old.spawnReadyPollMs(), fresh.spawnReadyPollMs())) {
|
||||
changed.add("spawnReady*");
|
||||
@@ -260,14 +575,114 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically. Compares every component
|
||||
* the launcher reads at spawn; {@code weight}, {@code maxLoad} and {@code credentialId} are
|
||||
* excluded because those are read live (by the placement policy and, for credentialId, by
|
||||
* {@code CompositePeerLauncher}/the CB-578 stage B exhaustion sink) and really do take effect on
|
||||
* the next spawn.
|
||||
* Split keys whose value differs between the running config and the candidate — see the class
|
||||
* doc's Split bullet (fleetd #330, extended for {@code fleet:} by fleetd #333). Unlike
|
||||
* {@link #changedDeferredKeys}, this does not try to tell which sub-field moved for {@code
|
||||
* health:} or {@code coordinator:}: any change to either gets the same fixed message, because
|
||||
* the message already names both halves every time, so there is no "which half changed"
|
||||
* question left for the caller to answer. {@code fleet:} is different on purpose — see below.
|
||||
*
|
||||
* <p>Package-private (not {@code private}) so {@code ConfigRefTopLevelReportingCoverageTest}
|
||||
* can call it directly with a reflection-built {@code FleetConfig} pair, the same reason
|
||||
* {@link #sameLaunchSettings} is package-private — see that test's class doc (fleetd #333). That
|
||||
* test exists because membership in {@link #SPLIT_KEYS} proves nothing about this method on its
|
||||
* own: fleetd #333 measured that dropping the {@code coordinator} branch out of this method
|
||||
* while leaving {@code "coordinator"} in {@code SPLIT_KEYS} left the whole suite green except a
|
||||
* hand-written {@code ConfigRefTest} case — neither {@code ConfigRefTopLevelCoverageTest} (it
|
||||
* only reads the set) nor the "kept in step" assert below (it only checks the reported keys are
|
||||
* a SUBSET of {@code SPLIT_KEYS}, never that every {@code SPLIT_KEYS} member has a branch here)
|
||||
* would have caught it.
|
||||
*/
|
||||
private static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
static List<String> changedSplitKeys(FleetConfig old, FleetConfig fresh) {
|
||||
List<String> changed = new ArrayList<>();
|
||||
if (!Objects.equals(old.health(), fresh.health())) {
|
||||
changed.add("health: the monitor itself (enabled, interval, workingSuspectAfter) is "
|
||||
+ "frozen at startup and needs a restart; the coverage status fleet_profiles "
|
||||
+ "reports is read live and already applied");
|
||||
}
|
||||
if (!Objects.equals(old.coordinator(), fresh.coordinator())) {
|
||||
changed.add("coordinator: the LeadMailbox connection (uri, uriEnv, selfId, prefetch) is "
|
||||
+ "opened once and needs a restart; the broker URI env-var name kept out of a "
|
||||
+ "member's environment is read live on every spawn and already applied");
|
||||
}
|
||||
// fleetd #333: unlike health/coordinator above, most of `fleet:` (developers, reviewers,
|
||||
// charters, tabLabel) is genuinely hot — ConfigRefTest.aHotChangeIsAppliedAndRead-
|
||||
// ThroughGet and aCharterChangeIsHotAndReachesTheLiveConfig prove it reaches the live config
|
||||
// with no restart note. `architects` is hot too, and — since fleetd #424 — hot for BOTH of
|
||||
// its consumers, not just the one this comment used to name: CompositePeerLauncher reads it
|
||||
// live for PLACEMENT through the () -> config.get().fleet() supplier named in the class doc's
|
||||
// Hot bullet, and MemberRegistry separately reads it live for IDENTITY (which slot a spawn
|
||||
// may bind to, AND what a slot already bound still grants) through its own instance of that
|
||||
// same supplier shape — see MemberRegistry.live and its class doc for the binding rule:
|
||||
// removing a slot revokes ARCHITECT on the bound pane's very next request, and only the slot
|
||||
// OCCUPANCY survives, so the demoted session keeps its slot key until it unbinds. Only
|
||||
// fleet.leaders is frozen (Fleetd.java:281 reads cfg.fleet().leaders() off the startup
|
||||
// snapshot to build both the LeadTabScanner's tab-label-to-name map, wired into
|
||||
// CallerResolver.withLeadsAndMembers at Fleetd.java:620/624, and — when herdr answered —
|
||||
// LeadLauncher(...).ensureLeads() at Fleetd.java:315, which auto-launches each lead up to its
|
||||
// `instances` count; neither is rebuilt on reload). So this compares fleet.leaders alone, not
|
||||
// the whole Fleet record: comparing the whole record would report "split" for a tabLabel-only
|
||||
// or architects-only change that is actually fully hot, which is the over-claim mirror of the
|
||||
// under-claim bug this class exists to prevent.
|
||||
if (!Objects.equals(leadersOf(old), leadersOf(fresh))) {
|
||||
changed.add("fleet: fleet.leaders (each lead's tab, workspace, cwd, profile and "
|
||||
+ "instances count) is read once at startup to build the LeadTabScanner's "
|
||||
+ "identity map and to auto-launch leads, and neither is rebuilt on reload, so a "
|
||||
+ "lead added, removed, or given a new tab: label needs a restart — until then it "
|
||||
+ "stays unrecognised, and a caller from its new tab resolves as a worker, not a "
|
||||
+ "lead; the rest of fleet: (developers, reviewers, charters, tabLabel) is read "
|
||||
+ "live through the supplier on CompositePeerLauncher, and architects is read "
|
||||
+ "live through that same supplier for placement AND through a separate supplier "
|
||||
+ "on MemberRegistry for spawn-time identity — both already applied");
|
||||
}
|
||||
// Kept in step with SPLIT_KEYS the same way changedColdKeys is kept in step with COLD_KEYS —
|
||||
// every message here must be traceable to one of the split keys the class doc documents.
|
||||
// NOTE what this does NOT prove, per the javadoc above: it does not catch a SPLIT_KEYS
|
||||
// member with no branch above at all, only a branch whose message is mis-worded relative to
|
||||
// the set. ConfigRefTopLevelReportingCoverageTest is what proves the former.
|
||||
assert changed.stream().allMatch(m -> SPLIT_KEYS.stream().anyMatch(k -> m.startsWith(k + ":")))
|
||||
: "a split entry was reported that does not start with a SPLIT_KEYS name: " + changed;
|
||||
return changed;
|
||||
}
|
||||
|
||||
/** {@code cfg.fleet().leaders()}, defensively, in case a caller hands in a non-defaulted config. */
|
||||
private static Map<String, FleetConfig.Leader> leadersOf(FleetConfig cfg) {
|
||||
return cfg.fleet() == null ? Map.of() : cfg.fleet().leaders();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link FleetConfig.Profile} record components deliberately left out of
|
||||
* {@link #sameLaunchSettings} because they are read <em>live</em>, not baked in at spawn — see
|
||||
* the class doc's <em>Hot</em> bullet. {@code weight} and {@code maxLoad} are read live by the
|
||||
* placement policy on every spawn; {@code credentialId} is read live by
|
||||
* {@code CompositePeerLauncher} and the CB-578 stage B exhaustion sink; {@code exhaustedPattern}
|
||||
* (fleetd #446) is read live, cached by profile name, by {@code LiveExhaustedPatterns} — see
|
||||
* that class's doc and the class doc's <em>Hot</em> bullet for the history (it used to be
|
||||
* compared here, deferred, like its sibling {@code errorPattern} still is). Nothing else is
|
||||
* excluded — see {@code sameLaunchSettingsComparesEveryProfileComponentOrExcludesIt} in
|
||||
* {@code ConfigRefProfileCoverageTest}, which enumerates every {@code Profile} record component
|
||||
* by reflection and fails the build if one is neither compared below nor named here.
|
||||
*/
|
||||
static final Set<String> LAUNCH_SETTINGS_EXCLUDED =
|
||||
Set.of("weight", "maxLoad", "credentialId", "exhaustedPattern");
|
||||
|
||||
/**
|
||||
* Whether two versions of a profile would launch a peer identically.
|
||||
*
|
||||
* <p>This must compare every {@link FleetConfig.Profile} record component except the three in
|
||||
* {@link #LAUNCH_SETTINGS_EXCLUDED}. That is not a claim this javadoc can make good on by
|
||||
* itself — a javadoc saying "compares every component" is exactly what fleetd #323 found to be
|
||||
* false for three fields (and a sibling method's field list, for a fourth). The actual
|
||||
* guarantee comes from {@code ConfigRefProfileCoverageTest}: it enumerates every record
|
||||
* component of {@code FleetConfig.Profile} by reflection, mutates each one not in
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} on a base profile, and asserts this method reports a
|
||||
* difference — so a new component that is neither compared here nor added to
|
||||
* {@code LAUNCH_SETTINGS_EXCLUDED} (with a reason) fails that test by name, rather than
|
||||
* silently reporting "config reloaded" for a value the daemon never picked up.
|
||||
*/
|
||||
static boolean sameLaunchSettings(FleetConfig.Profile a, FleetConfig.Profile b) {
|
||||
return Objects.equals(a.profile(), b.profile())
|
||||
&& Objects.equals(a.baseUrl(), b.baseUrl())
|
||||
&& Objects.equals(a.model(), b.model())
|
||||
&& Objects.equals(a.configDir(), b.configDir())
|
||||
&& Objects.equals(a.tokenEnv(), b.tokenEnv())
|
||||
@@ -286,9 +701,25 @@ public final class ConfigRef implements Supplier<FleetConfig> {
|
||||
&& Objects.equals(a.kind(), b.kind())
|
||||
&& Objects.equals(a.env(), b.env())
|
||||
&& Objects.equals(a.subscription(), b.subscription())
|
||||
// CB-578 stage B: exhaustedPattern is compiled once into Fleetd.main's pattern map
|
||||
// at startup (see ExhaustedPatternLookup wiring) — a reload never re-reads it, so a
|
||||
// changed pattern must be reported as deferred, exactly like model/baseUrl/argv.
|
||||
&& Objects.equals(a.exhaustedPattern(), b.exhaustedPattern());
|
||||
// fleetd #446: exhaustedPattern moved to LAUNCH_SETTINGS_EXCLUDED — it is now read
|
||||
// live, cached by profile name, through LiveExhaustedPatterns (see that class's doc
|
||||
// and ConfigRef's class doc Hot bullet), so it must NOT be compared here any more: a
|
||||
// reload that changes only exhaustedPattern must report "config reloaded", not
|
||||
// "these changes need a restart".
|
||||
// fleetd #201 Unit 5: errorPattern is compiled once into Fleetd.main's backend-error
|
||||
// pattern map at startup (see BackendErrorPatternLookup wiring) — unlike its sibling
|
||||
// exhaustedPattern above (fleetd #446), a reload still never re-reads it; fleetd #446
|
||||
// scoped errorPattern out on purpose (see the class doc's Hot bullet).
|
||||
&& Objects.equals(a.errorPattern(), b.errorPattern())
|
||||
// fleetd #323 instance 1: ideProjectDir and ideOpenCommand are read at spawn off the
|
||||
// same frozen profile map as ideMcpUrl above (ClaudeCodeLauncher.java:267/269,
|
||||
// OpenCodeLauncher.java:474/480/486) and were missing from this comparison.
|
||||
&& Objects.equals(a.ideProjectDir(), b.ideProjectDir())
|
||||
&& Objects.equals(a.ideOpenCommand(), b.ideOpenCommand())
|
||||
// fleetd #323 instance 1: autoCompactWindow is read at spawn the same way
|
||||
// (ClaudeCodeLauncher.java:926, OpenCodeLauncher.java:650). Comparing it here only
|
||||
// makes the reload REPORT that a restart is needed — it deliberately does not make
|
||||
// autoCompactWindow take effect live, which is a separate, larger change.
|
||||
&& Objects.equals(a.autoCompactWindow(), b.autoCompactWindow());
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -14,6 +14,7 @@ import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.BiConsumer;
|
||||
@@ -28,17 +29,65 @@ public final class FleetHealthMonitor {
|
||||
static final int MAX_FAIL_TARGET_ATTEMPTS = 3;
|
||||
// CB-641: Match the injector's 60s readiness gate so health allows a full first boot.
|
||||
static final long READINESS_GRACE_NANOS = TimeUnit.SECONDS.toNanos(60);
|
||||
/**
|
||||
* fleetd #280: how long after a terminal transition to wait before the one bounded re-check
|
||||
* fires. Must exceed the worst-case reverse-rendezvous {@code fleet_ask} window (55-115s, see
|
||||
* {@code FleetMcp.ASK_DEFAULT_TIMEOUT_MS} / {@code FleetApp.MAX_ASK_TIMEOUT_MS}) so that, if the
|
||||
* target was genuinely {@code ASKING} when {@code state} was first observed, its own ask has had
|
||||
* time to lapse (clearing {@code Task#question} back to {@code null}) before this fires.
|
||||
*/
|
||||
static final long ASK_LAPSE_RECHECK_DELAY_SECONDS = 120;
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code System.nanoTime()} (or whatever {@link #clock} is) does not advance while
|
||||
* macOS sleeps, so a raw {@code nowNanos - lastActivityAtNanos} comparison freezes with the
|
||||
* host and can never cross {@link #workingSuspectAfterNanos}. This is a second, wall-clock
|
||||
* source used ONLY inside the stall check ({@link #stallElapsedNanos}) to detect and correct
|
||||
* for that freeze. Nothing else in this class reads it — every other decision (readiness grace,
|
||||
* the fault classification itself) stays exactly on {@link #clock}, as the ticket requires.
|
||||
*/
|
||||
private static final LongSupplier DEFAULT_REALTIME_CLOCK =
|
||||
() -> TimeUnit.MILLISECONDS.toNanos(System.currentTimeMillis());
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<List<MemberSession>> roster;
|
||||
private final MessageService messages;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final LongSupplier clock;
|
||||
private final LongSupplier realtimeClock;
|
||||
private final long intervalSeconds;
|
||||
private final long tickIntervalNanos;
|
||||
private final long workingSuspectAfterNanos;
|
||||
private final BiConsumer<String, String> failTarget;
|
||||
private final Map<String, HealthPrior> priors = new HashMap<>();
|
||||
private final Map<String, HealthState> states = new HashMap<>();
|
||||
/**
|
||||
* fleetd #386 clock-drift bookkeeping. {@code haveClockBaseline}/{@code lastTickMonoNanos}/
|
||||
* {@code lastTickRealNanos} track the previous tick's pair of readings so each new tick can
|
||||
* measure how far the two clocks moved apart since then. {@code accumulatedDriftNanos} is the
|
||||
* running total of every such divergence observed since this monitor started (never decreases —
|
||||
* the monotonic clock can only lag real time, never lead it). {@code busyDriftBaselineNanos}/
|
||||
* {@code busyBaselineActivityNanos} record, per target, the value of {@code accumulatedDriftNanos}
|
||||
* at the moment this monitor first saw that target's CURRENT {@code lastActivityAtNanos} while
|
||||
* BUSY — so {@link #stallElapsedNanos} adds back only the drift observed DURING this BUSY span,
|
||||
* never drift from a sleep that happened before the member went busy. All five fields are touched
|
||||
* only from {@code tick()}, like {@link #priors}.
|
||||
*/
|
||||
private boolean haveClockBaseline = false;
|
||||
private long lastTickMonoNanos;
|
||||
private long lastTickRealNanos;
|
||||
private long accumulatedDriftNanos = 0;
|
||||
private final Map<String, Long> busyDriftBaselineNanos = new HashMap<>();
|
||||
private final Map<String, Long> busyBaselineActivityNanos = new HashMap<>();
|
||||
/**
|
||||
* The live classification per member, and the only one of this class's three maps that more
|
||||
* than one scheduler task touches. {@code tick} writes it (and prunes it to the roster);
|
||||
* fleetd #280's delayed {@link #recheckTerminalTarget} reads it from its own separate scheduled
|
||||
* task. Both run on the single-threaded scheduler {@code Fleetd} passes in today, so they are
|
||||
* serialised — but nothing in this class enforces that, and an unsynchronised {@link HashMap}
|
||||
* read racing a resize can spin a CPU forever rather than fail visibly. {@code priors} and
|
||||
* {@code orphanStreaks} stay plain maps because {@code tick} is still their only toucher.
|
||||
*/
|
||||
private final Map<String, HealthState> states = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* CB-643: consecutive ticks on which a target looked like an orphaned delegation. The fact
|
||||
* {@link MessageService#hasOrphanedDelegation} reports is a true snapshot, but it can read true
|
||||
@@ -70,12 +119,29 @@ public final class FleetHealthMonitor {
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, long intervalSeconds,
|
||||
long workingSuspectAfterSeconds, BiConsumer<String, String> failTarget) {
|
||||
this(agents, roster, messages, scheduler, clock, DEFAULT_REALTIME_CLOCK, intervalSeconds,
|
||||
workingSuspectAfterSeconds, failTarget);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param realtimeClock fleetd #386: a wall-clock nanosecond source (e.g.
|
||||
* {@code System.currentTimeMillis()} converted to nanos) that keeps
|
||||
* advancing while {@code clock} is frozen by a host sleep. Used only to
|
||||
* correct the stall check — see the class-level javadoc on the
|
||||
* clock-drift fields.
|
||||
*/
|
||||
public FleetHealthMonitor(AgentControl agents, Supplier<List<MemberSession>> roster, MessageService messages,
|
||||
ScheduledExecutorService scheduler, LongSupplier clock, LongSupplier realtimeClock,
|
||||
long intervalSeconds, long workingSuspectAfterSeconds,
|
||||
BiConsumer<String, String> failTarget) {
|
||||
this.agents = agents;
|
||||
this.roster = roster;
|
||||
this.messages = messages;
|
||||
this.scheduler = scheduler;
|
||||
this.clock = clock;
|
||||
this.realtimeClock = Objects.requireNonNull(realtimeClock, "realtimeClock");
|
||||
this.intervalSeconds = intervalSeconds;
|
||||
this.tickIntervalNanos = TimeUnit.SECONDS.toNanos(intervalSeconds);
|
||||
this.workingSuspectAfterNanos = TimeUnit.SECONDS.toNanos(workingSuspectAfterSeconds);
|
||||
this.failTarget = Objects.requireNonNull(failTarget, "failTarget");
|
||||
}
|
||||
@@ -105,6 +171,7 @@ public final class FleetHealthMonitor {
|
||||
for (Agent agent : agentsNow) live.put(agent.terminalId(), agent);
|
||||
HashSet<String> current = new HashSet<>();
|
||||
long nowNanos = clock.getAsLong();
|
||||
long driftBeforeThisTick = observeClockDrift(nowNanos);
|
||||
for (MemberSession session : rosterNow) {
|
||||
current.add(session.terminalId());
|
||||
Agent agent = live.get(session.terminalId());
|
||||
@@ -115,7 +182,7 @@ public final class FleetHealthMonitor {
|
||||
&& session.state() != MemberSession.State.SPAWNING;
|
||||
boolean readinessGraceElapsed = nowNanos - session.spawnedAtNanos() >= READINESS_GRACE_NANOS;
|
||||
boolean stalled = session.state() == MemberSession.State.BUSY
|
||||
&& nowNanos - session.lastActivityAtNanos() >= workingSuspectAfterNanos;
|
||||
&& stallElapsedNanos(session, nowNanos, driftBeforeThisTick) >= workingSuspectAfterNanos;
|
||||
// CB-643: the three message-layer facts CB-640 published. Read them here rather than
|
||||
// leaving them false — that constant is what made 8 of the 9 fault states dead.
|
||||
boolean queuedDelivery = messages.hasQueuedDelivery(session.terminalId());
|
||||
@@ -133,6 +200,8 @@ public final class FleetHealthMonitor {
|
||||
priors.keySet().retainAll(current);
|
||||
states.keySet().retainAll(current);
|
||||
orphanStreaks.keySet().retainAll(current);
|
||||
busyDriftBaselineNanos.keySet().retainAll(current);
|
||||
busyBaselineActivityNanos.keySet().retainAll(current);
|
||||
} catch (Throwable error) {
|
||||
// Any unclassified collection failure must never kill the monitor's only scheduler task.
|
||||
log.warn("fleet health collection failed; will retry next tick", error);
|
||||
@@ -157,6 +226,65 @@ public final class FleetHealthMonitor {
|
||||
return streak >= ORPHAN_CONFIRM_TICKS;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: compare this tick's monotonic and real-time readings against the previous
|
||||
* tick's, and fold any positive divergence into {@link #accumulatedDriftNanos} (a ratchet — it
|
||||
* never decreases, since the monotonic clock can only fall behind real time, never ahead of
|
||||
* it). Logs once, at WARN, when that single tick's divergence exceeds one full tick interval —
|
||||
* the signature of a host that slept between the two ticks (a tick literally cannot run while
|
||||
* the process itself is suspended, so the whole sleep duration lands inside one tick's gap).
|
||||
*
|
||||
* @return {@link #accumulatedDriftNanos} as it stood BEFORE this tick's divergence was folded
|
||||
* in — the baseline {@link #stallElapsedNanos} needs when a target is observed BUSY
|
||||
* for the first time this tick, so a sleep that happened before this member went busy
|
||||
* is not attributed to it.
|
||||
*/
|
||||
private long observeClockDrift(long nowNanos) {
|
||||
long nowRealNanos = realtimeClock.getAsLong();
|
||||
long driftBeforeThisTick = accumulatedDriftNanos;
|
||||
if (haveClockBaseline) {
|
||||
long monoDelta = nowNanos - lastTickMonoNanos;
|
||||
long realDelta = nowRealNanos - lastTickRealNanos;
|
||||
long tickDrift = realDelta - monoDelta;
|
||||
if (tickDrift > tickIntervalNanos) {
|
||||
log.warn("fleet health: the monotonic clock did not advance for about {}s that the "
|
||||
+ "real clock did since the last tick (host likely slept); the stall "
|
||||
+ "detector could not see that time", TimeUnit.NANOSECONDS.toSeconds(tickDrift));
|
||||
}
|
||||
if (tickDrift > 0) {
|
||||
accumulatedDriftNanos = driftBeforeThisTick + tickDrift;
|
||||
}
|
||||
}
|
||||
lastTickMonoNanos = nowNanos;
|
||||
lastTickRealNanos = nowRealNanos;
|
||||
haveClockBaseline = true;
|
||||
return driftBeforeThisTick;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #386: {@code nowNanos - lastActivityAtNanos} alone freezes across a host sleep, since
|
||||
* both come from the monotonic {@link #clock}. This adds back the real-time drift observed
|
||||
* since this BUSY span started — not the monitor's whole lifetime, so a sleep that happened
|
||||
* before this member went busy never leaks into its stall reading (see the class-level javadoc
|
||||
* on the drift fields). The baseline resets whenever {@code lastActivityAtNanos} changes (a new
|
||||
* turn) or the member is not currently BUSY.
|
||||
*/
|
||||
private long stallElapsedNanos(MemberSession session, long nowNanos, long driftBeforeThisTick) {
|
||||
String target = session.terminalId();
|
||||
if (session.state() != MemberSession.State.BUSY) {
|
||||
busyDriftBaselineNanos.remove(target);
|
||||
busyBaselineActivityNanos.remove(target);
|
||||
return nowNanos - session.lastActivityAtNanos();
|
||||
}
|
||||
Long baselineActivity = busyBaselineActivityNanos.get(target);
|
||||
if (baselineActivity == null || baselineActivity != session.lastActivityAtNanos()) {
|
||||
busyBaselineActivityNanos.put(target, session.lastActivityAtNanos());
|
||||
busyDriftBaselineNanos.put(target, driftBeforeThisTick);
|
||||
}
|
||||
long driftSinceBusyStart = accumulatedDriftNanos - busyDriftBaselineNanos.get(target);
|
||||
return (nowNanos - session.lastActivityAtNanos()) + driftSinceBusyStart;
|
||||
}
|
||||
|
||||
void reportTransition(String target, HealthState next) {
|
||||
HealthState previous = states.put(target, next);
|
||||
if (previous == next) return;
|
||||
@@ -171,6 +299,7 @@ public final class FleetHealthMonitor {
|
||||
// member stayed terminal.
|
||||
if (terminal(next)) {
|
||||
failTerminalTarget(target, next);
|
||||
scheduleTerminalRecheck(target, next);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -191,6 +320,48 @@ public final class FleetHealthMonitor {
|
||||
target, state, MAX_FAIL_TARGET_ATTEMPTS, last);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: schedule the one bounded, delayed follow-up for a terminal transition — never a
|
||||
* per-tick retry (CB-580 rejected that shape; {@link #reportTransition} still fires
|
||||
* {@link #failTerminalTarget} exactly once per transition, unconditionally on the tick loop).
|
||||
* This is a single one-shot task, scheduled once per transition into GONE/NEVER_READY, so a
|
||||
* member stuck terminal for the rest of its life gets exactly one extra attempt, not one per
|
||||
* tick. See {@link #recheckTerminalTarget} for why the extra attempt is safe.
|
||||
*/
|
||||
private void scheduleTerminalRecheck(String target, HealthState state) {
|
||||
if (scheduler.isShutdown()) return;
|
||||
try {
|
||||
scheduler.schedule(() -> recheckTerminalTarget(target, state),
|
||||
ASK_LAPSE_RECHECK_DELAY_SECONDS, TimeUnit.SECONDS);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("fleet health: could not schedule terminal re-check for member={} state={}",
|
||||
target, state, e);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #280: the delayed re-check {@link #scheduleTerminalRecheck} scheduled for one terminal
|
||||
* transition. By now, a {@code fleet_ask} that was still open when {@code state} was first
|
||||
* observed has had time to lapse on its own (see {@link #ASK_LAPSE_RECHECK_DELAY_SECONDS}),
|
||||
* clearing {@code Task#question} back to {@code null} — which is exactly what
|
||||
* {@link MessageService#abandon(String, String, boolean)}'s {@code sweepAsking=false} filter
|
||||
* needs to finally match it. Calling {@link #failTerminalTarget} again is safe only because
|
||||
* {@code sweepAsking} stays {@code false}: a task genuinely still {@code ASKING} is skipped
|
||||
* exactly as it was on the very first attempt — this never fails a ticket whose ask has not yet
|
||||
* lapsed.
|
||||
*
|
||||
* <p><strong>Guarded on "target is still classified {@code state}."</strong> Without this guard,
|
||||
* a member that recovered (or was released and dropped from the roster) between the transition
|
||||
* and this re-check would still take a blind {@code failTarget} call — reaching into whatever
|
||||
* brand-new, unrelated turn it has since picked up and failing it too. {@link #states} already
|
||||
* carries the live classification (updated every tick, pruned to the current roster on release),
|
||||
* so a stale or recovered target simply reads as a mismatch here and this is a no-op.
|
||||
*/
|
||||
void recheckTerminalTarget(String target, HealthState state) {
|
||||
if (states.get(target) != state) return;
|
||||
failTerminalTarget(target, state);
|
||||
}
|
||||
|
||||
private static boolean terminal(HealthState state) {
|
||||
return state == HealthState.GONE || state == HealthState.NEVER_READY;
|
||||
}
|
||||
|
||||
@@ -38,6 +38,11 @@ public final class AgentControl {
|
||||
this.herdr = herdr;
|
||||
}
|
||||
|
||||
/** The herdr daemon this control object sends its agent calls to. */
|
||||
public HerdrClient herdr() {
|
||||
return herdr;
|
||||
}
|
||||
|
||||
/** One agent-targeted call, translating a terminal id to its pane id (retrying once fresh). */
|
||||
private JsonNode agentCall(String method, String target, Map<String, Object> extra) {
|
||||
String resolved = resolveTarget(target);
|
||||
|
||||
@@ -3,7 +3,7 @@ package dev.ltms.fleet.herdr;
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
/**
|
||||
* Client face onto the herdr daemon (protocol 14, herdr 0.7.0).
|
||||
* Client face onto the herdr daemon (protocol 19, herdr 0.8.0).
|
||||
*
|
||||
* <p>This is the ONLY thing in {@code fleetd} that speaks to herdr. Every method
|
||||
* maps to a herdr JSON-RPC call over its Unix domain socket. Requests are
|
||||
|
||||
@@ -8,7 +8,7 @@ import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import java.nio.charset.StandardCharsets;
|
||||
|
||||
/**
|
||||
* Wire codec for herdr's newline-delimited JSON-RPC (protocol 14).
|
||||
* Wire codec for herdr's newline-delimited JSON-RPC (protocol 19).
|
||||
*
|
||||
* <p>Split out from the socket so the framing rules — the ones that actually bit us
|
||||
* during the spike (id MUST be a string; response carries {@code result} or
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import java.util.Objects;
|
||||
import java.util.function.Predicate;
|
||||
|
||||
/** Routes lead operations and member operations to their owning herdr daemon. */
|
||||
public final class HerdrRouter implements AutoCloseable {
|
||||
private final HerdrClient lead;
|
||||
private final HerdrClient member;
|
||||
private final AgentControl leadAgents;
|
||||
private final AgentControl memberAgents;
|
||||
private final WorkspaceControl leadSpaces;
|
||||
private final WorkspaceControl memberSpaces;
|
||||
private final Predicate<String> isLead;
|
||||
|
||||
public HerdrRouter(HerdrClient lead, HerdrClient member, Predicate<String> isLead) {
|
||||
this.lead = Objects.requireNonNull(lead, "lead");
|
||||
this.member = member != null ? member : lead;
|
||||
this.isLead = Objects.requireNonNull(isLead, "isLead");
|
||||
leadAgents = new AgentControl(this.lead);
|
||||
memberAgents = this.member == this.lead ? leadAgents : new AgentControl(this.member);
|
||||
leadSpaces = new WorkspaceControl(this.lead);
|
||||
memberSpaces = this.member == this.lead ? leadSpaces : new WorkspaceControl(this.member);
|
||||
}
|
||||
|
||||
public AgentControl leadAgents() { return leadAgents; }
|
||||
public WorkspaceControl leadSpaces() { return leadSpaces; }
|
||||
public AgentControl memberAgents() { return memberAgents; }
|
||||
public WorkspaceControl memberSpaces() { return memberSpaces; }
|
||||
public AgentControl agentsFor(String targetId) { return isLead.test(targetId) ? leadAgents : memberAgents; }
|
||||
|
||||
HerdrClient leadClient() { return lead; }
|
||||
HerdrClient memberClient() { return member; }
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
lead.close();
|
||||
if (member != lead) member.close();
|
||||
}
|
||||
}
|
||||
@@ -5,6 +5,7 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.Collections;
|
||||
import java.util.HashSet;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Locale;
|
||||
import java.util.Map;
|
||||
@@ -53,17 +54,40 @@ import java.util.function.Supplier;
|
||||
* daemon and found again by this scan. The trust direction above is unaffected — fleetd writing a
|
||||
* name for a lead it just started is not a pane promoting itself — but <em>staleness</em> becomes
|
||||
* real: a label left behind by a session that has since died would read as a live lead forever.
|
||||
* This scanner does not solve that (its job is naming, and a stale name costs nothing here); the
|
||||
* launcher does, by requiring a running agent in the tab before it counts the lead as live. If you
|
||||
* ever make a decision that <em>removes</em> something based on this map, add the same check.
|
||||
* The remaining hazard is an <em>operator</em> one — a worker {@code tabLabel} template that
|
||||
* happens to start with the same prefix would promote the whole fleet — and that is refused at
|
||||
* startup by {@code FleetConfig.validateLeadTabPrefixes} rather than documented here.
|
||||
*
|
||||
* <p><strong>fleetd #359 — the staleness check this class used to skip.</strong> This used to say
|
||||
* "a stale name costs nothing here" and leave liveness to {@code LeadLauncher}, on the theory that
|
||||
* naming and removing are different decisions. That was wrong: {@code dev.ltms.fleet.msg.LeadCoordLoop}
|
||||
* makes exactly the kind of removal decision the old javadoc warned about, by reading this map to
|
||||
* pick which pane a peer message goes into — and a stale entry there is not free. On a host where
|
||||
* {@code fleetd} had restarted more than once, a labelled-but-dead tab from a previous life was
|
||||
* reported right alongside the live one; {@code LeadCoordLoop.resolveLocalLead()} saw more than one
|
||||
* candidate and refused to guess (safe), but the fix the daemon's own WARN suggests — name a lead
|
||||
* after {@code coordinator.selfId} — stops being safe once two tabs can share a label: step 1 of
|
||||
* that resolution picks whichever matching entry it finds first, which can be the dead one, and
|
||||
* typing a peer's message into a dead shell does not fail — it is silently gone instead of merely
|
||||
* held. {@link #scan()} now cross-checks every labelled tab against {@code agent.list} (the same
|
||||
* signal {@code LeadLauncher.countLeads} already trusts for the same purpose) and drops any tab
|
||||
* with no agent running in it, so a dead tab is never in the map for a caller to pick at all.
|
||||
*
|
||||
* <p><strong>Caching.</strong> {@link #get()} is on the request path (every resolve), so the scan
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan keeps
|
||||
* the previous answer instead of emptying it — a herdr hiccup must not silently demote a live lead
|
||||
* mid-session.
|
||||
* is TTL-cached and a stale-but-valid map is preferred to a herdr round-trip. A failed scan (herdr
|
||||
* throws) keeps the previous answer instead of emptying it — a herdr hiccup must not silently
|
||||
* demote a live lead mid-session.
|
||||
*
|
||||
* <p><strong>fleetd #359 review, finding 2 — a successful-but-wrong scan is the same hazard.</strong>
|
||||
* The catch above only fires when a call throws. It does nothing for a call that returns 200 with an
|
||||
* incomplete answer — exactly what the ticket's own evidence showed {@code agent.list} can do. Once
|
||||
* this class started trusting that signal, an empty read would otherwise get cached as fact and
|
||||
* silently drop a lead {@code CallerResolver} had, until then, correctly resolved — turning it into a
|
||||
* {@code Role.WORKER}, which refuses every orchestration call. So a terminal this class already
|
||||
* reported as live is not dropped the first time {@code agent.list} loses it: {@link #scan()} grants
|
||||
* it one grace scan (see {@code gracedTerminals}) and only drops it if a <em>later</em> scan still
|
||||
* finds no agent. A terminal never reported live before gets no grace — that would weaken the
|
||||
* original #359 fix itself, which this class's own test suite already pins.
|
||||
*/
|
||||
public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
|
||||
@@ -79,6 +103,15 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
private long scannedAtNanos;
|
||||
private boolean everScanned;
|
||||
|
||||
/**
|
||||
* Terminals currently on their one grace scan: {@code cached} reported them live, the most
|
||||
* recent {@link #scan()} found no agent for them, and they were re-included anyway. Cleared for
|
||||
* a terminal the instant it is seen live again; a terminal still here on the <em>next</em> scan
|
||||
* is finally dropped. Scoped separately from {@link #cached} so a graced terminal cannot renew
|
||||
* its own grace forever just by staying in the exposed map (fleetd #359 review, finding 2).
|
||||
*/
|
||||
private Set<String> gracedTerminals = Set.of();
|
||||
|
||||
/**
|
||||
* @param herdr the herdr client to query ({@code workspace.list},
|
||||
* {@code tab.list}, {@code pane.list} — all read-only)
|
||||
@@ -141,7 +174,7 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
return cached;
|
||||
}
|
||||
|
||||
/** One full pass: labelled tabs → their panes → those panes' terminals. */
|
||||
/** One full pass: labelled tabs → live agents in them → those panes' terminals. */
|
||||
private Map<String, String> scan() {
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
for (JsonNode w : herdr.call("workspace.list").path("workspaces")) {
|
||||
@@ -158,17 +191,47 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
if (!nameByTab.isEmpty()) {
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String name = nameByTab.get(p.path("tab_id").asText(null));
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name != null && terminal != null && !terminal.isBlank()) {
|
||||
byTerminal.put(terminal, name);
|
||||
}
|
||||
if (nameByTab.isEmpty()) {
|
||||
gracedTerminals = Set.of();
|
||||
return Map.of();
|
||||
}
|
||||
|
||||
// fleetd #359: a labelled tab is only a lead when herdr also reports a running agent in
|
||||
// it — the same liveness signal LeadLauncher.countLeads trusts for the identical purpose.
|
||||
// Without this, a tab left behind by a session that has since died reads as live forever.
|
||||
Set<String> tabsWithAgent = new HashSet<>();
|
||||
for (JsonNode a : herdr.call("agent.list").path("agents")) {
|
||||
String tabId = a.path("tab_id").asText(null);
|
||||
if (tabId != null) {
|
||||
tabsWithAgent.add(tabId);
|
||||
}
|
||||
}
|
||||
|
||||
Map<String, String> byTerminal = new LinkedHashMap<>();
|
||||
Set<String> stillGraced = new HashSet<>();
|
||||
// One pane.list for every tab: panes carry tab_id, so the join is local.
|
||||
for (JsonNode p : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String tabId = p.path("tab_id").asText(null);
|
||||
String name = nameByTab.get(tabId);
|
||||
String terminal = p.path("terminal_id").asText(null);
|
||||
if (name == null || terminal == null || terminal.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (tabsWithAgent.contains(tabId)) {
|
||||
byTerminal.put(terminal, name);
|
||||
continue;
|
||||
}
|
||||
// No agent reported for this tab, but its tab/pane are still here — this is the
|
||||
// ambiguous case review finding 2 named: a successful agent.list that came back short
|
||||
// does not prove the lead is dead. Grant one grace scan to a terminal we had already
|
||||
// reported as live; a terminal we never reported live gets none, so the original #359
|
||||
// fix (a genuinely dead tab is never reported) is unaffected for the common case.
|
||||
if (cached.containsKey(terminal) && !gracedTerminals.contains(terminal)) {
|
||||
byTerminal.put(terminal, name);
|
||||
stillGraced.add(terminal);
|
||||
}
|
||||
}
|
||||
gracedTerminals = stillGraced;
|
||||
return Collections.unmodifiableMap(byTerminal);
|
||||
}
|
||||
|
||||
@@ -177,12 +240,14 @@ public final class LeadTabScanner implements Supplier<Map<String, String>> {
|
||||
*
|
||||
* <p>Exact match (case-insensitive, ends stripped) against {@link #tabToName} — no prefix
|
||||
* stripping, so an operator's {@code "lead: something-else"} tab is never mistaken for a
|
||||
* configured lead just because it shares a prefix.
|
||||
* configured lead just because it shares a prefix. The match strips a trailing
|
||||
* {@link PendingCloseMarker} first, so a tab {@code LeadLauncher} has flagged as maybe-dead but
|
||||
* not yet closed keeps resolving normally while that reconcile is pending.
|
||||
*/
|
||||
private String leadNameOf(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
return tabToName.get(label.strip().toLowerCase(Locale.ROOT));
|
||||
return tabToName.get(PendingCloseMarker.strip(label).toLowerCase(Locale.ROOT));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,7 +2,11 @@ package dev.ltms.fleet.herdr;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* Resolves which herdr pane a process belongs to — the herdr half of connection-based MCP
|
||||
@@ -11,46 +15,130 @@ import java.util.Map;
|
||||
* is calling without the worker sending anything spoofable.
|
||||
*
|
||||
* <p>herdr owns the PID→pane truth: {@code pane.process_info} reports each pane's {@code shell_pid}
|
||||
* and foreground process PIDs. This scans agent panes; a spawn-time {@code pid→terminal} cache is
|
||||
* the obvious optimization once wired into {@code ClaudeCodeLauncher}.
|
||||
* and foreground process PIDs. A pid that is neither of those directly — e.g. a grandchild a
|
||||
* worker spawned, such as a {@code python3} or {@code curl} helper that opens its own MCP
|
||||
* connection — is resolved by walking its ancestry (via {@link ParentResolver}) up to the root and
|
||||
* matching any ancestor against a pane's {@code shell_pid} or foreground pids (CB-161). Without
|
||||
* this walk such a pid matches no pane, and the caller falls through to loopback-trust and is
|
||||
* resolved as the primary — a worker→primary privilege escalation.
|
||||
*
|
||||
* <p>This scans agent panes; a spawn-time {@code pid→terminal} cache is the obvious optimization
|
||||
* once wired into {@code ClaudeCodeLauncher}.
|
||||
*
|
||||
* <p>CB-185 split the fleet across two herdr daemons — lead operations on one, members on the
|
||||
* other ({@code memberHerdrSocket}). A caller's pane can live on <em>either</em> daemon (a lead's
|
||||
* MCP connection resolves against the lead daemon; a member's against the member daemon), so this
|
||||
* must be able to search more than one client. {@link #PaneLocator(HerdrClient, HerdrClient)}
|
||||
* searches the lead client first, then the member client, and collapses to a single scan when the
|
||||
* two are the same object (the historical single-daemon deployment). The caller's ancestor set is
|
||||
* computed once per {@link #terminalForPid} call and reused across every client searched — it
|
||||
* does not depend on which daemon a pane happens to live on.
|
||||
*/
|
||||
public final class PaneLocator {
|
||||
|
||||
private final HerdrClient herdr;
|
||||
/**
|
||||
* Bound on how many ancestor generations {@link #ancestorsOf} walks. This runs on every MCP
|
||||
* call, so a cycle or a pathologically deep process tree must not hang identity resolution;
|
||||
* 32 generations is far more than any real worker→helper process tree needs.
|
||||
*/
|
||||
private static final int MAX_ANCESTRY_DEPTH = 32;
|
||||
|
||||
private final List<HerdrClient> herdrs;
|
||||
private final ParentResolver parentResolver;
|
||||
|
||||
/** Search only this client — the single-daemon deployment. */
|
||||
public PaneLocator(HerdrClient herdr) {
|
||||
this.herdr = herdr;
|
||||
this(herdr, ParentResolver.PROCESS_HANDLE);
|
||||
}
|
||||
|
||||
/** Search only this client, resolving ancestry through {@code parentResolver} — for tests. */
|
||||
public PaneLocator(HerdrClient herdr, ParentResolver parentResolver) {
|
||||
this.herdrs = List.of(herdr);
|
||||
this.parentResolver = parentResolver;
|
||||
}
|
||||
|
||||
/**
|
||||
* Search {@code lead} first, then {@code member} — the two-daemon deployment (CB-185). When
|
||||
* the caller passes the same client for both (no {@code memberHerdrSocket} configured), this
|
||||
* collapses to one client and one scan, exactly {@link #PaneLocator(HerdrClient)}'s behaviour.
|
||||
*/
|
||||
public PaneLocator(HerdrClient lead, HerdrClient member) {
|
||||
this(lead, member, ParentResolver.PROCESS_HANDLE);
|
||||
}
|
||||
|
||||
/** Two-daemon deployment, resolving ancestry through {@code parentResolver} — for tests. */
|
||||
public PaneLocator(HerdrClient lead, HerdrClient member, ParentResolver parentResolver) {
|
||||
this.herdrs = lead == member ? List.of(lead) : List.of(lead, member);
|
||||
this.parentResolver = parentResolver;
|
||||
}
|
||||
|
||||
/**
|
||||
* The {@code terminal_id} of the agent pane whose process tree contains {@code pid}, or
|
||||
* {@code null} if no agent pane owns it (e.g. the caller is the primary, or off-host).
|
||||
* {@code null} if no agent pane on any searched daemon owns it (e.g. the caller is the
|
||||
* primary, or off-host).
|
||||
*/
|
||||
public String terminalForPid(long pid) {
|
||||
if (pid <= 0) {
|
||||
return null;
|
||||
}
|
||||
Set<Long> ancestry = ancestorsOf(pid);
|
||||
for (HerdrClient herdr : herdrs) {
|
||||
String terminal = terminalForPid(herdr, ancestry);
|
||||
if (terminal != null) {
|
||||
return terminal;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code pid} itself plus its ancestor chain, walked through {@link #parentResolver} up to
|
||||
* {@link #MAX_ANCESTRY_DEPTH} generations or pid 1, whichever comes first. A vanished ancestor
|
||||
* ({@link ParentResolver#parentOf} returning empty) ends the walk without error — it just means
|
||||
* the chain is shorter than the bound. A cycle in a fake resolver is caught by the "already
|
||||
* seen" check and also ends the walk, so this can never loop.
|
||||
*/
|
||||
private Set<Long> ancestorsOf(long pid) {
|
||||
Set<Long> ancestry = new LinkedHashSet<>();
|
||||
long current = pid;
|
||||
for (int depth = 0; depth < MAX_ANCESTRY_DEPTH; depth++) {
|
||||
if (current <= 0 || !ancestry.add(current)) {
|
||||
break; // vanished/invalid pid, or a cycle back to a pid already recorded
|
||||
}
|
||||
if (current == 1) {
|
||||
break; // reached the root of the process tree
|
||||
}
|
||||
OptionalLong parent = parentResolver.parentOf(current);
|
||||
if (parent.isEmpty()) {
|
||||
break; // vanished ancestor — not an error, just the end of the chain
|
||||
}
|
||||
current = parent.getAsLong();
|
||||
}
|
||||
return ancestry;
|
||||
}
|
||||
|
||||
private static String terminalForPid(HerdrClient herdr, Set<Long> ancestry) {
|
||||
for (JsonNode pane : herdr.call("pane.list", Map.of()).path("panes")) {
|
||||
String paneId = pane.path("pane_id").asText(null);
|
||||
if (paneId != null && paneOwnsPid(paneId, pid)) {
|
||||
if (paneId != null && paneOwnsAnyOf(herdr, paneId, ancestry)) {
|
||||
return pane.path("terminal_id").asText(null);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
private boolean paneOwnsPid(String paneId, long pid) {
|
||||
private static boolean paneOwnsAnyOf(HerdrClient herdr, String paneId, Set<Long> ancestry) {
|
||||
JsonNode info;
|
||||
try {
|
||||
info = herdr.call("pane.process_info", Map.of("pane_id", paneId)).path("process_info");
|
||||
} catch (HerdrException e) {
|
||||
return false; // pane vanished mid-scan — just skip it
|
||||
}
|
||||
if (info.path("shell_pid").asLong(-1) == pid) {
|
||||
if (ancestry.contains(info.path("shell_pid").asLong(-1))) {
|
||||
return true;
|
||||
}
|
||||
for (JsonNode p : info.path("foreground_processes")) {
|
||||
if (p.path("pid").asLong(-1) == pid) {
|
||||
if (ancestry.contains(p.path("pid").asLong(-1))) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
import java.util.OptionalLong;
|
||||
|
||||
/**
|
||||
* Resolves a pid's parent pid — the seam {@link PaneLocator} walks a process's ancestry through,
|
||||
* so its tests can drive the walk from a fake pid→parent map instead of spawning real processes.
|
||||
*
|
||||
* <p>{@link #PROCESS_HANDLE} is the production implementation, backed by {@link ProcessHandle}.
|
||||
*/
|
||||
public interface ParentResolver {
|
||||
|
||||
/** The parent pid of {@code pid}, or empty if {@code pid} is gone or has no known parent. */
|
||||
OptionalLong parentOf(long pid);
|
||||
|
||||
/** Production resolver: asks the JVM's {@link ProcessHandle} view of the OS process tree. */
|
||||
ParentResolver PROCESS_HANDLE = pid -> ProcessHandle.of(pid)
|
||||
.flatMap(ProcessHandle::parent)
|
||||
.map(parent -> OptionalLong.of(parent.pid()))
|
||||
.orElse(OptionalLong.empty());
|
||||
}
|
||||
@@ -0,0 +1,42 @@
|
||||
package dev.ltms.fleet.herdr;
|
||||
|
||||
/**
|
||||
* The suffix {@code dev.ltms.fleet.lead.LeadLauncher} appends to a lead tab's label the first time a
|
||||
* reconcile finds no running agent in it, before it is sure enough to close the tab outright.
|
||||
*
|
||||
* <p><strong>fleetd #359 review, finding 1.</strong> The daemon's own evidence showed
|
||||
* {@code agent.list} can read "no agent" for a tab that genuinely has one running — so a single such
|
||||
* reading must never be treated as proof a tab is dead. {@code LeadLauncher} now writes this marker
|
||||
* on the first miss, and only closes the tab if a <em>later</em>, independent reconcile still finds
|
||||
* it dead while the marker is still there. Two consecutive misses, one restart apart, is a much
|
||||
* stronger claim than one.
|
||||
*
|
||||
* <p>{@link LeadTabScanner} strips the same suffix before matching a label against a configured
|
||||
* lead's {@code tab}, so a flagged-but-actually-still-live tab keeps resolving normally — the marker
|
||||
* changes nothing about which pane {@code LeadCoordLoop} can reach while the flag is pending. Both
|
||||
* classes must use exactly this suffix, which is why it lives here rather than as a private constant
|
||||
* on either.
|
||||
*/
|
||||
public final class PendingCloseMarker {
|
||||
|
||||
public static final String SUFFIX = " [fleetd:pending-close]";
|
||||
|
||||
private PendingCloseMarker() {
|
||||
}
|
||||
|
||||
/** The label with any trailing pending-close marker removed, for name matching. */
|
||||
public static String strip(String label) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String stripped = label.strip();
|
||||
return stripped.endsWith(SUFFIX)
|
||||
? stripped.substring(0, stripped.length() - SUFFIX.length()).strip()
|
||||
: stripped;
|
||||
}
|
||||
|
||||
/** Whether a label currently carries the marker. */
|
||||
public static boolean isFlagged(String label) {
|
||||
return label != null && label.strip().endsWith(SUFFIX);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* Per-target lookup for a profile's configured backend-error pattern (fleetd#201 / #227): how
|
||||
* {@link CompletionResolver} recognizes a backend that failed a turn outright (a credential
|
||||
* outage, a provider 5xx) from whatever ends up on the member's pane, apart from a genuine
|
||||
* completion.
|
||||
*
|
||||
* <p>The pattern is always profile config, never a vendor string in Java source: every backend
|
||||
* words its failure differently, so a hardcoded sentence would only ever match one of them. A
|
||||
* target with no configured pattern is not "off" the way {@link ExhaustedPatternLookup#none()}
|
||||
* is — {@link CompletionResolver} falls back to its narrow {@code (?i)\bAPI Error\s*:}
|
||||
* compatibility pattern instead, so classification still happens, just without a profile-specific
|
||||
* match. {@link #legacy()} is the explicit stand-in existing (pre-fleetd#201) callers pass to get
|
||||
* exactly that fallback-only behavior until a caller wires a real, profile-driven lookup
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorPatternLookup {
|
||||
|
||||
/** The compiled pattern configured for {@code target}'s profile, or {@code null} if none. */
|
||||
Pattern patternFor(String target);
|
||||
|
||||
/**
|
||||
* Legacy lookup — no profile has a configured pattern, so {@link CompletionResolver} classifies
|
||||
* every target using only its built-in {@code (?i)\bAPI Error\s*:} compatibility pattern. The
|
||||
* explicit stand-in existing constructors pass so their behavior is unchanged until a caller
|
||||
* wires a real, profile-driven lookup.
|
||||
*/
|
||||
static BackendErrorPatternLookup legacy() {
|
||||
return target -> null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,37 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} has a backend-error match at the start of a pane line,
|
||||
* or both a match and its too-fast crash signature, for a waiting send (fleetd#201 / #227). A text
|
||||
* match inside ordinary pane prose can be a member's report about an error, so it fails the send
|
||||
* without notifying this sink.
|
||||
* {@link CompletionResolver} calls this only after {@code Rendezvous.resolveFailure} returns
|
||||
* {@code true} for that exact waiter, mirroring the win-only race rule {@link ExhaustionSink} uses.
|
||||
*
|
||||
* <p>The public send result is unchanged by this classification — it is still a failed send
|
||||
* ({@code Rendezvous.Kind#FAILED}); this sink is the internal seam a later stage (fleetd#201 Unit
|
||||
* 5, the policy that cools a credential off after two such failures in 60 seconds and tells the
|
||||
* lead) consumes. {@link CompletionResolver} knows only {@code target} (a herdr terminal id); it
|
||||
* has no notion of profiles or credentials, so mapping {@code target} to whatever should be
|
||||
* quarantined is entirely the sink's job — exactly like {@link ExhaustionSink}.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface BackendErrorSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified as a backend error
|
||||
* @param matchedLine the single pane line that matched the backend-error pattern
|
||||
* @param reason the full failure reason carried by the classification (the matched line
|
||||
* plus whatever pane context the classifying path attaches)
|
||||
*/
|
||||
void onBackendError(String target, String matchedLine, String reason);
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on a backend-error classification. The explicit stand-in a
|
||||
* caller (or a test not exercising this feature) passes instead of a defaulting overload,
|
||||
* exactly like {@link ExhaustionSink#none()}.
|
||||
*/
|
||||
static BackendErrorSink none() {
|
||||
return (target, matchedLine, reason) -> { };
|
||||
}
|
||||
}
|
||||
@@ -6,12 +6,15 @@ import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.time.Duration;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import java.util.TreeSet;
|
||||
import java.util.concurrent.CompletableFuture;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
@@ -61,13 +64,51 @@ public final class CompletionResolver implements TurnListener {
|
||||
/** Cap the scraped tail so a long transcript can't return an unbounded blob. */
|
||||
static final int MAX_SCRAPE_CHARS = 4000;
|
||||
|
||||
/**
|
||||
* fleetd#164: the floor below which a {@code BUSY -> DONE} transition cannot be real work. A
|
||||
* backend that rejects a turn outright (e.g. an HTTP 400 from the model, before the worker read
|
||||
* a single file or produced a token) drives the exact same confirmed {@code working -> idle}
|
||||
* transition a genuine completion does — just in about a second instead of the many seconds a
|
||||
* real turn costs. {@link #onTurnComplete} cannot tell those two cases apart from the transition
|
||||
* alone, so a turn that settles inside this floor is treated as a crash signature and resolved
|
||||
* as a failure, never as a (possibly empty) success.
|
||||
*/
|
||||
public static final long MIN_TURN_NANOS = Duration.ofSeconds(2).toNanos();
|
||||
|
||||
/**
|
||||
* fleetd#164 (part 2): one stable, explicit backend-failure marker seen on a Claude Code pane
|
||||
* when the backend itself rejected the turn (e.g. {@code "API Error: 400 invalid request body"}).
|
||||
* Kept deliberately narrow — a growing list of ad-hoc error strings rots as backends change their
|
||||
* wording; broader backend-error surfacing is out of scope here (fleetd#164 point 3).
|
||||
*/
|
||||
private static final Pattern BACKEND_ERROR = Pattern.compile("(?i)\\bAPI Error\\s*:");
|
||||
|
||||
private static final String CLIPPED_PANE_TAIL_MARKER =
|
||||
"[Pane tail clipped: member did not call fleet_reply.]";
|
||||
|
||||
/** A pane echo must be this large before it can replace a completion report. */
|
||||
static final int ECHO_MIN_CHARS = 400;
|
||||
|
||||
/** Normalised TUI chrome may add this many characters to an otherwise echoed brief. */
|
||||
static final int MAX_ECHO_EXCESS_CHARS = 160;
|
||||
|
||||
/** The explicit result returned instead of a lead's echoed injected brief. */
|
||||
public static final String NO_REPORT_PREFIX = "[no report — the member ended its turn without fleet_reply, "
|
||||
+ "and the pane still shows the injected brief. Nothing was produced on the pane. Check the "
|
||||
+ "member's worktree and branch for committed work before re-delegating.";
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Rendezvous rendezvous;
|
||||
private final ExhaustedPatternLookup exhaustedPatterns;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
private final BackendErrorPatternLookup backendErrorPatterns;
|
||||
private final BackendErrorSink backendErrorSink;
|
||||
private final LongSupplier nowNanos;
|
||||
private final Function<String, WorktreeBranch> worktreeBranches;
|
||||
|
||||
/** Known member location, used only to guide a lead after an echoed brief. */
|
||||
public record WorktreeBranch(String worktree, String branch) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Per-target record of the turn currently in flight: the exact {@link Rendezvous} waiter its
|
||||
@@ -79,8 +120,26 @@ public final class CompletionResolver implements TurnListener {
|
||||
* reference: a completion scrape equal to it means the worker produced no new output (the previous
|
||||
* turn's wind-down sampled as this boundary), so it is suppressed. Overwritten on each delivery;
|
||||
* cleared when the turn resolves. Package-private so tests can capture and replay a specific turn.
|
||||
*
|
||||
* <p>{@code deliveredAtNanos} (fleetd#164) is the {@link #nowNanos} reading taken at delivery —
|
||||
* the other half of the {@link #MIN_TURN_NANOS} floor check, compared against a fresh reading at
|
||||
* resolution time.
|
||||
*/
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
record InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos,
|
||||
String injectedText) {
|
||||
|
||||
/**
|
||||
* Convenience for tests exercising scrape/suppression logic that don't care about turn
|
||||
* timing: back-dates the delivery far enough that {@link #MIN_TURN_NANOS} can never fire.
|
||||
* Not used by production code — {@link #captureBaseline} always records a real reading.
|
||||
*/
|
||||
InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline) {
|
||||
this(waiter, baseline, Long.MIN_VALUE / 2, null);
|
||||
}
|
||||
|
||||
InFlight(CompletableFuture<Rendezvous.Resolution> waiter, String baseline, long deliveredAtNanos) {
|
||||
this(waiter, baseline, deliveredAtNanos, null);
|
||||
}
|
||||
}
|
||||
|
||||
private final ConcurrentHashMap<String, InFlight> inFlight = new ConcurrentHashMap<>();
|
||||
@@ -94,13 +153,86 @@ public final class CompletionResolver implements TurnListener {
|
||||
* classification actually resolves a waiter. Required for the same
|
||||
* reason as {@code exhaustedPatterns} — pass {@link ExhaustionSink#none()}
|
||||
* to opt out.
|
||||
*
|
||||
* <p>Transition constructor (fleetd#201 Unit 1): delegates to the full constructor below with
|
||||
* {@link BackendErrorPatternLookup#legacy()} and {@link BackendErrorSink#none()}, so every
|
||||
* existing caller keeps today's behavior — the narrow built-in {@code API Error:} match, no
|
||||
* sink notified — until a caller wires a real, profile-driven backend-error lookup and sink
|
||||
* (fleetd#201 Unit 5).
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime);
|
||||
}
|
||||
|
||||
/** Production constructor with a lookup for the member worktree and branch. */
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, Function<String, WorktreeBranch> worktreeBranches) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), System::nanoTime, worktreeBranches);
|
||||
}
|
||||
|
||||
/**
|
||||
* Transition constructor (fleetd#201 Unit 1): same legacy backend-error defaults as the 4-arg
|
||||
* constructor above, but with the injectable clock. Kept so existing fleetd#164 timing tests
|
||||
* (and any other caller that wants a controllable clock but not the backend-error feature)
|
||||
* keep compiling unchanged.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink,
|
||||
BackendErrorPatternLookup.legacy(), BackendErrorSink.none(), nowNanos);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor (fleetd#201 Unit 1): adds the target-keyed backend-error pattern
|
||||
* lookup and sink alongside the existing exhaustion pair. Required, like {@code exhaustedPatterns}
|
||||
* and {@code exhaustionSink} — pass {@link BackendErrorPatternLookup#legacy()} and
|
||||
* {@link BackendErrorSink#none()} to opt out.
|
||||
*
|
||||
* @param backendErrorPatterns fleetd#201: per-target lookup for a profile's configured
|
||||
* backend-error pattern; a target with none configured is matched
|
||||
* against the built-in compatibility pattern instead (never "off").
|
||||
* @param backendErrorSink fleetd#201: notified when a backend-error classification actually
|
||||
* resolves a waiter — never on a race that lost.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
System::nanoTime, _ -> null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Test constructor with an injectable clock (fleetd#164), matching the {@code LongSupplier}
|
||||
* pattern {@link dev.ltms.fleet.session.SessionManager} and {@link dev.ltms.fleet.msg.MessageService}
|
||||
* already use: lets a test place a turn's delivery and its resolution at an exact, controllable
|
||||
* distance apart around the {@link #MIN_TURN_NANOS} floor, without a real sleep. Public (rather
|
||||
* than package-private like those two) because callers that wire a full {@code MessageService}
|
||||
* fixture — e.g. {@code MessageServiceTest} — construct this resolver directly from another
|
||||
* package.
|
||||
*/
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos) {
|
||||
this(agents, rendezvous, exhaustedPatterns, exhaustionSink, backendErrorPatterns, backendErrorSink,
|
||||
nowNanos, _ -> null);
|
||||
}
|
||||
|
||||
/** Full constructor with injectable clock and member location lookup. */
|
||||
public CompletionResolver(AgentControl agents, Rendezvous rendezvous, ExhaustedPatternLookup exhaustedPatterns,
|
||||
ExhaustionSink exhaustionSink, BackendErrorPatternLookup backendErrorPatterns,
|
||||
BackendErrorSink backendErrorSink, LongSupplier nowNanos,
|
||||
Function<String, WorktreeBranch> worktreeBranches) {
|
||||
this.agents = agents;
|
||||
this.rendezvous = rendezvous;
|
||||
this.exhaustedPatterns = Objects.requireNonNull(exhaustedPatterns, "exhaustedPatterns");
|
||||
this.exhaustionSink = Objects.requireNonNull(exhaustionSink, "exhaustionSink");
|
||||
this.backendErrorPatterns = Objects.requireNonNull(backendErrorPatterns, "backendErrorPatterns");
|
||||
this.backendErrorSink = Objects.requireNonNull(backendErrorSink, "backendErrorSink");
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
this.worktreeBranches = Objects.requireNonNull(worktreeBranches, "worktreeBranches");
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -130,7 +262,7 @@ public final class CompletionResolver implements TurnListener {
|
||||
baseline = null; // fail open: no baseline ⇒ no suppression
|
||||
log.debug("delivery baseline for {} failed: {}", target, e.getMessage());
|
||||
}
|
||||
inFlight.put(target, new InFlight(waiter, baseline));
|
||||
inFlight.put(target, new InFlight(waiter, baseline, nowNanos.getAsLong(), token.injectedText()));
|
||||
}
|
||||
|
||||
/** The turn currently baselined for {@code target}, or {@code null} — a test hook for the captureBaseline path. */
|
||||
@@ -176,31 +308,61 @@ public final class CompletionResolver implements TurnListener {
|
||||
inFlight.remove(target, turn);
|
||||
return;
|
||||
}
|
||||
// fleetd#164: a BUSY -> DONE transition inside the floor cannot be real work — it's a crash
|
||||
// signature (e.g. a backend HTTP 400 before the worker did anything), not a fast answer. Fail
|
||||
// it before spending a scrape on the ordinary path; the reason still carries whatever is on
|
||||
// screen, since that is usually the backend's own error.
|
||||
long elapsedNanos = nowNanos.getAsLong() - turn.deliveredAtNanos();
|
||||
if (elapsedNanos < MIN_TURN_NANOS) {
|
||||
failTooFast(target, turn, waiter, elapsedNanos);
|
||||
return;
|
||||
}
|
||||
String tail;
|
||||
String assistantBlock = null;
|
||||
String rawScrape = null;
|
||||
int originalLength = 0;
|
||||
boolean clipped = false;
|
||||
boolean scrapeFailed = false;
|
||||
try {
|
||||
assistantBlock = lastAssistantBlock(agents.read(target, SCRAPE_SOURCE));
|
||||
rawScrape = agents.read(target, SCRAPE_SOURCE);
|
||||
assistantBlock = lastAssistantBlock(rawScrape);
|
||||
originalLength = assistantBlock.strip().length();
|
||||
clipped = originalLength > MAX_SCRAPE_CHARS;
|
||||
tail = clip(assistantBlock);
|
||||
} catch (RuntimeException e) {
|
||||
// The worker finished but we couldn't read its screen — still resolve the send so the
|
||||
// caller unblocks; an empty tail beats hanging until the caller's timeout.
|
||||
log.warn("completion scrape for {} failed; resolving with an empty tail: {}",
|
||||
target, e.getMessage());
|
||||
log.warn("completion scrape for {} failed: {}", target, e.getMessage());
|
||||
tail = "";
|
||||
scrapeFailed = true;
|
||||
}
|
||||
// fleetd#164: a scrape nobody could read, and a scrape that read cleanly but produced nothing,
|
||||
// both used to resolve the send as a SUCCESS carrying "" — indistinguishable from a worker that
|
||||
// genuinely finished with nothing to say. That is the defect: fail loudly instead, naming the
|
||||
// member, so a caller (including a lead deciding whether to delegate again) can tell a lost
|
||||
// turn from a real empty answer.
|
||||
if (scrapeFailed || tail.isEmpty()) {
|
||||
// fleetd#211: lastAssistantBlock() found nothing usable — most often a pane with no ⏺
|
||||
// marker at all, whose boundary scan then starts at the top of the raw screen and breaks
|
||||
// immediately on the first line of TUI chrome (╭, │, ❯, …). Before giving up as a lost
|
||||
// turn, run the same exhaustion/backend-error classification against the RAW scrape as a
|
||||
// fallback, ONLY here. A pane that already yielded a usable block never reaches this
|
||||
// branch, so the narrow (trimmed) match on the normal path below is completely unchanged
|
||||
// — zero new false positives there. Every pane this fallback examines was already headed
|
||||
// for the empty-scrape failure, so a wrong label here is strictly less bad than silently
|
||||
// losing an exhaustion signal: the alternative outcome is already a failure, just one that
|
||||
// never quarantines the credential. lastAssistantBlock stays the source of the reply
|
||||
// TEXT everywhere else; only classification ever consults the raw scrape, and only here.
|
||||
if (rawScrape != null && classifyRawScrapeFallback(target, turn, waiter, rawScrape)) {
|
||||
return;
|
||||
}
|
||||
fail(target, turn, emptyScrapeReason(target, scrapeFailed));
|
||||
return;
|
||||
}
|
||||
// Misattribution guard (CB-115): if the scrape is byte-identical to the pane content at
|
||||
// delivery, this turn produced no new output — the boundary belongs to the previous turn's
|
||||
// wind-down (common on rapid back-to-back sends). Suppress rather than resolve the send with
|
||||
// a stale answer; the real fleet_reply (or a later genuine completion) resolves it instead.
|
||||
// A scrape that failed to read is exempt — an empty tail there is "couldn't see", not "no change".
|
||||
String baseline = turn.baseline();
|
||||
if (!scrapeFailed && baseline != null && baseline.equals(tail)) {
|
||||
if (baseline != null && baseline.equals(tail)) {
|
||||
log.debug("suppressing misattributed completion for {} (no output change since delivery)",
|
||||
target);
|
||||
return; // keep the in-flight record: a later genuine completion still needs it
|
||||
@@ -208,23 +370,50 @@ public final class CompletionResolver implements TurnListener {
|
||||
// CB-578 stage A: a turn that ended with no fleet_reply AND whose scrape matches the
|
||||
// backend's configured usage-limit pattern is a refusal, not an answer. Classify it as
|
||||
// BACKEND_EXHAUSTED rather than handing the caller a scrape that reads like a real reply.
|
||||
if (!scrapeFailed) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(assistantBlock, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED (no fleet_reply; scrape "
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(assistantBlock, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED (no fleet_reply; scrape "
|
||||
+ "matched the profile's exhausted pattern): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
if (startsWithExhaustion(matchedLine, exhausted)) {
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#164 (part 2) / fleetd#201: a scrape that read cleanly and produced content still
|
||||
// isn't a real reply when that content is the backend's own rejection (e.g. an HTTP 400
|
||||
// before the worker did any work). Classify it as a failure naming the member, rather than
|
||||
// handing the caller a scrape that reads like a completed answer. A text match alone is not
|
||||
// enough to notify the typed backend-error sink: this assistant block can be a member's
|
||||
// normal prose about an error. A line that starts with the error match is stronger evidence;
|
||||
// the too-fast path below also has its crash signature before it records a credential failure.
|
||||
String backendError = firstMatchingLine(assistantBlock, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the whole scrape, not just the matched line. The pattern is a heuristic: a member
|
||||
// that forgot fleet_reply while reporting *about* a backend error matches it too. Failing
|
||||
// is still right — the caller must not read a scrape as an answer — but dropping the rest
|
||||
// of the pane would destroy the report, which is the same defect fleetd#164 is about.
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + tail;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
if (startsWithBackendError(backendError, backendErrorPatternOrFallback(target))) {
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
String completion = clipped ? tail + "\n" + CLIPPED_PANE_TAIL_MARKER : tail;
|
||||
if (echoesInjectedBrief(tail, turn.injectedText())) {
|
||||
completion = noReportMessage(target) + (clipped ? "\n" + CLIPPED_PANE_TAIL_MARKER : "");
|
||||
}
|
||||
if (rendezvous.resolveCompletion(waiter, completion)) {
|
||||
inFlight.remove(target, turn);
|
||||
if (clipped) {
|
||||
@@ -237,6 +426,86 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* A full echoed brief is at least 400 normalised characters. A scrape that contains the brief may
|
||||
* add no more than 160 normalised characters of TUI chrome. This accepts harmless status text, but
|
||||
* preserves a real report that restates the full brief before adding substantive content.
|
||||
*/
|
||||
static boolean echoesInjectedBrief(String scrape, String injectedText) {
|
||||
String normalScrape = normalize(scrape);
|
||||
String normalInjected = normalize(injectedText);
|
||||
if (normalScrape.length() < ECHO_MIN_CHARS || normalInjected.length() < ECHO_MIN_CHARS) {
|
||||
return false;
|
||||
}
|
||||
if (normalInjected.contains(normalScrape)) {
|
||||
return true;
|
||||
}
|
||||
return normalScrape.contains(normalInjected)
|
||||
&& normalScrape.length() - normalInjected.length() <= MAX_ECHO_EXCESS_CHARS;
|
||||
}
|
||||
|
||||
private static String normalize(String text) {
|
||||
return text == null ? "" : text.toLowerCase().replaceAll("[^a-z0-9]+", "");
|
||||
}
|
||||
|
||||
private String noReportMessage(String target) {
|
||||
WorktreeBranch location = worktreeBranches.apply(target);
|
||||
if (location == null || (location.worktree() == null && location.branch() == null)) {
|
||||
return NO_REPORT_PREFIX + "]";
|
||||
}
|
||||
String locationText = location.worktree() == null ? "" : " worktree=" + location.worktree();
|
||||
if (location.branch() != null) {
|
||||
locationText += " branch=" + location.branch();
|
||||
}
|
||||
return NO_REPORT_PREFIX + locationText + "]";
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#211: the raw-scrape fallback classification, run only when {@link #lastAssistantBlock}
|
||||
* found nothing usable (see the call site in {@link #resolve}). Mirrors the two classifications
|
||||
* the normal path already applies to the trimmed assistant block — exhaustion first, then the
|
||||
* narrow {@link #BACKEND_ERROR} pattern — against {@code raw} instead, and reports whether one of
|
||||
* them handled the turn (resolved the waiter or failed it) so the caller skips the empty-scrape
|
||||
* failure. Never runs on the normal (non-empty-block) path, and never touches the reply text.
|
||||
*/
|
||||
private boolean classifyRawScrapeFallback(String target, InFlight turn,
|
||||
CompletableFuture<Rendezvous.Resolution> waiter, String raw) {
|
||||
Pattern exhausted = exhaustedPatterns.patternFor(target);
|
||||
String matchedLine = exhausted == null ? null : firstMatchingLine(raw, exhausted);
|
||||
if (matchedLine != null) {
|
||||
String reason = "backend exhausted (usage limit): " + matchedLine;
|
||||
if (rendezvous.resolveExhausted(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("completion for {} classified BACKEND_EXHAUSTED from the raw scrape (no "
|
||||
+ "usable assistant block; no fleet_reply): {}", target, reason);
|
||||
// CB-578 stage B: only on the resolution that actually won the race — a late
|
||||
// duplicate must never quarantine a credential twice for one refusal.
|
||||
if (startsWithExhaustion(matchedLine, exhausted)) {
|
||||
exhaustionSink.onExhausted(target, reason);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
String backendError = firstMatchingLine(raw, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
// Carry the pane, not just the matched line — the same fleetd#164 rule the normal path
|
||||
// above applies. Here it matters more, not less: the trimmed block was empty, so the raw
|
||||
// scrape is the ONLY copy of whatever the member managed to say. Clipped to the same cap
|
||||
// the normal path uses, since a raw screen has no boundary trimming to bound it.
|
||||
String reason = "member " + target + " ended on a backend error: " + backendError
|
||||
+ "\n--- pane tail ---\n" + clip(raw);
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback from the raw scrape: {}", target, reason);
|
||||
if (startsWithBackendError(backendError, backendErrorPatternOrFallback(target))) {
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Synchronous fail (the unit-testable core of {@link #onTurnFailed}). */
|
||||
void fail(String target, InFlight turn) {
|
||||
fail(target, turn, null);
|
||||
@@ -272,6 +541,85 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164 (floor) / fleetd#201 (classification): fail a turn that settled inside
|
||||
* {@link #MIN_TURN_NANOS} — a crash signature (e.g. a backend HTTP 400 before the worker did
|
||||
* anything) that a bare {@code BUSY -> DONE} transition cannot be told apart from a genuinely
|
||||
* fast completion. Runs the same backend-error classification the normal and raw-scrape paths
|
||||
* apply, against whatever is on screen right now: a match together with the too-fast crash
|
||||
* signature notifies {@link #backendErrorSink} (only on the resolution that wins the race). A
|
||||
* non-match stays the generic too-fast failure, naming the member and both timings, with
|
||||
* whatever the pane shows appended so the caller sees the evidence, not just "it failed".
|
||||
*
|
||||
* <p>fleetd#376: <strong>this path must never resolve a completion.</strong> A fast backend can
|
||||
* genuinely answer inside the floor, so the failure is sometimes wrong — but it is wrong in the
|
||||
* loud direction, and the fix for that is honest wording, not a guess at the pane's meaning.
|
||||
* Reclassifying from the scrape was tried and rejected: there is no reliable positive marker for
|
||||
* "this is a real reply" across backends. {@link #lastAssistantBlock} falls back to the entire
|
||||
* pane when it finds no {@code ⏺} marker, so on a crash the candidate "reply" is the whole
|
||||
* screen; and {@code ⏺} itself is a Claude Code marker that an opencode pane never carries — the
|
||||
* very backend whose speed raised this ticket. Any weaker test (non-blank, or "contains sentence
|
||||
* punctuation") passes on almost every crash, because a pane holding a file path, a version
|
||||
* number or a hostname contains a full stop. That trades a loud wrong answer for a silent one.
|
||||
*/
|
||||
private void failTooFast(String target, InFlight turn, CompletableFuture<Rendezvous.Resolution> waiter,
|
||||
long elapsedNanos) {
|
||||
String scrape;
|
||||
try {
|
||||
scrape = agents.read(target, SCRAPE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
scrape = "";
|
||||
}
|
||||
String clippedScrape = clip(scrape);
|
||||
String timing = String.format(
|
||||
"member %s went BUSY -> DONE in %dms (floor %dms)",
|
||||
target, elapsedNanos / 1_000_000, MIN_TURN_NANOS / 1_000_000);
|
||||
String backendError = firstMatchingLine(scrape, backendErrorPatternOrFallback(target));
|
||||
if (backendError != null) {
|
||||
String reason = timing + " — too fast to be real work, and the pane carries a backend "
|
||||
+ "error: " + clippedScrape;
|
||||
if (rendezvous.resolveFailure(waiter, reason)) {
|
||||
inFlight.remove(target, turn);
|
||||
log.warn("failing send to {} via turn-stall fallback: {}", target, reason);
|
||||
// fleetd#201 Unit 1: only on the resolution that actually won the race.
|
||||
backendErrorSink.onBackendError(target, backendError, reason);
|
||||
}
|
||||
return;
|
||||
}
|
||||
// fleetd#376: no pattern matched, so the cause is genuinely unknown. Say that, rather than
|
||||
// asserting a backend error the way this message used to — a fast backend really can finish
|
||||
// inside the floor, and a reader who trusts a wrong cause stops looking at the pane.
|
||||
String reason = timing + " — inside the floor. That is usually a backend error before any "
|
||||
+ "work started, but a fast backend can answer inside it too, and nothing here can "
|
||||
+ "tell those apart, so the turn is reported failed rather than guessed. Read the "
|
||||
+ "pane below before deciding which it was";
|
||||
fail(target, turn, clippedScrape.isBlank() ? reason : reason + ": " + clippedScrape);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#201: the pattern to classify a backend error against for {@code target} — its
|
||||
* profile's configured {@link BackendErrorPatternLookup} entry when there is one, else the
|
||||
* built-in {@link #BACKEND_ERROR} compatibility pattern. A target with no configured pattern is
|
||||
* never "off": it always falls back to this narrow default, exactly like the classification
|
||||
* behaved before fleetd#201 (Unit 5 reports a target relying on this fallback separately, as
|
||||
* legacy-default coverage rather than full per-profile coverage).
|
||||
*/
|
||||
private Pattern backendErrorPatternOrFallback(String target) {
|
||||
Pattern configured = backendErrorPatterns.patternFor(target);
|
||||
return configured != null ? configured : BACKEND_ERROR;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd#164: the failure reason for a scrape that produced zero characters — names the member
|
||||
* and says plainly that the turn produced nothing, so a caller (a lead deciding whether to
|
||||
* delegate again included) never mistakes a lost turn for a genuinely empty reply.
|
||||
*/
|
||||
private static String emptyScrapeReason(String target, boolean scrapeFailed) {
|
||||
return "member " + target + " turn completed with an empty scrape (0 chars) — "
|
||||
+ (scrapeFailed ? "its pane could not be read; " : "")
|
||||
+ "treating as a lost turn, not a real answer";
|
||||
}
|
||||
|
||||
/**
|
||||
* The first line of {@code text} matching {@code pattern}, stripped — the CB-578 stage A
|
||||
* evidence carried in a {@code BACKEND_EXHAUSTED} reason so the operator sees the real refusal
|
||||
@@ -288,17 +636,119 @@ public final class CompletionResolver implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for the CB-578 stage A exhausted-pattern classification, logged at startup
|
||||
* True when nothing before the match on this pane line ends a sentence — that is, the match is
|
||||
* still inside the line's first sentence rather than inside prose a member wrote about it.
|
||||
* Used to decide whether an exhaustion match may quarantine a credential (fleetd #348).
|
||||
*
|
||||
* <p><strong>Why this is looser than {@link #startsWithBackendError}.</strong> An
|
||||
* {@code exhaustedPattern} is written per profile and may name only the decisive words of a
|
||||
* provider message — {@code "usage limit has been reached"} without its leading {@code "The"}.
|
||||
* A start-of-line check would then reject the genuine refusal. That is the false negative
|
||||
* fleetd #348's invariant 1 calls the worse direction: an unrecorded exhaustion leaves the
|
||||
* fleet spawning into a credential with no capacity, and a quarantine runs 1800s against the
|
||||
* backend-error cooldown's fixed 60s.
|
||||
*
|
||||
* <p>This rule accepts a superset of what a start-of-line check accepts: if the match begins
|
||||
* right after the chrome, there is nothing in front of it, so there is no sentence ending
|
||||
* either. So moving to it cannot add a false negative.
|
||||
*
|
||||
* <p><strong>No chrome skipping here, deliberately.</strong> The first version of this method
|
||||
* copied {@code startsWithBackendError}'s leading-chrome loop. Measured on merge: deleting that
|
||||
* loop left all 1369 tests green, and it must — the scan only looks for {@code . ! ?}, and no
|
||||
* terminal chrome character is one of those. A step that cannot change the result is worse than
|
||||
* no step, because the next reader takes it as evidence that chrome was handled.
|
||||
*
|
||||
* <p>It stays a heuristic. Prose whose <em>first</em> sentence carries the pattern still
|
||||
* notifies the sink, and a genuine refusal behind an earlier full stop (a hostname, a version
|
||||
* number) still does not. Both are known and neither is fixed here.
|
||||
*/
|
||||
private static boolean startsWithExhaustion(String line, Pattern pattern) {
|
||||
var matcher = pattern.matcher(line);
|
||||
if (!matcher.find()) {
|
||||
return false;
|
||||
}
|
||||
for (int prefix = 0; prefix < matcher.start(); prefix++) {
|
||||
if (".!?".indexOf(line.charAt(prefix)) >= 0) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when the error pattern begins the matched pane line, rather than appearing in prose.
|
||||
*
|
||||
* <p>Leading terminal chrome is skipped first — box-drawing characters, bullets, gutter bars and
|
||||
* spaces. #339 introduced this check with a bare {@code lookingAt}, and that rejected a genuine
|
||||
* error line rendered as {@code "| 503 Service Unavailable: ..."}: the send still failed, but the
|
||||
* credential outage was never recorded. That is the false negative #339's own invariant 3 called
|
||||
* worse than the false positive it set out to fix — measured with a throwaway probe on the
|
||||
* raw-scrape path, which is exactly the path whose comment says to expect leading chrome.
|
||||
*
|
||||
* <p>Skipping only a leading run of non-letter, non-digit characters keeps the fix's intent. A
|
||||
* member's prose ({@code "I checked the retry path. An API Error: makes it back off."}) still
|
||||
* does not match, because there the pattern sits after words, not after chrome.
|
||||
*/
|
||||
private static boolean startsWithBackendError(String line, Pattern pattern) {
|
||||
int i = 0;
|
||||
while (i < line.length() && !Character.isLetterOrDigit(line.charAt(i))) {
|
||||
i++;
|
||||
}
|
||||
return pattern.matcher(line.substring(i)).lookingAt();
|
||||
}
|
||||
|
||||
/**
|
||||
* What an unset pattern key means for the classification it configures (fleetd#415).
|
||||
* {@code coverage()} cannot infer this from the key's name — the two keys it currently
|
||||
* describes disagree on it, and a string comparison on the name would just move the same bug
|
||||
* to a new spot — so every caller must state it explicitly.
|
||||
*
|
||||
* <p><strong>This alone does not prove a caller passes the right one for its key.</strong> A
|
||||
* test that calls {@code coverage()} directly and supplies the meaning itself only proves this
|
||||
* enum is worded correctly, never that {@code Fleetd}'s two call sites pair each key with its
|
||||
* true meaning — that pairing is #415's actual defect. Measured on review: swapping the two
|
||||
* {@code UnsetMeaning} arguments at those call sites (giving {@code exhaustedPattern} the
|
||||
* built-in-default wording and {@code errorPattern} the off wording — #415's exact defect with
|
||||
* the keys exchanged) compiled with 0 errors and left all 1506 existing tests green. See
|
||||
* {@code dev.ltms.fleet.Fleetd#exhaustedPatternCoverageLine}/{@code #errorPatternCoverageLine}
|
||||
* and {@code FleetdPatternCoverageLineTest}, which exists specifically to catch that swap.
|
||||
*/
|
||||
public enum UnsetMeaning {
|
||||
/** No fallback exists: a profile with no configured pattern truly has this classification off. */
|
||||
OFF,
|
||||
/** A built-in pattern applies when unset: the classification still runs for that profile. */
|
||||
BUILT_IN_DEFAULT
|
||||
}
|
||||
|
||||
/**
|
||||
* Coverage summary for a fleetd#201/CB-578-style pattern-key classification, logged at startup
|
||||
* the way {@link dev.ltms.fleet.health.FleetHealthMonitor#coverage} is — so an operator can
|
||||
* see whether the classification is on, and for which profiles, without reading every
|
||||
* profile's config by hand.
|
||||
*
|
||||
* <p>fleetd#415: this method measures <em>pattern coverage</em> — how many profiles set the
|
||||
* key — which is not the same thing as <em>feature state</em> for a key with a fallback. For
|
||||
* {@code errorPattern}, an empty {@code configuredProfiles} still runs the classification
|
||||
* against {@code CompletionResolver}'s built-in compatibility pattern ({@link #BACKEND_ERROR}
|
||||
* at line ~84); for {@code exhaustedPattern} there is no fallback, so empty really does mean
|
||||
* off. {@code unsetMeaning} is the single, required source of that fact — see
|
||||
* {@link dev.ltms.fleet.config.FleetConfig#rejectMalformedProfilePatterns} lines ~2029-2032 for
|
||||
* where it is documented for config authors. It is a required parameter, not a defaulted
|
||||
* overload: a third pattern key added later must supply one to compile at all, rather than
|
||||
* silently inheriting whichever wording this method happened to default to.
|
||||
*
|
||||
* @param allProfiles every configured profile name
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry an exhausted pattern
|
||||
* @param configuredProfiles the subset of {@code allProfiles} that carry the pattern
|
||||
*/
|
||||
public static String coverage(Set<String> allProfiles, Set<String> configuredProfiles) {
|
||||
public static String coverage(String patternKey, UnsetMeaning unsetMeaning, Set<String> allProfiles,
|
||||
Set<String> configuredProfiles) {
|
||||
if (configuredProfiles.isEmpty()) {
|
||||
return "off (no profile has an exhaustedPattern configured; profiles: " + sorted(allProfiles) + ")";
|
||||
return switch (unsetMeaning) {
|
||||
case OFF -> "off (no profile has an " + patternKey + " configured; profiles: "
|
||||
+ sorted(allProfiles) + ")";
|
||||
case BUILT_IN_DEFAULT -> "built-in default for all profiles (no profile customises "
|
||||
+ patternKey + "; profiles: " + sorted(allProfiles) + ")";
|
||||
};
|
||||
}
|
||||
Set<String> unconfigured = new TreeSet<>(allProfiles);
|
||||
unconfigured.removeAll(configuredProfiles);
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* Notified when {@link CompletionResolver} actually delivers a {@code BACKEND_EXHAUSTED}
|
||||
* classification to a waiting send (CB-578 stage B) — never on a race that lost (see
|
||||
@@ -10,22 +12,81 @@ package dev.ltms.fleet.inject;
|
||||
* profiles or credentials, so mapping {@code target} to whatever should be quarantined is entirely
|
||||
* the sink's job — see {@code Fleetd.main}'s wiring, which resolves target → session → profile →
|
||||
* {@code effectiveCredentialId()} and calls {@code BackendQuarantine.quarantine} on it.
|
||||
*
|
||||
* <p><strong>The 3-arg overload is the single abstract method — fleetd #234, round 4.</strong> Two
|
||||
* earlier rounds each shipped a caller that silently dropped the profile hint (see {@link
|
||||
* #onExhausted(String, String, String)}): a lambda written against this interface can only ever
|
||||
* implement whichever overload is abstract, and while the 2-arg form held that position, EVERY
|
||||
* lambda site — a call-site forwarder in {@code Fleetd.java}, a hand-built test double — bound to
|
||||
* it and silently inherited the profile-dropping default, whether or not its author remembered the
|
||||
* hazard. Making the 3-arg form abstract instead removes the shape entirely: a lambda declared
|
||||
* against this interface today is <em>forced</em> by the compiler to take {@code (target, reason,
|
||||
* profile)}, so there is no overload left for it to bind to that can drop the hint. This is a type
|
||||
* change, not a test — it holds even for a caller that has never heard of fleetd #234.
|
||||
*/
|
||||
@FunctionalInterface
|
||||
public interface ExhaustionSink {
|
||||
|
||||
/**
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
* @param target the herdr terminal id whose turn was classified {@code BACKEND_EXHAUSTED}
|
||||
* @param reason the matched-line reason carried by the classification
|
||||
* @param profile the profile the caller already knows should be quarantined, or {@code null}
|
||||
* when the caller has no better answer than {@code target} alone (fleetd #234):
|
||||
* a caller whose {@code target} is not yet resolvable through whatever roster
|
||||
* the sink's implementation consults — {@link
|
||||
* dev.ltms.fleet.member.OpenCodeLauncher}'s model-mismatch check fires from
|
||||
* {@code SessionAwareHandle.agentSessionId()}, which runs during {@code
|
||||
* SessionManager.acquire()} <em>before</em> that session is registered, so a
|
||||
* target -> session -> profile lookup finds nothing at that point. That launcher
|
||||
* already has its own {@code FleetConfig.Profile} in hand and does not need the
|
||||
* roster to know which profile to quarantine, so it supplies this directly
|
||||
* instead of leaving the sink to guess.
|
||||
*/
|
||||
void onExhausted(String target, String reason);
|
||||
void onExhausted(String target, String reason, String profile);
|
||||
|
||||
/**
|
||||
* Convenience for a caller with no profile to offer — every existing call site that predates
|
||||
* the hint (fleetd #234): {@link dev.ltms.fleet.inject.CompletionResolver}'s two call sites
|
||||
* always call with a {@code target} that IS live in the roster at the time of the call, so they
|
||||
* need no hint and keep working exactly as before, unchanged by this default.
|
||||
*
|
||||
* @param target as {@link #onExhausted(String, String, String)}
|
||||
* @param reason as {@link #onExhausted(String, String, String)}
|
||||
*/
|
||||
default void onExhausted(String target, String reason) {
|
||||
onExhausted(target, reason, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert sink — nothing happens on exhaustion. The explicit stand-in a caller (or a test not
|
||||
* exercising this feature) passes instead of a defaulting overload, exactly like
|
||||
* {@link ExhaustedPatternLookup#none()}.
|
||||
* {@link ExhaustedPatternLookup#none()}. Safe as a lambda: the 3-arg form is now the interface's
|
||||
* single abstract method, so a lambda here has no other overload to silently bind to instead —
|
||||
* it simply does nothing with all three arguments.
|
||||
*/
|
||||
static ExhaustionSink none() {
|
||||
return (target, reason) -> { };
|
||||
return (target, reason, profile) -> { };
|
||||
}
|
||||
|
||||
/**
|
||||
* A sink that forwards to whatever {@code target} currently supplies (fleetd #234). Exists to
|
||||
* break a genuine construction-order cycle: {@code Fleetd.main} builds its adapters (including
|
||||
* {@link dev.ltms.fleet.member.OpenCodeLauncher}) before {@code sessions} exists, so it cannot
|
||||
* hand them the real sink yet — it hands them a forwarder pointed at an {@code
|
||||
* AtomicReference<ExhaustionSink>} that starts at {@link #none()} and gets {@code .set()} to the
|
||||
* real sink once {@code sessions} is built. {@code target} is evaluated on every call, never
|
||||
* cached, so the forwarder keeps working after the reference is repointed.
|
||||
*
|
||||
* <p>Now safe as a one-line lambda (round 4): forwarding the single 3-arg abstract method
|
||||
* forwards everything a caller can supply — there is no separate 2-arg overload left for a
|
||||
* forwarder to bind to instead and silently lose the hint. Kept as a named factory rather than
|
||||
* written inline at each call site anyway, so a test can call the exact object {@code
|
||||
* Fleetd.java} builds instead of asserting a rebuilt copy of its shape (round 3's lesson).
|
||||
*
|
||||
* @param target supplies the sink to forward to, evaluated fresh on every call
|
||||
* @return a sink whose call delegates to {@code target.get()}
|
||||
*/
|
||||
static ExhaustionSink forwardingTo(Supplier<ExhaustionSink> target) {
|
||||
return (t, r, p) -> target.get().onExhausted(t, r, p);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import dev.ltms.fleet.msg.TurnToken;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
@@ -89,6 +90,7 @@ public final class Injector {
|
||||
public static final long POLL_INTERVAL_MILLIS = 250;
|
||||
|
||||
private final AgentControl agents;
|
||||
private final HerdrRouter router;
|
||||
private final TurnListener turnListener;
|
||||
private final Predicate<String> ready; // CB-113: a target is deliverable only when available
|
||||
private final Consumer<String> forget; // CB-114: clear a gone worker's readiness/presence
|
||||
@@ -124,13 +126,77 @@ public final class Injector {
|
||||
public Injector(AgentControl agents, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this.agents = agents;
|
||||
this.router = null;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
}
|
||||
|
||||
public Injector(HerdrRouter router, TurnListener turnListener, Predicate<String> ready,
|
||||
Consumer<String> forget) {
|
||||
this.agents = null;
|
||||
this.router = router;
|
||||
this.turnListener = turnListener;
|
||||
this.ready = ready;
|
||||
this.forget = forget;
|
||||
}
|
||||
|
||||
private AgentControl agentsFor(String target) {
|
||||
return router != null ? router.agentsFor(target) : agents;
|
||||
}
|
||||
|
||||
/** The result of trying to remove an undelivered message from the injector. */
|
||||
public enum Cancellation {
|
||||
CANCELLED,
|
||||
DELIVERED,
|
||||
NOT_DELIVERED
|
||||
}
|
||||
|
||||
/**
|
||||
* An identity handle for one queued delivery. It is the only value accepted by
|
||||
* {@link #cancel(Delivery)}, so a caller cannot cancel a different message with the same target
|
||||
* or text.
|
||||
*/
|
||||
public static final class Delivery {
|
||||
private final Pending pending;
|
||||
|
||||
private Delivery(Pending pending) {
|
||||
this.pending = pending;
|
||||
}
|
||||
|
||||
public CompletableFuture<Void> completion() {
|
||||
return pending.delivered;
|
||||
}
|
||||
}
|
||||
|
||||
/** A pending message and the future that completes when it has been delivered. */
|
||||
private record Pending(String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
private static final class Pending {
|
||||
enum State { QUEUED, DELIVERED, NOT_DELIVERED, CANCELLED }
|
||||
|
||||
final String target;
|
||||
final String text;
|
||||
final TurnToken token;
|
||||
final CompletableFuture<Void> delivered;
|
||||
volatile State state = State.QUEUED; // written under the owning Target monitor
|
||||
|
||||
Pending(String target, String text, TurnToken token, CompletableFuture<Void> delivered) {
|
||||
this.target = target;
|
||||
this.text = text;
|
||||
this.token = token;
|
||||
this.delivered = delivered;
|
||||
}
|
||||
|
||||
String text() {
|
||||
return text;
|
||||
}
|
||||
|
||||
TurnToken token() {
|
||||
return token;
|
||||
}
|
||||
|
||||
CompletableFuture<Void> delivered() {
|
||||
return delivered;
|
||||
}
|
||||
}
|
||||
|
||||
/** Per-worker delivery state, guarded by its own monitor (single writer per worker). */
|
||||
@@ -141,6 +207,7 @@ public final class Injector {
|
||||
boolean awaitingCompletion; // a delivered message's turn is not yet known-complete
|
||||
boolean turnObserved; // saw a real `working` sample since that delivery (turn ran)
|
||||
int unknownSinceTurn; // consecutive `unknown` samples while a delegation is outstanding (CB-109)
|
||||
int unknownSincePostTurn; // the same, for the post-turn housekeeping phase (fleetd #306)
|
||||
int notReadySincePoll; // consecutive injectable samples a queued message waited on the readiness gate (CB-114)
|
||||
boolean postTurnPending; // completion observed; adapter housekeeping has not started yet
|
||||
boolean awaitingPostTurnPickup;
|
||||
@@ -160,15 +227,47 @@ public final class Injector {
|
||||
* <p>Uses an atomic map update so a concurrent {@link #drop} cannot slip between "find the
|
||||
* target" and "queue the message" and orphan it in a target it just removed.
|
||||
*/
|
||||
public CompletableFuture<Void> enqueue(String target, String text, TurnToken token) {
|
||||
public Delivery enqueue(String target, String text, TurnToken token) {
|
||||
CompletableFuture<Void> delivered = new CompletableFuture<>();
|
||||
Pending p = new Pending(text, token, delivered);
|
||||
Pending p = new Pending(target, text, token, delivered);
|
||||
targets.compute(target, (_, existing) -> {
|
||||
Target t = (existing != null) ? existing : new Target();
|
||||
t.add(p); // synchronized on the Target monitor — atomic with a concurrent drop
|
||||
return t;
|
||||
});
|
||||
return delivered;
|
||||
return new Delivery(p);
|
||||
}
|
||||
|
||||
/**
|
||||
* Cancel this exact queued delivery. The target monitor serializes this operation with
|
||||
* {@link #onStatus}: if delivery wins that race, this returns {@link Cancellation#DELIVERED}
|
||||
* rather than claiming the message remained queued.
|
||||
*/
|
||||
public Cancellation cancel(Delivery delivery) {
|
||||
Pending p = delivery.pending;
|
||||
Target t = targets.get(p.target);
|
||||
if (t == null) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
synchronized (t) {
|
||||
if (p.state != Pending.State.QUEUED || !t.queue.remove(p)) {
|
||||
return cancellationOf(p);
|
||||
}
|
||||
p.state = Pending.State.CANCELLED;
|
||||
if (isQuiescent(t)) {
|
||||
targets.remove(p.target, t);
|
||||
}
|
||||
return Cancellation.CANCELLED;
|
||||
}
|
||||
}
|
||||
|
||||
private static Cancellation cancellationOf(Pending p) {
|
||||
return p.state == Pending.State.DELIVERED ? Cancellation.DELIVERED : Cancellation.NOT_DELIVERED;
|
||||
}
|
||||
|
||||
private static boolean isQuiescent(Target t) {
|
||||
return t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -201,10 +300,12 @@ public final class Injector {
|
||||
t.awaitingPickup = false;
|
||||
t.injectableSincePickup = 0;
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
t.notReadySincePoll = 0;
|
||||
if (t.awaitingCompletion) t.turnObserved = true;
|
||||
} else if (status.injectable()) { // IDLE or BLOCKED
|
||||
t.unknownSinceTurn = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
if (t.awaitingPostTurnPickup) {
|
||||
if (++t.injectableSincePostTurnPickup >= PICKUP_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
@@ -253,8 +354,9 @@ public final class Injector {
|
||||
if (p != null && ready.test(target)) {
|
||||
t.notReadySincePoll = 0;
|
||||
try {
|
||||
agents.send(target, p.text());
|
||||
agentsFor(target).send(target, p.text());
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.DELIVERED;
|
||||
t.awaitingPickup = true;
|
||||
t.awaitingCompletion = true;
|
||||
t.turnObserved = false;
|
||||
@@ -264,6 +366,7 @@ public final class Injector {
|
||||
// Delivery failed at herdr; drop the poisoned message and surface it
|
||||
// rather than blocking the queue behind it.
|
||||
t.queue.poll();
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
sent = p;
|
||||
sendError = e;
|
||||
}
|
||||
@@ -274,6 +377,9 @@ public final class Injector {
|
||||
// fail every queued message and release the target (CB-114) instead of
|
||||
// polling it indefinitely with the caller's future never completing.
|
||||
notReady = new ArrayList<>(t.queue);
|
||||
for (Pending pending : notReady) {
|
||||
pending.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
log.warn("readiness grace for {} expired after {} polls ({}s): target never "
|
||||
+ "became deliverable, so failing {} queued message(s) that never "
|
||||
+ "reached its pane",
|
||||
@@ -299,12 +405,29 @@ public final class Injector {
|
||||
t.unknownSinceTurn = 0;
|
||||
turnFailed = true;
|
||||
}
|
||||
// fleetd #306: the same escape for the post-turn housekeeping phase. Four latches
|
||||
// gate delivery (awaitingCompletion, postTurnPending, awaitingPostTurnPickup,
|
||||
// postTurnObserved) and only the first had a way out of a sustained unknown streak —
|
||||
// a gate that closed one direction only. The other two below are released here as
|
||||
// well; postTurnPending needs no escape because it is cleared unconditionally on the
|
||||
// line after the listener call that sets it.
|
||||
//
|
||||
// This does NOT set turnFailed. The delegated turn already completed and its waiter
|
||||
// already resolved — what is outstanding is adapter housekeeping (the `/clear`).
|
||||
// Reporting a turn failure here would drive SessionManager.onFailed on a session
|
||||
// that genuinely finished its work, which is a worse lie than the wedge.
|
||||
if ((t.awaitingPostTurnPickup || t.postTurnObserved)
|
||||
&& ++t.unknownSincePostTurn >= TURN_STALL_GRACE_POLLS) {
|
||||
t.awaitingPostTurnPickup = false;
|
||||
t.postTurnObserved = false;
|
||||
t.injectableSincePostTurnPickup = 0;
|
||||
t.unknownSincePostTurn = 0;
|
||||
}
|
||||
}
|
||||
|
||||
// Reclaim the entry once the worker is fully quiescent (nothing queued, no pickup or
|
||||
// completion awaited), so the map cannot grow without bound across short-lived workers.
|
||||
if (t.queue.isEmpty() && !t.awaitingPickup && !t.awaitingCompletion
|
||||
&& !t.postTurnPending && !t.awaitingPostTurnPickup && !t.postTurnObserved) {
|
||||
if (isQuiescent(t)) {
|
||||
targets.remove(target, t);
|
||||
}
|
||||
}
|
||||
@@ -313,7 +436,7 @@ public final class Injector {
|
||||
// thread while it holds the target lock.
|
||||
if (resubmit) {
|
||||
try {
|
||||
agents.submit(target); // nudge a raced Enter so the pending paste submits
|
||||
agentsFor(target).submit(target); // nudge a raced Enter so the pending paste submits
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("resubmit to {} failed (will retry next poll): {}", target, e.getMessage());
|
||||
}
|
||||
@@ -395,6 +518,9 @@ public final class Injector {
|
||||
boolean hadDeliveredTurn;
|
||||
synchronized (t) {
|
||||
pending = new ArrayList<>(t.queue);
|
||||
for (Pending p : pending) {
|
||||
p.state = Pending.State.NOT_DELIVERED;
|
||||
}
|
||||
t.queue.clear();
|
||||
hadDeliveredTurn = t.awaitingCompletion;
|
||||
t.awaitingCompletion = false;
|
||||
|
||||
@@ -0,0 +1,91 @@
|
||||
package dev.ltms.fleet.inject;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* fleetd #446: the single live source for "is {@code exhaustedPattern} configured for this
|
||||
* profile, and what does it compile to". Read fresh off the config supplier on every call — the
|
||||
* same reason {@code CompositePeerLauncher#models0} is a live supplier read rather than a value
|
||||
* captured at construction (see that class's doc) — so an operator can arm or disarm usage-limit
|
||||
* detection for a profile by editing {@code exhaustedPattern} and reloading, with no restart.
|
||||
*
|
||||
* <p>Before this class, {@code exhaustedPattern} was compiled once into a {@code Map<String,
|
||||
* Pattern>} built inside {@code Fleetd.main} at startup ({@code ConfigRef}'s class doc used to
|
||||
* list it under <em>Deferred</em>, CB-578 stage A) — the asymmetry fleetd #446 exists to close:
|
||||
* an operator could turn a model off at runtime ({@code models.allow}'s {@code enabled: false},
|
||||
* hot since fleetd #422) but could not arm the detector that would tell them to, without a
|
||||
* restart. That was backwards for a feature whose whole point is to react while the fleet runs.
|
||||
*
|
||||
* <p>{@link #patternFor} is what {@link CompletionResolver} enforces on (via the {@link
|
||||
* ExhaustedPatternLookup} production wiring in {@code Fleetd.main}); {@link #armed} is what
|
||||
* {@code fleet_profiles}/{@code GET /profiles} report as {@code exhaustionDetectionArmed}. Both
|
||||
* read this ONE object, so the report can never disagree with the behaviour — the same rule
|
||||
* {@code CompositePeerLauncher#modelGateState()}'s javadoc states for the model gate's own
|
||||
* armed/off pair (fleetd #404): "armed" and "which models are off" must come from one read of the
|
||||
* same accessor the gate enforces on.
|
||||
*
|
||||
* <h2>Cache eviction</h2>
|
||||
* Cached by profile NAME, not by pattern text. A {@code Map<String, Pattern>} keyed by the
|
||||
* pattern STRING would grow by one entry per distinct regex ever typed for any profile across the
|
||||
* daemon's uptime — unbounded in practice, because tuning a regex to match a backend's exact
|
||||
* wording is exactly the kind of edit an operator makes several times while getting it right, and
|
||||
* every edit-and-reload cycle would leave the previous attempt's compiled {@link Pattern} behind
|
||||
* forever. Keyed by profile name instead, this cache holds at most one entry per profile name that
|
||||
* has ever been looked up — and that key space is bounded by the (small, human-authored) set of
|
||||
* configured profiles, which changes only on a restart: adding or removing a profile is itself a
|
||||
* deferred key (a new backend needs its own launcher, built once — see {@code ConfigRef}'s class
|
||||
* doc), so profile names do not churn the way pattern text does. Re-editing an EXISTING profile's
|
||||
* {@code exhaustedPattern} — the case this class exists to make hot — simply overwrites that
|
||||
* profile's one cache entry; it never adds a new one.
|
||||
*/
|
||||
public final class LiveExhaustedPatterns {
|
||||
|
||||
/** A compiled pattern paired with the source string it was compiled from, for change detection. */
|
||||
private record Cached(String source, Pattern compiled) {
|
||||
}
|
||||
|
||||
private final Supplier<Map<String, FleetConfig.Profile>> profiles;
|
||||
private final ConcurrentHashMap<String, Cached> cache = new ConcurrentHashMap<>();
|
||||
|
||||
public LiveExhaustedPatterns(Supplier<Map<String, FleetConfig.Profile>> profiles) {
|
||||
this.profiles = Objects.requireNonNull(profiles, "profiles");
|
||||
}
|
||||
|
||||
/**
|
||||
* The compiled {@code exhaustedPattern} currently configured for {@code profileName}, or
|
||||
* {@code null} when that profile is unknown or has none configured. Recompiles only when the
|
||||
* live pattern text differs from what is cached for this profile name; {@code
|
||||
* FleetConfig#rejectMalformedProfilePatterns} already refuses a config (at load and at reload)
|
||||
* whose {@code exhaustedPattern} does not compile, so this is not expected to throw in
|
||||
* production — it is not defended against here for that reason, the same trust
|
||||
* {@code CompositePeerLauncher#models0} places in config validation having already run.
|
||||
*/
|
||||
public Pattern patternFor(String profileName) {
|
||||
if (profileName == null) {
|
||||
return null;
|
||||
}
|
||||
FleetConfig.Profile profile = profiles.get().get(profileName);
|
||||
if (profile == null || !profile.hasExhaustedPattern()) {
|
||||
return null;
|
||||
}
|
||||
String source = profile.exhaustedPattern();
|
||||
Cached cached = cache.get(profileName);
|
||||
if (cached != null && cached.source().equals(source)) {
|
||||
return cached.compiled();
|
||||
}
|
||||
Cached fresh = new Cached(source, Pattern.compile(source));
|
||||
cache.put(profileName, fresh);
|
||||
return fresh.compiled();
|
||||
}
|
||||
|
||||
/** Whether {@code profileName} currently has a usage-limit pattern configured (live). */
|
||||
public boolean armed(String profileName) {
|
||||
return patternFor(profileName) != null;
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package dev.ltms.fleet.inject;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.HerdrRouter;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -22,6 +23,7 @@ public final class StatusPoller {
|
||||
private static final Logger log = LoggerFactory.getLogger(StatusPoller.class);
|
||||
|
||||
private final AgentControl agents;
|
||||
private final HerdrRouter router;
|
||||
private final Injector injector;
|
||||
private final StatusRefiner refiner;
|
||||
private final long intervalMillis;
|
||||
@@ -35,11 +37,24 @@ public final class StatusPoller {
|
||||
public StatusPoller(AgentControl agents, Injector injector, StatusRefiner refiner,
|
||||
long intervalMillis) {
|
||||
this.agents = agents;
|
||||
this.router = null;
|
||||
this.injector = injector;
|
||||
this.refiner = refiner;
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
public StatusPoller(HerdrRouter router, Injector injector, long intervalMillis) {
|
||||
this.agents = null;
|
||||
this.router = router;
|
||||
this.injector = injector;
|
||||
// CB-185: this refiner's own AgentControl (member) is only a default for the legacy 2-arg
|
||||
// refine() overload — the loop below always calls the 3-arg refine(target, raw, control)
|
||||
// with the per-target control from router.agentsFor(target), so a lead target is refined
|
||||
// against the LEAD daemon even though this field points at the member one.
|
||||
this.refiner = new StatusRefiner(router.memberAgents());
|
||||
this.intervalMillis = intervalMillis;
|
||||
}
|
||||
|
||||
/** Start the polling loop on a virtual thread. Idempotent. */
|
||||
public synchronized void start() {
|
||||
if (running) return;
|
||||
@@ -56,7 +71,11 @@ public final class StatusPoller {
|
||||
try {
|
||||
// herdr's agent_status can misreport a settled worker as `unknown`; refine it
|
||||
// against the pane content before it drives delivery/completion (CB-115).
|
||||
AgentStatus status = refiner.refine(target, agents.status(target));
|
||||
// CB-185: refine THROUGH the same control the raw status came from — a router
|
||||
// splits lead/member targets across two herdr daemons, and reading a lead's pane
|
||||
// through the (fixed) member refiner never finds it, wedging that lead at UNKNOWN.
|
||||
AgentControl control = router != null ? router.agentsFor(target) : agents;
|
||||
AgentStatus status = refiner.refine(target, control.status(target), control);
|
||||
injector.onStatus(target, status);
|
||||
} catch (HerdrException e) {
|
||||
// The worker's agent is gone — stop trying and unblock its waiters.
|
||||
|
||||
@@ -41,15 +41,32 @@ public final class StatusRefiner {
|
||||
}
|
||||
|
||||
/**
|
||||
* Return a trustworthy status for {@code target}. Any non-{@code UNKNOWN} {@code raw} is returned
|
||||
* unchanged; an {@code UNKNOWN} triggers a pane read and content classification. A read failure
|
||||
* leaves it {@code UNKNOWN} (the safe default: no delivery, and the stall path still applies).
|
||||
* Return a trustworthy status for {@code target}, reading its pane through this refiner's own
|
||||
* {@link AgentControl}. Equivalent to {@link #refine(String, AgentStatus, AgentControl)} with
|
||||
* that control — kept for callers that only ever talk to one herdr daemon.
|
||||
*/
|
||||
public AgentStatus refine(String target, AgentStatus raw) {
|
||||
return refine(target, raw, agents);
|
||||
}
|
||||
|
||||
/**
|
||||
* Return a trustworthy status for {@code target}. Any non-{@code UNKNOWN} {@code raw} is returned
|
||||
* unchanged; an {@code UNKNOWN} triggers a pane read (through {@code control}) and content
|
||||
* classification. A read failure leaves it {@code UNKNOWN} (the safe default: no delivery, and
|
||||
* the stall path still applies).
|
||||
*
|
||||
* <p>CB-185: {@code control} must be the {@link AgentControl} for the <em>same</em> daemon the
|
||||
* raw status was sampled from — a router splits lead and member targets across two herdr
|
||||
* daemons, and reading a lead's pane through the member client (or vice versa) fails to find
|
||||
* the pane and leaves the target wedged at {@code UNKNOWN} forever. Callers that route per
|
||||
* target (e.g. {@code StatusPoller}) must pass that target's control explicitly rather than
|
||||
* relying on the control fixed at construction.
|
||||
*/
|
||||
public AgentStatus refine(String target, AgentStatus raw, AgentControl control) {
|
||||
if (raw != AgentStatus.UNKNOWN) return raw;
|
||||
String pane;
|
||||
try {
|
||||
pane = agents.read(target, PROBE_SOURCE);
|
||||
pane = control.read(target, PROBE_SOURCE);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("status refine read for {} failed; leaving UNKNOWN: {}", target, e.getMessage());
|
||||
return AgentStatus.UNKNOWN;
|
||||
|
||||
@@ -4,6 +4,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.herdr.PendingCloseMarker;
|
||||
import dev.ltms.fleet.herdr.Tab;
|
||||
import dev.ltms.fleet.herdr.Workspace;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
@@ -12,9 +13,11 @@ import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Set;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
|
||||
/**
|
||||
@@ -44,8 +47,33 @@ import dev.ltms.fleet.peer.PeerLauncher;
|
||||
* privilege escalation (the tab label never granted anything a pane could take for itself; see that
|
||||
* class's javadoc), but <em>staleness</em>: a label left behind by a crashed session would otherwise
|
||||
* read as a live lead forever, and the lead would never be relaunched. So a lead counts as live only
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #liveLeads}. A labelled
|
||||
* when herdr also reports a <em>running agent</em> in that tab — see {@link #countLeads}. A labelled
|
||||
* tab with no agent in it is not a lead.
|
||||
*
|
||||
* <p><strong>fleetd #359 — a stale label used to pile up, not just mislead.</strong> Finding "not
|
||||
* live" here used to mean only one thing: launch another. The old label was left exactly where it
|
||||
* was, so a daemon that restarted enough times — or hit one herdr read that missed a genuinely
|
||||
* running agent — accumulated one more identically-labelled dead tab per occurrence, and
|
||||
* {@code LeadTabScanner} (before its own #359 fix) reported every one of them as a lead.
|
||||
*
|
||||
* <p><strong>Review finding 1 — closing on one reading is worse than the bug.</strong> The first
|
||||
* version of this fix closed a name's dead tabs the moment a single {@link #countLeads} reading
|
||||
* called them dead. The ticket's own live evidence rules that out: on a real host, {@code
|
||||
* agent.list} was seen reporting "0 live" for a tab that a plain {@code ps} confirmed was running a
|
||||
* real session. Closing on that reading would have destroyed the operator's actual lead — a worse
|
||||
* failure than the extra tab it replaces. So {@link #ensureLeads()} now needs the same dead reading
|
||||
* <em>twice</em>, one restart apart, before it closes anything: the first time a labelled tab reads
|
||||
* dead, it is only flagged ({@link PendingCloseMarker}), left running, untouched; it is closed only
|
||||
* if a <em>later</em>, independently-connected reconcile still finds it dead while the flag is
|
||||
* still there. A transient miss self-heals — the next reconcile sees the agent again and clears the
|
||||
* flag (see {@code toUnflag} below) — so the worst case for a single bad reading is one extra tab
|
||||
* surviving one more restart, never a live session destroyed. Two alternatives were considered and
|
||||
* rejected: corroborating {@code agent.list} against a second, truly independent signal was dropped
|
||||
* because nothing else herdr exposes proves "is a process attached to this pane" any better — a
|
||||
* second call to the same unreliable source is not independent evidence; capping the close to "all
|
||||
* but the most recent dead tab" was dropped because "most recent" has no reliable ordering across
|
||||
* tab ids and would leave the true failure mode (a name that is <em>never</em> reconfirmed) growing
|
||||
* by one tab per bad reading forever, which is the exact defect this ticket exists to fix.
|
||||
*/
|
||||
public final class LeadLauncher {
|
||||
|
||||
@@ -79,9 +107,9 @@ public final class LeadLauncher {
|
||||
return 0;
|
||||
}
|
||||
|
||||
Map<String, Integer> live;
|
||||
Map<String, LeadCount> live;
|
||||
try {
|
||||
live = liveLeads(leaders);
|
||||
live = countLeads(leaders);
|
||||
} catch (HerdrException e) {
|
||||
// Counting is the whole safety mechanism against double-spawning. If we cannot count, we
|
||||
// must not guess — spawning a second orchestrator is worse than starting none.
|
||||
@@ -93,9 +121,53 @@ public final class LeadLauncher {
|
||||
for (Map.Entry<String, FleetConfig.Leader> e : leaders.entrySet()) {
|
||||
String name = e.getKey();
|
||||
FleetConfig.Leader lead = e.getValue();
|
||||
int running = live.getOrDefault(name, 0);
|
||||
LeadCount state = live.getOrDefault(name, LeadCount.NONE);
|
||||
int running = state.running();
|
||||
int wanted = lead.instances();
|
||||
|
||||
// fleetd #359 review finding 1: a tab already flagged pending-close, still labelled for
|
||||
// this lead, and STILL hosting no agent on this separate reconcile — two independent
|
||||
// readings agree, so close it. A tab found dead for the first time is only flagged below,
|
||||
// never closed on the spot.
|
||||
for (String tabId : state.toClose()) {
|
||||
log.info("lead '{}': closing tab {} — flagged pending-close on a previous reconcile "
|
||||
+ "and still no agent running in it", name, tabId);
|
||||
try {
|
||||
spaces.closeTab(tabId);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not close stale tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
// A tab labelled for this lead, with no agent running in it, seen dead for the first
|
||||
// time — flag it rather than closing it. One reading of `agent.list` is not enough
|
||||
// evidence to destroy a tab that might genuinely be live (see the class javadoc).
|
||||
for (String tabId : state.toFlag()) {
|
||||
String flagged = lead.tabLabel() + PendingCloseMarker.SUFFIX;
|
||||
log.info("lead '{}': tab {} has no agent running in it this reconcile — flagging it "
|
||||
+ "'{}' rather than closing; it is only closed if a later reconcile still "
|
||||
+ "finds it dead", name, tabId, flagged);
|
||||
try {
|
||||
spaces.renameTab(tabId, flagged);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not flag stale tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
// A previously-flagged tab that is running an agent again — the miss that flagged it was
|
||||
// transient. Clear the flag so a future, unrelated miss starts its own two-reading count
|
||||
// rather than closing on the strength of this one's already-spent flag.
|
||||
for (String tabId : state.toUnflag()) {
|
||||
log.info("lead '{}': tab {} is running an agent again — clearing its pending-close flag",
|
||||
name, tabId);
|
||||
try {
|
||||
spaces.renameTab(tabId, lead.tabLabel());
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("could not clear the pending-close flag on tab {} for lead '{}': {}",
|
||||
tabId, name, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
if (running >= wanted) {
|
||||
log.info("lead '{}': {} live, {} wanted — nothing to start", name, running, wanted);
|
||||
continue;
|
||||
@@ -126,18 +198,29 @@ public final class LeadLauncher {
|
||||
}
|
||||
|
||||
/**
|
||||
* How many live leads exist per configured name: a running agent in a tab labelled with that
|
||||
* lead's exact {@code tab} (CB-579). Member workspaces are excluded, exactly as the scanner
|
||||
* excludes them: a member must not be counted as a lead because it happens to sit in a matching
|
||||
* tab.
|
||||
* How many live leads exist per configured name, and which of that name's labelled tabs are
|
||||
* <em>not</em> live: a running agent in a tab labelled with that lead's exact {@code tab}
|
||||
* (CB-579). Member workspaces are excluded, exactly as the scanner excludes them: a member must
|
||||
* not be counted as a lead because it happens to sit in a matching tab.
|
||||
*
|
||||
* <p>There used to be a second path here — a running agent on the terminal a
|
||||
* {@code fleet.leaders.<name>.terminal} pin named, for a lead opened and pinned by hand. That
|
||||
* pin is retired: {@code tab} is now the only field identity depends on, and {@link Agent}
|
||||
* already carries {@link Agent#tabId()} directly, so a hand-opened lead is found the same way an
|
||||
* auto-launched one is — by labelling its tab to match.
|
||||
*
|
||||
* <p>fleetd #359 review finding 1: a labelled tab with nothing running in it is split into
|
||||
* {@code toClose} (already flagged pending-close by a previous reconcile, and still dead — two
|
||||
* independent readings agree) and {@code toFlag} (dead for the first time — not enough evidence
|
||||
* to close yet). {@code toUnflag} is the reverse: a tab flagged pending-close that is running an
|
||||
* agent again, so the flag it carries no longer means anything and {@link #ensureLeads()} clears
|
||||
* it.
|
||||
*/
|
||||
private Map<String, Integer> liveLeads(Map<String, FleetConfig.Leader> leaders) {
|
||||
private record LeadCount(int running, List<String> toClose, List<String> toFlag, List<String> toUnflag) {
|
||||
static final LeadCount NONE = new LeadCount(0, List.of(), List.of(), List.of());
|
||||
}
|
||||
|
||||
private Map<String, LeadCount> countLeads(Map<String, FleetConfig.Leader> leaders) {
|
||||
// A lead and the members share ONE workspace now (the operator asked for a single "session"
|
||||
// with many tabs), so a workspace can no longer be excluded wholesale — the lead lives in the
|
||||
// member workspace by design. The sole discriminator is the exact tab label: a lead carries
|
||||
@@ -145,6 +228,7 @@ public final class LeadLauncher {
|
||||
// profile's `worker: {profile} #{n}` template. These never collide, so an exact-label match
|
||||
// separates them without needing to know which workspace anyone is in.
|
||||
Map<String, String> nameByTab = new LinkedHashMap<>();
|
||||
Set<String> flaggedTabIds = new LinkedHashSet<>();
|
||||
for (Workspace ws : spaces.listWorkspaces()) {
|
||||
if (ws.workspaceId() == null) {
|
||||
continue;
|
||||
@@ -153,31 +237,65 @@ public final class LeadLauncher {
|
||||
String declared = leadNameOf(tab.label(), leaders);
|
||||
if (declared != null && tab.tabId() != null) {
|
||||
nameByTab.put(tab.tabId(), declared);
|
||||
if (PendingCloseMarker.isFlagged(tab.label())) {
|
||||
flaggedTabIds.add(tab.tabId());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
Set<String> liveTabIds = new LinkedHashSet<>();
|
||||
Map<String, Integer> counts = new LinkedHashMap<>();
|
||||
for (Agent a : agents.list()) {
|
||||
String name = nameByTab.get(a.tabId());
|
||||
if (name != null) {
|
||||
counts.merge(name, 1, Integer::sum);
|
||||
liveTabIds.add(a.tabId());
|
||||
}
|
||||
}
|
||||
return counts;
|
||||
|
||||
Map<String, List<String>> toCloseByName = new LinkedHashMap<>();
|
||||
Map<String, List<String>> toFlagByName = new LinkedHashMap<>();
|
||||
Map<String, List<String>> toUnflagByName = new LinkedHashMap<>();
|
||||
nameByTab.forEach((tabId, name) -> {
|
||||
boolean live = liveTabIds.contains(tabId);
|
||||
boolean flagged = flaggedTabIds.contains(tabId);
|
||||
if (live) {
|
||||
if (flagged) {
|
||||
toUnflagByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (flagged) {
|
||||
toCloseByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
} else {
|
||||
toFlagByName.computeIfAbsent(name, k -> new ArrayList<>()).add(tabId);
|
||||
}
|
||||
});
|
||||
|
||||
Map<String, LeadCount> out = new LinkedHashMap<>();
|
||||
for (String name : leaders.keySet()) {
|
||||
out.put(name, new LeadCount(counts.getOrDefault(name, 0),
|
||||
toCloseByName.getOrDefault(name, List.of()),
|
||||
toFlagByName.getOrDefault(name, List.of()),
|
||||
toUnflagByName.getOrDefault(name, List.of())));
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* The configured lead a tab label names, or {@code null} for a label that names none.
|
||||
*
|
||||
* <p>Matched exactly (case-insensitively) against each lead's configured {@code tab}, so an
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead.
|
||||
* operator's {@code "lead: something-else"} tab is not mistaken for a configured lead. A
|
||||
* trailing {@link PendingCloseMarker} is stripped first, so a tab this class flagged on a
|
||||
* previous reconcile is still recognised as the same lead's tab on this one.
|
||||
*/
|
||||
private String leadNameOf(String label, Map<String, FleetConfig.Leader> leaders) {
|
||||
if (label == null) {
|
||||
return null;
|
||||
}
|
||||
String l = label.strip();
|
||||
String l = PendingCloseMarker.strip(label);
|
||||
for (Map.Entry<String, FleetConfig.Leader> e : leaders.entrySet()) {
|
||||
String tab = e.getValue().tabLabel();
|
||||
if (tab != null && l.equalsIgnoreCase(tab.strip())) {
|
||||
|
||||
@@ -0,0 +1,649 @@
|
||||
package dev.ltms.fleet.lead;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
import java.util.UUID;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
|
||||
/**
|
||||
* fleetd #480: replace a lead session that has decided it is ready to be rolled over, without an
|
||||
* operator doing it by hand. A lead writes a handover file, calls {@link #open}, and then — once
|
||||
* every gate ({@link #confirm}'s own checks) has passed — a deferred, single-shot continuation
|
||||
* clears the lead's own pane and bootstraps a fresh session against that file.
|
||||
*
|
||||
* <p>This is the executor only. Nothing in this ticket wires an MCP tool onto {@link #open}/
|
||||
* {@link #confirm}/{@link #cancel} — that is a separate, later unit; until it lands, nothing calls
|
||||
* this class at all.
|
||||
*
|
||||
* <p><strong>{@code confirm()} cannot roll inline — a fleetd #480 correction.</strong> The first
|
||||
* version of this class called {@code agents.send(lead, "/clear")} directly from inside {@code
|
||||
* confirm()}, then polled for the pane to become injectable again. That is wrong, because {@code
|
||||
* confirm()} is called BY the lead, FROM the lead's own turn: the lead's pane is {@code WORKING}
|
||||
* for the whole duration of that call and cannot possibly report injectable until {@code confirm()}
|
||||
* itself returns. The poll always timed out — but only after the {@code /clear} had already been
|
||||
* sent and queued in the pane, where it fired the instant the turn ended anyway. The result was the
|
||||
* worst outcome this feature can produce: a silently destroyed lead context with no fresh session
|
||||
* ever started, and a refusal return value that claimed nothing had happened.
|
||||
*
|
||||
* <p>The fix: {@link #confirm} validates every gate, then does no I/O against the lead's own pane
|
||||
* at all — it only records that the request is approved and hands a one-shot continuation to
|
||||
* {@code continuationRunner} before returning. That continuation is what actually touches the pane,
|
||||
* once the calling turn has ended, in this order:
|
||||
* <ol>
|
||||
* <li>wait for the lead's own pane to report a real turn boundary — {@code IDLE} or {@code
|
||||
* DONE}, never merely {@code BLOCKED} — i.e. wait for the very {@code confirm()} call that
|
||||
* approved this roll to finish its turn — bounded by {@code turnSettleSeconds}. <strong>If
|
||||
* this never happens, nothing else in this list runs: no {@code /clear} is ever sent.</strong>
|
||||
* A lead that never goes idle is a lead still doing real work, and clearing it would throw
|
||||
* away live context — exactly the failure this correction exists to prevent.</li>
|
||||
* <li>{@code agents.send(lead, "/clear")}</li>
|
||||
* <li>wait for {@code /clear} to be picked up and settle, bounded by {@code clearSettleSeconds}
|
||||
* (fleetd #489: no longer a plain re-check of the same boundary — {@code /clear} starts no
|
||||
* turn of its own, so this instead nudges the submit keystroke while no pickup has been seen,
|
||||
* then waits for a real {@code WORKING} → {@code IDLE}/{@code DONE} boundary once one has;
|
||||
* see {@link #waitForClearPickupAndSettle})</li>
|
||||
* <li>{@code agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()))}</li>
|
||||
* </ol>
|
||||
* A {@link #confirm} that returns {@link RollDecision#approved()} therefore means <em>"every gate
|
||||
* passed and the roll is scheduled"</em>, never <em>"the pane has been cleared"</em> — the pane may
|
||||
* still be mid-turn, possibly for a long time, when the caller gets that answer back.
|
||||
*
|
||||
* <p><strong>The safety invariant survives this change, restated precisely.</strong> The ticket
|
||||
* that first defined this class required "no timer, no scheduler, no background thread" so that
|
||||
* nothing but an explicit {@link #confirm} call could ever cause a {@code /clear}. That invariant
|
||||
* is about INITIATIVE, not about synchronicity, and this correction keeps it: {@code
|
||||
* continuationRunner} launches a single-shot task that exists only because one specific,
|
||||
* already-approved {@link #confirm} call created it — it is not recurring, it is not started at
|
||||
* construction time or on any schedule, and no two invocations of it ever share state. A recurring
|
||||
* heartbeat or timer that could decide on its own initiative to roll a pane is still, and will
|
||||
* always be, absent from this class. <strong>Nothing but an explicit {@link #confirm} call that
|
||||
* passes every gate can ever cause a {@code /clear} — that call may simply finish its own work
|
||||
* slightly later than the method return, as a continuation of the same approved request, rather
|
||||
* than entirely inside the method body.</strong>
|
||||
*
|
||||
* <p><strong>Identity is resolved by the caller, never looked up here — a second fleetd #480
|
||||
* correction.</strong> The first version resolved the pane to clear via {@code
|
||||
* PrimaryRegistry#primaryTerminal()}. That is correct for a background loop with no caller (see
|
||||
* {@code dev.ltms.fleet.msg.LeadHeartbeatLoop}), but wrong here and a violation of this project's
|
||||
* own charter invariant 3 — "identity comes from the connection, never an argument." This daemon
|
||||
* can hold more than one labelled lead tab (see {@code LeadLauncher}'s fleetd #359 two-reading
|
||||
* dead-tab cleanup), so a single-slot lookup lets lead X's {@link #confirm} clear lead Y's pane: an
|
||||
* unrecoverable loss of someone else's context, and a different lead's context at that. Both
|
||||
* {@link #open} and {@link #confirm} now take the lead's terminal id as a parameter instead —
|
||||
* {@link #open} stores it on the {@link PendingRollover}, and {@link #confirm} refuses with {@link
|
||||
* RefusalReason#NOT_YOUR_ROLLOVER} unless the caller's terminal matches the one {@link #open}
|
||||
* recorded. <strong>The terminal id passed to both methods must come from the MCP layer's own
|
||||
* connection-based caller resolution — the same source {@code auth/CallerResolver#resolve} uses to
|
||||
* build a {@code Principal.leader(...)} (see its use of {@code ConnectionIdentity.Caller#terminal})
|
||||
* — never a value the client supplies or chooses.</strong> The later MCP-tool unit that wires
|
||||
* {@link #open}/{@link #confirm} must pass the resolved caller terminal, not a request field.
|
||||
*
|
||||
* <p><strong>A relative {@code handoverPath} resolves against the CALLING lead's workspace, never
|
||||
* the daemon's own cwd — a fleetd #480 follow-up.</strong> The daemon and a lead's own pane can
|
||||
* have different working directories (this repo nests {@code fleetd/} inside its own root, so the
|
||||
* daemon's cwd and {@code fleet.leaders.<name>.cwd} already differ on this host). {@link #open}
|
||||
* resolves {@code cfg.handoverPath()} to an ABSOLUTE path exactly once — against {@code
|
||||
* leadWorkspace.apply(leadTerminal)} when that lookup returns a non-null, non-blank workspace, and
|
||||
* against {@code System.getProperty("user.dir")} otherwise (the same fallback {@code
|
||||
* LeadLauncher#launch} already uses for a lead with no configured {@code cwd}) — and stores only
|
||||
* that absolute path on {@link PendingRollover}. Every later read of {@code
|
||||
* PendingRollover#handoverPath()} (the freshness/exists/empty checks in {@link #checkHandover},
|
||||
* the value {@code FleetMcp} hands back to the lead in the {@code open} response so it knows where
|
||||
* to WRITE the file, and {@link FleetConfig.LeadRollover#bootstrapTextFor} which names it in the
|
||||
* text typed into the fresh session) therefore already sees the resolved absolute form and never
|
||||
* needs to resolve anything itself.
|
||||
*/
|
||||
public final class LeadRollover {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadRollover.class);
|
||||
|
||||
/** Poll interval while waiting for the lead's pane to settle after {@code /clear}. */
|
||||
static final long SETTLE_POLL_MS = 250;
|
||||
|
||||
/**
|
||||
* How many consecutive not-yet-picked-up polls {@link #waitForClearPickupAndSettle} allows
|
||||
* before releasing rather than wedging the roll — the same constant and the same
|
||||
* release-not-wedge choice {@link dev.ltms.fleet.inject.Injector} already makes for its own
|
||||
* post-turn {@code /clear} housekeeping (fleetd #306). <strong>This bounds the number of
|
||||
* consecutive polls, not the number of nudges:</strong> the first {@code PICKUP_GRACE_POLLS - 1}
|
||||
* of those polls each send a nudge, and the {@code PICKUP_GRACE_POLLS}th releases instead of
|
||||
* nudging again — so 8 polls produce 7 nudges, not 8.
|
||||
*/
|
||||
static final int PICKUP_GRACE_POLLS = 8;
|
||||
|
||||
/**
|
||||
* One request opened by {@link #open}, pending its {@link #confirm} (or {@link #cancel}).
|
||||
*
|
||||
* @param leadTerminal the lead pane that opened this request — the only terminal that may
|
||||
* later {@link #confirm} it (see {@link RefusalReason#NOT_YOUR_ROLLOVER})
|
||||
* @param handoverPath the ABSOLUTE, resolved handover path — never the raw configured value,
|
||||
* which may have been relative. {@link #open} resolves it once, against the
|
||||
* calling lead's workspace, before storing it here; see this class's
|
||||
* javadoc. This is the value the MCP layer hands back to the lead as
|
||||
* "write your file here", so callers may rely on it always being absolute.
|
||||
*/
|
||||
public record PendingRollover(String token, String leadTerminal, String handoverPath,
|
||||
long requestedAtMillis) {}
|
||||
|
||||
/** Which check refused a {@link #confirm} call, named so a caller can act on it. */
|
||||
public enum RefusalReason {
|
||||
/** {@code leadRollover:} is not configured — absent at construction, or removed since. */
|
||||
NOT_CONFIGURED,
|
||||
/** {@code token} names no pending request: never opened, already confirmed, or cancelled. */
|
||||
UNKNOWN_TOKEN,
|
||||
/**
|
||||
* The caller's terminal does not match the terminal that {@link #open} recorded for this
|
||||
* token. Only the lead that opened a request may confirm it (fleetd #480 correction 2).
|
||||
*/
|
||||
NOT_YOUR_ROLLOVER,
|
||||
/** {@code requireOperatorConfirm: true} and the caller passed {@code operatorConfirmed: false}. */
|
||||
OPERATOR_NOT_CONFIRMED,
|
||||
/** The handover file does not exist. */
|
||||
HANDOVER_MISSING,
|
||||
/** The handover file exists but is empty. */
|
||||
HANDOVER_EMPTY,
|
||||
/**
|
||||
* The handover file's modified time is not after {@link #open}'s request timestamp, or is
|
||||
* older than {@code maxDocAgeSeconds}.
|
||||
*/
|
||||
HANDOVER_STALE
|
||||
}
|
||||
|
||||
/**
|
||||
* The outcome of a {@link #confirm} call. {@link #approved()} means every gate passed and the
|
||||
* roll has been handed to a one-shot continuation — <strong>not</strong> that the pane has been
|
||||
* cleared; the continuation may still be waiting for the calling turn to end when this returns.
|
||||
* Whether the deferred roll itself later goes on to clear the pane, refuse for never going
|
||||
* idle, or refuse for never re-settling after {@code /clear} is logged only (see this class's
|
||||
* javadoc) — there is deliberately no synchronous caller left by that point to hand a result to.
|
||||
*/
|
||||
public record RollDecision(boolean accepted, RefusalReason reason, String detail) {
|
||||
static RollDecision approved() {
|
||||
return new RollDecision(true, null, "confirmed; the roll will run once the calling turn ends");
|
||||
}
|
||||
|
||||
static RollDecision refused(RefusalReason reason, String detail) {
|
||||
return new RollDecision(false, reason, detail);
|
||||
}
|
||||
}
|
||||
|
||||
private final AgentControl agents;
|
||||
private final Supplier<FleetConfig.LeadRollover> configSupplier;
|
||||
/**
|
||||
* Terminal id → that lead's configured workspace directory (their {@code
|
||||
* fleet.leaders.<name>.cwd}), or {@code null} when the terminal names no currently-recognised
|
||||
* lead. {@link #open} calls this to resolve a relative {@code handoverPath} — see this class's
|
||||
* javadoc. Required: there is no sane default that would not silently reintroduce the
|
||||
* daemon-cwd bug this parameter exists to fix.
|
||||
*/
|
||||
private final Function<String, String> leadWorkspace;
|
||||
private final LongSupplier nowMillis;
|
||||
private final Runnable settleSleeper;
|
||||
/**
|
||||
* Launches the post-{@code confirm()} continuation. Production uses a single unstarted virtual
|
||||
* thread per confirmed request — see this class's javadoc for why that is a single-shot task,
|
||||
* not a background scheduler. Tests inject {@code Runnable::run} to make the continuation run
|
||||
* synchronously and deterministically on the calling thread.
|
||||
*/
|
||||
private final Consumer<Runnable> continuationRunner;
|
||||
private final Map<String, PendingRollover> pending = new ConcurrentHashMap<>();
|
||||
|
||||
/** Production constructor — wall clock, real sleep between settle polls, a real virtual thread. */
|
||||
public LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace) {
|
||||
this(agents, configSupplier, leadWorkspace, System::currentTimeMillis,
|
||||
() -> sleepUninterruptibly(SETTLE_POLL_MS),
|
||||
r -> Thread.ofVirtual().name("lead-rollover-continuation-").start(r));
|
||||
}
|
||||
|
||||
/**
|
||||
* Full constructor — an injectable wall-clock supplier, settle-poll sleeper, and continuation
|
||||
* runner, for tests. {@code nowMillis} MUST be a wall-clock source (e.g. {@code
|
||||
* System.currentTimeMillis()}), never {@code System.nanoTime()}: the freshness check compares
|
||||
* against a file's modified time, which only a wall clock is comparable to, and {@code
|
||||
* nanoTime} freezes while the host sleeps (fleetd #386).
|
||||
*/
|
||||
LeadRollover(AgentControl agents, Supplier<FleetConfig.LeadRollover> configSupplier,
|
||||
Function<String, String> leadWorkspace, LongSupplier nowMillis,
|
||||
Runnable settleSleeper, Consumer<Runnable> continuationRunner) {
|
||||
this.agents = agents;
|
||||
this.configSupplier = configSupplier;
|
||||
this.leadWorkspace = leadWorkspace;
|
||||
this.nowMillis = nowMillis;
|
||||
this.settleSleeper = settleSleeper;
|
||||
this.continuationRunner = continuationRunner;
|
||||
}
|
||||
|
||||
private static void sleepUninterruptibly(long ms) {
|
||||
try {
|
||||
Thread.sleep(ms);
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
// preserve the interrupt flag but continue — this poll loop should not be aborted by an
|
||||
// interrupt that was not meant for it.
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The lead says it is ready to be replaced. Generates a token and records the resolved
|
||||
* handover path, this moment's wall-clock timestamp (the baseline {@link #confirm} checks the
|
||||
* handover file's modified time against), and {@code leadTerminal} — only that exact terminal
|
||||
* may later {@link #confirm} this token.
|
||||
*
|
||||
* @param leadTerminal the calling lead's terminal id, resolved by the MCP layer from the
|
||||
* connection (see this class's javadoc) — never a client-supplied value
|
||||
* @param reason free-text audit note (logged only; not otherwise interpreted or stored)
|
||||
* @throws IllegalStateException if {@code leadRollover:} is not configured
|
||||
* @throws IllegalArgumentException if {@code leadTerminal} is null or blank
|
||||
*/
|
||||
public PendingRollover open(String leadTerminal, String reason) {
|
||||
FleetConfig.LeadRollover cfg = configSupplier.get();
|
||||
if (cfg == null) {
|
||||
throw new IllegalStateException("leadRollover: is not configured");
|
||||
}
|
||||
if (leadTerminal == null || leadTerminal.isBlank()) {
|
||||
throw new IllegalArgumentException("leadTerminal is required — it must be resolved from "
|
||||
+ "the caller's connection, never accepted as a client-chosen argument");
|
||||
}
|
||||
String token = UUID.randomUUID().toString();
|
||||
long requestedAt = nowMillis.getAsLong();
|
||||
String resolvedPath = resolveHandoverPath(cfg.handoverPath(), leadTerminal);
|
||||
PendingRollover p = new PendingRollover(token, leadTerminal, resolvedPath, requestedAt);
|
||||
pending.put(token, p);
|
||||
if (resolvedPath.equals(cfg.handoverPath())) {
|
||||
log.info("lead-rollover: open token={} lead={} handoverPath={} reason={}",
|
||||
token, leadTerminal, resolvedPath, reason);
|
||||
} else {
|
||||
// The configured value was relative (or merely un-normalized) and resolved to a
|
||||
// different string — log both, so an operator reading this line can see which
|
||||
// directory the daemon actually looked in, not just the value it was given.
|
||||
log.info("lead-rollover: open token={} lead={} configuredHandoverPath={} "
|
||||
+ "resolvedHandoverPath={} reason={}",
|
||||
token, leadTerminal, cfg.handoverPath(), resolvedPath, reason);
|
||||
}
|
||||
return p;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve {@code configured} to an absolute path exactly once, here, so nothing downstream
|
||||
* ({@link #checkHandover}, the MCP layer's {@code open} response, {@link
|
||||
* FleetConfig.LeadRollover#bootstrapTextFor}) ever has to resolve — or worse, silently
|
||||
* mis-resolve — a relative path again.
|
||||
*
|
||||
* <p><strong>The return value is GUARANTEED absolute, not merely usually absolute.</strong>
|
||||
* {@code leadWorkspace.apply(leadTerminal)} returns an operator-configured {@code
|
||||
* fleet.leaders.<name>.cwd} string, and nothing forces an operator to write an absolute one —
|
||||
* a relative {@code cwd} resolved with plain {@link Path#resolve} would still yield a relative
|
||||
* result, silently reopening the exact bug this class exists to fix (every later reader back to
|
||||
* interpreting an ambiguous string against ITS OWN working directory). {@link
|
||||
* Path#toAbsolutePath()} closes that: it resolves any remaining relative path against {@code
|
||||
* user.dir} (the JVM's own cwd), which is the correct base for an operator-written path the
|
||||
* daemon process itself is meant to interpret, exactly like the {@code user.dir} fallback used
|
||||
* below. Applying it unconditionally on both branches means the ALREADY-absolute branch stays a
|
||||
* no-op (an absolute path is unaffected by {@code toAbsolutePath()}) while the relative-{@code
|
||||
* cwd} branch above is closed the same way.
|
||||
*
|
||||
* <ul>
|
||||
* <li>already absolute → returned unchanged (normalized)</li>
|
||||
* <li>relative → resolved against {@code leadWorkspace.apply(leadTerminal)} when that is
|
||||
* non-null and non-blank; otherwise against {@code System.getProperty("user.dir")} — the
|
||||
* same fallback {@code LeadLauncher#launch} uses for a lead with no configured {@code
|
||||
* cwd}. If {@code leadWorkspace}'s own answer is itself relative (an operator wrote a
|
||||
* relative {@code cwd:}), the result is finished off against the daemon's own
|
||||
* {@code user.dir} — see the paragraph above.</li>
|
||||
* </ul>
|
||||
*/
|
||||
private String resolveHandoverPath(String configured, String leadTerminal) {
|
||||
Path path = Path.of(configured);
|
||||
if (path.isAbsolute()) {
|
||||
// toAbsolutePath() is a no-op for an already-absolute path — kept here anyway so both
|
||||
// branches call the exact same guarantee, rather than one branch relying on
|
||||
// isAbsolute() alone to already imply what toAbsolutePath() enforces.
|
||||
return path.toAbsolutePath().normalize().toString();
|
||||
}
|
||||
String workspace = leadWorkspace.apply(leadTerminal);
|
||||
Path base = (workspace == null || workspace.isBlank())
|
||||
? Path.of(System.getProperty("user.dir"))
|
||||
: Path.of(workspace);
|
||||
return base.resolve(path).toAbsolutePath().normalize().toString();
|
||||
}
|
||||
|
||||
/**
|
||||
* Validate every gate, then — if and only if all of them pass — hand a one-shot continuation
|
||||
* that performs the actual roll to {@code continuationRunner} and return. <strong>This method
|
||||
* never itself sends anything to the lead's pane</strong> — see this class's javadoc for why
|
||||
* (it is called FROM the lead's own turn, so the pane cannot possibly be injectable yet).
|
||||
*
|
||||
* <p>Order: token lookup, then ownership ({@code callerTerminal} must match the terminal
|
||||
* {@link #open} recorded — {@link RefusalReason#NOT_YOUR_ROLLOVER}), then the
|
||||
* operator-confirmation gate, then the three handover-file checks (exists, not empty, fresh —
|
||||
* see {@link #checkHandover}). The first failing check is returned and {@code token} stays
|
||||
* pending (so a caller can fix the problem — e.g. rewrite the handover file — and retry with
|
||||
* the same token); it is consumed only once every gate passes and the continuation is launched.
|
||||
*
|
||||
* @param callerTerminal the CALLING lead's terminal id, resolved by the MCP layer from the
|
||||
* connection — never a client-supplied value (see this class's javadoc)
|
||||
* @param token the token {@link #open} returned
|
||||
* @param operatorConfirmed the caller's answer to "has an operator confirmed this roll" —
|
||||
* consulted only when the live config's {@code requireOperatorConfirm}
|
||||
* is true
|
||||
*/
|
||||
public RollDecision confirm(String callerTerminal, String token, boolean operatorConfirmed) {
|
||||
FleetConfig.LeadRollover cfg = configSupplier.get();
|
||||
if (cfg == null) {
|
||||
return RollDecision.refused(RefusalReason.NOT_CONFIGURED, "leadRollover: is not configured");
|
||||
}
|
||||
PendingRollover p = pending.get(token);
|
||||
if (p == null) {
|
||||
return RollDecision.refused(RefusalReason.UNKNOWN_TOKEN,
|
||||
"token " + token + " names no pending rollover request");
|
||||
}
|
||||
if (!p.leadTerminal().equals(callerTerminal)) {
|
||||
return RollDecision.refused(RefusalReason.NOT_YOUR_ROLLOVER,
|
||||
"token " + token + " was opened by a different lead terminal");
|
||||
}
|
||||
if (cfg.requireOperatorConfirm() && !operatorConfirmed) {
|
||||
return RollDecision.refused(RefusalReason.OPERATOR_NOT_CONFIRMED,
|
||||
"requireOperatorConfirm is true and operatorConfirmed was false");
|
||||
}
|
||||
|
||||
RollDecision docCheck = checkHandover(p, cfg);
|
||||
if (docCheck != null) {
|
||||
return docCheck;
|
||||
}
|
||||
|
||||
pending.remove(token);
|
||||
log.info("lead-rollover: confirmed token={} lead={} — roll scheduled once the calling turn ends",
|
||||
token, callerTerminal);
|
||||
continuationRunner.accept(() -> runRollover(p, cfg));
|
||||
return RollDecision.approved();
|
||||
}
|
||||
|
||||
/**
|
||||
* The single-shot continuation {@link #confirm} hands to {@code continuationRunner}. Runs
|
||||
* entirely after {@link #confirm} has returned to its caller — see this class's javadoc for the
|
||||
* four-step order. There is no result to return to by this point, so every outcome is logged
|
||||
* only.
|
||||
*/
|
||||
private void runRollover(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
||||
String lead = p.leadTerminal();
|
||||
long rollStartMillis = nowMillis.getAsLong();
|
||||
TurnSettleResult turnResult = waitUntilAtTurnBoundary(lead, cfg.turnSettleSeconds());
|
||||
if (!turnResult.settled()) {
|
||||
// fleetd #494 follow-up: this line had the SAME defect as the /clear-timeout line below
|
||||
// — cfg.turnSettleSeconds() is the CONFIGURED budget, not how long this wait actually
|
||||
// ran. Print the measured elapsed time alongside it, labelled, exactly like the /clear
|
||||
// path already does.
|
||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) after "
|
||||
+ "confirm() — refusing to send /clear at all; the calling lead's own "
|
||||
+ "turn is still live and clearing it now would destroy live context "
|
||||
+ "(token={}, configured={}s elapsed={}ms)",
|
||||
lead, p.token(), cfg.turnSettleSeconds(), turnResult.elapsedMillis());
|
||||
return;
|
||||
}
|
||||
|
||||
// This deliberately bypasses Injector, exactly like ClaudeCodeLauncher#clearContext:
|
||||
// /clear is housekeeping, not a delegated turn, and routing it through Injector wedges the
|
||||
// pane forever (see this class's javadoc).
|
||||
agents.send(lead, "/clear");
|
||||
ClearSettleResult clearResult = waitForClearPickupAndSettle(lead, cfg.clearSettleSeconds());
|
||||
if (!clearResult.settled()) {
|
||||
// fleetd #494: cfg.clearSettleSeconds() is the CONFIGURED budget, not how long the wait
|
||||
// actually ran — an operator reading only that number wrongly believes it is a measured
|
||||
// duration. Print the measured elapsed time and nudge count alongside it, each labelled,
|
||||
// so the two can be compared at a glance.
|
||||
log.warn("lead-rollover: pane {} did not reach a turn boundary (IDLE or DONE) after "
|
||||
+ "/clear — NOT sending bootstrapText (token={}, configured={}s "
|
||||
+ "elapsed={}ms nudges={})",
|
||||
lead, p.token(), cfg.clearSettleSeconds(), clearResult.elapsedMillis(),
|
||||
clearResult.nudges());
|
||||
return;
|
||||
}
|
||||
agents.send(lead, cfg.bootstrapTextFor(p.handoverPath()));
|
||||
long rollElapsedMillis = nowMillis.getAsLong() - rollStartMillis;
|
||||
log.info("lead-rollover: rolled token={} lead={} elapsedMs={}", p.token(), lead, rollElapsedMillis);
|
||||
}
|
||||
|
||||
/** Drop a pending request without rolling. @return whether a pending request existed for {@code token} */
|
||||
public boolean cancel(String token) {
|
||||
return pending.remove(token) != null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The three handover-file checks, in order: exists, not empty, fresh (modified after
|
||||
* {@link #open}'s timestamp and not older than {@code maxDocAgeSeconds}). Stats {@code
|
||||
* p.handoverPath()} directly — {@link #open} already resolved it to an absolute path, so this
|
||||
* never has to guess which directory it means.
|
||||
*
|
||||
* @return the first failing check's refusal, or {@code null} when all three pass
|
||||
*/
|
||||
private RollDecision checkHandover(PendingRollover p, FleetConfig.LeadRollover cfg) {
|
||||
Path path = Path.of(p.handoverPath());
|
||||
if (!Files.exists(path)) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_MISSING,
|
||||
"handover file " + p.handoverPath() + " does not exist");
|
||||
}
|
||||
long size;
|
||||
long mtimeMillis;
|
||||
try {
|
||||
size = Files.size(path);
|
||||
mtimeMillis = Files.getLastModifiedTime(path).toMillis();
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("failed to stat handover file " + p.handoverPath(), e);
|
||||
}
|
||||
if (size == 0) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_EMPTY,
|
||||
"handover file " + p.handoverPath() + " is empty");
|
||||
}
|
||||
if (mtimeMillis <= p.requestedAtMillis()) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_STALE,
|
||||
"handover file " + p.handoverPath() + " was not modified after the open() "
|
||||
+ "request (mtime=" + mtimeMillis + "ms, requestedAt=" + p.requestedAtMillis() + "ms)");
|
||||
}
|
||||
long ageMillis = nowMillis.getAsLong() - mtimeMillis;
|
||||
long maxAgeMillis = TimeUnit.SECONDS.toMillis(cfg.maxDocAgeSeconds());
|
||||
if (ageMillis > maxAgeMillis) {
|
||||
return RollDecision.refused(RefusalReason.HANDOVER_STALE,
|
||||
"handover file " + p.handoverPath() + " is " + TimeUnit.MILLISECONDS.toSeconds(ageMillis)
|
||||
+ "s old, older than maxDocAgeSeconds=" + cfg.maxDocAgeSeconds());
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Poll {@link AgentControl#status} until {@code target} reports a real turn boundary — {@link
|
||||
* AgentStatus#IDLE} or {@link AgentStatus#DONE} — bounded by {@code settleSeconds}. Used once by
|
||||
* {@link #runRollover}, to wait for the CALLING turn's own pane to settle before {@code /clear}
|
||||
* is ever sent at all — the {@code turnSettleSeconds} gate that makes this correction safe. The
|
||||
* SECOND wait, after {@code /clear}, is {@link #waitForClearPickupAndSettle} instead (fleetd
|
||||
* #489) — a plain boundary check is not enough there, because {@code /clear} starts no turn of
|
||||
* its own, so this method would (wrongly) report "settled" on its very first poll whether or not
|
||||
* {@code /clear} was actually picked up. A failed status read degrades to "not yet settled" and
|
||||
* is retried on the next poll, the same posture {@code LeadHeartbeatLoop} and {@code
|
||||
* HerdrPeerLauncher}'s readiness gate already take toward an unreadable status.
|
||||
*
|
||||
* <p><strong>Deliberately not {@link AgentStatus#injectable()}.</strong> {@code injectable()}
|
||||
* answers the {@code Injector}'s question — "may I deliver a message without stepping on a live
|
||||
* turn" — and it accepts {@link AgentStatus#BLOCKED} for that purpose, because a pane paused on
|
||||
* an approval prompt is safe to queue a message behind. This class asks a stricter question —
|
||||
* "has the turn actually ended" — and {@code BLOCKED} answers no: it is a live turn that is
|
||||
* merely paused, not one that has finished. Reusing {@code injectable()} here would let this
|
||||
* wait fire {@code /clear} while the lead's own {@code confirm()}-calling turn is still live and
|
||||
* paused on a prompt — exactly the live-context-destroying failure the {@code turnSettleSeconds}
|
||||
* gate exists to prevent. Do not "simplify" this back to {@code injectable()}. ({@link
|
||||
* #waitForClearPickupAndSettle} keeps the same exclusion of {@code BLOCKED}, for the same
|
||||
* reason, on the second wait.)
|
||||
*
|
||||
* @return a {@link TurnSettleResult} whose {@code settled()} is {@code true} once a real
|
||||
* boundary was observed, {@code false} if {@code settleSeconds} elapses first.
|
||||
* {@code elapsedMillis()} is a MEASURED value from the injected {@link #nowMillis}
|
||||
* clock, never the configured {@code settleSeconds} budget (fleetd #494 follow-up).
|
||||
*/
|
||||
private TurnSettleResult waitUntilAtTurnBoundary(String target, int settleSeconds) {
|
||||
long startMillis = nowMillis.getAsLong();
|
||||
long deadline = startMillis + TimeUnit.SECONDS.toMillis(settleSeconds);
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(target);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: status check failed while waiting for {} to settle: {}",
|
||||
target, e.toString());
|
||||
status = null;
|
||||
}
|
||||
if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||
return new TurnSettleResult(true, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
settleSleeper.run();
|
||||
}
|
||||
return new TurnSettleResult(false, nowMillis.getAsLong() - startMillis);
|
||||
}
|
||||
|
||||
/**
|
||||
* The measured outcome of {@link #waitUntilAtTurnBoundary} — fleetd #494 follow-up. The sibling
|
||||
* of {@link ClearSettleResult} for the FIRST wait, which never nudges, so it carries no nudge
|
||||
* count.
|
||||
*/
|
||||
private record TurnSettleResult(boolean settled, long elapsedMillis) {}
|
||||
|
||||
/**
|
||||
* The SECOND wait in {@link #runRollover} — after {@code /clear} has been sent, waits for it to
|
||||
* settle, bounded by {@code settleSeconds}. <strong>fleetd #489 — the paste-race fix.</strong>
|
||||
* {@code /clear} does not start a real turn of its own, so a pane with no submit race simply
|
||||
* stays {@link AgentStatus#IDLE} the whole time: {@link #waitUntilAtTurnBoundary} would (wrongly)
|
||||
* call that "settled" on its very first poll, whether or not the {@code /clear} Enter actually
|
||||
* landed. That was Fault 1, measured live on 2026-09-12 — the second gate was a no-op, so a
|
||||
* {@code bootstrapText} send followed immediately, racing Fault 2: {@link AgentControl#submit}'s
|
||||
* own javadoc already records that the submit accompanying a delivery "can race the paste —
|
||||
* especially right as the worker's TUI becomes interactive — leaving the text unsubmitted"
|
||||
* (CB-113). Because {@code runRollover} deliberately bypasses {@code Injector} for {@code
|
||||
* /clear} (see this class's javadoc), it inherited none of {@code Injector}'s nudging — so the
|
||||
* lost {@code /clear} Enter sat in the input box and {@code bootstrapText} was typed right after
|
||||
* it, landing as one concatenated line.
|
||||
*
|
||||
* <p>This method copies the pickup-nudge pattern {@link dev.ltms.fleet.inject.Injector} already
|
||||
* ships for exactly this, on its own post-turn {@code /clear} housekeeping (fleetd #306; see
|
||||
* {@code Injector.java:288-340} and {@code Injector.java:437-442}):
|
||||
* <ul>
|
||||
* <li>an {@link AgentStatus#WORKING} sample means {@code /clear} was picked up as a real
|
||||
* turn;</li>
|
||||
* <li>until that happens, each poll that still reports {@link AgentStatus#IDLE} or {@link
|
||||
* AgentStatus#DONE} re-sends the submit keystroke ({@link AgentControl#submit}) to nudge
|
||||
* the raced Enter — for the first {@code PICKUP_GRACE_POLLS - 1} of {@link
|
||||
* #PICKUP_GRACE_POLLS} consecutive such polls (i.e. {@code PICKUP_GRACE_POLLS - 1}
|
||||
* nudges: 7, not 8, given {@code PICKUP_GRACE_POLLS = 8}). A second Enter on an empty
|
||||
* Claude Code prompt is a no-op, so repeating it is safe;</li>
|
||||
* <li>the {@code PICKUP_GRACE_POLLS}th consecutive such poll, with {@code WORKING} still never
|
||||
* observed, releases rather than wedges the roll instead of nudging again — the same
|
||||
* choice {@code Injector} makes — and returns {@code settled() == true} anyway, logged at
|
||||
* {@code warn} with the measured elapsed time (fleetd #494) so an operator can see which
|
||||
* path ran and how long it actually took;</li>
|
||||
* <li>once {@code WORKING} has been observed, nudging stops and this instead waits for a real
|
||||
* {@code working → IDLE/DONE} completion boundary before returning {@code true}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>{@link AgentStatus#BLOCKED} is deliberately excluded from both the nudge and the
|
||||
* boundary check</strong> — the same reasoning as {@link #waitUntilAtTurnBoundary}'s own
|
||||
* javadoc: a paused live turn is not a settled one, and re-sending Enter into an open approval
|
||||
* prompt could wrongly answer it. A {@code BLOCKED} sample (or an unreadable/{@link
|
||||
* AgentStatus#UNKNOWN} one) simply keeps this polling, with no nudge and no release, until either
|
||||
* a real boundary is reached or {@code settleSeconds} runs out.
|
||||
*
|
||||
* <p>{@link AgentControl#submit} can itself throw; a {@link RuntimeException} from it is
|
||||
* swallowed and logged at {@code debug}, exactly like {@code Injector.java:437-442} — a failed
|
||||
* nudge must not abort the roll.
|
||||
*
|
||||
* @return a {@link ClearSettleResult} whose {@code settled()} is {@code true} once {@code
|
||||
* /clear} has settled, or once the nudge budget was exhausted with no pickup ever
|
||||
* observed (released rather than wedged); {@code false} if {@code settleSeconds} elapses
|
||||
* first — the caller must NOT send {@code bootstrapText} in that case, exactly as before
|
||||
* this fix. {@code elapsedMillis()} and {@code nudges()} are MEASURED values (from the
|
||||
* injected {@link #nowMillis} clock and an actual nudge count), never the configured
|
||||
* {@code settleSeconds} budget (fleetd #494).
|
||||
*/
|
||||
private ClearSettleResult waitForClearPickupAndSettle(String target, int settleSeconds) {
|
||||
long startMillis = nowMillis.getAsLong();
|
||||
long deadline = startMillis + TimeUnit.SECONDS.toMillis(settleSeconds);
|
||||
boolean pickedUp = false; // a WORKING sample has been observed since /clear was sent
|
||||
int idlePollsAwaitingPickup = 0;
|
||||
int nudges = 0;
|
||||
while (nowMillis.getAsLong() < deadline) {
|
||||
AgentStatus status;
|
||||
try {
|
||||
status = agents.status(target);
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: status check failed while waiting for {} to settle after "
|
||||
+ "/clear: {}", target, e.toString());
|
||||
status = null;
|
||||
}
|
||||
if (status == AgentStatus.WORKING) {
|
||||
pickedUp = true;
|
||||
} else if (status == AgentStatus.IDLE || status == AgentStatus.DONE) {
|
||||
if (pickedUp) {
|
||||
// a real WORKING -> IDLE/DONE completion boundary
|
||||
return new ClearSettleResult(true, nowMillis.getAsLong() - startMillis, nudges);
|
||||
}
|
||||
if (++idlePollsAwaitingPickup >= PICKUP_GRACE_POLLS) {
|
||||
long elapsedMillis = nowMillis.getAsLong() - startMillis;
|
||||
// fleetd #494: this release trades a possibly-unsubmitted /clear for progress
|
||||
// instead of wedging the roll — that trade is deliberate and stays. But it is
|
||||
// also exactly the case that reported false success in the real incident (the
|
||||
// whole roll "succeeded" after 438ms of a 20s budget), so raise it to WARN and
|
||||
// print the MEASURED elapsed time next to the target pane, not just the count.
|
||||
//
|
||||
// fleetd #494 follow-up (2nd pass): BOTH numbers in this line must come from
|
||||
// the loop's own counters, never from the PICKUP_GRACE_POLLS constant.
|
||||
// `idlePollsAwaitingPickup` and `nudges` each have exactly one write site in
|
||||
// this loop, on the same branch, so on this branch they cannot differ from
|
||||
// PICKUP_GRACE_POLLS / PICKUP_GRACE_POLLS - 1 today — no test can prove the
|
||||
// difference on this line, and printing the counters does not change that.
|
||||
// What it does buy: one source of truth instead of two, so a later change to
|
||||
// the loop (an early return, a second increment site, a different exit
|
||||
// condition) cannot leave this message reporting a number the loop no longer
|
||||
// produces. The place where `nudges` genuinely varies with the run — and is
|
||||
// covered by a test that can tell it apart from a constant — is the
|
||||
// /clear-timeout warn in runRollover, which prints clearResult.nudges().
|
||||
log.warn("lead-rollover: /clear on {} was never observed as WORKING after {} "
|
||||
+ "consecutive IDLE/DONE polls ({} of those were nudged) — "
|
||||
+ "releasing rather than wedging the roll (elapsed={}ms)",
|
||||
target, idlePollsAwaitingPickup, nudges, elapsedMillis);
|
||||
return new ClearSettleResult(true, elapsedMillis, nudges);
|
||||
}
|
||||
try {
|
||||
agents.submit(target); // nudge a raced Enter (CB-113) so /clear actually submits
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("lead-rollover: resubmit to {} failed (will retry next poll): {}",
|
||||
target, e.getMessage());
|
||||
} finally {
|
||||
nudges++; // an attempted nudge, whether or not the submit call itself threw
|
||||
}
|
||||
}
|
||||
// AgentStatus.BLOCKED or UNKNOWN (or an unreadable status, above): neither a pickup
|
||||
// signal nor a boundary — keep polling without nudging or releasing.
|
||||
settleSleeper.run();
|
||||
}
|
||||
return new ClearSettleResult(false, nowMillis.getAsLong() - startMillis, nudges);
|
||||
}
|
||||
|
||||
/**
|
||||
* The measured outcome of {@link #waitForClearPickupAndSettle} — fleetd #494. Carries the
|
||||
* MEASURED elapsed time (from the injected {@link #nowMillis} clock) and nudge count alongside
|
||||
* the settle/timeout decision, so callers can log them instead of the configured budget, which
|
||||
* is not how long the wait actually ran.
|
||||
*/
|
||||
private record ClearSettleResult(boolean settled, long elapsedMillis, int nudges) {}
|
||||
}
|
||||
@@ -0,0 +1,81 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.regex.Matcher;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
/**
|
||||
* fleetd #469: a launch charter that names an MCP tool the server does not register must stop the
|
||||
* daemon at startup, not wait for a member to discover the gap by calling something that is not
|
||||
* there.
|
||||
*
|
||||
* <p>Deliberately its own class outside {@code dev.ltms.fleet.config}, not a case in {@link
|
||||
* dev.ltms.fleet.config.FleetConfig#validateCharters()}. The canonical tool surface ({@link
|
||||
* FleetTool}) lives in the {@code mcp} package; config is loaded before the MCP server exists and
|
||||
* must not gain a dependency on it. So this check belongs at the seam that already holds both a
|
||||
* loaded {@code FleetConfig} and the {@code mcp} package: {@code Fleetd.main}, called right after
|
||||
* {@code cfg.validateAll()} and before anything opens a socket or spawns a member.
|
||||
*
|
||||
* <p>{@code #464}'s {@code CharterToolSurfaceTest} proved the same comparison against a charter
|
||||
* fixture it wrote itself into a {@code @TempDir}, which meant nothing anyone wrote into the live
|
||||
* {@code fleetd.yaml} could ever fail it. This class is what a real charter is actually checked
|
||||
* against at boot; {@code FleetdStartupValidationTest} exercises it through {@code Fleetd.main}
|
||||
* itself, the same way it proves every other {@code validateXxx()} still runs there.
|
||||
*
|
||||
* <p><strong>fleetd #474</strong> — startup was not the only door: {@code
|
||||
* dev.ltms.fleet.config.ConfigRef#reload()} used to run {@code FleetConfig#validateAll()} alone,
|
||||
* which does not look at what a charter's text names, so a charter naming an unregistered tool
|
||||
* that could not have booted the daemon could still be installed into a running one through a
|
||||
* reload. This class still knows nothing about {@code ConfigRef} — {@code Fleetd.main} wires a
|
||||
* small {@code FleetConfig -> void} adapter over {@link #assertChartersNameOnlyRegisteredTools}
|
||||
* ({@code Fleetd::assertChartersNameOnlyRegisteredTools}) into {@code ConfigRef}'s constructor as
|
||||
* its {@code Consumer<FleetConfig>} {@code extraValidation}, run inside {@code reload()}'s same
|
||||
* try/catch as {@code validateAll()}, so both call sites — {@code Fleetd.main} at startup and
|
||||
* {@code ConfigRef#reload()} afterwards — go through this one method and can never check different
|
||||
* things. {@code dev.ltms.fleet.config.ConfigRefTest} and {@code
|
||||
* FleetdConfigRefCharterToolSurfaceWiringTest} are what prove the reload call site, the same way
|
||||
* {@code FleetdStartupValidationTest} proves the startup one.
|
||||
*/
|
||||
public final class CharterToolSurface {
|
||||
|
||||
/** A {@code fleet_…} (current) or {@code bridge_…} (pre-CB-634) tool-shaped token in prose. */
|
||||
private static final Pattern TOOL_REFERENCE = Pattern.compile("(fleet_[a-z_]+|bridge_[a-z_]+)");
|
||||
|
||||
private CharterToolSurface() {
|
||||
}
|
||||
|
||||
/**
|
||||
* @param charters the configured {@code fleet.charters:} map (role wire name → charter text);
|
||||
* {@code null} or empty is a no-op, same as an absent {@code fleet:} block
|
||||
* @throws IllegalStateException naming the charter key and every tool it names that {@link
|
||||
* FleetTool} does not list, when any charter does so
|
||||
*/
|
||||
public static void assertChartersNameOnlyRegisteredTools(Map<String, String> charters) {
|
||||
if (charters == null || charters.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
Set<String> registered = FleetTool.wireNames();
|
||||
List<String> bad = new ArrayList<>();
|
||||
charters.forEach((key, text) -> {
|
||||
if (text == null) {
|
||||
return;
|
||||
}
|
||||
Set<String> named = new LinkedHashSet<>();
|
||||
Matcher m = TOOL_REFERENCE.matcher(text);
|
||||
while (m.find()) {
|
||||
named.add(m.group(1));
|
||||
}
|
||||
named.stream()
|
||||
.filter(t -> !registered.contains(t))
|
||||
.forEach(unknown -> bad.add("fleet.charters." + key + " names '" + unknown
|
||||
+ "', which the server does not register (registered: " + registered + ")."));
|
||||
});
|
||||
if (!bad.isEmpty()) {
|
||||
throw new IllegalStateException("refusing to start: " + String.join(" ", bad));
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -36,6 +36,25 @@ public final class ConnectionIdentity {
|
||||
* primary / an off-host client) and its {@code pid} (or {@code -1} if not resolvable).
|
||||
*/
|
||||
public record Caller(String terminal, long pid) {
|
||||
|
||||
/**
|
||||
* Whether the OS peer-PID lookup actually succeeded — {@code false} means {@code pid} is
|
||||
* the {@code -1} sentinel, not a real process id, so this caller's identity could not be
|
||||
* established at all. That is a different fact from a real pid that simply owns no worker
|
||||
* pane (the primary's own connection): the primary is {@code resolved()} and has a
|
||||
* {@code null terminal}; an unresolvable caller is {@code !resolved()} and also has a
|
||||
* {@code null terminal}. The two look identical through {@link #terminal} alone, which is
|
||||
* exactly how fleetd #317 happened — a failed {@code lsof} lookup and a genuine primary both
|
||||
* fell through to {@code Principal.primary(...)}.
|
||||
*
|
||||
* <p>Centralised here, next to the sentinel it tests, for the same reason
|
||||
* {@link ConnectionIdentity#isLoopback} is centralised rather than left for each caller to
|
||||
* reimplement: a raw {@code pid > 0} check duplicated at every call site is precisely the
|
||||
* "one rule, two copies" shape that let #305 drift.
|
||||
*/
|
||||
public boolean resolved() {
|
||||
return pid > 0;
|
||||
}
|
||||
}
|
||||
|
||||
/** Resolve the caller's terminal and PID from one peer-PID lookup. */
|
||||
@@ -60,7 +79,44 @@ public final class ConnectionIdentity {
|
||||
return pid > 0 ? cwds.cwdForPid(pid) : null;
|
||||
}
|
||||
|
||||
private static boolean isLoopback(String addr) {
|
||||
return "127.0.0.1".equals(addr) || "::1".equals(addr) || "0:0:0:0:0:0:0:1".equals(addr);
|
||||
/**
|
||||
* Whether {@code addr} is a same-host address, and therefore one whose peer PID is worth
|
||||
* looking up. <strong>This is the one definition of loopback in the daemon</strong> —
|
||||
* {@code CallerResolver} calls it rather than keeping its own, because the two used to differ
|
||||
* and that difference was a privilege escalation (fleetd #305).
|
||||
*
|
||||
* <p>The whole of {@code 127.0.0.0/8} counts, not just {@code 127.0.0.1}. On Linux every
|
||||
* address in that range is bound to {@code lo} by default, so a process can connect to
|
||||
* {@code 127.0.0.1:8765} with a source address of {@code 127.0.0.2} — measured on the Linux
|
||||
* fleet host, where binding that source succeeds.
|
||||
*
|
||||
* <p><strong>What excluding an address costs, stated as it is today.</strong> This paragraph
|
||||
* used to say that narrowing this range turned a worker into the lead, and that widening the
|
||||
* check was what closed the hole. That was true only while there were <em>two</em> definitions
|
||||
* that disagreed: {@code ConnectionIdentity} skipped the identity lookup for {@code 127.0.0.2}
|
||||
* while {@code CallerResolver} read the same address as loopback and granted the primary role.
|
||||
* #305 removed the second copy, and with one shared definition the old sentence no longer holds.
|
||||
*
|
||||
* <p>Measured on 2026-09-04 by narrowing this method back to exactly {@code 127.0.0.1} and
|
||||
* running {@code CallerResolverTest} and {@code ConnectionIdentityTest}: a caller from
|
||||
* {@code 127.0.0.2} then resolves to {@code ANONYMOUS}, not {@code PRIMARY} — for a worker
|
||||
* ({@code aWorkerOnAnyLoopbackSourceAddressIsStillAWorkerNotThePrimary}) and for a non-worker
|
||||
* ({@code aNonWorkerOnAnyLoopbackSourceAddressIsStillThePrimary}) alike. Excluding an address
|
||||
* now <em>refuses</em> its caller; it does not promote one.
|
||||
*
|
||||
* <p>So keep the whole range, but for the plain reason: a genuine worker or primary that
|
||||
* connects from {@code 127.0.0.2} must be identifiable at all, and narrowing this predicate
|
||||
* locks it out. That is an outage, and an outage is the direction to fail in — which is exactly
|
||||
* why the range must not be narrowed casually and also why doing so is no longer a security
|
||||
* hole. This predicate still does not decide whether a caller is trusted; it decides whether the
|
||||
* caller's identity is <em>resolved at all</em>. What makes an unresolved caller safe is
|
||||
* {@link Caller#resolved()} (#317), not this method.
|
||||
*/
|
||||
public static boolean isLoopback(String addr) {
|
||||
if (addr == null) {
|
||||
return false;
|
||||
}
|
||||
String a = addr.startsWith("::ffff:") ? addr.substring(7) : addr; // IPv4-mapped IPv6
|
||||
return a.startsWith("127.") || "::1".equals(a) || "0:0:0:0:0:0:0:1".equals(a);
|
||||
}
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,83 @@
|
||||
package dev.ltms.fleet.mcp;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
|
||||
/**
|
||||
* The one canonical set of MCP tool names this daemon registers (fleetd #469, follow-up to #464).
|
||||
*
|
||||
* <p>Before this enum, the tool surface was written twice with nothing tying the copies together:
|
||||
* once as the literal {@code "fleet_…"} string passed to each tool-schema builder in
|
||||
* {@link FleetMcp}, and again as the case labels of {@link FleetMcp}'s authorization switch. A
|
||||
* reader that needed "what does this server register" — a charter check, in particular — had no
|
||||
* source to ask except scraping {@code FleetMcp.java}'s source text for {@code tool("…")} calls: a
|
||||
* third copy of the same list, and the weakest of the three forms.
|
||||
*
|
||||
* <p>Every reader that needs the registered tool surface now asks this enum instead:
|
||||
*
|
||||
* <ul>
|
||||
* <li>the tool-schema builders in {@code FleetMcp} pass {@code wireName()} rather than a literal;
|
||||
* <li>{@code FleetMcp}'s constructor asserts, at startup, that the set of tool names it actually
|
||||
* registers with the MCP SDK equals {@link #wireNames()} exactly — a canonical entry that is
|
||||
* never registered, or a registration with no canonical entry backing it, fails the daemon's
|
||||
* own boot rather than only a test's;
|
||||
* <li>{@code FleetMcp.toolAction}'s dispatch onto {@code Authz.Action} switches on the enum
|
||||
* (not the raw string) with no {@code default}, so adding a tool here without pinning its
|
||||
* action is a compile error, not a run-time throw;
|
||||
* <li>{@link CharterToolSurface} asks {@link #wireNames()} to check a configured launch charter
|
||||
* against the live tool surface, instead of scraping source text a third time.
|
||||
* </ul>
|
||||
*/
|
||||
public enum FleetTool {
|
||||
|
||||
SEND("fleet_send"),
|
||||
REPLY("fleet_reply"),
|
||||
ASK("fleet_ask"),
|
||||
STATUS("fleet_status"),
|
||||
POLL("fleet_poll"),
|
||||
ACK("fleet_ack"),
|
||||
SPAWN("fleet_spawn"),
|
||||
LIST("fleet_list"),
|
||||
STOP("fleet_stop"),
|
||||
PROFILES("fleet_profiles"),
|
||||
WHOAMI("fleet_whoami"),
|
||||
HANDOVER("fleet_handover");
|
||||
|
||||
private final String wireName;
|
||||
|
||||
FleetTool(String wireName) {
|
||||
this.wireName = wireName;
|
||||
}
|
||||
|
||||
/** The name this tool is registered under, and called by, on the wire ({@code "fleet_send"}, …). */
|
||||
public String wireName() {
|
||||
return wireName;
|
||||
}
|
||||
|
||||
private static final Map<String, FleetTool> BY_WIRE_NAME;
|
||||
private static final Set<String> WIRE_NAMES;
|
||||
|
||||
static {
|
||||
Map<String, FleetTool> byName = new LinkedHashMap<>();
|
||||
Set<String> names = new LinkedHashSet<>();
|
||||
for (FleetTool tool : values()) {
|
||||
byName.put(tool.wireName, tool);
|
||||
names.add(tool.wireName);
|
||||
}
|
||||
BY_WIRE_NAME = Map.copyOf(byName);
|
||||
WIRE_NAMES = Set.copyOf(names);
|
||||
}
|
||||
|
||||
/** The tool named {@code wireName}, or empty when this daemon registers no such tool. */
|
||||
public static Optional<FleetTool> byWireName(String wireName) {
|
||||
return Optional.ofNullable(BY_WIRE_NAME.get(wireName));
|
||||
}
|
||||
|
||||
/** Every wire name this daemon registers — the canonical tool surface. */
|
||||
public static Set<String> wireNames() {
|
||||
return WIRE_NAMES;
|
||||
}
|
||||
}
|
||||
@@ -41,6 +41,14 @@ public final class LsofPeerPidLookup implements PeerPidLookup {
|
||||
if (!p.waitFor(2, TimeUnit.SECONDS)) {
|
||||
p.destroyForcibly();
|
||||
}
|
||||
if (found < 0) {
|
||||
// fleetd #317: this is the silent path — lsof ran clean and simply reported no
|
||||
// matching process (e.g. queried before the OS socket table settles). Previously
|
||||
// this logged nothing at all, which is exactly why the escalation went unnoticed;
|
||||
// the exception path below already logs. A caller now refused because of this is
|
||||
// still refused (never promoted) — this line only makes the refusal diagnosable.
|
||||
log.debug("lsof peer-pid lookup for port {} found no matching process", port);
|
||||
}
|
||||
return found;
|
||||
} catch (Exception e) {
|
||||
log.debug("lsof peer-pid lookup for port {} failed: {}", port, e.getMessage());
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import com.fasterxml.jackson.databind.node.ObjectNode;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
@@ -14,6 +17,9 @@ import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.StandardCopyOption;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
import java.util.Arrays;
|
||||
import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -47,6 +53,21 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(ClaudeCodeLauncher.class);
|
||||
|
||||
/** JSON codec for the additive workspace-trust seed (fleetd #149) — Jackson's default settings. */
|
||||
private static final ObjectMapper TRUST_JSON = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* Serialises every {@link #seedTrustDialog} read-modify-write for the whole daemon process.
|
||||
* Two claude-code spawns starting at once are normal (fleetd runs several members in parallel
|
||||
* routinely) and both would otherwise read the same {@code .claude.json}, add their own entry
|
||||
* to their own in-memory copy, and write — the second write wins and the first spawn's trust
|
||||
* entry silently disappears. A single process-wide lock is enough because every spawn on this
|
||||
* daemon runs in this one JVM; it does not protect against a second daemon process or the
|
||||
* operator's own Claude Code process writing at the same instant, which {@link #writeAtomically}
|
||||
* covers instead (each writer only ever sees a fully-old or fully-new file, never a torn one).
|
||||
*/
|
||||
private static final Object TRUST_JSON_LOCK = new Object();
|
||||
|
||||
private final SubscriptionGuard guard;
|
||||
|
||||
/**
|
||||
@@ -94,13 +115,26 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, null, null);
|
||||
}
|
||||
|
||||
/** Production constructor, plus the live config for URI environment exclusions. */
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
fleet, memberCredentials);
|
||||
fleet, memberCredentials, hostEnvNames, config);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -153,10 +187,22 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, guard, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
fleet, memberCredentials, null, null);
|
||||
}
|
||||
|
||||
/** Full testability constructor, plus the live config for URI environment exclusions. */
|
||||
public ClaudeCodeLauncher(AgentControl agents, WorkspaceControl spaces, SubscriptionGuard guard,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames,
|
||||
Supplier<FleetConfig> config) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames, config);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
@@ -174,7 +220,7 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<Set<String>> hostEnvNames) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames);
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, hostEnvNames, null);
|
||||
this.guard = guard;
|
||||
}
|
||||
|
||||
@@ -233,11 +279,15 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
workerEnv.remove("ANTHROPIC_AUTH_TOKEN");
|
||||
} else {
|
||||
workerEnv.put("ANTHROPIC_BASE_URL", baseUrl);
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", env.apply(cfg.tokenEnv()));
|
||||
putIfPresent(workerEnv, "ANTHROPIC_AUTH_TOKEN", resolveEnv(cfg.tokenEnv()));
|
||||
}
|
||||
putIfPresent(workerEnv, "ANTHROPIC_MODEL", cfg.model());
|
||||
putIfPresent(workerEnv, "CLAUDE_CONFIG_DIR", cfg.configDir());
|
||||
applyGitToken(workerEnv, cfg);
|
||||
// fleetd #149: seed the workspace-trust entry BEFORE this spawn ever reaches herdr — see
|
||||
// seedTrustDialog for why, and isProvisionedWorktree for why this is gated to a worktree
|
||||
// fleetd itself provisioned (never a real checkout, never an un-configured fallback cwd).
|
||||
seedTrustDialog(cfg.configDir(), spec.cwd());
|
||||
|
||||
// CB-547a: Claude Code can MINT its own session id, so fleetd chooses it — a fresh spawn
|
||||
// gets a UUID we pass as --session-id and return from agentSessionId(), so the resume
|
||||
@@ -253,18 +303,22 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
/**
|
||||
* Add the Claude-specific session-identity flags to {@code argv} and return the peer's OWN
|
||||
* session id — the resume handle. A resume request passes the prior id via {@code -r} and
|
||||
* returns that id; a fresh named session mints a new UUID, passes it via {@code --session-id},
|
||||
* and returns the mint. The bridge's logical name rides along as {@code -n} when present. When
|
||||
* <em>no</em> identity is requested (sessionName and resumeSessionId both blank) this adds
|
||||
* nothing and returns {@code null}, keeping the legacy no-identity launch byte-identical.
|
||||
* session id — the resume handle. A resume passes the prior id via {@code -r} and returns
|
||||
* that id; every other spawn mints a new UUID, passes it via {@code --session-id}, and
|
||||
* returns the mint. The bridge's logical name rides along as {@code -n} when present.
|
||||
*
|
||||
* <p>fleetd #214: the mint is unconditional. A plain {@code fleet_spawn} passes neither
|
||||
* sessionName nor resumeSessionId, yet the member must still be resumable, and this id is
|
||||
* the only resume handle a claude-code member has — unlike opencode, nothing resolves it
|
||||
* after the launch. Checked against the real binary (claude 2.1.252): the flag is safe on
|
||||
* every spawn. The binary takes only a valid UUID — it refuses any other value at argument
|
||||
* parsing ("Invalid session ID. Must be a valid UUID.") — so the {@code UUID.randomUUID()}
|
||||
* mint is required, not incidental. The only flag interaction the binary documents is with
|
||||
* {@code -r} (both claim the session id), and the resume branch above never combines the two.
|
||||
*/
|
||||
private static String applySessionIdentity(List<String> argv, String sessionName, String resumeSessionId) {
|
||||
boolean resuming = resumeSessionId != null && !resumeSessionId.isBlank();
|
||||
boolean named = sessionName != null && !sessionName.isBlank();
|
||||
if (!resuming && !named) {
|
||||
return null; // no identity requested — keep the legacy launch byte-identical
|
||||
}
|
||||
if (named) {
|
||||
argv.add("-n");
|
||||
argv.add(sessionName);
|
||||
@@ -295,11 +349,19 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
*
|
||||
* <p>CB-618: Claude Code refuses to start when BOTH {@code --append-system-prompt} and
|
||||
* {@code --append-system-prompt-file} are on the command line ("Cannot use both ... Please use
|
||||
* only one"), so the two charters can never travel on separate flags. When both are present they
|
||||
* are concatenated into the one file, role charter first and reply charter last — last is where
|
||||
* the reply rule must sit, because it is the rule that must survive. When only the reply charter
|
||||
* is present it keeps its proven inline {@code --append-system-prompt} delivery, which is also
|
||||
* the only form that reaches a member with no repo checkout.
|
||||
* only one"), so the two charters can never travel on separate flags. They are concatenated
|
||||
* into the one file, role charter first and reply charter last — last is where the reply rule
|
||||
* must sit, because it is the rule that must survive.
|
||||
*
|
||||
* <p>fleetd #220: a lone reply charter used to ride inline on {@code --append-system-prompt},
|
||||
* which put ~800 bytes of prose on the command line herdr types into the pane. That line is
|
||||
* capped at {@value HerdrPeerLauncher#PANE_COMMAND_BYTE_LIMIT} bytes by the pty itself, and
|
||||
* everything past the cap is dropped with no error from any layer. The charter alone left about
|
||||
* 50 bytes of headroom, so adding one flag ({@code --session-id}, fleetd #214) truncated the
|
||||
* LAST argument instead — {@code --autocompact 250000} arrived as {@code --autocompact 25},
|
||||
* claude rejected it, and every claude-code spawn died as an unexplained readiness timeout.
|
||||
* The charter now always travels as a file, which takes the prose off the command line for
|
||||
* good; {@link HerdrPeerLauncher#checkPaneCommandFits} is the backstop for whatever grows next.
|
||||
*/
|
||||
private List<String> argvWithFleet(FleetConfig.Profile cfg, LaunchSpec spec) {
|
||||
String roleCharter = nonBlank(spec.roleCharter());
|
||||
@@ -316,16 +378,13 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
argv.add("--mcp-config");
|
||||
argv.add(mcpConfigJson(cfg));
|
||||
}
|
||||
// Combine the charters in order role -> reply, dropping any that are absent. When two
|
||||
// or more survive they must ride one --append-system-prompt-file (CB-618 forbids the inline
|
||||
// flag and the file flag together). A lone reply charter keeps its proven inline delivery.
|
||||
// Combine the charters in order role -> reply, dropping any that are absent. They ride one
|
||||
// --append-system-prompt-file (CB-618 forbids the inline flag and the file flag together),
|
||||
// always — fleetd #220: charter prose on the command line overruns the pane's byte cap.
|
||||
List<String> charters = new java.util.ArrayList<>(2);
|
||||
if (roleCharter != null) charters.add(roleCharter);
|
||||
if (replyCharter != null) charters.add(replyCharter);
|
||||
if (charters.size() == 1 && replyCharter != null && roleCharter == null) {
|
||||
argv.add("--append-system-prompt");
|
||||
argv.add(replyCharter);
|
||||
} else if (!charters.isEmpty()) {
|
||||
if (!charters.isEmpty()) {
|
||||
argv.add("--append-system-prompt-file");
|
||||
argv.add(writeCharterFile(String.join("\n\n", charters)).toString());
|
||||
}
|
||||
@@ -378,12 +437,12 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private static void writeIdeOverlay(String cwd, String projectPath) {
|
||||
try {
|
||||
Path dotGit = Path.of(cwd, ".git");
|
||||
if (!Files.isRegularFile(dotGit)) {
|
||||
if (!isProvisionedWorktree(cwd)) {
|
||||
// Not a provisioned worktree (primary's real checkout has a .git directory, or the
|
||||
// cwd is not a repo at all). Never write into it.
|
||||
return;
|
||||
}
|
||||
Path dotGit = Path.of(cwd, ".git");
|
||||
// The overlay FILE lives at the worktree root (claude-code's cwd), but its CONTENT pins
|
||||
// project_path to the module dir the IDE opened (projectPath), not the worktree root.
|
||||
Files.writeString(Path.of(cwd, "CLAUDE.local.md"), PeerLauncher.ideOverlayText(projectPath));
|
||||
@@ -413,6 +472,321 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #149: pre-seed the workspace-trust entry for {@code cwd} in Claude Code's own config
|
||||
* file, BEFORE this launch ever reaches herdr (called from {@link #buildLaunch}, which always
|
||||
* runs before the base class starts the process). Claude Code asks an interactive, un-timed
|
||||
* "Is this a project you created or one you trust?" the first time it starts in a directory it
|
||||
* has not seen before, and every {@code worktree: true} spawn lands in a brand-new directory —
|
||||
* so without this seed the member sits on that dialog forever, never mounts the bridge MCP, and
|
||||
* never calls {@code fleet_reply}. herdr reports it as healthy the whole time ({@code
|
||||
* agent_status: blocked}, {@code interactive_ready: true}), so nothing else catches it. Measured
|
||||
* live on fleet01 2026-08-23: an unseeded fresh cwd sat on the dialog indefinitely; a seeded one
|
||||
* reached {@code idle} clean.
|
||||
*
|
||||
* <p>This is not a new grant — the operator already trusted this repo by configuring the
|
||||
* profile against it, and a worktree is a checkout of that same repo.
|
||||
*
|
||||
* <p>The key Claude Code reads is per-project, in {@code .claude.json}: {@code
|
||||
* projects.<cwd>.hasTrustDialogAccepted}. The file lives at {@code <configDir>/.claude.json}
|
||||
* when the profile sets {@code CLAUDE_CONFIG_DIR} (mirrors this launcher's own env var above),
|
||||
* else the default {@code ~/.claude.json} — the same file Claude Code itself would read either
|
||||
* way, so this seeds exactly what the spawned peer is about to open.
|
||||
*
|
||||
* <p><b>Additive, not a rewrite.</b> {@code .claude.json} is large (tens of KB, dozens of
|
||||
* projects) and Claude Code itself rewrites it while running, so this reads the file as a JSON
|
||||
* tree (missing or unreadable → treated as an empty object) and changes only
|
||||
* {@code projects.<cwd>.hasTrustDialogAccepted} (that key alone — see fleetd #247) —
|
||||
* every other top-level key and every other project entry is written back untouched. Only the
|
||||
* one project entry for {@code cwd} is replaced/created; an existing entry for a DIFFERENT cwd
|
||||
* (or the operator's own project history) is never touched.
|
||||
*
|
||||
* <p>Best-effort, like {@link #writeIdeOverlay}: a failure here (unwritable configDir, a
|
||||
* corrupt existing file, …) must never fail the spawn — it is logged at debug and swallowed. A
|
||||
* peer that starts without the seed still starts; it just may hit the dialog fleetd #149
|
||||
* describes.
|
||||
*
|
||||
* <p><b>Gated to a provisioned worktree</b> ({@link HerdrPeerLauncher#isProvisionedWorktree}) — see that
|
||||
* method's javadoc for the incident that made this gate mandatory, not optional: this must
|
||||
* never run against a real checkout or an un-configured fallback cwd, only the exact
|
||||
* always-fresh-directory population fleetd #149 describes.
|
||||
*
|
||||
* <p><b>Atomic and lock-protected.</b> {@code .claude.json} is a live file — Claude Code itself
|
||||
* rewrites it while running, and this daemon routinely spawns several members at once, each
|
||||
* calling this method for its own cwd. Every write goes through {@link #writeAtomically} (a
|
||||
* sibling-temp-file + {@code ATOMIC_MOVE}, never a truncate-in-place) so a crash mid-write or a
|
||||
* concurrent reader never observes a half-written file, and through {@link #TRUST_JSON_LOCK} so
|
||||
* two concurrent spawns' entries both survive instead of the second write silently discarding
|
||||
* the first. Both exist because of a real incident: see {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc
|
||||
* and {@link #writeAtomically}'s javadoc.
|
||||
*
|
||||
* <p><b>Compare-and-swap against a writer the lock cannot reach (fleetd #247).</b>
|
||||
* {@code TRUST_JSON_LOCK} only serialises calls this launcher itself makes inside this one JVM.
|
||||
* It does nothing about a writer outside it — and on a host where the profile's
|
||||
* {@code configDir} is the operator's own {@code CLAUDE_CONFIG_DIR}, the file this method writes
|
||||
* IS the operator's own live Claude Code session's config file, being read and written by that
|
||||
* session while it runs. Measured 2026-09-03: its mtime moved minutes after a spawn while that
|
||||
* session was active. A plain read-modify-write there is a routine lost update, not a rare one:
|
||||
* fleetd reads v1, the operator's session reads v1 and writes v2 with their own change, fleetd's
|
||||
* {@code ATOMIC_MOVE} then lands v3 built from v1 — atomic, but v2's change is gone. So before
|
||||
* the move this method re-reads {@code target}'s exact bytes and compares them with the bytes it
|
||||
* built its update from; a mismatch means someone else wrote in between, and it discards its
|
||||
* work and rebuilds from the fresh bytes, up to {@link #MAX_TRUST_JSON_CAS_ATTEMPTS} times.
|
||||
* <b>Exhausting the retries writes nothing</b> — see the WARN at the end of the loop for why
|
||||
* that, not a last write-anyway, is the safe failure: the member shows the trust dialog and
|
||||
* fails to reach an injectable state, which is visible, logged and recoverable; overwriting the
|
||||
* operator's live config with a stale copy is neither. This narrows the lost-update window, it
|
||||
* does not close it — a write landing between the final re-read and the {@code ATOMIC_MOVE}
|
||||
* itself is still lost, because there is no OS-level compare-and-swap on a plain file, only this
|
||||
* cooperative narrowing of the gap.
|
||||
*
|
||||
* <p><b>fleetd #285: refuses under {@code memberHerdrSocket} rather than writing somewhere the
|
||||
* member cannot read.</b> Under {@code memberHerdrSocket:} the member pane runs as a
|
||||
* <em>different OS user with its own {@code $HOME}</em> — the same reason {@link
|
||||
* #writeCharterFile} routes the role/reply charter under {@code worktreeRoot} instead of
|
||||
* {@code java.io.tmpdir} and refuses the spawn when it cannot. This method has no equivalent
|
||||
* relocation available: unlike the charter (fleetd's own content, free to place anywhere and
|
||||
* hand to the peer via an argv flag), {@code .claude.json} is a file Claude Code looks up for
|
||||
* ITSELF at a fixed location — {@code CLAUDE_CONFIG_DIR/.claude.json}, or else the member OS
|
||||
* user's own {@code ~/.claude.json}, a path fleetd has no channel to learn. So when {@code
|
||||
* configDir} is unset, there is no member-readable target to seed at all — writing the
|
||||
* unqualified default would land in <em>fleetd's own</em> {@code ~/.claude.json} instead, the
|
||||
* exact defect this fix closes, not a workable fallback. And even with {@code configDir} set,
|
||||
* the file this method itself just wrote is {@code 0600} (owner-only — see {@link
|
||||
* #copyPosixPermissionsIfPresent}), unreadable by a different-uid member unless shared with
|
||||
* {@code worktreeGroup}, the same group {@link EnvAllowListScrub#shareWithGroup} already uses
|
||||
* for the ZDOTDIR scrub (fleetd #213) and the charter file (fleetd #219/#222). So under {@code
|
||||
* memberHerdrSocket} this method requires BOTH {@code configDir} and {@code worktreeGroup}
|
||||
* before it ever touches a file, and refuses the spawn — naming exactly which one is missing —
|
||||
* rather than silently corrupt fleetd's own home or hand the member an unreadable path. This
|
||||
* mirrors {@link #writeCharterFile}'s "refuse, don't degrade" decision: a member spawned without
|
||||
* a readable trust seed is not degraded, it sits on the interactive dialog forever and never
|
||||
* calls {@code fleet_reply} — exactly the failure fleetd #149 exists to prevent, so trading it
|
||||
* for "spawn something" is not worth it. When {@code configDir} and {@code worktreeGroup} are
|
||||
* both present, the write proceeds exactly as below and the resulting file is additionally
|
||||
* chgrp'd/chmod'd group-readable ({@code rw-r-----}) via {@link
|
||||
* EnvAllowListScrub#shareFileWithGroup(Path, String)} so the member's OS user can actually
|
||||
* open it — the
|
||||
* directory itself (unlike {@code worktreeRoot} or the charter's per-spawn directory) is not
|
||||
* fleetd-managed, so its own traversal permissions remain the operator's setup, same as they
|
||||
* already must be for the member to read anything else fleetd points {@code CLAUDE_CONFIG_DIR}
|
||||
* at. With {@code memberHerdrSocket} ABSENT (today's only live mode) every branch below is
|
||||
* byte-identical to before this fix.
|
||||
*
|
||||
* @param configDir the profile's {@code CLAUDE_CONFIG_DIR} ({@code cfg.configDir()}), or
|
||||
* {@code null}/blank to target the default {@code ~/.claude.json} — refused
|
||||
* outright when {@code memberHerdrSocket} is configured, see above
|
||||
* @param cwd the spawn's resolved working directory — the exact key Claude Code will look
|
||||
* up for itself once it starts there
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* configDir} and/or {@code worktreeGroup} is not — the same
|
||||
* refusal shape as {@link #writeCharterFile}
|
||||
*/
|
||||
private void seedTrustDialog(String configDir, String cwd) {
|
||||
if (!isProvisionedWorktree(cwd)) {
|
||||
return;
|
||||
}
|
||||
boolean unsetConfigDir = configDir == null || configDir.isBlank();
|
||||
boolean memberHerdrSocket = memberHerdrSocketConfigured();
|
||||
String group = memberHerdrSocket ? memberGroup() : null;
|
||||
if (memberHerdrSocket && (unsetConfigDir || group == null)) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the workspace-trust "
|
||||
+ "seed (.claude.json, which gates Claude Code's interactive trust dialog) must be "
|
||||
+ "placed where the member's OS user can read it — configDir, shared via "
|
||||
+ "worktreeGroup — but " + (unsetConfigDir ? "configDir" : "worktreeGroup")
|
||||
+ " is not configured. Refusing to spawn rather than write fleetd's own default "
|
||||
+ "'~/.claude.json' or hand the member a config file it cannot read: that member "
|
||||
+ "would sit on the interactive trust dialog forever and never reach an "
|
||||
+ "injectable state. Configure configDir on this profile and worktreeGroup on the "
|
||||
+ "fleet to enable claude-code member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path target = unsetConfigDir
|
||||
? Path.of(System.getProperty("user.home"), ".claude.json")
|
||||
: Path.of(configDir, ".claude.json");
|
||||
if (unsetConfigDir) {
|
||||
// fleetd #247: configDir unset is the ONLY path that targets ~/.claude.json — the
|
||||
// operator's own home file, not a per-profile one — and it is the default, so a
|
||||
// profile that simply forgot to set configDir gets no signal at all short of the
|
||||
// operator noticing their own file changing. Say so loudly, every time it is about to
|
||||
// happen, rather than only once ever: each occurrence is a live write to a real
|
||||
// person's home config and deserves its own log line. (Reached only when
|
||||
// memberHerdrSocket is absent — the block above already refused otherwise.)
|
||||
log.warn("seedTrustDialog: profile has no configDir set, so the workspace-trust seed "
|
||||
+ "for cwd '{}' is about to write the operator's own default '{}' — set "
|
||||
+ "configDir on this profile to target a per-member config file instead",
|
||||
cwd, target);
|
||||
}
|
||||
synchronized (TRUST_JSON_LOCK) {
|
||||
boolean written = false;
|
||||
try {
|
||||
if (target.getParent() != null) {
|
||||
Files.createDirectories(target.getParent());
|
||||
}
|
||||
for (int attempt = 1; attempt <= MAX_TRUST_JSON_CAS_ATTEMPTS && !written; attempt++) {
|
||||
byte[] before = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
ObjectNode root = parseTrustJsonOrEmpty(before);
|
||||
JsonNode projectsNode = root.get("projects");
|
||||
ObjectNode projects = projectsNode instanceof ObjectNode projectsObject
|
||||
? projectsObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectsNode instanceof ObjectNode)) {
|
||||
root.set("projects", projects);
|
||||
}
|
||||
JsonNode projectNode = projects.get(cwd);
|
||||
ObjectNode project = projectNode instanceof ObjectNode projectObject
|
||||
? projectObject : TRUST_JSON.createObjectNode();
|
||||
if (!(projectNode instanceof ObjectNode)) {
|
||||
projects.set(cwd, project);
|
||||
}
|
||||
// fleetd #247: ONLY hasTrustDialogAccepted. We used to write
|
||||
// hasCompletedProjectOnboarding beside it; do not put it back. Measured on
|
||||
// 2026-09-03, minutes after a live spawn seeded this file: 28 of 28 project
|
||||
// entries carried hasTrustDialogAccepted and 0 of 28 carried the onboarding key
|
||||
// — including the 27 entries Claude Code wrote for itself. Claude Code
|
||||
// normalises the whole file when it saves and drops that key every time, so
|
||||
// writing it achieved nothing except making the next reader think it mattered.
|
||||
// The member reached idle with the trust flag alone, which is the only outcome
|
||||
// this seed exists for. If a future Claude Code needs the second flag the
|
||||
// symptom returns as the trust dialog fleetd #149 describes — re-measure then,
|
||||
// do not restore it on a guess.
|
||||
project.put("hasTrustDialogAccepted", true);
|
||||
String newContent = TRUST_JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root);
|
||||
|
||||
trustJsonCasTestHook.run();
|
||||
|
||||
// fleetd #247 CAS: re-read immediately before the move and compare with what
|
||||
// this attempt built its update from. A mismatch means another writer (most
|
||||
// plausibly the operator's own live Claude Code — see this method's javadoc)
|
||||
// landed a change in between; discard this attempt's work and rebuild from the
|
||||
// fresh bytes rather than blindly overwriting it.
|
||||
byte[] atMove = Files.isRegularFile(target) ? Files.readAllBytes(target) : null;
|
||||
if (!Arrays.equals(before, atMove)) {
|
||||
continue;
|
||||
}
|
||||
writeAtomically(target, newContent);
|
||||
written = true;
|
||||
}
|
||||
if (!written) {
|
||||
// fleetd #247: deliberately do NOT write here. A member that starts without the
|
||||
// seed still starts — it may hit the trust dialog fleetd #149 describes and fail
|
||||
// to reach an injectable state, but that failure is visible (herdr reports it,
|
||||
// the spawn-readiness gate times out) and recoverable (retry the spawn). Writing
|
||||
// our stale copy over whatever the other writer left would be silent and, if that
|
||||
// other writer is the operator's own live session, could destroy real
|
||||
// configuration — fail toward the recoverable outcome, not the silent one.
|
||||
log.warn("seedTrustDialog: gave up seeding workspace-trust for cwd '{}' into '{}' "
|
||||
+ "after {} attempts — another writer (most plausibly the operator's own "
|
||||
+ "live Claude Code sharing this file) kept changing it faster than we "
|
||||
+ "could re-read it, so nothing was written; the member may show the "
|
||||
+ "trust dialog instead", cwd, target, MAX_TRUST_JSON_CAS_ATTEMPTS);
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("cannot seed workspace-trust entry for cwd '{}' into '{}'", cwd, target, e);
|
||||
return;
|
||||
}
|
||||
// fleetd #285: the write above lands as fleetd's own OS user; under memberHerdrSocket
|
||||
// that is NOT the member's OS user, so without this the member still cannot read the
|
||||
// file it exists to seed — a silent readiness timeout with the write looking "done".
|
||||
// Deliberately OUTSIDE the swallow-all catch above: a group that fails to resolve here
|
||||
// means the seed is unreadable despite a successful write, which must fail as loudly as
|
||||
// writeCharterFile's own EnvAllowListScrub.shareWithGroup call already does.
|
||||
if (written && memberHerdrSocket) {
|
||||
EnvAllowListScrub.shareFileWithGroup(target, group);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
/** Bound on {@link #seedTrustDialog}'s fleetd #247 compare-and-swap retry loop. */
|
||||
private static final int MAX_TRUST_JSON_CAS_ATTEMPTS = 5;
|
||||
|
||||
/**
|
||||
* Test-only seam for fleetd #247: invoked once per CAS attempt inside {@link #seedTrustDialog}'s
|
||||
* retry loop, after that attempt has read the target's bytes and built its replacement content,
|
||||
* but immediately before the final re-read/compare that decides whether to write. A no-op in
|
||||
* production. Package-visible (not {@code private}) so {@code ClaudeCodeLauncherTest} can install
|
||||
* a hook here that writes to the target file, deterministically simulating a writer racing
|
||||
* fleetd's own read-modify-write at the exact instant the CAS is meant to catch — the same race
|
||||
* an external process (most plausibly the operator's own live Claude Code) creates, without
|
||||
* depending on real thread scheduling to land the interleaving. A test that sets this MUST
|
||||
* restore it to the no-op default in a {@code finally} block — it is shared, static state.
|
||||
*/
|
||||
static Runnable trustJsonCasTestHook = () -> {};
|
||||
|
||||
/**
|
||||
* Parse {@code bytes} as a {@code .claude.json} tree, or hand back a fresh empty object when
|
||||
* {@code bytes} is {@code null} (no file yet) or does not parse to a JSON object — the same
|
||||
* missing-or-unreadable-is-empty fallback {@link #seedTrustDialog} always used, factored out so
|
||||
* the fleetd #247 CAS loop can call it once per attempt.
|
||||
*/
|
||||
private static ObjectNode parseTrustJsonOrEmpty(byte[] bytes) throws IOException {
|
||||
if (bytes == null) {
|
||||
return TRUST_JSON.createObjectNode();
|
||||
}
|
||||
JsonNode existing = TRUST_JSON.readTree(bytes);
|
||||
return existing instanceof ObjectNode existingObject ? existingObject : TRUST_JSON.createObjectNode();
|
||||
}
|
||||
|
||||
/**
|
||||
* Write {@code content} to {@code target} atomically: serialise to a sibling temp file in the
|
||||
* <strong>same directory</strong> as {@code target} (an atomic move is only guaranteed within
|
||||
* one filesystem — a different directory could mean a different filesystem), then
|
||||
* {@link StandardCopyOption#ATOMIC_MOVE} it into place. A reader — Claude Code itself, or
|
||||
* another {@code seedTrustDialog} call — only ever observes the fully-old file or the
|
||||
* fully-new one, never a truncated or half-written one.
|
||||
*
|
||||
* <p><b>fleetd #149 incident.</b> The original implementation used
|
||||
* {@code Files.writeString(target, content)} directly, which truncates {@code target} in place
|
||||
* before writing the replacement bytes. Combined with an ungated {@code cwd} (see
|
||||
* {@link HerdrPeerLauncher#isProvisionedWorktree}'s javadoc), a mutation-testing run hit that truncation window
|
||||
* against the operator's real {@code ~/.claude.json} and left it at 178 bytes. The gate closes
|
||||
* <em>which file</em> this can ever target; this closes <em>how</em> the target is written, so
|
||||
* that even a legitimate write against a real, live, concurrently-read {@code .claude.json}
|
||||
* cannot leave it observably empty or partial.
|
||||
*
|
||||
* <p>Preserves {@code target}'s existing POSIX permissions (Claude Code ships {@code
|
||||
* .claude.json} as {@code 0600}) when the filesystem reports them; a freshly created temp file
|
||||
* already defaults to owner-only permissions on a POSIX filesystem, so a first-ever write (no
|
||||
* existing {@code target}) is no less private without this. On a non-POSIX filesystem (e.g.
|
||||
* Windows) the permission copy is a silent no-op rather than a failure.
|
||||
*
|
||||
* <p>Package-visible (not {@code private}) so a test can drive it directly with a concurrent
|
||||
* reader thread and prove the torn-file property this method exists for — the LOCK in
|
||||
* {@link #seedTrustDialog} already fully serialises every call this launcher itself makes, so a
|
||||
* test that only ever goes through {@code seedTrustDialog}/{@code spawn()} could never observe
|
||||
* a torn file regardless of whether this method is atomic; it would be proving the lock, not
|
||||
* this method. Atomicity's actual job is protecting against a writer the lock cannot reach at
|
||||
* all — a second daemon process, or the operator's own live Claude Code — so the test for it
|
||||
* has to reach this method on its own.
|
||||
*/
|
||||
static void writeAtomically(Path target, String content) throws IOException {
|
||||
Path parent = target.getParent();
|
||||
Path tmp = Files.createTempFile(parent, target.getFileName() + ".", ".tmp");
|
||||
try {
|
||||
Files.writeString(tmp, content);
|
||||
copyPosixPermissionsIfPresent(target, tmp);
|
||||
Files.move(tmp, target, StandardCopyOption.ATOMIC_MOVE, StandardCopyOption.REPLACE_EXISTING);
|
||||
} catch (IOException e) {
|
||||
Files.deleteIfExists(tmp);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/** Copy {@code target}'s POSIX permissions onto {@code tmp}, or no-op where either is unsupported. */
|
||||
private static void copyPosixPermissionsIfPresent(Path target, Path tmp) {
|
||||
try {
|
||||
if (!Files.isRegularFile(target)) {
|
||||
return; // nothing to inherit from — first-ever write, temp file's own default stands
|
||||
}
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(target, PosixFileAttributeView.class);
|
||||
if (view == null) {
|
||||
return; // non-POSIX filesystem — nothing this JVM can read/set here
|
||||
}
|
||||
Files.setPosixFilePermissions(tmp, Files.getPosixFilePermissions(target));
|
||||
} catch (IOException e) {
|
||||
log.debug("cannot preserve permissions of '{}' onto its replacement", target, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** {@code s}, or {@code null} when {@code s} is null/blank — the charter-presence test used above. */
|
||||
private static String nonBlank(String s) {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
@@ -425,12 +799,83 @@ public final class ClaudeCodeLauncher extends HerdrPeerLauncher {
|
||||
* {@link OpenCodeLauncher#writeConfig} already uses for its charter file, since the process that
|
||||
* reads this file (the spawned peer) outlives this JVM call and there is no spawn-scoped teardown
|
||||
* hook to delete it synchronously.
|
||||
*
|
||||
* <p><b>fleetd #222.</b> {@code Files.createTempFile(prefix, suffix)} with no directory argument
|
||||
* resolves against {@code java.io.tmpdir} — on macOS the per-user {@code $TMPDIR} under
|
||||
* {@code /var/folders/...}, mode {@code 0700}, both resolved against FLEETD's own OS user. Under
|
||||
* {@code memberHerdrSocket:} the member pane runs as a DIFFERENT OS user, so that user cannot even
|
||||
* traverse the directory, let alone read the file — and since fleetd #220 the charter file is the
|
||||
* ONLY delivery path for {@code --append-system-prompt-file}, always, not merely the fallback it
|
||||
* used to be. A member handed a path it cannot read is not degraded, it is broken: see this
|
||||
* method's refusal branch below.
|
||||
*
|
||||
* <p><b>Measured severity (fleetd #222 real-binary check, claude 2.1.258):</b> an unreadable
|
||||
* {@code --append-system-prompt-file} is the LOUD failure, not the silent one. {@code claude}
|
||||
* checks the file before touching auth or the network — invoked with a bogus API key against a
|
||||
* {@code chmod 000} file, it printed {@code Error reading append system prompt file: EACCES:
|
||||
* permission denied, open '<path>'} and exited 1 immediately (a nonexistent path gets {@code
|
||||
* Error: Append system prompt file not found: <path>}, same exit code). So the pre-fix bug did
|
||||
* NOT leave a charter-less member silently occupying a pane and never calling {@code
|
||||
* fleet_reply} — it made the herdr pane exit immediately, which the CB-306 spawn-readiness gate
|
||||
* (this launcher's {@code spawnReadyTimeoutMs} poll) would have surfaced as "did not reach
|
||||
* injectable state", the same unexplained-timeout shape fleetd #220 already describes. Still a
|
||||
* real defect (every claude-code member under {@code memberHerdrSocket} would have failed to
|
||||
* spawn), but not the worse, undetectable failure mode.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only live mode): byte-identical to before this
|
||||
* fix — {@code Files.createTempFile("fleetd-role-charter-", ".md")} with no directory
|
||||
* argument, i.e. still resolved against {@code java.io.tmpdir}.</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: a fresh per-spawn directory is created under {@code
|
||||
* worktreeRoot} (never {@code java.io.tmpdir}) holding just the charter file, then shared
|
||||
* read-only with {@code worktreeGroup} via {@link EnvAllowListScrub#shareWithGroup} — the
|
||||
* SAME mechanism fleetd #213 built for the ZDOTDIR scrub and fleetd #219 reused for {@link
|
||||
* OpenCodeLauncher#writeConfig}'s {@code opencode.json} directory, reused here rather than
|
||||
* duplicated a third time. A per-spawn subdirectory (not {@code worktreeRoot} itself) is the
|
||||
* unit {@code shareWithGroup} chmods, so this never touches permissions on anything else
|
||||
* under {@code worktreeRoot}.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><b>Follows fleetd #219's REFUSAL decision, not #213's degrade decision.</b> The ZDOTDIR
|
||||
* scrub is a credential CONTROL — a degraded control (the CB-596 sentinel overlay) still has
|
||||
* value, so #213 falls back rather than refusing. A charter is NOT a control, it is the member's
|
||||
* TURN CONTRACT (the rule that ends every turn with {@code fleet_reply}). A member spawned with no
|
||||
* charter is not degraded, it is broken: either claude-code exits on the unreadable
|
||||
* {@code --append-system-prompt-file} path and the spawn dies at the readiness gate (loud), or it
|
||||
* starts anyway with no charter and never calls {@code fleet_reply} — the sender silently gets
|
||||
* nothing (silent). Neither outcome is worth trading for "spawn something." So a missing {@code
|
||||
* worktreeRoot}/{@code worktreeGroup} under {@code memberHerdrSocket} refuses the spawn here,
|
||||
* naming the missing key, exactly like {@link OpenCodeLauncher#configParentDir()}.
|
||||
*
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* worktreeRoot} and/or {@code worktreeGroup} is not
|
||||
*/
|
||||
private static Path writeCharterFile(String charterText) {
|
||||
private Path writeCharterFile(String charterText) {
|
||||
try {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
if (!memberHerdrSocketConfigured()) {
|
||||
Path file = Files.createTempFile("fleetd-role-charter-", ".md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
return file;
|
||||
}
|
||||
Path parentDir = memberScrubParentDir();
|
||||
String group = memberGroup();
|
||||
if (parentDir == null || group == null) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so the role/reply "
|
||||
+ "charter file (mounted via --append-system-prompt-file) must be placed where "
|
||||
+ "the member's OS user can read it — worktreeRoot, shared via worktreeGroup — "
|
||||
+ "but " + (parentDir == null ? "worktreeRoot" : "worktreeGroup") + " is not "
|
||||
+ "configured. Refusing to spawn rather than hand the member a charter path it "
|
||||
+ "cannot read: that member's turn contract (the fleet_reply rule) would never "
|
||||
+ "reach it. Configure both worktreeRoot and worktreeGroup to enable claude-code "
|
||||
+ "member spawns under memberHerdrSocket.");
|
||||
}
|
||||
Path dir = Files.createTempDirectory(parentDir, "fleetd-role-charter-");
|
||||
dir.toFile().deleteOnExit();
|
||||
Path file = dir.resolve("charter.md");
|
||||
Files.writeString(file, charterText);
|
||||
file.toFile().deleteOnExit();
|
||||
EnvAllowListScrub.shareWithGroup(dir, group);
|
||||
return file;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot write role charter temp file", e);
|
||||
|
||||
@@ -2,15 +2,19 @@ package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementCandidate;
|
||||
import dev.ltms.fleet.placement.PlacementContext;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.placement.PlacementPolicy;
|
||||
@@ -21,6 +25,7 @@ import java.util.ArrayList;
|
||||
import java.util.Collections;
|
||||
import java.util.EnumSet;
|
||||
import java.util.HashSet;
|
||||
import java.util.IdentityHashMap;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
@@ -44,12 +49,17 @@ import java.util.stream.Collectors;
|
||||
* the single adapter that declares it. Profiles partition cleanly across adapters: the
|
||||
* constructor rejects a name claimed by two.</li>
|
||||
* <li><strong>By pane id</strong> — {@link #stop} routes to the adapter that spawned that pane
|
||||
* (recorded at spawn time). A pane the composite never spawned (only real for a caller that
|
||||
* hand-rolls an id) falls back to the first delegate; teardown is pane-id addressed and
|
||||
* tab cleanup is single-occupant guarded, so it is safe either way.</li>
|
||||
* (recorded at spawn time). A pane the composite never spawned, or one whose record was lost
|
||||
* to a daemon restart (CB-185 blocker 1 — {@link #spawnedBy} is in-memory only), can use the
|
||||
* fallback route in a one-daemon fleet. With more than one herdr daemon, {@link #probeOwner}
|
||||
* asks each configured daemon which one actually knows the pane: exactly one match routes
|
||||
* (and caches); no match is treated as already-gone; more than one match is a genuine
|
||||
* ambiguity (pane ids are per-daemon counters, so two daemons really can both hold, say,
|
||||
* {@code w1:p1}) and stop refuses rather than closing a pane on an arbitrary herdr daemon.</li>
|
||||
* <li><strong>Fleet-wide</strong> — {@link #reapOrphanWorkers} and {@link #capabilities} fan out
|
||||
* and combine. {@link #list} is deduplicated by pane id because every herdr-backed delegate
|
||||
* shares one herdr connection and so reports the same global agent set.</li>
|
||||
* and combine. {@link #list} is deduplicated by (owning daemon, pane id): delegates that share
|
||||
* one herdr connection report the same global agent set, but two daemons can each hold a pane
|
||||
* called {@code w1:p1}, so the daemon has to be part of the key.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>CB-518: an unqualified spawn is routed through a {@link PlacementPolicy}. The default
|
||||
@@ -81,9 +91,34 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
private final Supplier<Map<String, FleetConfig.Profile>> profileConfigs;
|
||||
private final Supplier<PlacementPolicy> placementPolicy;
|
||||
|
||||
/**
|
||||
* fleetd #422: the central model allow-list's on/off state, read live per spawn — same reason
|
||||
* {@link #profileConfigs} is a supplier rather than a captured map (see the class doc above and
|
||||
* {@link #enforceModelEnabled}). A caller with no {@code models:} block to read from (the
|
||||
* simpler, map-based constructors used throughout this class's own tests) wires this to a
|
||||
* constant {@code null}, which {@link #models0} treats as "nothing configured, gate never
|
||||
* fires" — the pre-#422 behaviour.
|
||||
*/
|
||||
private final Supplier<FleetConfig.Models> models;
|
||||
|
||||
/** The value {@link #models0} normalizes a {@code null} supplier result to. */
|
||||
private static final FleetConfig.Models NO_MODELS_CONFIGURED = new FleetConfig.Models(List.of());
|
||||
|
||||
/** CB-578 stage B: credential cooldown, checked before an explicit spawn and filtered into placement. */
|
||||
private final BackendQuarantine quarantine;
|
||||
|
||||
/**
|
||||
* fleetd #201 Unit 5: credential cool-off after repeated backend errors, checked before an
|
||||
* explicit spawn and filtered into placement — a SEPARATE, shorter-lived source from
|
||||
* {@link #quarantine}. A never-{@code record}-called instance is naturally inert (its
|
||||
* {@code remainingCoolOffSeconds} always returns empty), so back-compat constructors that predate
|
||||
* this feature share one fixed instance rather than needing a {@code none()} sentinel.
|
||||
*/
|
||||
private final BackendOutagePolicy outagePolicy;
|
||||
|
||||
/** The shared inert instance back-compat constructors wire in — never {@code record}-called. */
|
||||
private static final BackendOutagePolicy NO_OUTAGE_POLICY = new BackendOutagePolicy(System::nanoTime);
|
||||
|
||||
/**
|
||||
* CB-557: the role pools an unqualified spawn draws its candidates from. A supplier that yields
|
||||
* {@code null}, and an empty pool for a role, both fall back to every configured profile — the
|
||||
@@ -122,7 +157,8 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null, BackendQuarantine.none());
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, null,
|
||||
BackendQuarantine.none(), NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -140,13 +176,14 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, BackendQuarantine.none());
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet,
|
||||
BackendQuarantine.none(), NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B). The full-featured
|
||||
* non-reloading form; {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine)}
|
||||
* is what {@code Fleetd.main} actually wires up.
|
||||
* Production constructor with role pools and quarantine (CB-578 stage B), no cool-off (fleetd
|
||||
* #201 Unit 5). Kept for callers that predate the cool-off feature; use the 8-arg overload below
|
||||
* to wire a real {@link BackendOutagePolicy}.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
@@ -158,17 +195,74 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, quarantine,
|
||||
NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor with role pools, quarantine (CB-578 stage B), and cool-off (fleetd
|
||||
* #201 Unit 5). The full-featured non-reloading form;
|
||||
* {@link #CompositePeerLauncher(List, String, Supplier, Function, BackendQuarantine, BackendOutagePolicy)}
|
||||
* is what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param quarantine required — pass {@link BackendQuarantine#none()} for a caller that does not
|
||||
* want the feature, never a defaulting overload (CB-578 stage B's own rule).
|
||||
* @param outagePolicy required — pass a fresh, never-{@code record}-called {@link
|
||||
* BackendOutagePolicy} for a caller that does not want the feature; same
|
||||
* "explicit opt-out, never a silent default" rule as {@code quarantine}.
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
// No models: block to read from a plain profiles Map — fleetd #422's gate is wired to a
|
||||
// constant null (see enforceModelEnabled/models0), the pre-#422 behaviour for every caller
|
||||
// of this overload. Use the 9-arg overload below to test the gate against a LIVE supplier.
|
||||
this(delegates, defaultProfile, profileConfigs, placementPolicy, liveCount, fleet, quarantine,
|
||||
outagePolicy, constant(null));
|
||||
}
|
||||
|
||||
/**
|
||||
* As above, plus a LIVE model-gate source (fleetd #422). The 8-arg overload above wires
|
||||
* {@code models} to a constant {@code null} because it has only a static {@code Map<String,
|
||||
* Profile>}, never a full {@code FleetConfig}, to read one from; this overload exists so a test
|
||||
* can prove {@link #enforceModelEnabled} and its candidate filter re-read {@link
|
||||
* FleetConfig.Models} on every call rather than a value captured once at construction — the
|
||||
* exact distinction fleetd #422 exists to get right (see the class doc's CB-559 note on {@link
|
||||
* #profileConfigs}, which this follows). Production wiring uses the
|
||||
* {@code Supplier<FleetConfig>} constructor below instead, which already threads a live
|
||||
* {@code config.get().models()} through.
|
||||
*
|
||||
* @param models required — pass a supplier returning {@code null} for a caller that has no
|
||||
* {@code models:} block to gate against, never a defaulting overload (the same
|
||||
* "explicit opt-out, never a silent default" rule {@code quarantine}/
|
||||
* {@code outagePolicy} already follow).
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Map<String, FleetConfig.Profile> profileConfigs,
|
||||
PlacementPolicy placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
FleetConfig.Fleet fleet,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy,
|
||||
Supplier<FleetConfig.Models> models) {
|
||||
// LinkedHashMap, not Map.copyOf: candidates() promises definition order and the weighted
|
||||
// policy breaks exact-weight ties on it, so a salted iteration order would make placement
|
||||
// differ from one JVM run to the next.
|
||||
this(delegates, defaultProfile,
|
||||
constant(Collections.unmodifiableMap(new LinkedHashMap<>(profileConfigs))),
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine);
|
||||
constant(placementPolicy), liveCount, constant(fleet), quarantine, outagePolicy, models);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), so a config
|
||||
* reload retargets the next member without a restart.
|
||||
* reload retargets the next member without a restart. No cool-off (fleetd #201 Unit 5); use the
|
||||
* 6-arg overload below to wire a real {@link BackendOutagePolicy}.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
@@ -178,12 +272,34 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Supplier<FleetConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine) {
|
||||
this(delegates, defaultProfile, config, liveCount, quarantine, NO_OUTAGE_POLICY);
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor that re-reads its placement inputs per spawn (CB-559), with cool-off
|
||||
* (fleetd #201 Unit 5). This is what {@code Fleetd.main} actually wires up.
|
||||
*
|
||||
* @param config the live configuration — read at every spawn, never captured
|
||||
* @param quarantine required — CB-578 stage B; pass {@link BackendQuarantine#none()} to opt out
|
||||
* @param outagePolicy required — fleetd #201 Unit 5; pass a fresh, never-{@code record}-called
|
||||
* {@link BackendOutagePolicy} to opt out
|
||||
*/
|
||||
public CompositePeerLauncher(List<HerdrPeerLauncher> delegates,
|
||||
String defaultProfile,
|
||||
Supplier<FleetConfig> config,
|
||||
Function<String, Integer> liveCount,
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy) {
|
||||
this(delegates, defaultProfile,
|
||||
() -> config.get().profiles(),
|
||||
() -> PlacementPolicies.fromName(config.get().placement()),
|
||||
liveCount,
|
||||
() -> config.get().fleet(),
|
||||
quarantine);
|
||||
quarantine,
|
||||
outagePolicy,
|
||||
// fleetd #422: read live, same as profiles/placement/fleet above — a models.allow
|
||||
// edit (on/off or otherwise) is visible to the very next spawn, no restart needed.
|
||||
() -> config.get().models());
|
||||
}
|
||||
|
||||
/** The all-suppliers form every other constructor funnels into. */
|
||||
@@ -193,9 +309,12 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
Supplier<PlacementPolicy> placementPolicy,
|
||||
Function<String, Integer> liveCount,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
BackendQuarantine quarantine) {
|
||||
BackendQuarantine quarantine,
|
||||
BackendOutagePolicy outagePolicy,
|
||||
Supplier<FleetConfig.Models> models) {
|
||||
this.fleet = fleet;
|
||||
this.quarantine = Objects.requireNonNull(quarantine, "quarantine");
|
||||
this.outagePolicy = Objects.requireNonNull(outagePolicy, "outagePolicy");
|
||||
if (delegates.isEmpty()) {
|
||||
throw new IllegalArgumentException("at least one peer adapter must be configured");
|
||||
}
|
||||
@@ -204,6 +323,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
this.profileConfigs = profileConfigs;
|
||||
this.placementPolicy = placementPolicy;
|
||||
this.liveCount = liveCount;
|
||||
this.models = Objects.requireNonNull(models, "models");
|
||||
Map<String, HerdrPeerLauncher> index = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : this.delegates) {
|
||||
for (String profile : d.profiles()) {
|
||||
@@ -234,6 +354,16 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return m == null ? Map.of() : m;
|
||||
}
|
||||
|
||||
/**
|
||||
* The currently-configured {@code models:} block, never null (fleetd #422). Read fresh on
|
||||
* every call, the same reason {@link #profiles0} is — a config reload's on/off edit must reach
|
||||
* the very next spawn.
|
||||
*/
|
||||
private FleetConfig.Models models0() {
|
||||
FleetConfig.Models m = models.get();
|
||||
return m == null ? NO_MODELS_CONFIGURED : m;
|
||||
}
|
||||
|
||||
/** The adapter owning {@code profileName} (null/blank → the default). Throws on an unknown profile. */
|
||||
private HerdrPeerLauncher route(String profileName) {
|
||||
String resolved = (profileName == null || profileName.isBlank()) ? defaultProfile : profileName;
|
||||
@@ -258,8 +388,12 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// the charter makes explicit-profile spawns the normal path — so skipping the check
|
||||
// here would leave the cap dead config in real operation.
|
||||
HerdrPeerLauncher d = route(requestedProfile);
|
||||
// Checked in this order so exhaustion quarantine wins when both are active: quarantine
|
||||
// throws first and short-circuits before the cool-off check ever runs (fleetd #201 Unit 5).
|
||||
enforceNotQuarantined(requestedProfile);
|
||||
enforceNotCoolingOff(requestedProfile);
|
||||
enforceMaxLoad(requestedProfile);
|
||||
enforceModelEnabled(requestedProfile);
|
||||
PeerHandle handle = d.spawn(req);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
@@ -269,15 +403,10 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// the whole profile list. An EXPLICIT profile (above) is left alone on purpose — it is the
|
||||
// operator overriding, and refusing it would break `fleet_spawn{profile:"opus"}`, which
|
||||
// carries no role and so would be judged against the dev pool it was never meant for.
|
||||
List<PlacementCandidate> candidates = candidates(req.role());
|
||||
String roleDefault = defaultProfileFor(req.role());
|
||||
Set<String> unreachable = new HashSet<>();
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
PlacementContext ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
PlacementContext ctx = placementContextFor(req.role(), unreachable);
|
||||
|
||||
int maxAttempts = candidates.isEmpty() ? 1 : candidates.size();
|
||||
int maxAttempts = ctx.candidates().isEmpty() ? 1 : ctx.candidates().size();
|
||||
for (int attempt = 0; attempt < maxAttempts; attempt++) {
|
||||
// Deliberately uncaught: when no candidate is left (all at cap, or all unreachable) the
|
||||
// policy already throws a clear message. Catching it to rethrow a generic
|
||||
@@ -294,8 +423,7 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
// CB-547a: route the chosen profile but keep the caller's session identity — dropping it
|
||||
// here would silently sever the resume handle on every policy-routed spawn. CB-557: the
|
||||
// role rides along for the same reason, or a routed spawn would be labelled as a dev.
|
||||
SpawnRequest routedReq = new SpawnRequest(chosen.profile(), req.requestedCwd(), req.callerCwd(),
|
||||
req.sessionName(), req.resumeSessionId(), req.role());
|
||||
SpawnRequest routedReq = req.withProfile(chosen.profile());
|
||||
try {
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
@@ -305,13 +433,16 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
chosen.profile(), e.getMessage());
|
||||
unreachable.add(chosen.profile());
|
||||
// Update the context for the next selection so the policy excludes this profile.
|
||||
ctx = new PlacementContext(roleDefault, candidates, liveCount, unreachable, quarantined);
|
||||
ctx = placementContextFor(req.role(), unreachable);
|
||||
}
|
||||
}
|
||||
|
||||
// unreachable.size() counts DISTINCT profiles, not attempts (a HashSet dedupes a profile
|
||||
// added twice) — say "distinct" so the count matches the sentence and the profile list that
|
||||
// follows, rather than reading as a count of attempts made (fleetd #315).
|
||||
throw new PeerUnreachableException(
|
||||
"no reachable worker profile available after trying " + unreachable.size()
|
||||
+ " candidate(s): " + String.join(", ", unreachable));
|
||||
+ " distinct candidate(s): " + String.join(", ", unreachable));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -371,6 +502,31 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose credential is cooling off after repeated backend errors
|
||||
* (fleetd #201 Unit 5 — {@link BackendOutagePolicy}): a SEPARATE, shorter-lived source from
|
||||
* {@link #enforceNotQuarantined}'s exhaustion quarantine. Checked after quarantine so exhaustion
|
||||
* wins when both are active — see the call site in {@link #spawn}.
|
||||
*
|
||||
* @throws PlacementException naming the profile, its credential, and the remaining cool-off
|
||||
*/
|
||||
private void enforceNotCoolingOff(String profile) {
|
||||
String credentialId = credentialIdFor(profile);
|
||||
outagePolicy.remainingCoolOffSeconds(credentialId).ifPresent(remaining -> {
|
||||
throw new PlacementException("worker profile '" + profile + "' credential '" + credentialId
|
||||
+ "' is cooling off after repeated backend errors; ~" + remaining
|
||||
+ "s remaining — refusing spawn");
|
||||
});
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose credential is currently cooling off (fleetd #201 Unit 5). */
|
||||
private Set<String> coolingOffProfiles(List<PlacementCandidate> candidates) {
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> outagePolicy.remainingCoolOffSeconds(credentialIdFor(p)).isPresent())
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
private void enforceMaxLoad(String profile) {
|
||||
// Absent config, or a config whose maxLoad normalized to null (ABSENT ⇒ unlimited at load),
|
||||
// means no cap — never cap what wasn't configured. Note "non-positive ⇒ unlimited" was true
|
||||
@@ -389,6 +545,49 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Refuse an explicit-profile spawn whose {@code model:} the operator has turned off in the
|
||||
* central {@code models.allow:} list (fleetd #422). Deliberately worded apart from {@link
|
||||
* #enforceNotQuarantined} and {@link #enforceNotCoolingOff}: those two report a BACKEND-reported
|
||||
* outage (exhaustion, repeated errors); this one reports an OPERATOR decision, so the message
|
||||
* says "turned off" and names the model, never "quarantined" or "cooling off". A FOURTH,
|
||||
* independent reason to refuse a spawn — never layered onto {@code BackendQuarantine} or {@code
|
||||
* BackendOutagePolicy}, which would misattribute an operator's own choice to the backend.
|
||||
*
|
||||
* <p>Reads {@link #models0()} fresh on every call — the same liveness {@link #profiles0()}
|
||||
* already has — so flipping {@code enabled: false} and reloading takes effect on the very next
|
||||
* spawn, no restart (criterion 4). A profile naming no model, or a model absent from {@code
|
||||
* models.allow:} entirely (nothing to gate against), is never refused here.
|
||||
*
|
||||
* @throws PlacementException naming the model and the profile, distinct from quarantine/cool-off
|
||||
*/
|
||||
private void enforceModelEnabled(String profile) {
|
||||
FleetConfig.Profile cfg = profiles0().get(profile);
|
||||
String model = (cfg == null) ? null : cfg.model();
|
||||
if (model == null) {
|
||||
return;
|
||||
}
|
||||
if (models0().offIds().contains(model)) {
|
||||
throw new PlacementException("worker profile '" + profile + "' names model '" + model
|
||||
+ "', which the operator has turned off in models.allow — refusing spawn");
|
||||
}
|
||||
}
|
||||
|
||||
/** The subset of {@code candidates} whose {@code model:} is currently turned off (fleetd #422). */
|
||||
private Set<String> modelOffProfiles(List<PlacementCandidate> candidates) {
|
||||
Set<String> off = models0().offIds();
|
||||
if (off.isEmpty()) {
|
||||
return Set.of();
|
||||
}
|
||||
return candidates.stream()
|
||||
.map(PlacementCandidate::profile)
|
||||
.filter(p -> {
|
||||
FleetConfig.Profile cfg = profiles0().get(p);
|
||||
return cfg != null && cfg.model() != null && off.contains(cfg.model());
|
||||
})
|
||||
.collect(Collectors.toSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile names {@code role} may be placed on, in definition order.
|
||||
*
|
||||
@@ -405,8 +604,23 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return known.isEmpty() ? List.copyOf(configured.keySet()) : known;
|
||||
}
|
||||
|
||||
/** The profile an unqualified spawn for {@code role} falls back to under {@code fixed} placement. */
|
||||
private String defaultProfileFor(MemberRole role) {
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Live: reads {@link #poolFor}, which reads {@link #profileConfigs} and {@link #fleet} fresh
|
||||
* on every call, so a config reload is visible without a restart (fleetd #425) — unlike {@link
|
||||
* #defaultProfile}, the field captured once at construction, which this falls back to only when
|
||||
* {@link #poolFor} has nothing to offer at all (no profiles configured for this composite).
|
||||
*
|
||||
* <p>Exact only under the {@code fixed} placement policy — the one that reads this value
|
||||
* ({@code FixedPlacementPolicy}, package-private, hence not linked) as its first, preferred
|
||||
* candidate. {@code weighted}/{@code round-robin} placement can choose a different candidate
|
||||
* from {@code role}'s pool even on the very first spawn; this method does not simulate that
|
||||
* choice, matching what the {@code defaultProfile:}-derived reporting this replaces has always
|
||||
* done.
|
||||
*/
|
||||
@Override
|
||||
public String defaultProfileFor(MemberRole role) {
|
||||
List<String> pool = poolFor(role);
|
||||
return pool.isEmpty() ? defaultProfile : pool.getFirst();
|
||||
}
|
||||
@@ -423,6 +637,142 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the {@link PlacementContext} an unqualified spawn of {@code role} would be judged
|
||||
* against right now — the single source both {@link #spawn} and {@link #place} read, so the two
|
||||
* can never disagree about which conditions (quarantine, cool-off, model-off) apply to which
|
||||
* candidate (fleetd #425 rework: round 1 duplicated this into a second, blind resolver —
|
||||
* {@link #defaultProfileFor} — which is why it regressed; round 2 found that even a single
|
||||
* shared resolver is not enough on its own if the CALLER re-resolves through an explicit
|
||||
* profile afterwards — see {@link PlacementDecision}).
|
||||
*
|
||||
* @param unreachable the caller's mutable unreachable set; {@link #spawn} grows this across
|
||||
* retries and rebuilds the context from it, {@link #place} passes a fresh
|
||||
* empty one since it never retries
|
||||
*/
|
||||
private PlacementContext placementContextFor(MemberRole role, Set<String> unreachable) {
|
||||
List<PlacementCandidate> candidates = candidates(role);
|
||||
String roleDefault = defaultProfileFor(role);
|
||||
// CB-578 stage B: computed once up front — a quarantine's expiry cannot pass within one spawn
|
||||
// call, so re-deriving it per retry would only cost work, never change the answer.
|
||||
Set<String> quarantined = quarantinedProfiles(candidates);
|
||||
// fleetd #201 Unit 5: a distinct set from quarantined — see PlacementContext.coolingOff.
|
||||
Set<String> coolingOff = coolingOffProfiles(candidates);
|
||||
// fleetd #422: read live per spawn, same as quarantined/coolingOff above — a config reload
|
||||
// that flips a model's enabled state is visible to the very next unqualified spawn.
|
||||
Set<String> modelOff = modelOffProfiles(candidates);
|
||||
return new PlacementContext(roleDefault, candidates, liveCount, unreachable,
|
||||
quarantined, coolingOff, modelOff);
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #425 rework, round 2: runs the exact same selection {@link #spawn} uses for a
|
||||
* blank-profile request — {@link #placementContextFor} plus one {@link PlacementPolicy#select}
|
||||
* — rather than {@link #defaultProfileFor}'s blind "pool's first entry", so a quarantined,
|
||||
* cooling-off, or model-off pool-first candidate is routed around here exactly as it would be
|
||||
* by a real spawn. Unlike {@link #spawn}, this never retries on {@link
|
||||
* PeerUnreachableException}: there is no spawn attempt to fail, so "unreachable" never grows
|
||||
* past the empty set it starts with, and a single {@link PlacementPolicy#select} call already
|
||||
* reflects the live quarantine/cool-off/model-off state.
|
||||
*
|
||||
* <p>Deliberately does <em>not</em> apply {@link #enforceMaxLoad} (or any of the other three
|
||||
* {@code enforce*} checks): those belong to {@link #spawn}'s EXPLICIT-profile branch, the
|
||||
* operator-override path, and this method answers a different question — "where would an
|
||||
* UNQUALIFIED spawn land". That is not the same as {@code select} ignoring these conditions —
|
||||
* every condition {@code select} filters on (quarantine, cooling off, {@code maxLoad} under
|
||||
* every placement policy including the default {@code fixed}, since fleetd #435, model-off,
|
||||
* unreachable, weight-0) is already reflected in the {@link PlacementDecision} this method
|
||||
* returns, because {@code select} walked past every excluded candidate to find it. What this
|
||||
* method's caller must not do is take that resolved name and hand it back to {@link
|
||||
* #spawn(SpawnRequest)} as an explicit profile: the explicit-profile branch treats the same
|
||||
* exclusion conditions as a reason to REFUSE, where {@code select} had already treated them as
|
||||
* a reason to fall through — round 1 of this fix did exactly that, turning a fall-through this
|
||||
* method had already resolved around into a refusal one call later. Round 2 fixes that at the
|
||||
* caller: {@link #spawn(SpawnRequest, PlacementDecision)} carries this exact decision to the
|
||||
* spawn without re-resolving or re-checking it, through the same routing path {@code select}
|
||||
* itself was consulted from.
|
||||
*
|
||||
* @throws PlacementException if no candidate in {@code role}'s pool is currently placeable
|
||||
* (mirrors what an actual unqualified spawn would throw)
|
||||
*/
|
||||
@Override
|
||||
public PlacementDecision place(MemberRole role) {
|
||||
PlacementContext ctx = placementContextFor(role, new HashSet<>());
|
||||
return new PlacementDecision(placementPolicy.get().select(ctx).profile());
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Delegates to {@link #place}, so the two can never disagree about the answer for the same
|
||||
* {@code role} at the same instant — kept as a convenience for a caller that only wants the
|
||||
* resolved name (a status report, a log line), never for a caller that will act on it by
|
||||
* spawning: that caller must hold the {@link PlacementDecision} itself and pass it to {@link
|
||||
* #spawn(SpawnRequest, PlacementDecision)} — see {@link PlacementDecision}'s javadoc for why
|
||||
* resolving here and spawning separately, with the name fed back in as an explicit profile,
|
||||
* regressed fleetd #425 twice.
|
||||
*/
|
||||
@Override
|
||||
public String routedProfileFor(MemberRole role) {
|
||||
return place(role).profile();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>Routes {@code decision.profile()} directly to its owning delegate — the identical
|
||||
* {@code d.spawn(routedReq)} call {@link #spawn(SpawnRequest)}'s blank-profile branch makes for
|
||||
* its first pick — WITHOUT re-running {@link #enforceNotQuarantined}, {@link
|
||||
* #enforceNotCoolingOff}, {@link #enforceMaxLoad}, or {@link #enforceModelEnabled}: those are
|
||||
* the EXPLICIT-profile branch's checks, and {@code decision} did not come from an operator
|
||||
* naming a profile — it came from {@link #place}, which already applied whichever of these
|
||||
* conditions {@link PlacementPolicy#select} actually filters on (fleetd #425 rework, round 2).
|
||||
*
|
||||
* <p>The two branches disagree on purpose about what an excluded profile means, and that
|
||||
* disagreement is not what this method removes. The blank-profile routing branch (and
|
||||
* {@link #place}) treats a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0 profile
|
||||
* as a reason to fall through to the next candidate; the EXPLICIT-profile branch treats naming
|
||||
* that same profile as a reason to refuse outright — someone who names a profile should get a
|
||||
* refusal, not a silent substitution onto a different backend. That is still correct after
|
||||
* fleetd #435. What round 1 got wrong, and what this method exists to stop happening again, is
|
||||
* turning a fall-through into a refusal by accident: resolving a name via {@link #place} and
|
||||
* then handing that same name back to {@link #spawn(SpawnRequest)} as an explicit profile takes
|
||||
* the refusing branch on a decision the routing branch had already approved by falling through
|
||||
* past everything else.
|
||||
*
|
||||
* <p>Before fleetd #435, this exact accident was reachable through {@code maxLoad} specifically:
|
||||
* {@code FixedPlacementPolicy} — the default policy — did not evaluate {@code maxLoad} at all
|
||||
* for automatic selection, so {@link #place} could approve an at-cap profile that {@link
|
||||
* #enforceMaxLoad} would then refuse one call later. fleetd #435 closed that: {@code
|
||||
* FixedPlacementPolicy} now walks past an at-cap candidate exactly like {@code weighted}/
|
||||
* {@code round-robin} already did, so {@link #place} can no longer return one, and this specific
|
||||
* failure — an approved placement dying at {@code enforceMaxLoad} — cannot happen any more.
|
||||
* What this method still buys, now that {@code maxLoad} can no longer cause it: it never
|
||||
* re-evaluates a condition {@link #place} already decided, and it closes the window between
|
||||
* that decision and the spawn in which the underlying state (another spawn landing on the same
|
||||
* profile, a config reload) could otherwise move and make a stale explicit re-check wrong.
|
||||
*
|
||||
* <p>Deliberately does not retry on {@link PeerUnreachableException} across candidates the way
|
||||
* {@link #spawn(SpawnRequest)}'s blank-profile branch does: retrying here would silently
|
||||
* re-place the caller onto a different profile than the one {@code decision} named, behind the
|
||||
* back of a caller that may already have provisioned something (a worktree's {@code repoRoot},
|
||||
* parity overlay) specifically for that name. A caller that wants the composite's own failover
|
||||
* should call {@link #spawn(SpawnRequest)} with a blank profile directly, not resolve through
|
||||
* {@link #place} first. Losing that retry on a resolve-then-spawn path is an accepted, unrelated
|
||||
* cost — see {@code SessionManager.acquireWithWorktree}'s own comment on it — never widened by
|
||||
* this round to include {@code maxLoad}, which is what round 1 actually lost.
|
||||
*/
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||
HerdrPeerLauncher d = route(decision.profile());
|
||||
SpawnRequest routedReq = req.withProfile(decision.profile());
|
||||
PeerHandle handle = d.spawn(routedReq);
|
||||
spawnedBy.put(handle.id(), d);
|
||||
return handle;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
return route(req.profileName()).effectiveCwd(req);
|
||||
@@ -433,14 +783,126 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return route(profileName).parityOverlay(profileName);
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: the single live read that answers both "is the models.allow: gate
|
||||
* armed" and "which models are off", off the exact same accessor ({@link #models0()}) {@link
|
||||
* #enforceModelEnabled} and {@link #modelOffProfiles} read — so {@code fleet_profiles}/{@code
|
||||
* GET /profiles} (via {@link PeerLauncher#disabledModels()}, which now delegates here) can
|
||||
* never report a different answer than the gate enforces (the fleetd #404 lesson), and the
|
||||
* startup log line built from this can never disagree with either.
|
||||
*
|
||||
* <p>{@link #models0()} itself normalizes a {@code null} {@link #models} read to the shared
|
||||
* {@link #NO_MODELS_CONFIGURED} sentinel — deliberately the one object no config-supplied
|
||||
* {@code Models} instance can ever be identical to, since it is private to this class — so
|
||||
* comparing by reference here recovers exactly the fact {@code models0()}'s normalization
|
||||
* would otherwise erase: whether the live source was {@code null} (no {@code models:} block,
|
||||
* armed = false) or a real, config-supplied block (armed = true, even one whose {@code allow:}
|
||||
* is itself empty or absent — {@link FleetConfig.Models}'s "absent or empty allow: is off"
|
||||
* wording governs config-load validation, a distinct question from whether this gate is armed
|
||||
* for reporting).
|
||||
*/
|
||||
@Override
|
||||
public PeerLauncher.ModelGateState modelGateState() {
|
||||
FleetConfig.Models m = models0();
|
||||
return m == NO_MODELS_CONFIGURED
|
||||
? PeerLauncher.ModelGateState.notConfigured()
|
||||
: PeerLauncher.ModelGateState.armed(m.offIds());
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
HerdrPeerLauncher d = spawnedBy.remove(id);
|
||||
HerdrPeerLauncher d = spawnedBy.get(id);
|
||||
if (d == null) {
|
||||
log.debug("stop({}) — no recorded owner, routing to the first adapter (pane-addressed)", id);
|
||||
d = delegates.getFirst();
|
||||
if (herdrDaemonCount() == 1) {
|
||||
log.debug("stop({}) — no recorded owner in a single-daemon fleet", id);
|
||||
d = delegates.getFirst();
|
||||
} else {
|
||||
d = probeOwner(id);
|
||||
if (d == null) {
|
||||
// No configured herdr daemon has ever heard of this pane. CB-185 blocker 1: this
|
||||
// is the normal case right after a daemon restart empties spawnedBy for a member
|
||||
// that has ALREADY been torn down since — the caller retried a stop that already
|
||||
// succeeded. Nothing to close and no owner to cache; matching the tolerance
|
||||
// HerdrPeerLauncher#stop already gives an already-gone pane (agent.close swallows
|
||||
// that as success), stop() here is a no-op rather than a refusal.
|
||||
log.debug("stop({}) — no configured herdr daemon knows this pane; "
|
||||
+ "treating as already stopped", id);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
// Drop the owner record only after the delegate accepted the stop. Removing it first meant a
|
||||
// delegate that threw left the pane alive with its owner forgotten, so the retry fell into
|
||||
// the ambiguous branch above and refused the id for good.
|
||||
d.stop(id);
|
||||
spawnedBy.remove(id);
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-185 blocker 1: recover a spawnedBy cache miss by asking every distinct herdr daemon which
|
||||
* one actually knows {@code id} — the fix for "after a restart, every surviving member becomes
|
||||
* un-stoppable" (spawnedBy is in-memory only, so a restart empties it, and members intentionally
|
||||
* outlive the daemon).
|
||||
*
|
||||
* <p>Grouped by daemon identity, not by delegate, for the same reason {@link #list()} groups
|
||||
* that way: two adapters (claude-code, opencode) sharing one herdr connection would otherwise be
|
||||
* probed twice, and a pane on their shared daemon would look owned by two adapters instead of
|
||||
* one daemon.
|
||||
*
|
||||
* <p>A daemon that fails to answer {@code list()} (e.g. it is down) is treated as "does not know
|
||||
* this pane" rather than aborting the whole probe — one unreachable daemon must never make a
|
||||
* pane that a <em>different</em>, healthy daemon actually owns un-stoppable too, which would
|
||||
* resurrect the exact bug this method exists to fix.
|
||||
*
|
||||
* @return the owning delegate — cached into {@link #spawnedBy} so the next call is free — or
|
||||
* {@code null} when no daemon knows the pane
|
||||
* @throws IllegalArgumentException when more than one daemon claims the pane: pane ids are
|
||||
* per-daemon counters, so two daemons really can both hold, say, {@code w1:p1}, and there
|
||||
* is no way to tell which one the caller means
|
||||
*/
|
||||
private HerdrPeerLauncher probeOwner(String id) {
|
||||
Map<HerdrClient, HerdrPeerLauncher> byDaemon = new IdentityHashMap<>();
|
||||
for (HerdrPeerLauncher delegate : delegates) {
|
||||
byDaemon.putIfAbsent(delegate.herdr(), delegate);
|
||||
}
|
||||
List<HerdrPeerLauncher> owners = new ArrayList<>();
|
||||
for (HerdrPeerLauncher representative : byDaemon.values()) {
|
||||
List<Agent> agents;
|
||||
try {
|
||||
agents = representative.list();
|
||||
} catch (HerdrException e) {
|
||||
log.warn("stop({}) probe: a configured herdr daemon was unreachable ({}); "
|
||||
+ "treating it as not knowing this pane", id, e.getClass().getSimpleName());
|
||||
continue;
|
||||
}
|
||||
boolean knows = agents.stream().anyMatch(a -> id.equals(a.paneId()));
|
||||
if (knows) {
|
||||
owners.add(representative);
|
||||
}
|
||||
}
|
||||
if (owners.size() > 1) {
|
||||
throw new IllegalArgumentException("ambiguous paneId '" + id + "': "
|
||||
+ owners.size() + " configured herdr daemons report this pane — "
|
||||
+ "no way to tell which one the caller means");
|
||||
}
|
||||
if (owners.isEmpty()) {
|
||||
return null;
|
||||
}
|
||||
HerdrPeerLauncher owner = owners.get(0);
|
||||
spawnedBy.put(id, owner);
|
||||
return owner;
|
||||
}
|
||||
|
||||
/**
|
||||
* Count actual herdr daemons, not peer adapter kinds. Identity is intentional: separate client
|
||||
* objects may represent different daemons even if a client later implements value equality.
|
||||
*/
|
||||
private int herdrDaemonCount() {
|
||||
Set<HerdrClient> daemons = Collections.newSetFromMap(new IdentityHashMap<>());
|
||||
for (HerdrPeerLauncher delegate : delegates) {
|
||||
daemons.add(delegate.herdr());
|
||||
}
|
||||
return daemons.size();
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -458,9 +920,21 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return byProfile.keySet();
|
||||
}
|
||||
|
||||
/**
|
||||
* {@inheritDoc}
|
||||
*
|
||||
* <p>fleetd #425: reports the <em>live</em> {@code dev} pool's first entry — the same value
|
||||
* {@link #defaultProfileFor} computes for {@link MemberRole#DEV} — not the {@link
|
||||
* #defaultProfile} field captured at construction. An unqualified {@code fleet_spawn} defaults
|
||||
* to {@code MemberRole#DEV} (see {@link dev.ltms.fleet.peer.SpawnRequest}), so "the dev pool's
|
||||
* live first entry" is exactly the profile such a spawn actually lands on right now — the
|
||||
* question {@code fleet_profiles}' {@code "default"} field exists to answer. The frozen field is
|
||||
* a role-agnostic fallback used only when {@link #poolFor} has nothing to report at all (no
|
||||
* profiles configured), which {@link #defaultProfileFor} already handles.
|
||||
*/
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return defaultProfile;
|
||||
return defaultProfileFor(MemberRole.DEV);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -475,14 +949,24 @@ public final class CompositePeerLauncher implements PeerLauncher {
|
||||
return route(profileName).capabilities();
|
||||
}
|
||||
|
||||
/** Every herdr agent, deduplicated by pane id (all delegates share one herdr and list globally). */
|
||||
/**
|
||||
* Every herdr agent, deduplicated by (owning daemon, pane id).
|
||||
*
|
||||
* <p>Delegates that share one {@link HerdrClient} see the same global agent set, so listing them
|
||||
* both would report every agent twice — that is what the dedupe is for. But pane ids are
|
||||
* per-daemon counters, so two daemons really can both hold {@code w1:p1} on different panes.
|
||||
* Keying on the pane id alone would silently drop one of them from {@code fleet_list} and from
|
||||
* every status view built on it. The daemon is part of the key for exactly that reason.
|
||||
*/
|
||||
@Override
|
||||
public List<Agent> list() {
|
||||
Map<HerdrClient, Integer> daemonIndex = new IdentityHashMap<>();
|
||||
Map<String, Agent> byPane = new LinkedHashMap<>();
|
||||
for (HerdrPeerLauncher d : delegates) {
|
||||
int daemon = daemonIndex.computeIfAbsent(d.herdr(), _ -> daemonIndex.size());
|
||||
for (Agent a : d.list()) {
|
||||
if (a.paneId() != null) {
|
||||
byPane.putIfAbsent(a.paneId(), a);
|
||||
byPane.putIfAbsent(daemon + "\u0000" + a.paneId(), a);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -7,9 +7,13 @@ import java.io.IOException;
|
||||
import java.io.UncheckedIOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.nio.file.attribute.GroupPrincipal;
|
||||
import java.nio.file.attribute.PosixFileAttributeView;
|
||||
import java.nio.file.attribute.PosixFilePermissions;
|
||||
import java.time.Duration;
|
||||
import java.time.Instant;
|
||||
import java.util.ArrayList;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
import java.util.stream.Stream;
|
||||
@@ -25,33 +29,85 @@ import java.util.stream.Stream;
|
||||
* control lacked: herdr applies that overlay BEFORE the shell starts, so any sourced file can undo
|
||||
* it — and did.
|
||||
*
|
||||
* <p><b>Which file is last depends on the platform, so the scrub runs from two of them.</b> zsh
|
||||
* <p><b>zsh reads its four startup files under three different conditions, so no single file is
|
||||
* guaranteed to run — the scrub has to cover the gap between them, not just the platforms.</b> zsh
|
||||
* reads {@code .zshenv} always, {@code .zprofile} and {@code .zlogin} only for a LOGIN shell, and
|
||||
* {@code .zshrc} only for an INTERACTIVE one. herdr does not open the same kind of shell
|
||||
* everywhere — measured on herdr 0.8.0: macOS panes run {@code -zsh} (login, so {@code .zlogin}
|
||||
* runs), Linux panes run a plain {@code /usr/bin/zsh} (interactive but NOT login, so
|
||||
* {@code .zlogin} never runs at all). A scrub in {@code .zlogin} alone is therefore a control that
|
||||
* silently does nothing on Linux — the exact failure this class exists to remove, one platform
|
||||
* over.
|
||||
* {@code .zshrc} only for an INTERACTIVE one. A pane shell that is at least one of login or
|
||||
* interactive is covered by sourcing the scrub from {@code .zshrc} and {@code .zlogin} (below), but
|
||||
* a pane shell that is NEITHER reads only {@code .zshenv} and stops — fleetd #388, measured: a herdr
|
||||
* pane can be neither login nor interactive, and such a pane read {@code .zshenv}, never reached
|
||||
* {@code scrub.zsh}, and left no report at all. A bare {@code argv[0]} of {@code /usr/bin/zsh}
|
||||
* proves the shell is NOT a login shell; it says nothing about whether it is interactive, so it
|
||||
* must never be read as "therefore interactive" — that wrong inference is what let #388 ship.
|
||||
*
|
||||
* <p>So both {@code .zshrc} and {@code .zlogin} source the same generated {@code scrub.zsh} after
|
||||
* sourcing their {@code $HOME} counterpart. On Linux only the first fires; on macOS both do, and
|
||||
* the second pass is deliberate rather than merely harmless — it re-scrubs anything the operator's
|
||||
* own {@code ~/.zlogin} exported after {@code .zshrc} had finished. Re-running is idempotent: a
|
||||
* name already blank is blanked again, and the report is rewritten with the same counts.
|
||||
* <p>So {@code .zshenv} carries a THIRD pass, guarded by the exact condition that defines the gap:
|
||||
* {@code [[ ! -o login && ! -o interactive ]]}. That guard is why this pass cannot double-scrub a
|
||||
* pane that {@code .zshrc} or {@code .zlogin} will also cover — one of {@code -o login}/
|
||||
* {@code -o interactive} is always true there, so the {@code .zshenv} pass never fires for them, and
|
||||
* their own unconditional sourcing is untouched. The guard also carries a sentinel
|
||||
* ({@value #SCRUB_SENTINEL}) so it fires once per PANE and not once per PROCESS: {@code .zshenv} is
|
||||
* read by every zsh a member's own tooling forks (a plain {@code zsh -c '...'} for a single
|
||||
* command is itself neither login nor interactive), and those children inherit variables their
|
||||
* parent deliberately set for them (git hooks get {@code GIT_DIR}, a venv gets
|
||||
* {@code VIRTUAL_ENV}, a build tool gets {@code NODE_OPTIONS} or {@code JAVA_TOOL_OPTIONS}).
|
||||
* Re-scrubbing every such child would blank all of that, and would also make the pane's own
|
||||
* {@code scrub-report.txt} — rewritten on every pass — describe whichever child exited last
|
||||
* instead of the pane. The sentinel is exported only AFTER {@code scrub.zsh} runs, so the pass
|
||||
* that sets it never sees it and cannot blank it; it must also be on the scrub's own allow-list
|
||||
* (see {@link #generate(Path, Set)}) so a later pass, in the same pane, cannot blank it back to
|
||||
* empty — an exported-but-empty sentinel reads as unset to the {@code -z} guard and would silently
|
||||
* re-enable scrubbing for every subsequent child of that pane.
|
||||
*
|
||||
* <p>So all three of {@code .zshenv} (gap only, guarded), {@code .zshrc}, and {@code .zlogin}
|
||||
* source the same generated {@code scrub.zsh} after sourcing their {@code $HOME} counterpart. A
|
||||
* login-and-interactive pane runs the {@code .zshrc} and {@code .zlogin} passes, and the second is
|
||||
* deliberate rather than merely harmless — it re-scrubs anything the operator's own
|
||||
* {@code ~/.zlogin} exported after {@code .zshrc} had finished. A pane that is neither runs only the
|
||||
* {@code .zshenv} pass. Re-running is idempotent: a name already blank is blanked again, and the
|
||||
* report is rewritten with the same counts.
|
||||
*
|
||||
* <p>Each generated file sources its {@code $HOME} counterpart FIRST, so {@code PATH} and every
|
||||
* toolchain binary still resolve exactly as the operator configured them; only afterwards does
|
||||
* {@code .zlogin} run the scrub: every EXPORTED variable not on the derived allow-list is re-exported
|
||||
* toolchain binary still resolve exactly as the operator configured them; only afterwards does the
|
||||
* scrub run: every EXPORTED variable not on the derived allow-list is re-exported
|
||||
* blank. Blank, not credential-shaped-pattern-filtered: a pattern list ({@code *TOKEN*}, …) is an
|
||||
* enumeration and misses what it did not think of — a username is the other half of a credential and
|
||||
* is shaped like none. Credential-SHAPED names among the blanked set go to the WARN log only,
|
||||
* never to the control.
|
||||
*
|
||||
* <p>The scrub also writes {@code scrub-report.txt} into its own directory: one {@code allowed N of
|
||||
* M} line (N = exports left untouched, M = exports present when the scrub ran), then the blanked
|
||||
* NAMES — never values. The launcher reads this back at teardown and logs it, because a blocked
|
||||
* count next to an unknown denominator is not a finding.
|
||||
* M failed F} line (N = exports left untouched, M = exports present when the scrub ran, F = names
|
||||
* the scrub attempted to blank but could not), then the NAMES — blanked ones bare, unblankable ones
|
||||
* {@code !}-prefixed — never values. The launcher reads this back at teardown and logs it, because a
|
||||
* blocked count next to an unknown denominator is not a finding.
|
||||
*
|
||||
* <p><b>fleetd #394:</b> plain {@code export "$n="} is a FATAL error for a zsh read-only or special
|
||||
* parameter (for example {@code UID}) — it aborts the whole sourced file, so every name still to
|
||||
* come is never blanked and the report above is never written at all. The blanking loop instead
|
||||
* routes each attempt through {@code eval}, which contains that error to the single iteration: the
|
||||
* loop always finishes, and a name that could not be blanked is counted as {@code failed} and
|
||||
* listed {@code !}-prefixed rather than silently disappearing. This is deliberately not a skip-list
|
||||
* of known-bad names — every enumerated name is still attempted, so a name nobody has thought of
|
||||
* yet still gets tried and, if it fails, still gets counted.
|
||||
*
|
||||
* <p>The blanking loop also re-asserts, on its own, the same {@code [A-Za-z_][A-Za-z0-9_]*} shape
|
||||
* check the enumeration loop already applied. Before {@code eval} was introduced a non-conforming
|
||||
* name reaching {@code export "$n="} was harmless either way — the quoting made it inert. With
|
||||
* {@code eval}, the name is spliced into a string and interpreted as shell syntax, so the enumeration
|
||||
* loop's check is no longer sufficient on its own to keep that call site safe — it is a guard on a
|
||||
* different loop, and the two must not silently drift apart. Re-checking right before the
|
||||
* {@code eval} keeps that call site safe by its own reading, independent of whatever the enumeration
|
||||
* loop does or stops doing in a later change.
|
||||
*
|
||||
* <p><b>fleetd #400:</b> {@code eval}'s exit status is not proof that the blank actually happened.
|
||||
* zsh coerces a bare {@code NAME=} assignment on an integer special parameter (measured on macOS zsh
|
||||
* 5.9: {@code SECONDS}, {@code RANDOM}, {@code SHLVL}, {@code HISTSIZE}, {@code COLUMNS},
|
||||
* {@code LINES}, {@code USERNAME}) to a number instead of failing — {@code eval} returns success,
|
||||
* the value is untouched, and a status-based classification reports it as blanked when it was not.
|
||||
* The fix classifies on the observed effect instead: after the attempt, the name's value is read
|
||||
* back with the {@code (P)} indirection flag and the decision is made from whether that is now
|
||||
* empty. This one check covers all three shapes a name can take at this point — a genuine blank, a
|
||||
* fatal read-only error {@code eval} merely contained, and this silent no-op — and the exit status
|
||||
* plays no part in the decision at all.
|
||||
*/
|
||||
public final class EnvAllowListScrub {
|
||||
|
||||
@@ -60,7 +116,11 @@ public final class EnvAllowListScrub {
|
||||
/** Name of the report file written into the generated directory by the scrub itself. */
|
||||
static final String REPORT_FILE = "scrub-report.txt";
|
||||
|
||||
/** The scrub body, generated once and sourced from both {@code .zshrc} and {@code .zlogin}. */
|
||||
/**
|
||||
* The scrub body, generated once and sourced from {@code .zshrc} and {@code .zlogin}
|
||||
* unconditionally, and from {@code .zshenv} when the pane shell is neither login nor
|
||||
* interactive (fleetd #388) — see the class javadoc.
|
||||
*/
|
||||
static final String SCRUB_FILE = "scrub.zsh";
|
||||
|
||||
/** Prefix of every generated directory — also what {@link #reapOrphans} matches on. */
|
||||
@@ -76,14 +136,42 @@ public final class EnvAllowListScrub {
|
||||
private static final String SOURCE_SCRUB =
|
||||
"source \"$ZDOTDIR/" + SCRUB_FILE + "\"\n";
|
||||
|
||||
/**
|
||||
* fleetd #388: marks a pane, not a process, as already scrubbed. Set only by the guarded
|
||||
* {@code .zshenv} pass (see {@link #NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB}) after
|
||||
* {@code scrub.zsh} has run, so it must also be folded into that pass's own allow-list — see
|
||||
* the class javadoc's "must also be on the scrub's own allow-list" paragraph.
|
||||
*/
|
||||
static final String SCRUB_SENTINEL = "_CB633_SCRUBBED";
|
||||
|
||||
/**
|
||||
* Appended to {@code .zshenv}, after its {@code $HOME} source: the third pass, guarded on the
|
||||
* exact condition that defines the gap {@code .zshrc}/{@code .zlogin} do not cover — a shell
|
||||
* that is neither login nor interactive. The sentinel export happens only once the scrub has
|
||||
* already run, and only for as long as the current pane's environment has not been rebuilt from
|
||||
* scratch (a fresh {@code env -i} child would not inherit it — that is out of scope here, since
|
||||
* such a child is no longer running under the pane's own environment at all).
|
||||
*/
|
||||
private static final String NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB =
|
||||
"if [[ ! -o login && ! -o interactive && -z \"${" + SCRUB_SENTINEL + ":-}\" ]]; then\n"
|
||||
+ " " + SOURCE_SCRUB
|
||||
+ " export " + SCRUB_SENTINEL + "=1\n"
|
||||
+ "fi\n";
|
||||
|
||||
private EnvAllowListScrub() {
|
||||
}
|
||||
|
||||
/**
|
||||
* A parsed {@code scrub-report.txt}: how many exported variables existed when the scrub ran,
|
||||
* how many were left untouched (allowed), and the NAMES that were blanked. Values never appear.
|
||||
* A parsed {@code scrub-report.txt}: how many exported variables existed when the scrub ran
|
||||
* ({@code total}), how many were left untouched ({@code allowed}), how many the scrub attempted
|
||||
* to blank but could not ({@code failed} — fleetd #394: a zsh read-only/special parameter such
|
||||
* as {@code UID} fatally errors on plain {@code export NAME=}, so those attempts go through
|
||||
* {@code eval} instead so the loop keeps going and the failure is counted rather than left
|
||||
* invisible), and the NAMES in each of the latter two categories. {@code allowed +
|
||||
* blanked.size() + unblankable.size() == total}, and {@code unblankable.size() == failed}.
|
||||
* Values never appear.
|
||||
*/
|
||||
record ScrubReport(int allowed, int total, List<String> blanked) {
|
||||
record ScrubReport(int allowed, int total, int failed, List<String> blanked, List<String> unblankable) {
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -102,11 +190,16 @@ public final class EnvAllowListScrub {
|
||||
reapOrphans(parentDir);
|
||||
Path dir = Files.createTempDirectory(parentDir, DIR_PREFIX);
|
||||
dir.toFile().deleteOnExit();
|
||||
// The report is written by zsh, after these hooks are registered, so register its path
|
||||
// too — otherwise the directory is non-empty at JVM exit and cannot be removed at all.
|
||||
dir.resolve(REPORT_FILE).toFile().deleteOnExit();
|
||||
write(dir, SCRUB_FILE, scrubScript(allowedNames));
|
||||
write(dir, ".zshenv", homeSourcingFile(".zshenv"));
|
||||
// zsh truncates this pre-created receipt after these hooks are registered. Register its
|
||||
// path too — otherwise the directory is non-empty at JVM exit and cannot be removed.
|
||||
Files.createFile(dir.resolve(REPORT_FILE)).toFile().deleteOnExit();
|
||||
// fleetd #388: scrub.zsh's OWN allow-list must also keep SCRUB_SENTINEL, or a later
|
||||
// pass in the same pane blanks it back to empty and the .zshenv guard below thinks it
|
||||
// was never scrubbed — see the class javadoc.
|
||||
Set<String> namesForScrubScript = new HashSet<>(allowedNames);
|
||||
namesForScrubScript.add(SCRUB_SENTINEL);
|
||||
write(dir, SCRUB_FILE, scrubScript(namesForScrubScript));
|
||||
write(dir, ".zshenv", homeSourcingFile(".zshenv") + NEITHER_LOGIN_NOR_INTERACTIVE_SCRUB);
|
||||
write(dir, ".zprofile", homeSourcingFile(".zprofile"));
|
||||
write(dir, ".zshrc", homeSourcingFile(".zshrc") + SOURCE_SCRUB);
|
||||
write(dir, ".zlogin", homeSourcingFile(".zlogin") + SOURCE_SCRUB);
|
||||
@@ -116,6 +209,111 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #213: as {@link #generate(Path, Set)}, plus share the generated directory with
|
||||
* {@code group} — the member OS user's group (the operator's existing {@code worktreeGroup:}
|
||||
* name, reused rather than inventing a second one) — so a member running under a different OS
|
||||
* user than fleetd's own process can still read what it needs from a directory placed outside
|
||||
* {@code java.io.tmpdir}. {@code group} null/blank ⇒ identical to {@link #generate(Path, Set)};
|
||||
* this is the single-daemon (no {@code memberHerdrSocket}) shape, where the pane is fleetd's own
|
||||
* uid and no group sharing is needed.
|
||||
*
|
||||
* @throws UncheckedIOException also when {@code group} does not resolve on this host, or a
|
||||
* group-ownership/permission call is refused — the same "fail
|
||||
* loudly rather than start unprotected" contract as above: a scrub
|
||||
* the configured member user cannot even read is not a working
|
||||
* control.
|
||||
*/
|
||||
public static Path generate(Path parentDir, Set<String> allowedNames, String group) {
|
||||
Path dir = generate(parentDir, allowedNames);
|
||||
if (group != null && !group.isBlank()) {
|
||||
shareWithGroup(dir, group);
|
||||
}
|
||||
return dir;
|
||||
}
|
||||
|
||||
/**
|
||||
* chgrp/chmod-equivalent over a freshly generated directory and the flat files already written
|
||||
* into it: owner keeps full access, {@code group} gets traverse+read on the directory ({@code
|
||||
* rwxr-x---}, so a member process — a login shell reading it via {@code ZDOTDIR}, or another
|
||||
* process simply opening a file under it — running under that group can find and read the
|
||||
* files) and read-only on each file ({@code rw-r-----}), except the pre-created ZDOTDIR
|
||||
* {@code scrub-report.txt}. That receipt gets group write ({@code rw-rw----}), so
|
||||
* {@code scrub.zsh} can truncate and write it without granting group write on the directory.
|
||||
* If its optional permission change fails, the member cannot write a receipt and the launcher
|
||||
* keeps its existing WARN rather than failing the spawn.
|
||||
*
|
||||
* <p>Package-private and named generically on purpose: fleetd #213 built this for the ZDOTDIR
|
||||
* scrub directory, and fleetd #219 reuses it verbatim for {@link
|
||||
* dev.ltms.fleet.member.OpenCodeLauncher}'s ephemeral {@code opencode.json} directory — both are
|
||||
* "a fleetd-generated directory of flat files that a different-uid member process must read but
|
||||
* never write," so the sharing mechanism is shared rather than copied a second time.
|
||||
*/
|
||||
static void shareWithGroup(Path dir, String group) {
|
||||
try {
|
||||
GroupPrincipal principal = dir.getFileSystem().getUserPrincipalLookupService()
|
||||
.lookupPrincipalByGroupName(group);
|
||||
setGroupAndPermissions(dir, principal, "rwxr-x---");
|
||||
try (Stream<Path> entries = Files.list(dir)) {
|
||||
for (Path file : entries.toList()) {
|
||||
if (REPORT_FILE.equals(file.getFileName().toString())) {
|
||||
try {
|
||||
setGroupAndPermissions(file, principal, "rw-rw----");
|
||||
} catch (IOException | UnsupportedOperationException ignored) {
|
||||
// The receipt is optional. Its absence keeps the existing WARN path.
|
||||
}
|
||||
} else {
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot share generated directory " + dir + " with group '"
|
||||
+ group + "' — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
} catch (UnsupportedOperationException e) {
|
||||
throw new UncheckedIOException("cannot share generated directory " + dir + " with group '"
|
||||
+ group + "' — this filesystem does not support POSIX group ownership",
|
||||
new IOException(e));
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #285: share ONE file with {@code group}, read-only ({@code rw-r-----}) — the same
|
||||
* per-file mode {@link #shareWithGroup} applies to a directory's entries, and the same error
|
||||
* shapes, but without touching a parent directory. Used for a file fleetd writes into a
|
||||
* directory it does NOT own — {@code configDir}'s own traversal permissions stay the
|
||||
* operator's setup — where the directory-wide {@link #shareWithGroup} would be wrong.
|
||||
*
|
||||
* @throws UncheckedIOException when {@code group} does not resolve on this host, the
|
||||
* filesystem has no POSIX group ownership, or a
|
||||
* group-ownership/permission call is refused
|
||||
*/
|
||||
static void shareFileWithGroup(Path file, String group) {
|
||||
try {
|
||||
GroupPrincipal principal = file.getFileSystem().getUserPrincipalLookupService()
|
||||
.lookupPrincipalByGroupName(group);
|
||||
setGroupAndPermissions(file, principal, "rw-r-----");
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — the group must exist, and the fleetd operator ("
|
||||
+ System.getProperty("user.name") + ") must be a member of it", e);
|
||||
} catch (UnsupportedOperationException e) {
|
||||
throw new UncheckedIOException("cannot share generated file " + file + " with group '"
|
||||
+ group + "' — this filesystem does not support POSIX group ownership",
|
||||
new IOException(e));
|
||||
}
|
||||
}
|
||||
|
||||
private static void setGroupAndPermissions(Path path, GroupPrincipal group, String perms) throws IOException {
|
||||
PosixFileAttributeView view = Files.getFileAttributeView(path, PosixFileAttributeView.class);
|
||||
if (view == null) {
|
||||
throw new IOException("POSIX file attributes are not supported for " + path);
|
||||
}
|
||||
view.setGroup(group);
|
||||
Files.setPosixFilePermissions(path, PosixFilePermissions.fromString(perms));
|
||||
}
|
||||
|
||||
/** One operator-sourcing startup file: source the {@code $HOME} counterpart, change nothing else. */
|
||||
private static String homeSourcingFile(String name) {
|
||||
return """
|
||||
@@ -144,11 +342,13 @@ public final class EnvAllowListScrub {
|
||||
}
|
||||
return """
|
||||
# generated by fleetd (CB-633 memberCredentials policy=allow-list) — do not edit.
|
||||
# Sourced from .zshrc and again from .zlogin, each time AFTER that file has sourced
|
||||
# its $HOME counterpart — so this runs after everything the operator sourced, on a
|
||||
# login shell (macOS panes) and on a plain interactive one (Linux panes) alike.
|
||||
# Running twice is idempotent and deliberate: the second pass catches anything
|
||||
# ~/.zlogin exported after ~/.zshrc had finished.
|
||||
# Sourced unconditionally from .zshrc and again from .zlogin, each time AFTER that
|
||||
# file has sourced its $HOME counterpart — so this runs after everything the
|
||||
# operator sourced, on any pane that is login and/or interactive. Running twice is
|
||||
# idempotent and deliberate: the second pass catches anything ~/.zlogin exported
|
||||
# after ~/.zshrc had finished. Also sourced, once, from a guarded pass in .zshenv
|
||||
# (fleetd #388) when the pane shell is NEITHER login nor interactive — the one gap
|
||||
# those two files do not cover.
|
||||
|
||||
typeset -A _cb633_allowed
|
||||
for _cb633_n in %s; do _cb633_allowed[$_cb633_n]=1; done
|
||||
@@ -170,15 +370,57 @@ public final class EnvAllowListScrub {
|
||||
_cb633_blank+=("$_cb633_n")
|
||||
done
|
||||
|
||||
{ for _cb633_n in "${_cb633_blank[@]}"; do export "$_cb633_n="; done; } 2>/dev/null
|
||||
# fleetd #394: plain `export "$n="` is FATAL for a zsh read-only/special parameter
|
||||
# (e.g. UID) and aborts this whole sourced file — every name still to come is never
|
||||
# blanked, and the report below is never written, silently. `eval` contains that
|
||||
# error to the single iteration instead: it still fails for that one name, but the
|
||||
# loop continues and we can tell allowed / blanked / unblankable apart afterwards.
|
||||
# This is not a skip-list of known-bad names (that would miss the next one nobody
|
||||
# thought of) — every name in _cb633_blank is still attempted, unconditionally.
|
||||
# Every name reaching this loop already passed the identical identifier check in the
|
||||
# enumeration loop above — but that guard is 20 lines away in a different loop, and
|
||||
# this line is about to splice the name into a string handed to `eval`. Before this
|
||||
# fix the name only ever reached `export` quoted ("$n="), which is inert on a
|
||||
# non-identifier string either way; `eval` makes THIS line the only thing standing
|
||||
# between such a string and code execution in the member's pane, so it re-asserts the
|
||||
# same check on its own rather than trusting a guard it does not own. Under normal
|
||||
# operation this can never fire (the enumeration guard already filtered everything
|
||||
# reaching _cb633_blank), so a name caught here is counted as unblankable rather than
|
||||
# silently dropped — it is real evidence that the upstream guard was bypassed.
|
||||
#
|
||||
# fleetd #400: the attempt's own exit status is NOT proof of its effect. zsh coerces
|
||||
# a bare `NAME=` assignment on an integer special parameter (SECONDS, RANDOM, SHLVL,
|
||||
# HISTSIZE, COLUMNS, LINES, USERNAME on this host) to a number instead of failing —
|
||||
# `eval` returns 0, the value is untouched, and the old exit-status check reported it
|
||||
# as blanked when it was not. Classify on the observed effect instead: attempt the
|
||||
# export, then read the name's value back with the `(P)` indirection flag and decide
|
||||
# from whether it is now empty. One check then covers all three shapes a name can
|
||||
# take here — a genuine blank, a fatal read-only error `eval` merely contained, and
|
||||
# this silent no-op — without the exit status entering the decision at all.
|
||||
typeset -a _cb633_ok _cb633_unblankable
|
||||
_cb633_ok=()
|
||||
_cb633_unblankable=()
|
||||
for _cb633_n in "${_cb633_blank[@]}"; do
|
||||
if [[ ! "$_cb633_n" =~ ^[A-Za-z_][A-Za-z0-9_]*$ ]]; then
|
||||
_cb633_unblankable+=("$_cb633_n")
|
||||
continue
|
||||
fi
|
||||
eval "export ${_cb633_n}=" 2>/dev/null
|
||||
if [[ -z "${(P)_cb633_n}" ]]; then
|
||||
_cb633_ok+=("$_cb633_n")
|
||||
else
|
||||
_cb633_unblankable+=("$_cb633_n")
|
||||
fi
|
||||
done
|
||||
|
||||
integer _cb633_kept=$(( _cb633_total - ${#_cb633_blank} ))
|
||||
{
|
||||
print -r -- "allowed $_cb633_kept of $_cb633_total"
|
||||
for _cb633_n in "${_cb633_blank[@]}"; do print -r -- "$_cb633_n"; done
|
||||
print -r -- "allowed $_cb633_kept of $_cb633_total failed ${#_cb633_unblankable}"
|
||||
for _cb633_n in "${_cb633_ok[@]}"; do print -r -- "$_cb633_n"; done
|
||||
for _cb633_n in "${_cb633_unblankable[@]}"; do print -r -- "!$_cb633_n"; done
|
||||
} > "$ZDOTDIR/%s" 2>/dev/null
|
||||
|
||||
unset _cb633_allowed _cb633_names _cb633_blank _cb633_n _cb633_total _cb633_kept
|
||||
unset _cb633_allowed _cb633_names _cb633_blank _cb633_ok _cb633_unblankable _cb633_n _cb633_total _cb633_kept
|
||||
""".formatted(names, MemberEnvAllowList.zshCasePattern(), REPORT_FILE);
|
||||
}
|
||||
|
||||
@@ -192,6 +434,11 @@ public final class EnvAllowListScrub {
|
||||
* Read and parse {@link #REPORT_FILE} out of a generated ZDOTDIR directory. Returns {@code null}
|
||||
* when absent or unreadable (the pane may have been torn down before its login shell ever got to
|
||||
* the scrub) — callers treat that as "no measurement available", never as success.
|
||||
*
|
||||
* <p>First line is {@code "allowed <N> of <M> failed <F>"} (fleetd #394 added the trailing
|
||||
* {@code failed <F>} — a count of names the scrub attempted to blank but could not, e.g. a zsh
|
||||
* read-only/special parameter). Every following non-blank line is a name: a bare name was
|
||||
* blanked, a {@code !}-prefixed name was attempted and failed. Values never appear on either.
|
||||
*/
|
||||
static ScrubReport readReport(Path zdotdir) {
|
||||
Path report = zdotdir.resolve(REPORT_FILE);
|
||||
@@ -204,17 +451,24 @@ public final class EnvAllowListScrub {
|
||||
return null;
|
||||
}
|
||||
String[] parts = lines.getFirst().substring("allowed ".length()).trim().split("\\s+");
|
||||
if (parts.length != 3 || !"of".equals(parts[1])) {
|
||||
if (parts.length != 5 || !"of".equals(parts[1]) || !"failed".equals(parts[3])) {
|
||||
return null;
|
||||
}
|
||||
List<String> blanked = new ArrayList<>();
|
||||
List<String> unblankable = new ArrayList<>();
|
||||
for (int i = 1; i < lines.size(); i++) {
|
||||
if (!lines.get(i).isBlank()) {
|
||||
blanked.add(lines.get(i));
|
||||
String line = lines.get(i);
|
||||
if (line.isBlank()) {
|
||||
continue;
|
||||
}
|
||||
if (line.startsWith("!")) {
|
||||
unblankable.add(line.substring(1));
|
||||
} else {
|
||||
blanked.add(line);
|
||||
}
|
||||
}
|
||||
return new ScrubReport(Integer.parseInt(parts[0]), Integer.parseInt(parts[2]),
|
||||
List.copyOf(blanked));
|
||||
Integer.parseInt(parts[4]), List.copyOf(blanked), List.copyOf(unblankable));
|
||||
} catch (IOException | NumberFormatException e) {
|
||||
return null;
|
||||
}
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,75 @@
|
||||
package dev.ltms.fleet.member;
|
||||
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): a single, testable read of the {@code memberCredentials:} policy — names
|
||||
* and counts only, never a value. The daemon never holds a credential's <em>value</em> in the
|
||||
* first place (only the names configured under {@code known:}/{@code allow:}), so there is
|
||||
* nothing here to redact by construction; the point of this class is that it is the ONE place
|
||||
* that turns a policy into names-and-counts, so nothing else hand-counts a second time.
|
||||
*
|
||||
* <p>Before this class, {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} computed these
|
||||
* same counts inline for the startup log line, and {@code scripts/probe-member-credentials.sh}
|
||||
* carried its own hardcoded {@code NAMES} array that the live policy could grow past silently
|
||||
* (#111) — the exact "hand-maintained second copy drifts" shape #114 fixed for the tool
|
||||
* catalogue. Both now read this class: the startup log via {@link
|
||||
* dev.ltms.fleet.Fleetd#reportMemberCredentialsGap}, and a live daemon via the {@code
|
||||
* GET /member-credentials} REST endpoint ({@link dev.ltms.fleet.rest.FleetApp}), which the probe
|
||||
* script fetches instead of carrying its own list.
|
||||
*
|
||||
* @param present policy configured with at least one {@code known} name. {@code false} for an
|
||||
* absent or empty {@code memberCredentials:} block — represented honestly as "no
|
||||
* policy", never as "nothing blocked" (an empty {@link #blocked} could otherwise be
|
||||
* misread as a clean bill of health).
|
||||
* @param policy the normalized policy mode ({@link FleetConfig.MemberCredentials#policy()}), or
|
||||
* {@code null} when {@link #present} is {@code false}.
|
||||
* @param known every name the policy declares, in configured order. Names only, never a value.
|
||||
* @param allowed the subset of {@link #known} explicitly let through. Names only.
|
||||
* @param blocked {@link #known} minus {@link #allowed} — the names an actual spawn shadows. Names
|
||||
* only.
|
||||
*/
|
||||
public record MemberCredentialPolicyView(boolean present, String policy, List<String> known,
|
||||
List<String> allowed, List<String> blocked) {
|
||||
|
||||
private static final MemberCredentialPolicyView ABSENT =
|
||||
new MemberCredentialPolicyView(false, null, List.of(), List.of(), List.of());
|
||||
|
||||
public MemberCredentialPolicyView {
|
||||
known = known == null ? List.of() : List.copyOf(known);
|
||||
allowed = allowed == null ? List.of() : List.copyOf(allowed);
|
||||
blocked = blocked == null ? List.of() : List.copyOf(blocked);
|
||||
}
|
||||
|
||||
/** The honest "no policy configured" view. */
|
||||
public static MemberCredentialPolicyView absent() {
|
||||
return ABSENT;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build the view straight from the live config. {@code creds} may be {@code null} (no {@code
|
||||
* memberCredentials:} block at all) — treated the same as a present-but-empty block, exactly
|
||||
* like {@link dev.ltms.fleet.Fleetd#reportMemberCredentialsGap} already did.
|
||||
*/
|
||||
public static MemberCredentialPolicyView of(FleetConfig.MemberCredentials creds) {
|
||||
if (creds == null || creds.known().isEmpty()) {
|
||||
return ABSENT;
|
||||
}
|
||||
return new MemberCredentialPolicyView(true, creds.policy(), creds.known(), creds.allow(),
|
||||
List.copyOf(creds.blockedSet()));
|
||||
}
|
||||
|
||||
public int knownCount() {
|
||||
return known.size();
|
||||
}
|
||||
|
||||
public int allowedCount() {
|
||||
return allowed.size();
|
||||
}
|
||||
|
||||
public int blockedCount() {
|
||||
return blocked.size();
|
||||
}
|
||||
}
|
||||
@@ -34,10 +34,29 @@ import java.util.TreeSet;
|
||||
*
|
||||
* <p>{@code SSH_AUTH_SOCK} is deliberately NOT here. It is a handle to the operator's ssh-agent — a
|
||||
* member holding it can sign with the operator's keys — so keeping it is a config decision
|
||||
* ({@code memberCredentials.sshAuthSock: allow}), not a derivation default.
|
||||
* ({@code memberCredentials.sshAgentEnv: inherit}), not a derivation default.
|
||||
*
|
||||
* <p><b>CB-633 follow-up:</b> the union also includes {@code memberCredentials.allow:} — the
|
||||
* operator's own explicit list. Before this, {@code policy: allow-list} silently ignored every name
|
||||
* an operator wrote under {@code allow:} unless a profile happened to carry it too, which meant
|
||||
* turning the policy on could blank credentials working members already depended on. {@code
|
||||
* SSH_AUTH_SOCK} and configured broker URI environment names are exceptions: even when the operator
|
||||
* lists them under {@code allow:}, they are excluded here. {@code SSH_AUTH_SOCK} is added back ONLY
|
||||
* by the caller when {@code sshAgentEnv: inherit} is explicitly set
|
||||
* (see {@link #SSH_AUTH_SOCK}'s javadoc) — it is a live handle to the operator's own ssh-agent, not
|
||||
* a value, so treating it like any other allow-listed name would hand a member every key the
|
||||
* operator's agent holds the moment they typed the name under {@code allow:} for an unrelated
|
||||
* reason.
|
||||
*/
|
||||
public final class MemberEnvAllowList {
|
||||
|
||||
/**
|
||||
* The operator's ssh-agent socket path. Deliberately excluded from {@link #derive}'s union of
|
||||
* {@code memberCredentials.allow:} — see the class javadoc's CB-633 follow-up note. Governed
|
||||
* ONLY by {@code memberCredentials.sshAgentEnv}, never by appearing in {@code allow:}.
|
||||
*/
|
||||
public static final String SSH_AUTH_SOCK = "SSH_AUTH_SOCK";
|
||||
|
||||
/**
|
||||
* Names that are not credentials and that a login shell or agent binary genuinely needs.
|
||||
*
|
||||
@@ -73,9 +92,31 @@ public final class MemberEnvAllowList {
|
||||
|
||||
/**
|
||||
* Derive the allowed NAME set from the given profiles plus {@link #INFRASTRUCTURE_PASSTHROUGH}.
|
||||
* Deterministic (sorted) so generated scrub files are diffable run-to-run.
|
||||
* Equivalent to {@link #derive(Collection, Set)} with no operator-configured names — kept for
|
||||
* callers (and existing tests) that only care about the profile-derived half.
|
||||
*/
|
||||
public static Set<String> derive(Collection<FleetConfig.Profile> profiles) {
|
||||
return derive(profiles, Set.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* Derive the allowed NAME set: the profile-derived union above, PLUS {@code configuredAllow} —
|
||||
* the operator's own {@code memberCredentials.allow:} list (CB-633 follow-up). {@code
|
||||
* SSH_AUTH_SOCK} is dropped from {@code configuredAllow} even if the operator listed it there;
|
||||
* see the class javadoc for why. Deterministic (sorted) so generated scrub files are diffable
|
||||
* run-to-run.
|
||||
*/
|
||||
public static Set<String> derive(Collection<FleetConfig.Profile> profiles, Set<String> configuredAllow) {
|
||||
return derive(profiles, configuredAllow, Set.of());
|
||||
}
|
||||
|
||||
/**
|
||||
* As {@link #derive(Collection, Set)}, while excluding names that fleetd knows carry credentials.
|
||||
* A configured broker URI contains its AMQP password inline, so it must never reach a member,
|
||||
* even when an operator put its variable name in {@code memberCredentials.allow:}.
|
||||
*/
|
||||
public static Set<String> derive(Collection<FleetConfig.Profile> profiles, Set<String> configuredAllow,
|
||||
Set<String> excludedNames) {
|
||||
Set<String> derived = new TreeSet<>(INFRASTRUCTURE_PASSTHROUGH);
|
||||
if (profiles != null) {
|
||||
for (FleetConfig.Profile p : profiles) {
|
||||
@@ -87,9 +128,44 @@ public final class MemberEnvAllowList {
|
||||
}
|
||||
}
|
||||
}
|
||||
if (configuredAllow != null) {
|
||||
for (String name : configuredAllow) {
|
||||
if (name != null && !name.isBlank() && !SSH_AUTH_SOCK.equals(name)) {
|
||||
derived.add(name);
|
||||
}
|
||||
}
|
||||
}
|
||||
if (excludedNames != null) {
|
||||
derived.removeAll(excludedNames);
|
||||
}
|
||||
return Set.copyOf(derived);
|
||||
}
|
||||
|
||||
/**
|
||||
* The host environment names whose values are AMQP URIs with inline passwords. Both broker
|
||||
* connections belong to fleetd, never to a member pane. Blank and absent configuration changes
|
||||
* nothing.
|
||||
*
|
||||
* <p><b>How strong this exclusion is depends on the policy, and the difference matters.</b>
|
||||
* Under {@code policy: allow-list} it is enforced by the generated ZDOTDIR scrub, which runs
|
||||
* AFTER the pane's shell has sourced the operator's chain — so a login shell that re-exports the
|
||||
* name is still blanked. Under the deny-list policy there is no scrub: the name is only removed
|
||||
* from the pre-shell env map, and a login shell that sources the operator's secret store
|
||||
* re-exports it. That is the long-standing weakness of deny-list (a sourced file can undo it),
|
||||
* not something this exclusion introduces, but it means deny-list deployments do NOT get this
|
||||
* guarantee. The same caveat applies to the non-zsh path, which has no scrub at all — see
|
||||
* {@code HerdrPeerLauncher#applyEnvironmentAllowListPolicy}.
|
||||
*/
|
||||
public static Set<String> brokerUriEnvNames(FleetConfig config) {
|
||||
if (config == null) {
|
||||
return Set.of();
|
||||
}
|
||||
Set<String> names = new TreeSet<>();
|
||||
addUriEnvIfPresent(names, config.broker());
|
||||
addUriEnvIfPresent(names, config.coordinator());
|
||||
return Set.copyOf(names);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code name} survives the scrub when {@code allowedNames} is the derived set: an exact
|
||||
* match, or an infrastructure-prefixed name ({@code LC_*}). Prefix rules live ONLY here and in
|
||||
@@ -109,4 +185,16 @@ public final class MemberEnvAllowList {
|
||||
into.add(name);
|
||||
}
|
||||
}
|
||||
|
||||
private static void addUriEnvIfPresent(Set<String> into, FleetConfig.Broker broker) {
|
||||
if (broker != null && broker.hasUriEnv()) {
|
||||
into.add(broker.uriEnv());
|
||||
}
|
||||
}
|
||||
|
||||
private static void addUriEnvIfPresent(Set<String> into, FleetConfig.Coordinator coordinator) {
|
||||
if (coordinator != null && coordinator.hasUriEnv()) {
|
||||
into.add(coordinator.uriEnv());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -6,6 +6,7 @@ import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.CharterReceipt;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
@@ -20,6 +21,10 @@ import java.util.EnumSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.BooleanSupplier;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.LongSupplier;
|
||||
import java.util.function.Supplier;
|
||||
@@ -71,6 +76,19 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
*/
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
|
||||
/**
|
||||
* fleetd #175: notified when {@link SessionAwareHandle} detects, on the same late-resolve
|
||||
* read that discovers the session id, that the live opencode session is running a DIFFERENT
|
||||
* model than the profile requested — opencode does not fail on an unknown {@code -m}, it
|
||||
* silently falls back to a default (potentially paid) model. Reused exactly as
|
||||
* {@code CompletionResolver}'s {@code BACKEND_EXHAUSTED} path uses it: this launcher supplies
|
||||
* only the herdr terminal id and a reason string; mapping target → session → profile →
|
||||
* credential stays entirely the sink's job (see {@code Fleetd.main}'s wiring). Defaults to
|
||||
* {@link ExhaustionSink#none()} for a caller (an older constructor, or a test not exercising
|
||||
* this) that opts out.
|
||||
*/
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
|
||||
/**
|
||||
* Production constructor — disables the spawn-ready gate ({@code spawnReadyTimeoutMs == 0}) so it
|
||||
* matches the legacy non-blocking spawn semantics. Config dirs are created under the JVM temp dir.
|
||||
@@ -116,12 +134,39 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env,
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, null);
|
||||
}
|
||||
|
||||
/** Production constructor, plus the live config for URI environment exclusions. */
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, spawnReadyPollMs,
|
||||
fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Production constructor, plus the live config for URI environment exclusions and the fleetd
|
||||
* #175 model-mismatch quarantine sink.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs, long spawnReadyPollMs,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs,
|
||||
System::currentTimeMillis, () -> sleepUninterruptibly(spawnReadyPollMs),
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials);
|
||||
defaultConfigRoot(), defaultDiscoveryRoot(), fleet, memberCredentials, config,
|
||||
exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -169,6 +214,7 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = ExhaustionSink.none();
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -182,17 +228,84 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
configRoot, discoveryRoot, fleet, memberCredentials, null);
|
||||
}
|
||||
|
||||
/** Full testability constructor, plus the live config for URI environment exclusions. */
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config) {
|
||||
this(agents, spaces, profiles, defaultProfile, env, spawnReadyTimeoutMs, nowMillis, sleeper,
|
||||
configRoot, discoveryRoot, fleet, memberCredentials, config, ExhaustionSink.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* Full testability constructor, plus the live config for URI environment exclusions and the
|
||||
* fleetd #175 model-mismatch quarantine sink — the constructor a test drives directly to
|
||||
* observe {@link ExhaustionSink#onExhausted} without going through {@code Fleetd.main}'s
|
||||
* wiring.
|
||||
*/
|
||||
public OpenCodeLauncher(AgentControl agents, WorkspaceControl spaces,
|
||||
Map<String, FleetConfig.Profile> profiles, String defaultProfile,
|
||||
Function<String, String> env, long spawnReadyTimeoutMs,
|
||||
LongSupplier nowMillis, Runnable sleeper, Path configRoot, Path discoveryRoot,
|
||||
Supplier<FleetConfig.Fleet> fleet,
|
||||
Supplier<FleetConfig.MemberCredentials> memberCredentials,
|
||||
Supplier<FleetConfig> config,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
super(NAME_PREFIX, agents, spaces, profiles, defaultProfile, env,
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials);
|
||||
spawnReadyTimeoutMs, nowMillis, sleeper, fleet, memberCredentials, null, config);
|
||||
this.configRoot = configRoot;
|
||||
this.discovery = new OpenCodeSessionDiscovery(discoveryRoot);
|
||||
this.exhaustionSink = exhaustionSink == null ? ExhaustionSink.none() : exhaustionSink;
|
||||
}
|
||||
|
||||
private static Path defaultConfigRoot() {
|
||||
return Path.of(System.getProperty("java.io.tmpdir"));
|
||||
}
|
||||
|
||||
/** The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir). */
|
||||
/**
|
||||
* The default opencode storage root: {@code ~/.local/share/opencode} (the XDG data dir) —
|
||||
* always FLEETD's OWN {@code user.home}, whichever OS user runs the daemon.
|
||||
*
|
||||
* <p><b>fleetd #219 site 2 — a decision, not a patch.</b> Under {@code memberHerdrSocket:} the
|
||||
* member pane runs as a <em>different</em> OS user, and opencode writes {@code opencode.db}
|
||||
* under <em>that</em> user's {@code $HOME}, not fleetd's. Scanning fleetd's own {@code
|
||||
* user.home} is therefore looking in the wrong place — a wrong-LOCATION failure, not a
|
||||
* wrong-PERMISSION one like site 1, and it fails quietly: {@link
|
||||
* SessionAwareHandle#agentSessionId()} would keep returning {@code null} forever, which reads
|
||||
* as "opencode does not support resume" rather than "fleetd looked in the wrong home." fleetd
|
||||
* #209 is the reason that silence is unacceptable.
|
||||
*
|
||||
* <p>Three ways to close the gap were weighed:
|
||||
* <ol>
|
||||
* <li><b>Make the member's home configurable.</b> Correct in principle, but this ticket's
|
||||
* scope is the two existing call sites, not a new config key — {@code memberHerdrSocket}
|
||||
* already carries the second herdr's socket path, not its user's home, and inventing a
|
||||
* parallel key here without also wiring it through discovery's actual callers is a
|
||||
* half-shipped feature (the exact shape CB-596/CB-611 warn against).</li>
|
||||
* <li><b>Derive it</b> (e.g. from {@code worktreeRoot}'s owner, or {@code getent passwd}).
|
||||
* Rejected: nothing in this codebase resolves a Unix username to a home directory today,
|
||||
* and guessing wrong would silently point discovery at a THIRD wrong location — worse
|
||||
* than the current gap, because it would look like it should work.</li>
|
||||
* <li><b>Declare discovery unavailable</b> under {@code memberHerdrSocket}, and say so once,
|
||||
* loudly, instead of scanning a directory that structurally cannot hold the answer.</li>
|
||||
* </ol>
|
||||
*
|
||||
* <p>Option 3 is taken — the one this ticket says to default to when unsure. {@link
|
||||
* OpenCodeLauncher#spawn} routes {@link SessionAwareHandle#agentSessionId()} through {@link
|
||||
* HerdrPeerLauncher#memberHerdrSocketConfigured()} before ever calling {@link
|
||||
* OpenCodeSessionDiscovery#sessionIdForDirectory}, so under {@code memberHerdrSocket} the
|
||||
* database at this root is never even opened, and one WARN per launcher instance names the gap
|
||||
* instead of the {@code null} return reading as "unsupported." Capability advertising is
|
||||
* unaffected: {@link #capabilities()} always includes {@code SESSION_RESUME}, since {@code
|
||||
* memberHerdrSocket} absent (today's only live mode) is unchanged by this decision.
|
||||
*/
|
||||
private static Path defaultDiscoveryRoot() {
|
||||
return Path.of(System.getProperty("user.home"), ".local", "share", "opencode");
|
||||
}
|
||||
@@ -218,11 +331,22 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
+ "form — opencode's per-model context limit could not be applied for this profile",
|
||||
cfg.profile(), cfg.model());
|
||||
}
|
||||
// fleetd #393: memberSkills seeding (GitWorktrees#seedSkills) copies skill folders into
|
||||
// EVERY provisioned worktree's .claude/skills/ regardless of which kind ultimately spawns
|
||||
// into it — that copy step cannot know the kind, only the caller of GitWorktrees#add does
|
||||
// (see that method's own javadoc). .claude/skills/ is a Claude Code CLI convention the CLI
|
||||
// discovers on its own; opencode has no such discovery, so without this, a seeded skill
|
||||
// never reaches an opencode member even though GitWorktrees logged it as seeded. Read
|
||||
// whatever landed under <cwd>/.claude/skills/ here — the one place in this launcher that
|
||||
// knows both the kind (opencode, by construction: this IS OpenCodeLauncher) and the cwd.
|
||||
List<Path> skillInstructionFiles = skillInstructionFiles(spec.cwd());
|
||||
// A config file is needed for the bridge MCP mount, a member charter, the IDE MCP (+ its
|
||||
// guidance overlay, CB-634), a pinned endpoint (CB-508), or a resolvable autoCompactWindow.
|
||||
// guidance overlay, CB-634), a pinned endpoint (CB-508), a resolvable autoCompactWindow, or
|
||||
// at least one seeded skill to deliver via instructions[] (fleetd #393).
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp() || spec.charter() != null || hasCustomProvider(cfg)
|
||||
|| wantsContextLimit) {
|
||||
workerEnv.put("OPENCODE_CONFIG", writeConfig(cfg, spec.charter(), spec.cwd()).toString());
|
||||
|| wantsContextLimit || !skillInstructionFiles.isEmpty()) {
|
||||
workerEnv.put("OPENCODE_CONFIG",
|
||||
writeConfig(cfg, spec.charter(), spec.cwd(), skillInstructionFiles).toString());
|
||||
}
|
||||
applyGitToken(workerEnv, cfg);
|
||||
List<String> argv = argvWithResume(argvWithModel(argvWithAuto(cfg), cfg), spec.resumeSessionId());
|
||||
@@ -245,6 +369,65 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
return withAgent;
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #393: the {@code SKILL.md} paths under {@code <cwd>/.claude/skills/} this launcher can
|
||||
* turn into {@code instructions[]} entries, plus the honest log this ticket asks for — emitted
|
||||
* here, at the one point a skill's fate for THIS spawn is actually known, rather than trusting
|
||||
* {@code GitWorktrees#seedSkills}'s kind-blind "N of M" line to mean "and it will be read."
|
||||
*
|
||||
* <p>Every non-hidden subdirectory of {@code .claude/skills/} is a candidate, whether it got
|
||||
* there via {@code memberSkills:} seeding or because the target repo ships its own copy — this
|
||||
* launcher does not care which; it only cares what it can find at spawn time. A candidate with
|
||||
* a {@code SKILL.md} at its top level (the same shape {@link #writeConfig} already requires for
|
||||
* the charter and IDE-rules instructions entries) is delivered; anything else is a directory
|
||||
* this launcher cannot turn into a flat instructions entry, named explicitly in the log rather
|
||||
* than silently dropped, so a caller sees a real "cannot consume" reason and not just a smaller
|
||||
* number than {@code GitWorktrees}' own count.
|
||||
*
|
||||
* <p>No candidates at all (directory absent or empty) logs nothing — the same
|
||||
* no-log-when-nothing-to-say shape {@link #hasCustomProvider} and friends already follow, and
|
||||
* the shape {@code GitWorktrees#seedSkills} itself uses when {@code memberSkills:} is unset.
|
||||
* A failure to even list the directory is logged and treated as "nothing delivered" — best
|
||||
* effort, must never fail the spawn, matching {@code GitWorktrees#seedSkills}'s own contract.
|
||||
*/
|
||||
private List<Path> skillInstructionFiles(String cwd) {
|
||||
if (cwd == null || cwd.isBlank()) {
|
||||
return List.of();
|
||||
}
|
||||
Path skillsDir = Path.of(cwd, ".claude", "skills");
|
||||
if (!Files.isDirectory(skillsDir)) {
|
||||
return List.of();
|
||||
}
|
||||
List<Path> candidates;
|
||||
try (var listing = Files.list(skillsDir)) {
|
||||
candidates = listing.filter(Files::isDirectory)
|
||||
.filter(p -> !p.getFileName().toString().startsWith("."))
|
||||
.sorted()
|
||||
.toList();
|
||||
} catch (IOException e) {
|
||||
log.warn("could not scan {} for skill folders to deliver to this opencode member: {}",
|
||||
skillsDir, e.getMessage());
|
||||
return List.of();
|
||||
}
|
||||
if (candidates.isEmpty()) {
|
||||
return List.of();
|
||||
}
|
||||
List<Path> delivered = candidates.stream()
|
||||
.map(dir -> dir.resolve("SKILL.md"))
|
||||
.filter(Files::isRegularFile)
|
||||
.toList();
|
||||
List<String> undeliverable = candidates.stream()
|
||||
.filter(dir -> !Files.isRegularFile(dir.resolve("SKILL.md")))
|
||||
.map(dir -> dir.getFileName().toString())
|
||||
.toList();
|
||||
log.info("skill delivery: {} of {} skill folder(s) under {} reached this opencode member via "
|
||||
+ "instructions[] (opencode does not read .claude/skills/ natively, unlike "
|
||||
+ "Claude Code){}",
|
||||
delivered.size(), candidates.size(), skillsDir,
|
||||
undeliverable.isEmpty() ? "" : "; no SKILL.md, could not be delivered: " + undeliverable);
|
||||
return delivered;
|
||||
}
|
||||
|
||||
/**
|
||||
* True when this profile pins its own OpenAI-compatible endpoint (CB-508) rather than using
|
||||
* whatever provider opencode resolves by default.
|
||||
@@ -305,10 +488,17 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* fresh per-spawn directory under {@link #configRoot}, and return the config file's path for
|
||||
* {@code OPENCODE_CONFIG}. The dir is unique per spawn so concurrent workers never race on it;
|
||||
* it is best-effort cleaned on JVM exit (worker config is disposable — regenerated every spawn).
|
||||
*
|
||||
* @param skillInstructionFiles fleetd #393: absolute {@code SKILL.md} paths from
|
||||
* {@link #skillInstructionFiles(String)}, appended to
|
||||
* {@code instructions[]} so a {@code memberSkills:}-seeded skill
|
||||
* reaches this opencode member the same way the charter and IDE
|
||||
* rules already do.
|
||||
*/
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd) {
|
||||
private Path writeConfig(FleetConfig.Profile cfg, String charterText, String cwd,
|
||||
List<Path> skillInstructionFiles) {
|
||||
try {
|
||||
Path dir = Files.createTempDirectory(configRoot, "fleetd-opencode-");
|
||||
Path dir = Files.createTempDirectory(configParentDir(), "fleetd-opencode-");
|
||||
dir.toFile().deleteOnExit();
|
||||
|
||||
ObjectNode root = JSON.createObjectNode();
|
||||
@@ -334,7 +524,39 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
Files.writeString(charter, charterText);
|
||||
charter.toFile().deleteOnExit();
|
||||
|
||||
root.putArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
// fleetd #393 follow-up: withArray, not putArray. putArray REPLACES whatever node
|
||||
// is already at "instructions"; withArray gets-or-creates. All three writers on
|
||||
// this array now use withArray so that write order stops being load-bearing for
|
||||
// whoever adds a fourth.
|
||||
//
|
||||
// Be precise about what this particular line is worth, because an earlier version
|
||||
// of this comment was wrong and claimed too much. THIS call is the one place where
|
||||
// the two idioms are equivalent, and no test can tell them apart: it runs first,
|
||||
// against a still-empty root, so there is never an existing node for putArray to
|
||||
// replace. Measured on the #393 merge: flipping this one call back to putArray
|
||||
// leaves the whole suite green (1618 tests), and always will. The edit is a
|
||||
// readability and future-proofing change with no test behind it, and that is not a
|
||||
// gap anyone can close.
|
||||
//
|
||||
// The ordering hazard is real, just not here. It is the LATER writers that can
|
||||
// destroy earlier entries. Measured on the same merge: flipping the skills writer
|
||||
// below to putArray deletes this charter entry and fails
|
||||
// OpenCodeLauncherTest.instructionsArrayHoldsCharterThenSkillsThenIdeRulesInOrder;
|
||||
// flipping the IDE-rules writer fails that test and
|
||||
// instructionsArrayHoldsCharterThenIdeRulesInOrder. Deleting this line altogether
|
||||
// fails instructionsArrayHoldsExactlyTheCharterWhenNothingElseWritesToIt — the
|
||||
// assertion that a role contract reaches an opencode member at all, which is the
|
||||
// hole that predates #393 and is what actually let the mutation hide.
|
||||
root.withArray("instructions").add(charter.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
// fleetd #393: each seeded skill's SKILL.md, delivered as a plain instructions[] entry
|
||||
// — the only mechanism opencode has for static guidance text. Unlike Claude Code's
|
||||
// Skill tool, opencode cannot load one of these on demand by name; the content is just
|
||||
// always part of the system prompt from spawn. That is a real difference in HOW the
|
||||
// content reaches the member, not a reason to withhold it.
|
||||
for (Path skillFile : skillInstructionFiles) {
|
||||
root.withArray("instructions").add(skillFile.toAbsolutePath().toString());
|
||||
}
|
||||
|
||||
if (cfg.hasMcp() || cfg.hasIdeMcp()) {
|
||||
@@ -379,6 +601,14 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
// carries operator-supplied values (URL, model id, api key), so escaping must be real.
|
||||
Files.writeString(cfgFile, JSON.writerWithDefaultPrettyPrinter().writeValueAsString(root));
|
||||
cfgFile.toFile().deleteOnExit();
|
||||
if (memberHerdrSocketConfigured()) {
|
||||
// fleetd #219: the same "different OS user" gap fleetd #213 closed for the ZDOTDIR
|
||||
// scrub — share read-only with worktreeGroup rather than leaving the directory under
|
||||
// fleetd's own 0700 java.io.tmpdir, where the member's OS user could not even
|
||||
// traverse it. memberGroup() cannot be null here: configParentDir() above already
|
||||
// refused this spawn if either worktreeRoot or worktreeGroup was missing.
|
||||
EnvAllowListScrub.shareWithGroup(dir, memberGroup());
|
||||
}
|
||||
return cfgFile;
|
||||
} catch (IOException e) {
|
||||
throw new UncheckedIOException(
|
||||
@@ -386,6 +616,57 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #219 site 1: where {@link #writeConfig} creates its per-spawn directory.
|
||||
*
|
||||
* <ul>
|
||||
* <li>{@code memberHerdrSocket} ABSENT (today's only mode): byte-identical to before this
|
||||
* fix — always {@link #configRoot} (defaults to {@code java.io.tmpdir}, fleetd's own
|
||||
* process).</li>
|
||||
* <li>{@code memberHerdrSocket} PRESENT: {@code java.io.tmpdir} is fleetd's own per-user temp
|
||||
* dir (mode {@code 0700} on macOS) — the member pane runs as a DIFFERENT OS user under
|
||||
* this config key and cannot even traverse it, so the directory holding {@code
|
||||
* opencode.json} (which tells the member where the bridge MCP is) and the member charter
|
||||
* would be unreadable to the very process it is written for. The directory instead goes
|
||||
* under {@code worktreeRoot}, shared read-only with {@code worktreeGroup} via {@link
|
||||
* EnvAllowListScrub#shareWithGroup} — the SAME mechanism fleetd #213 built for the ZDOTDIR
|
||||
* scrub, reused here rather than duplicated (see {@link
|
||||
* HerdrPeerLauncher#memberScrubParentDir()}).</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p><b>Unlike the ZDOTDIR scrub, a missing {@code worktreeRoot}/{@code worktreeGroup} here
|
||||
* REFUSES the spawn instead of degrading.</b> The ZDOTDIR scrub is a credential CONTROL: a
|
||||
* degraded control (CB-596's sentinel overlay) is still worth having. This config file is not a
|
||||
* control — it is the ONLY way the member learns where the bridge MCP lives. Writing it
|
||||
* somewhere the member cannot read would not degrade anything; it would spawn a member that
|
||||
* occupies a pane and never becomes deliverable, since {@code fleet_send} waits ~60s on the
|
||||
* readiness gate and then fails with nothing pointing at a temp directory as the cause. Refusing
|
||||
* up front, with a message that names the missing config key, is the honest failure — an
|
||||
* undeliverable member is not a working spawn either way, so nothing is lost by refusing loudly
|
||||
* instead of failing silently later.
|
||||
*
|
||||
* @throws IllegalStateException when {@code memberHerdrSocket} is configured but {@code
|
||||
* worktreeRoot} and/or {@code worktreeGroup} is not
|
||||
*/
|
||||
private Path configParentDir() {
|
||||
if (!memberHerdrSocketConfigured()) {
|
||||
return configRoot;
|
||||
}
|
||||
Path root = memberScrubParentDir();
|
||||
String group = memberGroup();
|
||||
if (root == null || group == null) {
|
||||
throw new IllegalStateException("memberHerdrSocket is configured, so opencode's config "
|
||||
+ "directory (opencode.json + member charter) must be placed where the member's "
|
||||
+ "OS user can read it — worktreeRoot, shared via worktreeGroup — but "
|
||||
+ (root == null ? "worktreeRoot" : "worktreeGroup") + " is not configured. "
|
||||
+ "Refusing to spawn rather than write a config the member cannot read: that "
|
||||
+ "member would occupy a pane and never become deliverable, with nothing "
|
||||
+ "pointing at the real cause. Configure both worktreeRoot and worktreeGroup to "
|
||||
+ "enable opencode member spawns under memberHerdrSocket.");
|
||||
}
|
||||
return root;
|
||||
}
|
||||
|
||||
/**
|
||||
* Declare a custom OpenAI-compatible provider so the worker talks to a pinned endpoint (a local
|
||||
* vLLM, say) instead of opencode's default gateway (CB-508).
|
||||
@@ -494,11 +775,48 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
return afterScheme.contains("/") ? trimmed : trimmed + "/v1";
|
||||
}
|
||||
|
||||
/** One WARN per launcher instance for the fleetd #219 site-2 discovery-unavailable gap. */
|
||||
private final AtomicBoolean discoveryUnavailableWarned =
|
||||
new AtomicBoolean();
|
||||
|
||||
/**
|
||||
* fleetd #267: one WARN per PROFILE (not per launcher instance — several profiles can each hit
|
||||
* this gap independently) for the model-mismatch check (fleetd #175) never getting to run
|
||||
* because the spawn was not given a fleetd-provisioned worktree (fleetd #249). Profile names
|
||||
* accumulate here for the life of this launcher instance and are never removed — the same
|
||||
* one-shot treatment {@link #discoveryUnavailableWarned} already gets, just keyed per profile
|
||||
* instead of globally.
|
||||
*/
|
||||
private final Set<String> modelCheckSkippedWarned = ConcurrentHashMap.newKeySet();
|
||||
|
||||
/** Add lazy on-disk session discovery to the base handle. */
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
String cwd = effectiveCwd(req);
|
||||
// fleetd #249: refuse rather than silently resume into unverifiable territory. opencode's
|
||||
// `-s <id>` flag itself resumes precisely — the resolved id is what fails, not the resume —
|
||||
// but resolvedSessionId() below can never confirm (or later re-report) this handle's own
|
||||
// identity without a fleetd-provisioned worktree (isProvisionedWorktree(cwd)), because the
|
||||
// directory is shared and sessionIdForDirectory's "most recently updated row" heuristic can
|
||||
// pick a sibling's session. Refusing here, before anything spawns, beats letting the member
|
||||
// start and only then discovering fleetd can never again verify who it actually is.
|
||||
if (req.resumeSessionId() != null && !req.resumeSessionId().isBlank()
|
||||
&& !isProvisionedWorktree(cwd)) {
|
||||
throw new IllegalArgumentException("resumeSessionId requires a fleetd-provisioned "
|
||||
+ "worktree for an opencode profile — without one, this member's cwd is shared "
|
||||
+ "with other sessions, so fleetd can never reliably confirm (now or later) which "
|
||||
+ "conversation it is actually running (fleetd #249). Pass fleet_spawn{worktree:"
|
||||
+ "<ticket-slug>} to resume this member.");
|
||||
}
|
||||
PeerHandle inner = super.spawn(req);
|
||||
return new SessionAwareHandle(inner, discovery, effectiveCwd(req));
|
||||
// fleetd #175: the same profile config buildLaunch resolved for this spawn (requireProfile
|
||||
// is deterministic on req.profileName(), so re-resolving here costs a map lookup, not a
|
||||
// second decision) — SessionAwareHandle needs cfg.model() to know what THIS session should
|
||||
// be running.
|
||||
FleetConfig.Profile cfg = requireProfile(req.profileName());
|
||||
return new SessionAwareHandle(inner, discovery, cwd, cfg,
|
||||
this::memberHerdrSocketConfigured, discoveryUnavailableWarned,
|
||||
modelCheckSkippedWarned, exhaustionSink);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -508,16 +826,63 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
* are untouched — only the opencode-specific identity answer is added. {@code sessionName()}
|
||||
* stays null: opencode has no display-name seam, so the logical name lives only in the bridge's
|
||||
* roster (see the SESSION_NAME capability).
|
||||
*
|
||||
* <p>fleetd #175: {@link #agentSessionId()} is also where the model-mismatch check lives (see
|
||||
* {@link #checkModelMatch()}) — it is the one method the real late-resolve path
|
||||
* ({@code SessionManager.resolveAgentSessionId}, via {@code get}/{@code rosterResolved}/
|
||||
* {@code release}) actually calls, and only while the session id is still unknown. Putting the
|
||||
* check anywhere else risks repeating PR #203's mistake: a check that runs before opencode has
|
||||
* written the row it needs, and so never fires.
|
||||
*/
|
||||
private static final class SessionAwareHandle implements PeerHandle {
|
||||
private final PeerHandle delegate;
|
||||
private final OpenCodeSessionDiscovery discovery;
|
||||
private final String cwd;
|
||||
private final FleetConfig.Profile cfg;
|
||||
private final BooleanSupplier discoveryUnavailable;
|
||||
private final AtomicBoolean discoveryUnavailableWarned;
|
||||
private final Set<String> modelCheckSkippedWarned;
|
||||
private final ExhaustionSink exhaustionSink;
|
||||
/** CAS'd true the first (and only) time a model mismatch is reported for this handle. */
|
||||
private final AtomicBoolean modelMismatchReported = new AtomicBoolean();
|
||||
/**
|
||||
* The session id, once {@link OpenCodeSessionDiscovery#sessionIdForDirectory} first
|
||||
* resolves a non-null answer for this handle (fleetd #234). Sticky on purpose: {@code
|
||||
* directory} is a shared-cwd heuristic (see {@link OpenCodeSessionDiscovery}'s class
|
||||
* javadoc) that can start returning a DIFFERENT row once another session shares the same
|
||||
* directory and writes a newer one — re-deriving it on every call would let this handle's
|
||||
* identity silently drift to a sibling's session. Once resolved, this IS the answer, and
|
||||
* {@link #checkModelMatch} reads only the row this id names, never "whatever is newest in
|
||||
* the directory right now."
|
||||
*/
|
||||
private final AtomicReference<String> resolvedSessionId = new AtomicReference<>();
|
||||
/**
|
||||
* fleetd #249: whether {@link #cwd} is a fleetd-provisioned git worktree
|
||||
* ({@link HerdrPeerLauncher#isProvisionedWorktree}), computed once at spawn time since
|
||||
* {@code cwd} never changes for this handle. When {@code false} the directory is shared
|
||||
* with other sessions (the default no-worktree spawn inherits the lead's own cwd), so
|
||||
* {@link OpenCodeSessionDiscovery#sessionIdForDirectory}'s "most recently updated row for
|
||||
* this directory" heuristic can and does pick another session's row — see that class's
|
||||
* javadoc. {@link #agentSessionId()} refuses to guess in that case: it reports absent
|
||||
* rather than a possibly-foreign id.
|
||||
*/
|
||||
private final boolean worktreeProvisioned;
|
||||
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd) {
|
||||
SessionAwareHandle(PeerHandle delegate, OpenCodeSessionDiscovery discovery, String cwd,
|
||||
FleetConfig.Profile cfg,
|
||||
BooleanSupplier discoveryUnavailable,
|
||||
AtomicBoolean discoveryUnavailableWarned,
|
||||
Set<String> modelCheckSkippedWarned,
|
||||
ExhaustionSink exhaustionSink) {
|
||||
this.delegate = delegate;
|
||||
this.discovery = discovery;
|
||||
this.cwd = cwd;
|
||||
this.cfg = cfg;
|
||||
this.discoveryUnavailable = discoveryUnavailable;
|
||||
this.discoveryUnavailableWarned = discoveryUnavailableWarned;
|
||||
this.modelCheckSkippedWarned = modelCheckSkippedWarned;
|
||||
this.exhaustionSink = exhaustionSink;
|
||||
this.worktreeProvisioned = isProvisionedWorktree(cwd);
|
||||
}
|
||||
|
||||
@Override
|
||||
@@ -542,11 +907,142 @@ public final class OpenCodeLauncher extends HerdrPeerLauncher {
|
||||
|
||||
@Override
|
||||
public String agentSessionId() {
|
||||
// fleetd #219 site 2: under memberHerdrSocket the member pane runs as a different OS
|
||||
// user, so opencode.db lives under THAT user's $HOME, not the one discoveryRoot was
|
||||
// built from (see OpenCodeLauncher#defaultDiscoveryRoot's javadoc for the full
|
||||
// reasoning). Scanning fleetd's own $HOME under that config would only ever find "no
|
||||
// row" and read as "resume unsupported" — declare it unavailable instead, once, loudly.
|
||||
// Checked before the fleetd #249 worktree gate below: this OS-user mismatch makes
|
||||
// discovery unusable regardless of whether cwd happens to be a provisioned worktree, so
|
||||
// it earns the one-time WARN either way.
|
||||
if (discoveryUnavailable.getAsBoolean()) {
|
||||
if (discoveryUnavailableWarned.compareAndSet(false, true)) {
|
||||
log.warn("opencode session discovery unavailable: memberHerdrSocket is "
|
||||
+ "configured, so opencode's on-disk session database lives under the "
|
||||
+ "MEMBER's own $HOME, not fleetd's ({}) — agentSessionId will stay null "
|
||||
+ "for every opencode member under this config, and SESSION_RESUME "
|
||||
+ "cannot be honored (fleetd #209/#219).",
|
||||
System.getProperty("user.home"));
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #249: cwd is shared with other sessions unless fleetd itself provisioned this
|
||||
// worktree, and sessionIdForDirectory's directory-keyed heuristic cannot tell this
|
||||
// member's row apart from a sibling's in that case (measured: a three-day-old row from
|
||||
// a different profile). Refuse to guess — absent is the honest answer, and it is what
|
||||
// this codebase already returns elsewhere for absent evidence (fleetd #175's UNKNOWN).
|
||||
// This IS the ordinary, expected shape of the large majority of spawns (no worktree
|
||||
// requested), not a configuration gap — but fleetd #267 found that same shape silently
|
||||
// switches off the fleetd #175 model-mismatch check for those spawns too, since
|
||||
// checkModelMatch's only call site is right below this gate. The check cannot be moved
|
||||
// off agentSessionId()'s resolved id: the id is the only safe way to key
|
||||
// actualModelForSessionId to THIS session's own row rather than "whatever is newest in
|
||||
// the shared directory" (fleetd #234) — re-deriving a second, independent answer via
|
||||
// `directory` here would reintroduce exactly the false-positive risk #234 fixed (a
|
||||
// sibling's differently-configured model looking like THIS profile's mismatch). So the
|
||||
// model genuinely is unknowable without a provisioned worktree, and unlike the silence
|
||||
// this branch used to keep, that gap now gets the same one-time, per-profile WARN
|
||||
// treatment discoveryUnavailable already gets above — but keyed by profile, since
|
||||
// several profiles can each hit this independently.
|
||||
if (!worktreeProvisioned) {
|
||||
if (cfg.model() != null && !cfg.model().isBlank()
|
||||
&& modelCheckSkippedWarned.add(cfg.profile())) {
|
||||
log.warn("opencode model-mismatch check (fleetd #175) cannot run for profile "
|
||||
+ "'{}': it was spawned without a fleetd-provisioned worktree (fleetd "
|
||||
+ "#249), so its cwd may be shared with other sessions and the actual "
|
||||
+ "model it is running cannot be safely told apart from a sibling's — "
|
||||
+ "spawn with worktree:true to enable the check for this profile.",
|
||||
cfg.profile());
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// fleetd #234: once resolved, stay resolved. Re-deriving from `directory` on every call
|
||||
// would let this handle's identity drift to a sibling session that later shares the
|
||||
// same cwd and writes a newer row — see resolvedSessionId's javadoc.
|
||||
String cached = resolvedSessionId.get();
|
||||
if (cached != null) {
|
||||
return cached;
|
||||
}
|
||||
// Lazy + retried, never a spawn-time blocker: opencode writes the session record only
|
||||
// when the session is first persisted, so null here is the correct interim answer and
|
||||
// the caller re-calls later (each call re-scans, picking up a record that has since
|
||||
// appeared).
|
||||
return discovery.sessionIdForDirectory(cwd);
|
||||
String id = discovery.sessionIdForDirectory(cwd);
|
||||
if (id != null) {
|
||||
resolvedSessionId.compareAndSet(null, id);
|
||||
}
|
||||
// fleetd #175: check on the SAME tick — while the caller (SessionManager's late-resolve
|
||||
// step) is still re-polling because the id is unknown, the row this id came from (once
|
||||
// it exists) is exactly the row that also carries the actual model. Once id resolves,
|
||||
// the caller stops calling agentSessionId() for this session, so this is naturally a
|
||||
// once-only check that happens right when the row first appears.
|
||||
checkModelMatch(id);
|
||||
return id;
|
||||
}
|
||||
|
||||
/**
|
||||
* Verify the live opencode session is running the model {@link #cfg} requested (fleetd
|
||||
* #175) and, on a real mismatch, log an ERROR and quarantine through {@link
|
||||
* #exhaustionSink}. A no-op when there is nothing to compare against — no model configured,
|
||||
* already reported once for this handle, {@code sessionId} itself is not resolved yet
|
||||
* (fleetd #234: absent evidence, not a mismatch), or the actual model is still UNKNOWN (no
|
||||
* row yet, unreadable database, or unparseable evidence). UNKNOWN must never be treated as
|
||||
* a mismatch: that is the single most important safety rule here — a false positive would
|
||||
* quarantine a perfectly working profile's credential.
|
||||
*
|
||||
* @param sessionId the id {@link #agentSessionId()} just resolved (or had cached) for THIS
|
||||
* handle — the model is read back for this exact session (fleetd #234's
|
||||
* {@link OpenCodeSessionDiscovery#actualModelForSessionId}), never
|
||||
* re-derived from {@code directory}
|
||||
*/
|
||||
private void checkModelMatch(String sessionId) {
|
||||
if (modelMismatchReported.get() || cfg.model() == null || cfg.model().isBlank()) {
|
||||
return;
|
||||
}
|
||||
if (sessionId == null || sessionId.isBlank()) {
|
||||
return; // id not resolved yet — UNKNOWN, never a mismatch (fleetd #175's rule)
|
||||
}
|
||||
OpenCodeSessionDiscovery.ActualModel actual = discovery.actualModelForSessionId(sessionId);
|
||||
if (actual == null) {
|
||||
return; // UNKNOWN evidence — never a mismatch
|
||||
}
|
||||
String[] requestedParts = splitProviderModel(cfg.model());
|
||||
String requestedProvider = requestedParts == null ? null : requestedParts[0];
|
||||
String requestedId = requestedParts == null ? cfg.model() : requestedParts[1];
|
||||
boolean idMatches = requestedId.equals(actual.id());
|
||||
// Compare the provider ONLY when BOTH sides have one. requestedProvider == null covers
|
||||
// a bare profile model with no "/" — the profile never asked for a specific provider.
|
||||
// actual.provider() == null covers opencode's model JSON having an id but no providerID
|
||||
// (a real shape parseModel accepts) — that is missing evidence, not a contradiction, and
|
||||
// acceptance rule 4 says missing evidence is UNKNOWN, never a mismatch. Narrowing the
|
||||
// provider comparison this way keeps the id comparison (the part that actually caught the
|
||||
// xf bug) fully intact — a genuine id mismatch is still caught either way (fleetd #175
|
||||
// review round 2).
|
||||
boolean providerMatches = requestedProvider == null || actual.provider() == null
|
||||
|| requestedProvider.equals(actual.provider());
|
||||
if (idMatches && providerMatches) {
|
||||
return;
|
||||
}
|
||||
if (!modelMismatchReported.compareAndSet(false, true)) {
|
||||
return; // another thread already reported this exact mismatch
|
||||
}
|
||||
String actualDisplay = actual.provider() == null
|
||||
? actual.id() : actual.provider() + "/" + actual.id();
|
||||
log.error("opencode profile '{}' requested model '{}' but the live session is actually "
|
||||
+ "running '{}' — opencode does not fail on an unknown -m, it silently "
|
||||
+ "falls back to a default model, which may be a PAID credential "
|
||||
+ "(fleetd #175); quarantining this profile's credential",
|
||||
cfg.profile(), cfg.model(), actualDisplay);
|
||||
// fleetd #234: pass our OWN profile name too. This check fires from agentSessionId(),
|
||||
// called during SessionManager.acquire() BEFORE this session is registered in
|
||||
// sessions.roster() — a roster-only sink (Fleetd's target -> session -> profile lookup)
|
||||
// finds nothing at this point and silently no-ops (defect 2). We already know exactly
|
||||
// which profile to quarantine without the roster; the sink is passed it explicitly.
|
||||
exhaustionSink.onExhausted(delegate.terminalId(),
|
||||
"opencode model mismatch: profile '" + cfg.profile() + "' requested '"
|
||||
+ cfg.model() + "' but the live session is running '" + actualDisplay
|
||||
+ "' (fleetd #175)",
|
||||
cfg.profile());
|
||||
}
|
||||
|
||||
@Override
|
||||
|
||||
@@ -2,57 +2,117 @@ package dev.ltms.fleet.member;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
import org.sqlite.SQLiteConfig;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.stream.Stream;
|
||||
import java.sql.Connection;
|
||||
import java.sql.PreparedStatement;
|
||||
import java.sql.ResultSet;
|
||||
import java.sql.SQLException;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
|
||||
/**
|
||||
* Resolves the opencode session id for a fleetd worker from opencode's on-disk storage — the
|
||||
* only place this adapter touches opencode's private layout, and deliberately the <em>only</em>
|
||||
* class that does.
|
||||
*
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not a
|
||||
* stable contract: opencode writes one JSON file per session under
|
||||
* {@code <storageRoot>/session/<projectID>/<ses_*.json>}, and each record carries a
|
||||
* {@code "version"} field (e.g. {@code "1.1.31"}), so the exact directory shape, file naming, and
|
||||
* field names can move between opencode releases. opencode also ships a headless HTTP server that
|
||||
* may supersede file scanning entirely. Everything this adapter knows about that private storage —
|
||||
* its shape, naming, and field names — lives here, so a layout change, or a switch to the HTTP
|
||||
* server, changes exactly one class and nothing in {@link OpenCodeLauncher}.
|
||||
* <p><strong>Why this is isolated behind one seam.</strong> The layout is version-coupled and not
|
||||
* a stable contract: opencode persists its session state in a SQLite database at
|
||||
* {@code <storageRoot>/opencode.db} (a {@code session} table, one row per session, keyed by id and
|
||||
* carrying a {@code directory} column). That schema can move between opencode releases exactly
|
||||
* like the JSON-file layout it replaced did (opencode migrated off a one-JSON-file-per-session
|
||||
* tree under {@code <storageRoot>/storage/session/<projectID>/ses_*.json} in January 2026 — that
|
||||
* tree is now a frozen migration artefact nothing writes, which is why this class no longer reads
|
||||
* it). opencode also ships a headless HTTP server that may supersede both of these entirely.
|
||||
* Everything this adapter knows about that private storage — its shape and column names — lives
|
||||
* here, so a layout change, or a switch to the HTTP server, changes exactly one class and nothing
|
||||
* in {@link OpenCodeLauncher}.
|
||||
*
|
||||
* <p>The determinism that makes this useful is structural, not a guess: every fleetd worker runs
|
||||
* in its own unique git worktree, so the record's {@code directory} (its project root) equals the
|
||||
* worker's cwd identifies <em>its</em> session unambiguously. We match on {@code directory} rather
|
||||
* than diffing {@code opencode session list} before/after — that races under concurrent spawns, and
|
||||
* the CLI listing does not even show the directory.
|
||||
* <p><strong>The {@code directory} match is a heuristic, not an identity — fleetd #234.</strong> A
|
||||
* worktree is opt-in: {@code fleet_spawn} only provisions one when the caller passes {@code
|
||||
* worktree:}; the default spawn inherits the lead's own cwd, which every other worker spawned the
|
||||
* same way (and every past session ever run there) shares. {@code directory} therefore does
|
||||
* <em>not</em> identify a session unambiguously in general — only in the special case of a fresh,
|
||||
* unique worktree does "most recently updated row for this directory" reliably mean "this worker's
|
||||
* own row." {@link #sessionIdForDirectory} still has to use this heuristic (the id has to come from
|
||||
* somewhere, and nothing else is available at this layer — see that method's javadoc), but a caller
|
||||
* that already holds a resolved id must never re-derive evidence about that same session via
|
||||
* {@code directory} again; see {@link #actualModelForSessionId}, which looks up by {@code id}
|
||||
* instead for exactly this reason. We match on {@code directory} rather than diffing
|
||||
* {@code opencode session list} before/after — that races under concurrent spawns, and the CLI
|
||||
* listing does not even show the directory.
|
||||
*
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable storage root, a record that
|
||||
* fails to parse, or a directory with no record yet all yield {@code null}, and the caller (the
|
||||
* session handle) treats that as "identity not resolved yet" and retries later.
|
||||
* <p>All reads are best-effort and never throw: a missing or unreadable database, a query that
|
||||
* fails, or a directory with no row yet all yield {@code null}, and the caller (the session
|
||||
* handle) treats that as "identity not resolved yet" and retries later. The database is opened
|
||||
* read-only and never written to: opencode itself may be running and writing it concurrently (WAL
|
||||
* mode), and this class must never disturb that.
|
||||
*/
|
||||
final class OpenCodeSessionDiscovery {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(OpenCodeSessionDiscovery.class);
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
|
||||
/**
|
||||
* The model opencode actually ran a session on, parsed from the {@code session.model} JSON
|
||||
* column (fleetd #175). {@code id} is never null/blank on a non-null {@code ActualModel} —
|
||||
* {@link #parseModel} returns {@code null} instead when {@code id} cannot be determined, so a
|
||||
* caller only ever sees a fully-known record or {@code null} (UNKNOWN). {@code provider} may
|
||||
* still be {@code null} on its own when the profile that requested the session named no
|
||||
* provider prefix, or opencode's JSON omitted {@code providerID}.
|
||||
*/
|
||||
record ActualModel(String provider, String id) {
|
||||
}
|
||||
|
||||
private final Path storageRoot; // e.g. ~/.local/share/opencode (injectable for tests)
|
||||
private final ObjectMapper json;
|
||||
private final Path databasePath;
|
||||
private final AtomicBoolean warnedMissingDatabase = new AtomicBoolean(false);
|
||||
|
||||
OpenCodeSessionDiscovery(Path storageRoot) {
|
||||
this.storageRoot = storageRoot;
|
||||
this.json = new ObjectMapper();
|
||||
this.databasePath = storageRoot.resolve("opencode.db");
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose record references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no record matches yet. When several records share the directory — e.g.
|
||||
* repeated spawns into the same worktree — the <em>most recently modified</em> one wins: it is
|
||||
* the session the pane most likely corresponds to.
|
||||
* A connection to {@link #databasePath} opened with SQLite's {@code SQLITE_OPEN_READONLY}
|
||||
* flag: it never creates the file, never writes, and never touches WAL or journal mode.
|
||||
* opencode may be running and writing this database concurrently, and this class must never
|
||||
* disturb it.
|
||||
*
|
||||
* <p>Never throws: a missing {@code storageRoot}, an unreadable/malformed record, or a
|
||||
* directory that has not been persisted yet all resolve to {@code null} rather than failing a
|
||||
* spawn. A fleetd worker's session record is written lazily (when the session is first
|
||||
* persisted), so {@code null} here is the normal answer right after the pane is ready, and the
|
||||
* caller retries later.
|
||||
* <p>Package-private so a test can hold the connection and prove it refuses a write. That is
|
||||
* the only way to pin this property: making the file unwritable does <em>not</em> work,
|
||||
* because SQLite silently downgrades a read-write open of an unwritable file to read-only, so
|
||||
* such a test passes whether or not the flag is set.
|
||||
*/
|
||||
Connection openReadOnly() throws SQLException {
|
||||
SQLiteConfig config = new SQLiteConfig();
|
||||
config.setReadOnly(true);
|
||||
return config.createConnection("jdbc:sqlite:" + databasePath);
|
||||
}
|
||||
|
||||
/**
|
||||
* The opencode session id whose row references {@code directory} (the worker's cwd), or
|
||||
* {@code null} when no row matches yet. When several rows share the directory — e.g. repeated
|
||||
* spawns into the same worktree, OR several workers sharing one cwd because none of them was
|
||||
* given a worktree (fleetd #234) — the row with the highest {@code time_updated} wins: it is
|
||||
* the session the pane most likely corresponds to. That "most likely" is a real caveat, not a
|
||||
* formality: when the directory is shared, this can and does pick another session's row (see
|
||||
* the class javadoc). This is the one place fleetd resolves an opencode session id at all —
|
||||
* nothing else is available at this layer to disambiguate further (no {@code opencode session
|
||||
* list} entry names the directory, and diffing before/after races under concurrent spawns) — so
|
||||
* the heuristic stays here unchanged. What must never happen is a SECOND, independent piece of
|
||||
* evidence about the same session being re-derived via {@code directory} once an id has already
|
||||
* come out of this method; see {@link #actualModelForSessionId}.
|
||||
*
|
||||
* <p>Never throws: a missing {@code opencode.db}, a locked/unreadable database, a query
|
||||
* failure, or a directory that has not been persisted yet all resolve to {@code null} rather
|
||||
* than failing a spawn. A fleetd worker's session row is written lazily (when the session is
|
||||
* first persisted), so {@code null} here is the normal answer right after the pane is ready,
|
||||
* and the caller retries later.
|
||||
*
|
||||
* @param directory the worker's cwd, as resolved for this spawn
|
||||
* @return the matching session id, or {@code null} if none is known yet
|
||||
@@ -61,62 +121,112 @@ final class OpenCodeSessionDiscovery {
|
||||
if (directory == null || directory.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
Path sessionRoot = storageRoot.resolve("session");
|
||||
if (!Files.isDirectory(sessionRoot)) {
|
||||
if (!Files.isRegularFile(databasePath)) {
|
||||
if (warnedMissingDatabase.compareAndSet(false, true)) {
|
||||
log.warn("opencode session database not found at {} — opencode's on-disk layout "
|
||||
+ "may have moved again; session discovery will keep returning null",
|
||||
databasePath);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
String best = null;
|
||||
long bestMtime = Long.MIN_VALUE;
|
||||
try (Stream<Path> projectDirs = Files.list(sessionRoot)) {
|
||||
for (Path projectDir : projectDirs.filter(Files::isDirectory).toList()) {
|
||||
try (Stream<Path> records = Files.list(projectDir)) {
|
||||
for (Path record : records.toList()) {
|
||||
String id = matchId(record, directory);
|
||||
if (id == null) {
|
||||
continue;
|
||||
}
|
||||
long mtime = lastModifiedEpochMillis(record);
|
||||
if (mtime > bestMtime) {
|
||||
bestMtime = mtime;
|
||||
best = id;
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// one project dir unreadable — skip it; another may still match
|
||||
String sql = "SELECT id FROM session WHERE directory = ? ORDER BY time_updated DESC LIMIT 1";
|
||||
try (Connection connection = openReadOnly();
|
||||
PreparedStatement statement = connection.prepareStatement(sql)) {
|
||||
statement.setString(1, directory);
|
||||
try (ResultSet rows = statement.executeQuery()) {
|
||||
if (rows.next()) {
|
||||
return rows.getString("id");
|
||||
}
|
||||
}
|
||||
} catch (IOException ignored) {
|
||||
// storage root vanished or became unreadable — "no session known yet"
|
||||
} catch (SQLException e) {
|
||||
// Locked, corrupt, or otherwise unreadable — never fatal to a spawn. Not the
|
||||
// "database moved" signal (the file exists), so this stays below WARN.
|
||||
log.debug("opencode session database unreadable at {}: {}", databasePath, e.toString());
|
||||
return null;
|
||||
}
|
||||
return best;
|
||||
log.debug("no opencode session row for directory (root={}, directory={})",
|
||||
storageRoot, directory);
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* The record's session id when it references {@code directory}, else {@code null}. A record
|
||||
* that is not JSON, lacks {@code id}/{@code directory}, or points at a different directory is
|
||||
* simply not our session; a malformed one is skipped, never fatal.
|
||||
* The model opencode actually ran the session {@code sessionId} on (fleetd #175/#234), read by
|
||||
* primary-key lookup — the ONE row that id names, and no other. Deliberately keyed on
|
||||
* {@code id} rather than {@code directory}: two independent {@code WHERE directory = ? ORDER BY
|
||||
* time_updated DESC LIMIT 1} queries (one for the id, one for the model) can each pick a
|
||||
* DIFFERENT row once more than one session shares a directory (fleetd #234 — the default
|
||||
* no-worktree spawn shares the lead's cwd with every other worker and every past session ever
|
||||
* run there), silently comparing a profile's requested model against a session that is not even
|
||||
* the one whose id was returned. Keying on {@code id} instead makes that impossible: the model
|
||||
* read back is always the SAME session {@link #sessionIdForDirectory} (or a cached copy of its
|
||||
* answer) already resolved.
|
||||
*
|
||||
* <p>Kept as its own query and its own connection, independent from {@link
|
||||
* #sessionIdForDirectory}: a database whose schema predates the {@code model} column (or any
|
||||
* other read failure on this column alone) can never take id resolution down with it. That
|
||||
* would be a regression of the id-resolution feature #209 shipped; this method degrades on its
|
||||
* own.
|
||||
*
|
||||
* <p>Never throws, and every failure mode — no matching row, a missing/unreadable database, a
|
||||
* missing {@code model} column, a null/blank {@code model} value, or JSON that does not parse
|
||||
* into {@code {"id": "...", "providerID": "..."}} with a non-blank {@code id} — resolves to
|
||||
* {@code null}. That is UNKNOWN evidence, not a mismatch signal: the caller must never
|
||||
* quarantine a profile on the strength of a {@code null} here. A blank/null {@code sessionId}
|
||||
* (the id is not resolved yet) is UNKNOWN too, for the same reason — never call this with one.
|
||||
*
|
||||
* @param sessionId the session id already resolved by {@link #sessionIdForDirectory} for this
|
||||
* spawn — never re-derived from {@code directory} here
|
||||
* @return the actual model, or {@code null} when unknown
|
||||
*/
|
||||
private String matchId(Path record, String directory) {
|
||||
ActualModel actualModelForSessionId(String sessionId) {
|
||||
if (sessionId == null || sessionId.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
if (!Files.isRegularFile(databasePath)) {
|
||||
// sessionIdForDirectory already WARNs once (shared warnedMissingDatabase) for this
|
||||
// exact condition — do not double-log it here.
|
||||
return null;
|
||||
}
|
||||
String sql = "SELECT model FROM session WHERE id = ?";
|
||||
try (Connection connection = openReadOnly();
|
||||
PreparedStatement statement = connection.prepareStatement(sql)) {
|
||||
statement.setString(1, sessionId);
|
||||
try (ResultSet rows = statement.executeQuery()) {
|
||||
if (rows.next()) {
|
||||
return parseModel(rows.getString("model"));
|
||||
}
|
||||
}
|
||||
} catch (SQLException e) {
|
||||
// Locked/corrupt database, or a `model` column this schema version does not have —
|
||||
// never fatal, and never a mismatch signal. See the class doc above.
|
||||
log.debug("opencode session model unreadable at {}: {}", databasePath, e.toString());
|
||||
return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse opencode's {@code model} column — {@code {"id":"...","providerID":"..."}} — into an
|
||||
* {@link ActualModel}, or {@code null} when {@code json} is null/blank, is not valid JSON, or
|
||||
* parses without a non-blank {@code id}. {@code providerID} may be absent; that alone does not
|
||||
* make the record unknown, since a caller comparing against a profile with no provider prefix
|
||||
* never looks at it.
|
||||
*/
|
||||
private static ActualModel parseModel(String json) {
|
||||
if (json == null || json.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
JsonNode node = json.readTree(record.toFile());
|
||||
JsonNode id = node == null ? null : node.get("id");
|
||||
JsonNode dir = node == null ? null : node.get("directory");
|
||||
if (id == null || dir == null || !directory.equals(dir.asText())) {
|
||||
JsonNode node = MAPPER.readTree(json);
|
||||
String id = node.path("id").asText(null);
|
||||
if (id == null || id.isBlank()) {
|
||||
return null;
|
||||
}
|
||||
return id.asText();
|
||||
String provider = node.path("providerID").asText(null);
|
||||
return new ActualModel(provider, id);
|
||||
} catch (IOException e) {
|
||||
log.debug("opencode session model JSON unparseable: {}", e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** The record's last-modified epoch ms, or {@code Long.MIN_VALUE} if unreadable (never wins). */
|
||||
private static long lastModifiedEpochMillis(Path record) {
|
||||
try {
|
||||
return Files.getLastModifiedTime(record).toMillis();
|
||||
} catch (IOException e) {
|
||||
return Long.MIN_VALUE;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -56,10 +56,13 @@ public final class FleetMetrics {
|
||||
m.describe(REPLIES, "counter",
|
||||
"Worker replies by delivery path (rendezvous=resolved an open send, inbox=stranded and held).");
|
||||
m.describe(PUSH_NUDGES, "counter",
|
||||
"CB-307 push-loop nudges to the primary (delivered|exhausted).");
|
||||
"CB-307 push-loop nudges to the primary (sent|exhausted). fleetd #365: \"sent\" means "
|
||||
+ "the herdr paste-and-submit call succeeded, not that the pane read it — this "
|
||||
+ "layer has no read-receipt concept.");
|
||||
m.describe(HEARTBEAT_NUDGES, "counter",
|
||||
"CB-551 idle-lead heartbeat nudges (delivered|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down.");
|
||||
"CB-551 idle-lead heartbeat nudges (sent|failed|exhausted). Quiet-cap exhaustion "
|
||||
+ "means the lead idled with nothing pending and was told to stand down. "
|
||||
+ "fleetd #365: \"sent\" means the herdr call succeeded, not that the lead read it.");
|
||||
m.describe(SPAWNS, "counter",
|
||||
"Worker spawn attempts by peer kind and outcome (ready|timeout|guard_rejected).");
|
||||
m.describe(HERDR_CALLS, "counter",
|
||||
|
||||
@@ -8,6 +8,7 @@ import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import com.rabbitmq.client.Return;
|
||||
import com.rabbitmq.client.impl.DefaultExceptionHandler;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -22,6 +23,7 @@ import java.util.concurrent.ConcurrentSkipListMap;
|
||||
import java.util.concurrent.ExecutionException;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.TimeoutException;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
/**
|
||||
* AMQP-backed {@link ReplyInbox} (CB-307 Stage 2): genuine cross-restart durability behind the same
|
||||
@@ -29,8 +31,9 @@ import java.util.concurrent.TimeoutException;
|
||||
*
|
||||
* <p><strong>Mapping — consume-and-hold with deferred manual ack.</strong> Each target has a durable
|
||||
* queue {@code agent.<target>.inbox}. The gateway that owns the target starts a manual-ack consumer
|
||||
* ({@link #own}) that pulls persistent messages off that queue into an in-memory <em>held</em> map
|
||||
* (keyed by {@code msgId}) but does <em>not</em> ack them. {@link #peek} returns that snapshot;
|
||||
* ({@link #own}) that pulls persistent messages, up to its prefetch window, off that queue into an
|
||||
* in-memory <em>held</em> map (keyed by {@code msgId}) but does <em>not</em> ack them.
|
||||
* {@link #peek} returns that snapshot;
|
||||
* {@link #ack} acks the broker delivery-tag and drops the entry. Because messages stay unacked until
|
||||
* the owning gateway actually drains them, a crash (or a {@code java -jar} bounce) before caller-ack
|
||||
* leaves them on the broker — it redelivers on reconnect. That is the durability the in-memory
|
||||
@@ -92,8 +95,23 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
private final Channel channel;
|
||||
/** All channel operations (publish/declare/ack/cancel) serialize on this — a Channel is not thread-safe. */
|
||||
private final Object channelLock = new Object();
|
||||
/** target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself. */
|
||||
/**
|
||||
* target → (msgId → held delivery). Per-target map is guarded by synchronizing on itself.
|
||||
*
|
||||
* <p><strong>CB-318 tombstone.</strong> The value {@link #RELEASED} is a reserved sentinel: it
|
||||
* marks a target whose {@link #release} has already run, so {@link #deliverCallback} can tell a
|
||||
* delivery landing after release() apart from a fresh target it has never seen. See both methods'
|
||||
* javadoc for why a plain {@code held.remove(target)} is not enough.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, LinkedHashMap<String, Held>> held = new ConcurrentHashMap<>();
|
||||
|
||||
/**
|
||||
* CB-318 sentinel stored in {@link #held} for a target whose {@link #release} has already run.
|
||||
* Never mutated — every read site compares it by reference ({@code ==}) before touching it as a
|
||||
* map, because it is a single object shared across every released target and calling a mutator on
|
||||
* it would corrupt state for all of them.
|
||||
*/
|
||||
private static final LinkedHashMap<String, Held> RELEASED = new LinkedHashMap<>();
|
||||
/** Targets whose queue is declared and consumer is running, mapped to their broker consumer tag. */
|
||||
private final ConcurrentHashMap<String, String> consumerTags = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -136,17 +154,22 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
/** As {@link #open(String)}, with an explicit consumer prefetch (CB-527: caps the held backlog per target). */
|
||||
public static AmqpReplyInbox open(String uri, int prefetch) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new AmqpReplyInbox(factory.newConnection("fleetd-reply-inbox"), prefetch);
|
||||
return new AmqpReplyInbox(connectionFactory(uri).newConnection(AmqpConnectionFailureLogger.REPLY_INBOX), prefetch);
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
static ConnectionFactory connectionFactory(String uri) throws Exception {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
factory.setExceptionHandler(new AmqpConnectionFailureLogger(AmqpConnectionFailureLogger.REPLY_INBOX, log));
|
||||
return factory;
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for the contract test). */
|
||||
AmqpReplyInbox(Connection connection) {
|
||||
this(connection, DEFAULT_PREFETCH);
|
||||
@@ -200,6 +223,12 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
String tag = channel.basicConsume(queue, false, deliverCallback(target), _ -> { });
|
||||
consumerTags.put(target, tag);
|
||||
// CB-318: drop a stale RELEASED tombstone from a prior ownership of this same target
|
||||
// string, so a delivery under this fresh consumer is held normally instead of being
|
||||
// nacked forever by deliverCallback's RELEASED check. Safe to do here, still under
|
||||
// channelLock: no delivery for the consumer tag just registered above can reach
|
||||
// deliverCallback before this basicConsume call returns.
|
||||
held.remove(target, RELEASED);
|
||||
log.debug("AMQP inbox owns queue {} for target {}", queue, target);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot own queue " + queue, e);
|
||||
@@ -207,18 +236,103 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release ownership of {@code target}: cancel its consumer, then nack-with-requeue every
|
||||
* delivery still held for it instead of just dropping the local record.
|
||||
*
|
||||
* <p><strong>Cancelling a consumer does not requeue its in-flight deliveries.</strong> In AMQP,
|
||||
* a delivery that was pushed to a consumer stays unacked, attached to the still-open
|
||||
* {@link #channel}, until that channel or the connection closes — {@code basicCancel} alone does
|
||||
* neither. So before this method existed with a requeue step, it dropped {@link #held}'s entries
|
||||
* for {@code target} while the broker still considered them outstanding: never acked, never
|
||||
* nacked, never requeued, and no longer reachable by {@link #peek} — permanently invisible. This
|
||||
* is unlike {@link #handleRecovery} and {@link #close()}, whose bare {@code held.clear()} is
|
||||
* correct because each has already made the broker requeue (a real connection drop, or
|
||||
* {@code channel.close()} respectively) before clearing local state.
|
||||
*
|
||||
* <p><strong>Order: cancel first, then nack.</strong> A delivery tag stays valid for
|
||||
* {@code basicNack} on this channel regardless of whether its consumer is still attached — only
|
||||
* a channel/connection close invalidates it — so cancelling {@code target}'s consumer first does
|
||||
* not risk the tags. Doing it the other way round does: nacking a delivery with {@code requeue}
|
||||
* while its consumer is still active hands the message straight back to that <em>same</em>
|
||||
* consumer the instant a prefetch slot frees up (confirmed against a real broker — see
|
||||
* {@code AmqpReplyInboxContractTest.releaseCancelsConsumerAndRequeuesHeldDeliveryForRecovery}),
|
||||
* which races this method's own {@code held.remove(target)}: the redelivery can land after the
|
||||
* clear and leave a stale entry behind, so {@link #peek} is no longer reliably empty right after
|
||||
* {@link #release}. Cancelling first closes that consumer, so the requeued message goes back to
|
||||
* the queue for whichever consumer picks it up next (a later {@link #own}), not this one.
|
||||
*
|
||||
* <p><strong>Failure of the requeue is best-effort, not fatal.</strong> {@link #release} runs
|
||||
* during teardown ({@code Fleetd} calls it right after {@code MessageService.abandon}), and a
|
||||
* throw here would abort cleanups the caller depends on — the same argument fleetd #293 settled
|
||||
* for {@code HerdrPeerLauncher.stop()}'s tab-close step. So a failed {@code basicNack} is logged
|
||||
* at WARN, naming the target and delivery tag that leaked, and release proceeds; the delivery
|
||||
* stays unacked on the broker rather than being silently dropped, so it is still recoverable by a
|
||||
* later connection drop even though this release did not manage to requeue it immediately. A
|
||||
* failed {@code basicCancel} still throws, unchanged from before this fix — that failure means
|
||||
* the consumer may still be attached, so best-effort requeue is not attempted underneath it.
|
||||
*
|
||||
* <p><strong>CB-318: {@code held.remove(target)} alone leaves a second window open.</strong> The
|
||||
* bullet above already explains why cancelling first does not save a tag from going stale — but
|
||||
* that only accounts for a delivery landing before this method starts touching {@link #held}.
|
||||
* {@code basicCancel} stops <em>new</em> dispatches; it does not flush one already handed to the
|
||||
* consumer work pool. So a delivery can still land on that pool's thread and reach
|
||||
* {@link #deliverCallback} at any point during, or after, this method's body — and a plain
|
||||
* {@code held.remove(target)} does nothing to stop it: {@code deliverCallback}'s
|
||||
* {@code computeIfAbsent} finds the key gone and happily creates a brand-new map under it, which
|
||||
* this method — already past its {@code remove} — never looks at again. That entry then sits
|
||||
* delivered-but-unacked on {@link #channel} until the whole inbox closes: never requeued, never
|
||||
* redelivered, and {@link #peek} is never called again for a target nothing owns any more.
|
||||
*
|
||||
* <p>The fix is {@link #held}{@code .compute(target, ...)} instead of {@code remove}: it takes
|
||||
* whatever was held (to nack, same as before) and, in the same atomic step, leaves the
|
||||
* {@link #RELEASED} tombstone behind instead of an absent key. {@code computeIfAbsent} and
|
||||
* {@code compute} calls for the same key are mutually exclusive in {@link ConcurrentHashMap} —
|
||||
* whichever of this call and a concurrent {@code deliverCallback} runs first is fully visible to
|
||||
* the other, with no gap between them. So a delivery that loses the race sees a real map here and
|
||||
* gets nacked by the loop below, same as always; a delivery that wins the race (runs first) is
|
||||
* itself nacked by that same loop, once it settles into {@code held}. A delivery that arrives once
|
||||
* this method has stored {@link #RELEASED} finds it via {@code computeIfAbsent} and refuses itself
|
||||
* — see {@link #deliverCallback}. Either way nothing is silently retained forever, satisfying the
|
||||
* ticket's invariant against dropping a message. This closes the window rather than merely
|
||||
* narrowing it — correctness does not depend on how much time elapses between the swap and this
|
||||
* method returning.
|
||||
*/
|
||||
@Override
|
||||
public void release(String target) {
|
||||
synchronized (channelLock) {
|
||||
String tag = consumerTags.remove(target);
|
||||
held.remove(target); // stale delivery tags must not survive release
|
||||
if (tag == null) {
|
||||
return;
|
||||
if (tag != null) {
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
}
|
||||
}
|
||||
try {
|
||||
channel.basicCancel(tag);
|
||||
} catch (IOException e) {
|
||||
throw new IllegalStateException("cannot cancel consumer for " + target, e);
|
||||
AtomicReference<LinkedHashMap<String, Held>> previouslyHeld = new AtomicReference<>();
|
||||
held.compute(target, (_, v) -> {
|
||||
previouslyHeld.set(v);
|
||||
return RELEASED;
|
||||
});
|
||||
var perTarget = previouslyHeld.get();
|
||||
if (perTarget != null && perTarget != RELEASED) {
|
||||
synchronized (perTarget) {
|
||||
for (Held h : perTarget.values()) {
|
||||
try {
|
||||
channel.basicNack(h.deliveryTag(), false, true); // requeue, don't drop
|
||||
} catch (IOException | RuntimeException e) {
|
||||
// Caught broadly (not just IOException) for the same reason #293 catches
|
||||
// RuntimeException in HerdrPeerLauncher.stop(): best-effort teardown must
|
||||
// not be guarded only against the expected failure and bare against any
|
||||
// other. The message stays unacked on the broker either way — not lost,
|
||||
// just not proactively requeued — until a connection drop frees it.
|
||||
log.warn("release({}): could not requeue held delivery (msgId={}, tag={})"
|
||||
+ " back to the broker — it stays unacked until a connection"
|
||||
+ " drop frees it: {}",
|
||||
target, h.message().msgId(), h.deliveryTag(), e.getMessage());
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -271,7 +385,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
@Override
|
||||
public List<InboxMessage> peek(String target) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return List.of();
|
||||
}
|
||||
synchronized (perTarget) {
|
||||
@@ -280,17 +394,17 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
public boolean ack(String target, String msgId) {
|
||||
var perTarget = held.get(target);
|
||||
if (perTarget == null) {
|
||||
return;
|
||||
if (perTarget == null || perTarget == RELEASED) {
|
||||
return false;
|
||||
}
|
||||
Held h;
|
||||
synchronized (perTarget) {
|
||||
h = perTarget.remove(msgId);
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
return false; // never held (or already acked) — no-op
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
@@ -304,6 +418,7 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
throw new IllegalStateException("cannot ack reply " + msgId + " on " + queueName(target), e);
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback(String target) {
|
||||
@@ -315,6 +430,20 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
String content = new String(delivery.getBody(), StandardCharsets.UTF_8);
|
||||
var perTarget = held.computeIfAbsent(target, _ -> new LinkedHashMap<>());
|
||||
if (perTarget == RELEASED) {
|
||||
// CB-318: release() already ran for this target and left the RELEASED tombstone in
|
||||
// held (see release()'s javadoc) — computeIfAbsent() is guaranteed to see it rather
|
||||
// than recreate a fresh map, because ConcurrentHashMap serializes compute/
|
||||
// computeIfAbsent calls for the same key against each other. Refuse the delivery
|
||||
// instead of holding it somewhere release() will never look at again: requeue it, the
|
||||
// same way release() nacks its own held entries, so a later owner (or a connection
|
||||
// drop) can still recover it. This does not need channelLock across a broker round
|
||||
// trip — basicNack, like the duplicate-ack case just below, does not wait for one.
|
||||
synchronized (channelLock) {
|
||||
channel.basicNack(tag, false, true);
|
||||
}
|
||||
return;
|
||||
}
|
||||
boolean duplicate;
|
||||
synchronized (perTarget) {
|
||||
if (perTarget.containsKey(msgId)) {
|
||||
@@ -449,3 +578,44 @@ public final class AmqpReplyInbox implements ReplyInbox, AutoCloseable {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Keeps RabbitMQ's forgiving exception behaviour while adding the connection identity that its
|
||||
* default logger drops. Package-private so both AMQP connections use the same two names.
|
||||
*/
|
||||
final class AmqpConnectionFailureLogger extends DefaultExceptionHandler {
|
||||
|
||||
static final String REPLY_INBOX = "fleetd-reply-inbox";
|
||||
static final String LEAD_MAILBOX = "fleetd-lead-mailbox";
|
||||
|
||||
private final String connectionName;
|
||||
private final Logger logger;
|
||||
|
||||
AmqpConnectionFailureLogger(String connectionName, Logger logger) {
|
||||
this.connectionName = connectionName;
|
||||
this.logger = logger;
|
||||
}
|
||||
|
||||
String connectionName() {
|
||||
return connectionName;
|
||||
}
|
||||
|
||||
@Override
|
||||
protected void log(String message, Throwable cause) {
|
||||
if (isSocketClosedOrConnectionReset(cause)) {
|
||||
logger.warn("AMQP connection {}: {} (Exception message: {})", connectionName, message, cause.getMessage());
|
||||
} else {
|
||||
logger.error("AMQP connection {}: {}", connectionName, message, cause);
|
||||
}
|
||||
}
|
||||
|
||||
private static boolean isSocketClosedOrConnectionReset(Throwable cause) {
|
||||
// Deliberate copy of ForgivingExceptionHandler's private static helper; check it on amqp-client upgrades.
|
||||
if (!(cause instanceof IOException)) {
|
||||
return false;
|
||||
}
|
||||
return "Connection reset".equals(cause.getMessage())
|
||||
|| "Socket closed".equals(cause.getMessage())
|
||||
|| "Connection reset by peer".equals(cause.getMessage());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -59,16 +59,17 @@ public final class InMemoryReplyInbox implements ReplyInbox {
|
||||
}
|
||||
|
||||
@Override
|
||||
public void ack(String target, String msgId) {
|
||||
public boolean ack(String target, String msgId) {
|
||||
if (!owned.contains(target)) {
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
var perTarget = store.get(target);
|
||||
if (perTarget != null) {
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
perTarget.remove(msgId);
|
||||
}
|
||||
if (perTarget == null) {
|
||||
return false;
|
||||
}
|
||||
//noinspection SynchronizationOnLocalVariableOrMethodParameter
|
||||
synchronized (perTarget) {
|
||||
return perTarget.remove(msgId) != null;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4,8 +4,8 @@ import java.util.List;
|
||||
|
||||
/**
|
||||
* The lead-to-lead message channel this daemon speaks, as its callers need it — one lead's own
|
||||
* mailbox: publish to a peer's coord-id, look at what has arrived for me, and ack what I have
|
||||
* delivered.
|
||||
* mailbox: publish to a peer's coord-id, look at what has arrived for me, ack what I have
|
||||
* delivered, and (fleetd #361) inspect any coord-id's mailbox from the outside without owning it.
|
||||
*
|
||||
* <p>Extracted from {@link LeadMailbox} purely as a seam. {@code LeadMailbox} is the one production
|
||||
* implementation and owns a live AMQP connection, so a test that wanted to exercise the routing in
|
||||
@@ -32,9 +32,99 @@ public interface LeadChannel {
|
||||
/** Non-destructive FIFO snapshot of the messages held for this daemon's own coord-id. */
|
||||
List<LeadMessage> peek();
|
||||
|
||||
/** Drop {@code msgId} from the held set and ack it on the broker. A no-op if it is not held. */
|
||||
/**
|
||||
* Drop {@code msgId} from the held set and ack it on the broker. A repeated ack that this
|
||||
* connection already completed may be a no-op. Any other unknown msgId must throw rather than
|
||||
* report an ack that did not reach the broker.
|
||||
*/
|
||||
void ack(String msgId);
|
||||
|
||||
/** This daemon's own lead coordination id — the mailbox it owns, and the {@code from} it sends as. */
|
||||
String selfCoordId();
|
||||
|
||||
/**
|
||||
* Whether a message sitting in {@link #peek}'s held set (fetched but not yet {@link #ack}ed) is
|
||||
* still safe if this daemon crashes or restarts right now — the conclusion of two independent
|
||||
* facts about how this channel owns its own queue: the queue was declared <em>durable</em>, and
|
||||
* the consumer that filled {@code held} uses <em>manual ack</em>, so an unacked delivery is still
|
||||
* owned by the broker rather than only in this process's memory. Both must hold for {@code true};
|
||||
* an implementation must derive this from what it actually did when it declared and consumed its
|
||||
* queue, never return a literal — fleetd #440 found {@code FleetMcp}'s {@code heldDurable} field
|
||||
* doing exactly that, unable to ever report {@code false} even after the fact stopped being true.
|
||||
*/
|
||||
boolean heldDurable();
|
||||
|
||||
/**
|
||||
* A non-destructive look at {@code coordId}'s mailbox — does it exist, how many messages are
|
||||
* waiting on it, and how many consumers are attached — without owning, consuming, or otherwise
|
||||
* changing it. {@code consumers == 0} on an existing mailbox is the observable form of "nobody
|
||||
* is reading this right now": a publish to it will sit queued rather than reach a pane.
|
||||
*
|
||||
* <p><strong>Never throws</strong> — this is a best-effort fact-finding call, not an operation a
|
||||
* caller must handle failing. But it must never turn "I could not check" into a false negative:
|
||||
* {@link MailboxState#absent(String)} means the broker positively confirmed there is no such
|
||||
* queue, and {@link MailboxState#unknown(String)} — a distinct value — means the look could not
|
||||
* be completed at all (broker unreachable, timed out, connection closed). A caller that
|
||||
* collapses those two into one, as fleetd #361 initially did, cannot tell "that peer is down"
|
||||
* from "I could not check", and a reader of {@code pending}/{@code consumers} cannot tell a
|
||||
* measured zero from a zero standing in for "not measured".
|
||||
*
|
||||
* <p><strong>Must never share fate with {@link #publish} or {@link #peek}/{@link #ack}.</strong>
|
||||
* fleetd #361: in AMQP 0-9-1 a passive queue declare of a queue that does not exist closes the
|
||||
* channel it was declared on with a 404. An implementation backed by a real broker connection
|
||||
* must inspect on a channel it can afford to lose — never the channel {@link #publish} or the
|
||||
* consume loop depends on — so that looking at a peer that happens to be down can never break
|
||||
* this daemon's own send or receive path.
|
||||
*/
|
||||
MailboxState inspect(String coordId);
|
||||
|
||||
/**
|
||||
* The result of {@link #inspect}. {@code presence} tells apart three states a caller must not
|
||||
* conflate: a confirmed-existing mailbox ({@link Presence#EXISTS}, the only case where
|
||||
* {@code pending}/{@code consumers} are measured facts), a confirmed-absent one
|
||||
* ({@link Presence#ABSENT} — the broker positively said "no such queue"), and one this call
|
||||
* simply could not determine ({@link Presence#UNKNOWN} — broker unreachable, timed out,
|
||||
* connection closed). {@code pending}/{@code consumers} are always {@code 0} and meaningless
|
||||
* outside {@link Presence#EXISTS}; a renderer must gate on {@link #exists()} (or {@code
|
||||
* presence} directly), never present them as measured otherwise.
|
||||
*
|
||||
* @param coordId the coord-id inspected
|
||||
* @param presence whether the mailbox is confirmed to exist, confirmed absent, or unknown
|
||||
* @param pending messages ready for delivery but not yet in a consumer's hands (0 unless EXISTS)
|
||||
* @param consumers how many consumers are attached (0 unless EXISTS)
|
||||
*/
|
||||
record MailboxState(String coordId, Presence presence, int pending, int consumers) {
|
||||
|
||||
/** Whether {@link #inspect} was able to reach a definite answer, of either kind. */
|
||||
public enum Presence { EXISTS, ABSENT, UNKNOWN }
|
||||
|
||||
/** {@code true} only when the broker confirmed this exact queue is currently declared. */
|
||||
public boolean exists() {
|
||||
return presence == Presence.EXISTS;
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code true} when {@link #inspect} reached a definite answer (exists or confirmed
|
||||
* absent); {@code false} when it could not determine either way. A caller must never treat
|
||||
* {@code !known()} the same as a confirmed absence — the mailbox may well exist.
|
||||
*/
|
||||
public boolean known() {
|
||||
return presence != Presence.UNKNOWN;
|
||||
}
|
||||
|
||||
/** The broker confirmed this queue exists, with these measured counts. */
|
||||
public static MailboxState exists(String coordId, int pending, int consumers) {
|
||||
return new MailboxState(coordId, Presence.EXISTS, pending, consumers);
|
||||
}
|
||||
|
||||
/** The broker positively confirmed there is no such queue (e.g. a 404 on passive declare). */
|
||||
public static MailboxState absent(String coordId) {
|
||||
return new MailboxState(coordId, Presence.ABSENT, 0, 0);
|
||||
}
|
||||
|
||||
/** The look could not be completed — broker unreachable, timed out, or connection closed. */
|
||||
public static MailboxState unknown(String coordId) {
|
||||
return new MailboxState(coordId, Presence.UNKNOWN, 0, 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,6 +5,7 @@ import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -43,6 +44,9 @@ public final class LeadCoordLoop {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(LeadCoordLoop.class);
|
||||
|
||||
/** A bounded window is enough: redelivery can only follow a recent failed ack or recovery. */
|
||||
private static final int RECENT_DELIVERY_LIMIT = 1_024;
|
||||
|
||||
/** How an arriving peer message is rendered into the lead's pane — the sender's coord-id, then its text. */
|
||||
static final String DELIVERY_FORMAT = "[lead %s] %s";
|
||||
|
||||
@@ -51,6 +55,8 @@ public final class LeadCoordLoop {
|
||||
private final Supplier<Map<String, String>> leads;
|
||||
private final ScheduledExecutorService scheduler;
|
||||
private final long intervalMs;
|
||||
/** msgIds already written to the pane, so recovery redelivery is acked without another pane write. */
|
||||
private final LinkedHashMap<String, Boolean> delivered = new LinkedHashMap<>();
|
||||
|
||||
private volatile boolean running;
|
||||
|
||||
@@ -117,6 +123,13 @@ public final class LeadCoordLoop {
|
||||
if (held.isEmpty()) {
|
||||
return;
|
||||
}
|
||||
LeadMessage msg = held.getFirst();
|
||||
if (wasDelivered(msg.msgId())) {
|
||||
// This lives here, rather than in LeadMailbox, because only this loop knows a pane write
|
||||
// happened. The mailbox only knows broker delivery tags and must still redeliver after a crash.
|
||||
ackDelivered(msg);
|
||||
return;
|
||||
}
|
||||
String lead = resolveLocalLead();
|
||||
if (lead == null) {
|
||||
// Left unacked on purpose: the broker keeps holding it until a lead pane exists.
|
||||
@@ -136,7 +149,6 @@ public final class LeadCoordLoop {
|
||||
lead, status, held.size());
|
||||
return;
|
||||
}
|
||||
LeadMessage msg = held.getFirst();
|
||||
try {
|
||||
agents.send(lead, DELIVERY_FORMAT.formatted(msg.from(), msg.content()));
|
||||
} catch (RuntimeException e) {
|
||||
@@ -145,6 +157,13 @@ public final class LeadCoordLoop {
|
||||
msg.msgId(), msg.from(), lead, e.toString());
|
||||
return;
|
||||
}
|
||||
rememberDelivered(msg.msgId());
|
||||
if (ackDelivered(msg)) {
|
||||
log.debug("lead coordination: delivered message {} from {} to lead {}", msg.msgId(), msg.from(), lead);
|
||||
}
|
||||
}
|
||||
|
||||
private boolean ackDelivered(LeadMessage msg) {
|
||||
try {
|
||||
channel.ack(msg.msgId());
|
||||
} catch (RuntimeException e) {
|
||||
@@ -152,9 +171,24 @@ public final class LeadCoordLoop {
|
||||
// deliberate direction of this trade.
|
||||
log.warn("lead coordination: delivered message {} but could not ack it: {}",
|
||||
msg.msgId(), e.toString());
|
||||
return;
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
private boolean wasDelivered(String msgId) {
|
||||
synchronized (delivered) {
|
||||
return delivered.containsKey(msgId);
|
||||
}
|
||||
}
|
||||
|
||||
private void rememberDelivered(String msgId) {
|
||||
synchronized (delivered) {
|
||||
delivered.put(msgId, Boolean.TRUE);
|
||||
if (delivered.size() > RECENT_DELIVERY_LIMIT) {
|
||||
delivered.remove(delivered.keySet().iterator().next());
|
||||
}
|
||||
}
|
||||
log.debug("lead coordination: delivered message {} from {} to lead {}", msg.msgId(), msg.from(), lead);
|
||||
}
|
||||
|
||||
/**
|
||||
|
||||
@@ -96,7 +96,14 @@ public final class LeadHeartbeatLoop {
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
/**
|
||||
* Count one nudge outcome when a registry is wired; a no-op in unit tests.
|
||||
*
|
||||
* <p>fleetd #365: the {@code "sent"} outcome (renamed from {@code "delivered"}) records only
|
||||
* that {@link #injectNudge} — a one-way herdr {@code agent.prompt} paste-and-submit — returned
|
||||
* without throwing, not that the lead's pane actually read or acted on the text. This layer has
|
||||
* no read-receipt concept, so "sent" is the honest word for what this call can ever establish.
|
||||
*/
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(FleetMetrics.HEARTBEAT_NUDGES, "outcome", outcome);
|
||||
@@ -245,7 +252,7 @@ public final class LeadHeartbeatLoop {
|
||||
agents.send(leadTerminal, fleet.nudgeText());
|
||||
log.debug("idle-heartbeat: nudge sent to lead {} (quiet nudges so far in this stretch: {})",
|
||||
leadTerminal, quietCount);
|
||||
countNudge("delivered");
|
||||
countNudge("sent");
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-heartbeat: failed to nudge lead {}: {}", leadTerminal, e.toString());
|
||||
countNudge("failed");
|
||||
|
||||
@@ -10,6 +10,7 @@ import com.rabbitmq.client.DeliverCallback;
|
||||
import com.rabbitmq.client.Recoverable;
|
||||
import com.rabbitmq.client.RecoveryListener;
|
||||
import com.rabbitmq.client.Return;
|
||||
import com.rabbitmq.client.ShutdownSignalException;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -58,7 +59,8 @@ import java.util.concurrent.TimeoutException;
|
||||
* <p><strong>Recovery.</strong> The connection is opened with automatic + topology recovery
|
||||
* enabled, mirroring {@code AmqpReplyInbox}: on reconnect the broker hands out fresh delivery tags,
|
||||
* so the held snapshot is cleared (dedup by {@code msgId} still prevents any double-queue on
|
||||
* redelivery) and any publish still awaiting its confirm is failed rather than left to idle out
|
||||
* redelivery). {@link LeadCoordLoop} separately deduplicates pane writes, since it alone knows
|
||||
* which messages reached a lead. Any publish still awaiting its confirm is failed rather than left to idle out
|
||||
* the confirm timeout against a sequence number that means nothing on the new channel.
|
||||
*/
|
||||
public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
@@ -84,6 +86,16 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
private final Object channelLock = new Object();
|
||||
/** msgId → held delivery, for this mailbox's own queue only (there is exactly one). */
|
||||
private final LinkedHashMap<String, Held> held = new LinkedHashMap<>();
|
||||
/**
|
||||
* fleetd #440: the answer to {@link #heldDurable()}, set once by {@link #own()} from the exact
|
||||
* booleans it passed to {@code queueDeclare}/{@code basicConsume} — never a separate literal that
|
||||
* could drift from what those calls actually did.
|
||||
*/
|
||||
private boolean heldDurable;
|
||||
/** Successful broker acks on this connection, retained only to make a repeated caller ack quiet. */
|
||||
private final LinkedHashMap<String, Boolean> recentlyAcked = new LinkedHashMap<>();
|
||||
/** Bounds {@link #recentlyAcked}: it is only an idempotency aid, never delivery state. */
|
||||
private static final int RECENT_ACK_LIMIT = 1_024;
|
||||
|
||||
/**
|
||||
* A dedicated channel for {@link #publish}, kept separate from {@link #channel} (consume + ack)
|
||||
@@ -122,17 +134,22 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
/** As {@link #open(String, String)}, with an explicit consumer prefetch. */
|
||||
public static LeadMailbox open(String uri, String selfCoordId, int prefetch) {
|
||||
try {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares the queue and re-attaches the consumer.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
return new LeadMailbox(factory.newConnection("fleetd-lead-mailbox"), selfCoordId, prefetch);
|
||||
return new LeadMailbox(connectionFactory(uri).newConnection(AmqpConnectionFailureLogger.LEAD_MAILBOX), selfCoordId, prefetch);
|
||||
} catch (Exception e) {
|
||||
throw new IllegalStateException("cannot connect to AMQP coordination broker at " + uri, e);
|
||||
}
|
||||
}
|
||||
|
||||
static ConnectionFactory connectionFactory(String uri) throws Exception {
|
||||
ConnectionFactory factory = new ConnectionFactory();
|
||||
factory.setUri(uri);
|
||||
// Self-heal transient blips; topology recovery re-declares queues and re-attaches consumers.
|
||||
factory.setAutomaticRecoveryEnabled(true);
|
||||
factory.setTopologyRecoveryEnabled(true);
|
||||
factory.setExceptionHandler(new AmqpConnectionFailureLogger(AmqpConnectionFailureLogger.LEAD_MAILBOX, log));
|
||||
return factory;
|
||||
}
|
||||
|
||||
/** Wrap an already-open connection with {@link #DEFAULT_PREFETCH} (injection seam for tests). */
|
||||
LeadMailbox(Connection connection, String selfCoordId) {
|
||||
this(connection, selfCoordId, DEFAULT_PREFETCH);
|
||||
@@ -156,16 +173,15 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
// On automatic recovery the broker redelivers unacked messages with FRESH delivery-tags; the
|
||||
// tags we were holding are now stale. Drop the held snapshot so the re-attached consumer
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). Any publish
|
||||
// repopulates it with valid tags (dedup by msgId still prevents any double-queue). LeadCoordLoop
|
||||
// remembers successful pane writes separately, so that redelivery cannot write a pane twice. Any publish
|
||||
// confirm still in flight when the connection dropped is equally stale — fail it now rather
|
||||
// than let it silently ride out CONFIRM_TIMEOUT_MS.
|
||||
if (connection instanceof Recoverable recoverable) {
|
||||
recoverable.addRecoveryListener(new RecoveryListener() {
|
||||
@Override
|
||||
public void handleRecovery(Recoverable recoverable) {
|
||||
synchronized (held) {
|
||||
held.clear();
|
||||
}
|
||||
clearHeldForRecovery();
|
||||
failPendingPublishesOnRecovery();
|
||||
log.info("AMQP lead mailbox connection recovered; cleared held messages for fresh redelivery");
|
||||
}
|
||||
@@ -181,13 +197,22 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
/** Declare + consume this daemon's own {@code lead.<selfCoordId>.inbox}. Called once, at construction. */
|
||||
private void own() throws IOException {
|
||||
String queue = queueName(selfCoordId);
|
||||
boolean durableQueue = true; // durable, non-exclusive, keep on idle
|
||||
boolean autoAck = false; // manual ack
|
||||
synchronized (channelLock) {
|
||||
channel.queueDeclare(queue, true, false, false, null); // durable, non-exclusive, keep on idle
|
||||
channel.basicConsume(queue, false, deliverCallback(), _ -> { }); // autoAck=false: manual ack
|
||||
channel.queueDeclare(queue, durableQueue, false, false, null);
|
||||
channel.basicConsume(queue, autoAck, deliverCallback(), _ -> { });
|
||||
}
|
||||
// fleetd #440: held mail is durable only while both hold — a durable queue AND manual ack.
|
||||
this.heldDurable = durableQueue && !autoAck;
|
||||
log.debug("lead mailbox owns queue {} for coord-id {}", queue, selfCoordId);
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean heldDurable() {
|
||||
return heldDurable;
|
||||
}
|
||||
|
||||
/**
|
||||
* Publish {@code msg} to {@code toCoordId}'s mailbox and block until the broker's publisher
|
||||
* confirm for it lands. Does <em>not</em> imply owning or consuming {@code toCoordId}'s queue.
|
||||
@@ -254,6 +279,89 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #361: look at {@code coordId}'s mailbox on a fresh, immediately-closed throwaway
|
||||
* channel — never {@link #channel} (consume/ack) or {@link #publishChannel} (publish). A
|
||||
* passive queue declare of a queue that does not exist closes the channel it was declared on
|
||||
* with a 404; using a disposable probe channel means that closure can never touch either
|
||||
* long-lived channel this instance depends on for {@link #publish} or the consume loop.
|
||||
*
|
||||
* <p>Classifies failures rather than collapsing them, both measured against a real broker in
|
||||
* {@code LeadMailboxTest} rather than assumed from the AMQP 0-9-1 spec text:
|
||||
* <ul>
|
||||
* <li>a genuine 404 — an {@link IOException} wrapping a {@link ShutdownSignalException} whose
|
||||
* {@link AMQP.Channel.Close#getReplyCode()} is {@code 404} — reports
|
||||
* {@link MailboxState#absent}; every other declare failure reports
|
||||
* {@link MailboxState#unknown} instead of quietly becoming the same "absent" value;
|
||||
* <li>{@code catch (RuntimeException e)} on both attempts matters as much as the checked
|
||||
* catches: a connection that is already closed makes {@link Connection#createChannel()}
|
||||
* throw {@link com.rabbitmq.client.AlreadyClosedException} (a {@link RuntimeException},
|
||||
* not an {@link IOException}) — an {@code inspect} that only caught {@code IOException}
|
||||
* would let that escape, breaking the "never throws" contract this method promises.
|
||||
* </ul>
|
||||
*
|
||||
* <p><strong>Honesty about which catch is measured and which is defensive:</strong> the
|
||||
* {@code createChannel()} catch above is exercised end-to-end against a real broker by
|
||||
* {@code LeadMailboxTest.inspectReportsUnknownRatherThanThrowingWhenTheConnectionIsAlreadyClosed}.
|
||||
* The second {@code catch (RuntimeException e)}, around the passive declare itself — for the
|
||||
* narrower race where the connection drops <em>between</em> {@code createChannel()} succeeding
|
||||
* and the declare landing — has no such test; reaching it needs a connection that dies at that
|
||||
* exact instant, which is not a scenario this suite drives on purpose. It stays purely
|
||||
* defensive: correct by the same reasoning as the first catch, but unproven the way the first
|
||||
* one is proven.
|
||||
*/
|
||||
@Override
|
||||
public MailboxState inspect(String coordId) {
|
||||
String queue = queueName(coordId);
|
||||
Channel probe;
|
||||
try {
|
||||
probe = connection.createChannel();
|
||||
} catch (IOException | RuntimeException e) {
|
||||
log.debug("lead mailbox inspect: cannot open a probe channel for {}: {}", coordId, e.toString());
|
||||
return MailboxState.unknown(coordId);
|
||||
}
|
||||
try {
|
||||
AMQP.Queue.DeclareOk declared = probe.queueDeclarePassive(queue);
|
||||
return MailboxState.exists(coordId, declared.getMessageCount(), declared.getConsumerCount());
|
||||
} catch (IOException e) {
|
||||
// The broker (or the client library) has already closed `probe` for us either way; only
|
||||
// a confirmed 404 means "no such queue" — anything else (a different declare failure) is
|
||||
// "could not determine", never silently reported as the same value as a genuine absence.
|
||||
return isMissingQueue(e) ? MailboxState.absent(coordId) : MailboxState.unknown(coordId);
|
||||
} catch (RuntimeException e) {
|
||||
// E.g. the connection dropped between createChannel() and the declare landing.
|
||||
log.debug("lead mailbox inspect: declare failed unexpectedly for {}: {}", coordId, e.toString());
|
||||
return MailboxState.unknown(coordId);
|
||||
} finally {
|
||||
try {
|
||||
if (probe.isOpen()) {
|
||||
probe.close();
|
||||
}
|
||||
} catch (Exception e) {
|
||||
log.debug("lead mailbox inspect: probe channel close for {}: {}", coordId, e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* {@code true} only for the specific shape a missing-queue passive declare actually produces —
|
||||
* measured against a real broker, not assumed from the spec text (see {@code
|
||||
* LeadMailboxTest.passiveDeclareOfAMissingQueueThrowsAnIOExceptionWrappingA404ShutdownSignal}):
|
||||
* an {@link IOException} whose cause is a {@link ShutdownSignalException} carrying an
|
||||
* {@link AMQP.Channel.Close} reason with {@code replyCode == 404}. Any other shape (a different
|
||||
* reply code, a {@code ShutdownSignalException} cause whose reason is not a
|
||||
* {@code Channel.Close}, or no cause at all) is a declare failure of some other kind and must
|
||||
* not be read as "confirmed absent" — pinned hermetically, with no broker needed, by
|
||||
* {@code LeadMailboxIsMissingQueueTest} for exactly those three false shapes. Package-private
|
||||
* (not {@code private}) so that test can call it directly.
|
||||
*/
|
||||
static boolean isMissingQueue(IOException e) {
|
||||
if (!(e.getCause() instanceof ShutdownSignalException sse)) {
|
||||
return false;
|
||||
}
|
||||
return sse.getReason() instanceof AMQP.Channel.Close close && close.getReplyCode() == AMQP.NOT_FOUND;
|
||||
}
|
||||
|
||||
/** Convenience: {@link #peek} the current snapshot, then {@link #ack} every message in it. */
|
||||
public List<LeadMessage> drain() {
|
||||
List<LeadMessage> snapshot = peek();
|
||||
@@ -261,15 +369,26 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
return snapshot;
|
||||
}
|
||||
|
||||
/** Remove the held message {@code msgId} and ack it on the broker. No-op if not held. */
|
||||
/**
|
||||
* Remove the held message {@code msgId} and ack it on the broker.
|
||||
*
|
||||
* <p>A repeated ack that this connection already completed is a no-op, tracked in the bounded
|
||||
* {@link #recentlyAcked} set. Any other missing entry throws: recovery clears {@link #held} while
|
||||
* the broker still owns the unacked delivery, and quiet success there would hide a required retry.
|
||||
* The set is bounded because it only distinguishes a recent duplicate caller ack from an unknown
|
||||
* delivery; it is not a substitute for broker state across a reconnect.
|
||||
*/
|
||||
@Override
|
||||
public void ack(String msgId) {
|
||||
Held h;
|
||||
synchronized (held) {
|
||||
h = held.remove(msgId);
|
||||
if (h == null && recentlyAcked.containsKey(msgId)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (h == null) {
|
||||
return; // never held (or already acked) — no-op
|
||||
throw new IllegalStateException("cannot ack lead message " + msgId + ": it is not held");
|
||||
}
|
||||
try {
|
||||
synchronized (channelLock) {
|
||||
@@ -283,6 +402,19 @@ public final class LeadMailbox implements LeadChannel, AutoCloseable {
|
||||
}
|
||||
throw new IllegalStateException("cannot ack lead message " + msgId, e);
|
||||
}
|
||||
synchronized (held) {
|
||||
recentlyAcked.put(msgId, Boolean.TRUE);
|
||||
if (recentlyAcked.size() > RECENT_ACK_LIMIT) {
|
||||
recentlyAcked.remove(recentlyAcked.keySet().iterator().next());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Clear stale delivery tags after recovery; package-private so the recovery contract test drives this exact path. */
|
||||
void clearHeldForRecovery() {
|
||||
synchronized (held) {
|
||||
held.clear();
|
||||
}
|
||||
}
|
||||
|
||||
private DeliverCallback deliverCallback() {
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -46,6 +46,13 @@ public interface ReplyInbox {
|
||||
/** Non-destructive snapshot of pending replies for {@code target} (FIFO), empty list if none. */
|
||||
List<InboxMessage> peek(String target);
|
||||
|
||||
/** Remove the reply {@code msgId} for {@code target} once the primary has taken it. No-op if absent. */
|
||||
void ack(String target, String msgId);
|
||||
/**
|
||||
* Remove the reply {@code msgId} for {@code target} once the primary has taken it.
|
||||
*
|
||||
* @return {@code true} if an entry was actually removed, {@code false} if there was nothing to
|
||||
* remove (unknown {@code target}, unowned {@code target}, or a {@code msgId} not held for
|
||||
* it). A {@code false} is not an error — acking a {@code target} this daemon does not own is
|
||||
* part of the normal contract, not a failure.
|
||||
*/
|
||||
boolean ack(String target, String msgId);
|
||||
}
|
||||
|
||||
@@ -2,6 +2,7 @@ package dev.ltms.fleet.msg;
|
||||
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.AgentStatus;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
@@ -9,8 +10,11 @@ import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.Collection;
|
||||
import java.util.HashSet;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
@@ -68,6 +72,12 @@ public final class ReplyPushLoop {
|
||||
static final String QUESTIONS_NUDGE_FORMAT =
|
||||
"%d workers are paused on a question — run fleet_poll(ticket=...) for each, then answer "
|
||||
+ "with fleet_send(turnId=..., content=...): %s";
|
||||
static final String BACKEND_INCIDENT_NUDGE_FORMAT =
|
||||
"Backend credential %s is cooling for %d remaining seconds; affected profiles: %s; "
|
||||
+ "affected workers: %s. Run fleet_list to see more.";
|
||||
static final String BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT =
|
||||
"A backend error on worker %s could not be mapped to a credential (%s) — no cool-off "
|
||||
+ "was applied. Run fleet_list to check that worker.";
|
||||
|
||||
private final PrimaryRegistry primaryRegistry;
|
||||
private final AgentControl agents;
|
||||
@@ -93,6 +103,13 @@ public final class ReplyPushLoop {
|
||||
* how depleted an older, still-open question's count is.
|
||||
*/
|
||||
private final ConcurrentHashMap<String, PendingQuestion> pendingQuestions = new ConcurrentHashMap<>();
|
||||
/** Backend incidents awaiting one successful delivery, keyed by incident id and owning lead. */
|
||||
private final ConcurrentHashMap<IncidentLead, PendingIncident> pendingIncidents = new ConcurrentHashMap<>();
|
||||
/** Incident/lead pairs already delivered. They make repeated incident reports one-shot. */
|
||||
private final Set<IncidentLead> deliveredIncidents = ConcurrentHashMap.newKeySet();
|
||||
private final ConcurrentHashMap<UnmappedTargetLead, PendingUnmappedTarget> pendingUnmappedTargets =
|
||||
new ConcurrentHashMap<>();
|
||||
private final Set<UnmappedTargetLead> deliveredUnmappedTargets = ConcurrentHashMap.newKeySet();
|
||||
/** CB-590: leads with an active combined reminder schedule (replies and/or tickets and/or questions). */
|
||||
private final ConcurrentHashMap<String, Boolean> activeLeads = new ConcurrentHashMap<>();
|
||||
|
||||
@@ -115,7 +132,15 @@ public final class ReplyPushLoop {
|
||||
this.metrics = metrics;
|
||||
}
|
||||
|
||||
/** Count one nudge outcome when a registry is wired; a no-op in unit tests. */
|
||||
/**
|
||||
* Count one nudge outcome when a registry is wired; a no-op in unit tests.
|
||||
*
|
||||
* <p>fleetd #365: the {@code "sent"} outcome (renamed from {@code "delivered"}) records only
|
||||
* that {@code agents.send} — a one-way herdr {@code agent.prompt} paste-and-submit — returned
|
||||
* without throwing. Nothing in this loop, or anywhere downstream of it, confirms the pane
|
||||
* actually read or acted on the text; there is no read-receipt concept at this layer. "Sent"
|
||||
* says exactly that; "delivered" claimed more than this call can ever establish.
|
||||
*/
|
||||
private void countNudge(String outcome) {
|
||||
if (metrics != null) {
|
||||
metrics.inc(FleetMetrics.PUSH_NUDGES, "outcome", outcome);
|
||||
@@ -178,7 +203,20 @@ public final class ReplyPushLoop {
|
||||
* tracked per question, not per lead per source).
|
||||
*/
|
||||
private record PendingQuestion(String turnId, String ticket, String target, String lead,
|
||||
String question, int nudgeCount) {
|
||||
String question, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record IncidentLead(String incidentId, String lead) {
|
||||
}
|
||||
|
||||
private record PendingIncident(IncidentLead key, String credential, List<String> profiles,
|
||||
List<String> targets, int remainingCoolOffSeconds, int nudgeCount) {
|
||||
}
|
||||
|
||||
private record UnmappedTargetLead(String target, String reason, String lead) {
|
||||
}
|
||||
|
||||
private record PendingUnmappedTarget(UnmappedTargetLead key, int nudgeCount) {
|
||||
}
|
||||
|
||||
/** Questions still open for {@code lead}, snapshotted fresh for one tick. */
|
||||
@@ -192,6 +230,44 @@ public final class ReplyPushLoop {
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* Test seam only (fleetd #418): carries no production behaviour, and nothing in this class
|
||||
* calls it. Exposes {@link #pendingQuestionTurnIdsFor} — the exact state {@link #decide} reads
|
||||
* to decide whether a question keeps a lead's schedule alive.
|
||||
*
|
||||
* <p>{@code MessageService.ask()} does three things in order before a question is fully open to
|
||||
* this loop: it flips the ticket's {@code poll()} phase to {@code Phase.ASKING}, then resolves
|
||||
* the reverse-rendezvous waiter, then calls {@link #onQuestionOpened}, which is what actually
|
||||
* populates {@link #pendingQuestions}. A test that barriers on {@code Phase.ASKING} observes only
|
||||
* the first of those three steps — under load the asker thread can be descheduled between steps
|
||||
* one and three, so the barrier releases before this method's underlying map is populated, and
|
||||
* {@link #decide} correctly reports nothing pending yet. A test that must order itself after the
|
||||
* state {@link #decide} actually reads waits on this instead of on the phase.
|
||||
*
|
||||
* @return an unmodifiable snapshot; empty for a lead with no open questions
|
||||
*/
|
||||
Set<String> pendingQuestionTurnIdsForTest(String lead) {
|
||||
return pendingQuestionTurnIdsFor(lead);
|
||||
}
|
||||
|
||||
private List<PendingIncident> pendingIncidentsFor(String lead) {
|
||||
return pendingIncidents.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<IncidentLead> pendingIncidentKeysFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().map(PendingIncident::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
private List<PendingUnmappedTarget> pendingUnmappedTargetsFor(String lead) {
|
||||
return pendingUnmappedTargets.values().stream().filter(i -> lead.equals(i.key().lead())).toList();
|
||||
}
|
||||
|
||||
private Set<UnmappedTargetLead> pendingUnmappedTargetKeysFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().map(PendingUnmappedTarget::key)
|
||||
.collect(Collectors.toUnmodifiableSet());
|
||||
}
|
||||
|
||||
/**
|
||||
* The reply-source reminder count {@link #decide} should see for {@code lead} on this tick:
|
||||
* the <em>minimum</em> nudge count among the reply targets currently pending for it (CB-598).
|
||||
@@ -236,6 +312,15 @@ public final class ReplyPushLoop {
|
||||
return min == Integer.MAX_VALUE ? 0 : min;
|
||||
}
|
||||
|
||||
private int minIncidentNudgeCountFor(String lead) {
|
||||
return pendingIncidentsFor(lead).stream().mapToInt(PendingIncident::nudgeCount).min().orElse(0);
|
||||
}
|
||||
|
||||
private int minUnmappedTargetNudgeCountFor(String lead) {
|
||||
return pendingUnmappedTargetsFor(lead).stream().mapToInt(PendingUnmappedTarget::nudgeCount)
|
||||
.min().orElse(0);
|
||||
}
|
||||
|
||||
/**
|
||||
* Pure decision function: examine everything pending for {@code lead} — reply targets and
|
||||
* tickets alike — and return what the loop should do.
|
||||
@@ -272,17 +357,35 @@ public final class ReplyPushLoop {
|
||||
* @param questionReminderCount the lowest nudge count among questions open for this lead
|
||||
*/
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
minIncidentNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with backend incidents as a fourth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount) {
|
||||
return decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, minUnmappedTargetNudgeCountFor(lead));
|
||||
}
|
||||
|
||||
/** As above, with unmapped backend targets as a fifth, independently bounded source. */
|
||||
Action decide(String lead, int replyReminderCount, int ticketReminderCount, int questionReminderCount,
|
||||
int incidentReminderCount, int unmappedTargetReminderCount) {
|
||||
boolean hasReplyWork = !pendingReplyTargetsFor(lead).isEmpty();
|
||||
boolean hasTicketWork = !pendingTicketIdsFor(lead).isEmpty();
|
||||
boolean hasQuestionWork = !pendingQuestionTurnIdsFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork) {
|
||||
boolean hasIncidentWork = !pendingIncidentKeysFor(lead).isEmpty();
|
||||
boolean hasUnmappedTargetWork = !pendingUnmappedTargetKeysFor(lead).isEmpty();
|
||||
if (!hasReplyWork && !hasTicketWork && !hasQuestionWork && !hasIncidentWork && !hasUnmappedTargetWork) {
|
||||
log.debug("push: nothing pending for lead {}, stopping reminder", lead);
|
||||
return Action.STOP;
|
||||
}
|
||||
boolean replyEligible = hasReplyWork && replyReminderCount < maxReminders;
|
||||
boolean ticketEligible = hasTicketWork && ticketReminderCount < maxReminders;
|
||||
boolean questionEligible = hasQuestionWork && questionReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible) {
|
||||
boolean incidentEligible = hasIncidentWork && incidentReminderCount < maxReminders;
|
||||
boolean unmappedTargetEligible = hasUnmappedTargetWork && unmappedTargetReminderCount < maxReminders;
|
||||
if (!replyEligible && !ticketEligible && !questionEligible && !incidentEligible && !unmappedTargetEligible) {
|
||||
log.debug("push: reminder cap ({}) reached for lead {} on every source with pending work, stopping",
|
||||
maxReminders, lead);
|
||||
countNudge("exhausted");
|
||||
@@ -302,6 +405,80 @@ public final class ReplyPushLoop {
|
||||
return Action.WAIT_BUSY;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve who to nudge about {@code target}, the way every public entry point below wants it:
|
||||
* {@link PrimaryRegistry#nudgeTargetFor}, but only after checking the delegating lead it names
|
||||
* is still actually there (fleetd #368).
|
||||
*
|
||||
* <p><strong>The bug this closes.</strong> {@code PrimaryRegistry.forgetDelegation} is wired to
|
||||
* exactly one event — a worker's release — because that is the only teardown the daemon already
|
||||
* observes for a session in this map. Nothing removes a binding when the LEAD half goes away: a
|
||||
* lead that is closed, crashes, or is relaunched leaves {@code leadByTarget} entries pointing at
|
||||
* a terminal that no longer exists. {@code nudgeTargetFor} falls back to the single known
|
||||
* primary only when the map holds nothing for {@code target} — a stale non-null entry beats the
|
||||
* fallback every time, which is exactly backwards: the fallback's own javadoc argues it is safe
|
||||
* precisely in the case a stale entry now hides.
|
||||
*
|
||||
* <p><strong>The fix.</strong> Before trusting a recorded delegation, probe the lead the same
|
||||
* way {@link #decide} already does every tick ({@code agents.status}) — cheap, since it is a
|
||||
* local herdr round-trip, and it is the same signal {@code AgentControl.paneByTerminal} already
|
||||
* trusts to tell a genuinely dead target from a live one. A lead that fails the probe is treated
|
||||
* as if it had never been recorded: the stale entry is forgotten (self-healing, exactly like
|
||||
* {@code AgentControl.paneByTerminal} already does on {@code agent_not_found}) and resolution is
|
||||
* retried, which now reaches the fallback {@code nudgeTargetFor} was built to reach — the same
|
||||
* empty-map state its javadoc already argues is correct.
|
||||
*
|
||||
* <p><strong>fleetd #368 review — only a positive "gone" reading forgets the binding.</strong>
|
||||
* The first version of this method treated <em>any</em> {@code RuntimeException} from the probe
|
||||
* as death, which is the #359 mistake repeated: a transient socket blip or a codec error on a
|
||||
* perfectly live lead would silently and permanently unbind it, with no re-record ever coming.
|
||||
* That is destructive on one bad reading, exactly what #359 shipped a two-reading guard to avoid
|
||||
* for the analogous lead-tab-liveness question. {@link #isLive} now matches
|
||||
* {@code AgentControl.agentCall}'s own narrower rule (see its {@code agent_not_found} check): only
|
||||
* that specific, affirmative "herdr has no such agent" signal counts as gone. Every other failure
|
||||
* — timeout, transport error, a decode error — is treated as still live and the binding is left
|
||||
* alone, because guessing wrong here is unrecoverable while guessing "live" merely costs one more
|
||||
* retry on the next tick, which {@link #decide} already tolerates.
|
||||
*/
|
||||
private Optional<String> resolveLiveLead(String target) {
|
||||
Optional<String> lead = primaryRegistry.nudgeTargetFor(target);
|
||||
if (lead.isEmpty() || isLive(lead.get())) {
|
||||
return lead;
|
||||
}
|
||||
log.debug("push: lead {} delegated to for {} is no longer live, forgetting the stale binding "
|
||||
+ "and falling back", lead.get(), target);
|
||||
primaryRegistry.forgetDelegation(target);
|
||||
return primaryRegistry.nudgeTargetFor(target);
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code lead} should still be trusted: {@code false} only when herdr affirmatively
|
||||
* reports the terminal gone ({@code agent_not_found}), never on a merely inconclusive failure.
|
||||
*
|
||||
* <p>fleetd #368 review: an earlier version returned {@code false} for any {@code RuntimeException},
|
||||
* which made a transient herdr hiccup on a live lead indistinguishable from the lead actually
|
||||
* being dead — and the caller's response to {@code false} ({@code forgetDelegation}) is
|
||||
* destructive and permanent. Narrowed to the one code {@code AgentControl.agentCall} itself
|
||||
* already treats as a genuine, resolvable absence (see its {@code agent_not_found} handling) —
|
||||
* every other {@code RuntimeException} is treated as "still live" and the binding survives to be
|
||||
* probed again next time, which costs nothing worse than one more retry.
|
||||
*/
|
||||
private boolean isLive(String lead) {
|
||||
try {
|
||||
agents.status(lead);
|
||||
return true;
|
||||
} catch (RuntimeException e) {
|
||||
boolean gone = e instanceof HerdrException he && "agent_not_found".equals(he.code());
|
||||
if (gone) {
|
||||
log.debug("push: lead {} no longer exists ({})", lead, e.toString());
|
||||
} else {
|
||||
log.debug("push: liveness check for lead {} was inconclusive ({}); treating as live "
|
||||
+ "rather than risk destroying a live binding", lead, e.toString());
|
||||
}
|
||||
return !gone;
|
||||
}
|
||||
}
|
||||
|
||||
// --- public entrypoints ----------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -312,7 +489,7 @@ public final class ReplyPushLoop {
|
||||
* backstop until a lead is recorded.
|
||||
*/
|
||||
public void onReplyQueued(String target) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}, skipping reminder", target);
|
||||
return;
|
||||
@@ -340,7 +517,7 @@ public final class ReplyPushLoop {
|
||||
* @param failed whether the ticket ended in a failure phase rather than {@code DONE}
|
||||
*/
|
||||
public void onTicketTerminal(String ticket, String target, boolean failed) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on ticket {} (target {}), skipping nudge",
|
||||
ticket, target);
|
||||
@@ -375,7 +552,7 @@ public final class ReplyPushLoop {
|
||||
* @param question the question text
|
||||
*/
|
||||
public void onQuestionOpened(String ticket, String target, String turnId, String question) {
|
||||
var lead = primaryRegistry.nudgeTargetFor(target);
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.debug("push: no lead is known to be waiting on {}'s question (turnId {}), skipping nudge",
|
||||
target, turnId);
|
||||
@@ -394,6 +571,48 @@ public final class ReplyPushLoop {
|
||||
pendingQuestions.remove(turnId);
|
||||
}
|
||||
|
||||
/**
|
||||
* Queue a one-shot backend credential outage notice for every distinct lead that owns an
|
||||
* affected worker. A successful injection records its {@code (incidentId, lead)} key, so a
|
||||
* repeat report never reminds that lead again.
|
||||
*/
|
||||
public void onBackendIncident(String incidentId, Collection<String> targets, String credential,
|
||||
Collection<String> profiles, int remainingCoolOffSeconds) {
|
||||
Map<String, List<String>> targetsByLead = new ConcurrentHashMap<>();
|
||||
for (String target : targets) {
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend incident {} has no known lead for target {}", incidentId, target);
|
||||
continue;
|
||||
}
|
||||
targetsByLead.computeIfAbsent(lead.get(), _ -> new ArrayList<>()).add(target);
|
||||
}
|
||||
List<String> profileNames = profiles.stream().sorted().toList();
|
||||
for (var entry : targetsByLead.entrySet()) {
|
||||
IncidentLead key = new IncidentLead(incidentId, entry.getKey());
|
||||
if (deliveredIncidents.contains(key)) continue;
|
||||
pendingIncidents.putIfAbsent(key, new PendingIncident(key, credential, profileNames,
|
||||
entry.getValue().stream().sorted().toList(), remainingCoolOffSeconds, 0));
|
||||
startOrCoalesce(entry.getKey());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Tell the owning lead that a classified backend error could not be tied to a credential.
|
||||
* Without an owning lead, emit a warning because no control can act on the target.
|
||||
*/
|
||||
public void onBackendTargetUnmapped(String target, String reason) {
|
||||
var lead = resolveLiveLead(target);
|
||||
if (lead.isEmpty()) {
|
||||
log.warn("push: backend target {} could not map to a credential: {}", target, reason);
|
||||
return;
|
||||
}
|
||||
UnmappedTargetLead key = new UnmappedTargetLead(target, reason, lead.get());
|
||||
if (deliveredUnmappedTargets.contains(key)) return;
|
||||
pendingUnmappedTargets.putIfAbsent(key, new PendingUnmappedTarget(key, 0));
|
||||
startOrCoalesce(lead.get());
|
||||
}
|
||||
|
||||
// --- the schedule ----------------------------------------------------------------------------
|
||||
|
||||
/** Start a reminder schedule for {@code lead}, or join the one already running. */
|
||||
@@ -425,18 +644,25 @@ public final class ReplyPushLoop {
|
||||
Set<String> repliesBefore = pendingReplyTargetsFor(lead);
|
||||
Set<String> ticketsBefore = pendingTicketIdsFor(lead);
|
||||
Set<String> questionsBefore = pendingQuestionTurnIdsFor(lead);
|
||||
Set<IncidentLead> incidentsBefore = pendingIncidentKeysFor(lead);
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore = pendingUnmappedTargetKeysFor(lead);
|
||||
int replyReminderCount = minReplyNudgeCountFor(lead);
|
||||
int ticketReminderCount = minTicketNudgeCountFor(lead);
|
||||
int questionReminderCount = minQuestionNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
int incidentReminderCount = minIncidentNudgeCountFor(lead);
|
||||
int unmappedTargetReminderCount = minUnmappedTargetNudgeCountFor(lead);
|
||||
var action = decide(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
switch (action) {
|
||||
case INJECT -> {
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount);
|
||||
injectNudge(lead, replyReminderCount, ticketReminderCount, questionReminderCount,
|
||||
incidentReminderCount, unmappedTargetReminderCount);
|
||||
scheduleNext(lead);
|
||||
}
|
||||
// Re-check after the configured backoff; the lead may become injectable soon.
|
||||
case WAIT_BUSY -> scheduleNext(lead);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore);
|
||||
case STOP -> stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
unmappedTargetsBefore);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -480,11 +706,25 @@ public final class ReplyPushLoop {
|
||||
* decision-to-release window reclaims the schedule slot exactly like a raced-in reply or ticket.
|
||||
*/
|
||||
void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore) {
|
||||
Set<String> questionsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, pendingIncidentKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore) {
|
||||
stopOrRestart(lead, repliesBefore, ticketsBefore, questionsBefore, incidentsBefore,
|
||||
pendingUnmappedTargetKeysFor(lead));
|
||||
}
|
||||
|
||||
private void stopOrRestart(String lead, Set<String> repliesBefore, Set<String> ticketsBefore,
|
||||
Set<String> questionsBefore, Set<IncidentLead> incidentsBefore,
|
||||
Set<UnmappedTargetLead> unmappedTargetsBefore) {
|
||||
activeLeads.remove(lead);
|
||||
boolean racedIn = pendingReplyTargetsFor(lead).stream().anyMatch(t -> !repliesBefore.contains(t))
|
||||
|| pendingTicketIdsFor(lead).stream().anyMatch(t -> !ticketsBefore.contains(t))
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t));
|
||||
|| pendingQuestionTurnIdsFor(lead).stream().anyMatch(t -> !questionsBefore.contains(t))
|
||||
|| pendingIncidentKeysFor(lead).stream().anyMatch(i -> !incidentsBefore.contains(i))
|
||||
|| pendingUnmappedTargetKeysFor(lead).stream().anyMatch(i -> !unmappedTargetsBefore.contains(i));
|
||||
if (racedIn && activeLeads.putIfAbsent(lead, Boolean.TRUE) == null) {
|
||||
log.debug("push: new work for lead {} raced the reminder loop's stop — restarting", lead);
|
||||
scheduleNext(lead);
|
||||
@@ -495,18 +735,22 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Send one combined nudge covering everything currently pending for {@code lead}. */
|
||||
private void injectNudge(String lead, int replyReminderCount, int ticketReminderCount,
|
||||
int questionReminderCount) {
|
||||
int questionReminderCount, int incidentReminderCount,
|
||||
int unmappedTargetReminderCount) {
|
||||
// Re-read rather than threading it down from decide(): a reply can drain, a ticket be
|
||||
// collected, or a question be answered (or another arrive), between the decision and the
|
||||
// injection.
|
||||
Set<String> replyTargets = pendingReplyTargetsFor(lead);
|
||||
List<PendingTicket> tickets = pendingTicketsFor(lead);
|
||||
List<PendingQuestion> questions = pendingQuestionsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty()) {
|
||||
List<PendingIncident> incidents = pendingIncidentsFor(lead);
|
||||
List<PendingUnmappedTarget> unmappedTargets = pendingUnmappedTargetsFor(lead);
|
||||
if (replyTargets.isEmpty() && tickets.isEmpty() && questions.isEmpty() && incidents.isEmpty()
|
||||
&& unmappedTargets.isEmpty()) {
|
||||
log.debug("push: pending work for lead {} drained before the nudge could be sent", lead);
|
||||
return;
|
||||
}
|
||||
String nudge = formatNudge(replyTargets, tickets, questions);
|
||||
String nudge = formatNudge(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
try {
|
||||
agents.send(lead, nudge);
|
||||
log.debug("push: nudge sent to lead {} (reply {}/{}, ticket {}/{}, question {}/{}; "
|
||||
@@ -514,7 +758,17 @@ public final class ReplyPushLoop {
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
questionReminderCount + 1, maxReminders,
|
||||
replyTargets.size(), tickets.size(), questions.size());
|
||||
countNudge("delivered");
|
||||
countNudge("sent");
|
||||
for (PendingIncident incident : incidents) {
|
||||
if (pendingIncidents.remove(incident.key(), incident)) {
|
||||
deliveredIncidents.add(incident.key());
|
||||
}
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
if (pendingUnmappedTargets.remove(unmappedTarget.key(), unmappedTarget)) {
|
||||
deliveredUnmappedTargets.add(unmappedTarget.key());
|
||||
}
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("push: failed to nudge lead {} (reply {}/{}, ticket {}/{}, question {}/{}): {}",
|
||||
lead, replyReminderCount + 1, maxReminders, ticketReminderCount + 1, maxReminders,
|
||||
@@ -526,12 +780,13 @@ public final class ReplyPushLoop {
|
||||
// item already at or over the cap keeps riding along in the text (still pending, still
|
||||
// named) but its extra bumps here are inert: decide() already treats it as ineligible once
|
||||
// its count reaches maxReminders.
|
||||
bumpNudgeCounts(replyTargets, tickets, questions);
|
||||
bumpNudgeCounts(replyTargets, tickets, questions, incidents, unmappedTargets);
|
||||
}
|
||||
|
||||
/** Record that every one of these items was just named in a sent (or attempted) nudge. */
|
||||
private void bumpNudgeCounts(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
for (String target : replyTargets) {
|
||||
pendingReplies.computeIfPresent(target, (t, e) -> new ReplyEntry(e.lead(), e.nudgeCount() + 1));
|
||||
}
|
||||
@@ -544,6 +799,14 @@ public final class ReplyPushLoop {
|
||||
new PendingQuestion(e.turnId(), e.ticket(), e.target(), e.lead(), e.question(),
|
||||
e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingIncident incident : incidents) {
|
||||
pendingIncidents.computeIfPresent(incident.key(), (id, e) -> new PendingIncident(e.key(),
|
||||
e.credential(), e.profiles(), e.targets(), e.remainingCoolOffSeconds(), e.nudgeCount() + 1));
|
||||
}
|
||||
for (PendingUnmappedTarget unmappedTarget : unmappedTargets) {
|
||||
pendingUnmappedTargets.computeIfPresent(unmappedTarget.key(), (id, e) ->
|
||||
new PendingUnmappedTarget(e.key(), e.nudgeCount() + 1));
|
||||
}
|
||||
}
|
||||
|
||||
/** Schedule the next tick on the scheduler thread pool. */
|
||||
@@ -556,7 +819,8 @@ public final class ReplyPushLoop {
|
||||
|
||||
/** Render everything pending for one lead as a single nudge line. */
|
||||
private static String formatNudge(Set<String> replyTargets, List<PendingTicket> tickets,
|
||||
List<PendingQuestion> questions) {
|
||||
List<PendingQuestion> questions, List<PendingIncident> incidents,
|
||||
List<PendingUnmappedTarget> unmappedTargets) {
|
||||
List<String> parts = new ArrayList<>();
|
||||
if (!replyTargets.isEmpty()) {
|
||||
parts.add(formatRepliesNudge(replyTargets));
|
||||
@@ -567,6 +831,12 @@ public final class ReplyPushLoop {
|
||||
if (!questions.isEmpty()) {
|
||||
parts.add(formatQuestionsNudge(questions));
|
||||
}
|
||||
if (!incidents.isEmpty()) {
|
||||
parts.addAll(incidents.stream().map(ReplyPushLoop::formatBackendIncidentNudge).toList());
|
||||
}
|
||||
if (!unmappedTargets.isEmpty()) {
|
||||
parts.addAll(unmappedTargets.stream().map(ReplyPushLoop::formatUnmappedTargetNudge).toList());
|
||||
}
|
||||
return String.join(" | ", parts);
|
||||
}
|
||||
|
||||
@@ -606,6 +876,16 @@ public final class ReplyPushLoop {
|
||||
return QUESTIONS_NUDGE_FORMAT.formatted(pending.size(), ids);
|
||||
}
|
||||
|
||||
private static String formatBackendIncidentNudge(PendingIncident incident) {
|
||||
return BACKEND_INCIDENT_NUDGE_FORMAT.formatted(incident.credential(), incident.remainingCoolOffSeconds(),
|
||||
String.join(", ", incident.profiles()), String.join(", ", incident.targets()));
|
||||
}
|
||||
|
||||
private static String formatUnmappedTargetNudge(PendingUnmappedTarget unmappedTarget) {
|
||||
return BACKEND_TARGET_UNMAPPED_NUDGE_FORMAT.formatted(unmappedTarget.key().target(),
|
||||
unmappedTarget.key().reason());
|
||||
}
|
||||
|
||||
// --- lifecycle -----------------------------------------------------------------------------
|
||||
|
||||
/**
|
||||
@@ -627,6 +907,10 @@ public final class ReplyPushLoop {
|
||||
pendingReplies.clear();
|
||||
pendingTickets.clear();
|
||||
pendingQuestions.clear();
|
||||
pendingIncidents.clear();
|
||||
deliveredIncidents.clear();
|
||||
pendingUnmappedTargets.clear();
|
||||
deliveredUnmappedTargets.clear();
|
||||
}
|
||||
|
||||
/** @see #stop() */
|
||||
|
||||
@@ -9,12 +9,19 @@ import java.util.concurrent.CompletableFuture;
|
||||
public final class TurnToken {
|
||||
private final String target;
|
||||
private final CompletableFuture<Rendezvous.Resolution> waiter;
|
||||
private final String injectedText;
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter) {
|
||||
this(target, waiter, null);
|
||||
}
|
||||
|
||||
public TurnToken(String target, CompletableFuture<Rendezvous.Resolution> waiter, String injectedText) {
|
||||
this.target = target;
|
||||
this.waiter = waiter;
|
||||
this.injectedText = injectedText;
|
||||
}
|
||||
|
||||
public String target() { return target; }
|
||||
public CompletableFuture<Rendezvous.Resolution> waiter() { return waiter; }
|
||||
public String injectedText() { return injectedText; }
|
||||
}
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
package dev.ltms.fleet.peer;
|
||||
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
import java.util.Set;
|
||||
@@ -143,9 +145,160 @@ public interface PeerLauncher {
|
||||
|
||||
/**
|
||||
* The profile a no-argument {@link #spawn(SpawnRequest)} uses, or {@code null} if none is configured.
|
||||
*
|
||||
* <p>fleetd #425: for an implementation with role pools (a no-argument spawn is read as {@link
|
||||
* MemberRole#DEV}, see {@link SpawnRequest}), this must be the profile a live spawn of that role
|
||||
* would actually be placed on right now, not a value captured once at startup — a caller such as
|
||||
* {@code fleet_profiles} relies on this to report a live, not frozen, fact.
|
||||
*/
|
||||
String defaultProfile();
|
||||
|
||||
/**
|
||||
* The profile an unqualified spawn of {@code role} would resolve to right now — the role-aware,
|
||||
* live counterpart of {@link #defaultProfile()} (fleetd #425).
|
||||
*
|
||||
* <p>A caller that must provision something profile-specific (working directory, parity overlay
|
||||
* files) <em>before</em> the actual spawn — {@code SessionManager.acquireWithWorktree} is the one
|
||||
* that exists today — needs the exact profile that spawn will use, for the caller's real role,
|
||||
* not a role-agnostic guess. Calling {@link #defaultProfile()} for that purpose reads {@code
|
||||
* MemberRole#DEV}'s answer regardless of the caller's actual role, which is wrong for any other
|
||||
* role and can provision for a profile the spawn never lands on.
|
||||
*
|
||||
* <p>Default implementation returns {@link #defaultProfile()}, ignoring {@code role} — the right
|
||||
* answer for a launcher with no role-pool concept of its own (e.g. a single {@code
|
||||
* HerdrPeerLauncher} adapter, which is never reached this way in production: {@code
|
||||
* CompositePeerLauncher} always fronts it and resolves roles itself).
|
||||
*
|
||||
* <p>fleetd #453: this default is deliberately <em>not</em> abstract — unlike {@link
|
||||
* #spawn(SpawnRequest, PlacementDecision)} (fleetd #450), there is no live defect in inheriting
|
||||
* it today, and the only current single-adapter implementer ({@code HerdrPeerLauncher}) is
|
||||
* correct to do so. But it stays correct only as long as that holds: <strong>if a launcher ever
|
||||
* routes more than one profile per role, it MUST override this method</strong>, or every role
|
||||
* silently resolves to {@link #defaultProfile()} with no error and no log line. {@code
|
||||
* HerdrPeerLauncher.spawn(SpawnRequest, PlacementDecision)} — the override in {@code
|
||||
* dev.ltms.fleet.member}, not the declaration below — names this method and {@link #place}
|
||||
* explicitly as "unoverridden here" for exactly this reason. Read it before adding role-pool
|
||||
* routing to any {@code HerdrPeerLauncher} subclass.
|
||||
*
|
||||
* <p>Who is forced to read which paragraph, because it is not symmetric. A new class that
|
||||
* implements this interface directly must write a body for {@link #spawn(SpawnRequest,
|
||||
* PlacementDecision)}, which is abstract here, so it lands on this javadoc. A subclass of
|
||||
* {@code HerdrPeerLauncher} does not: that class already implements the method, and the
|
||||
* subclass inherits the body. So for a subclass this paragraph is advice, not a gate.
|
||||
*/
|
||||
default String defaultProfileFor(MemberRole role) {
|
||||
return defaultProfile();
|
||||
}
|
||||
|
||||
/**
|
||||
* The profile an <em>unqualified</em> spawn of {@code role} would actually be routed to right
|
||||
* now — the same candidate list, the same {@code quarantined}/{@code coolingOff}/{@code
|
||||
* modelOff} filtering, and the same {@code PlacementPolicy} that {@link #spawn} itself
|
||||
* consults for a blank-profile request (fleetd #425 rework).
|
||||
*
|
||||
* <p>This is <em>not</em> {@link #defaultProfileFor}: that method answers "what is first in
|
||||
* {@code role}'s pool", blind to quarantine, cool-off, and the model on/off gate — the right
|
||||
* answer for a role-agnostic, best-effort report ({@code fleet_profiles}' {@code "default"}
|
||||
* field), but the wrong one for a caller that needs the profile a spawn will actually land on.
|
||||
* A quarantined or model-off pool-first profile makes {@link #defaultProfileFor} return a name
|
||||
* an unqualified spawn will never be routed to.
|
||||
*
|
||||
* <p>Just the resolved name, not the full {@link PlacementDecision} — a caller that only wants
|
||||
* to know the answer (a status report, a log line) can call this; a caller that will later
|
||||
* <em>act</em> on the answer by spawning — provisioning a worktree for a specific profile
|
||||
* before the peer exists is the one that matters — must call {@link #place} and carry the
|
||||
* {@link PlacementDecision} itself through to {@link #spawn(SpawnRequest, PlacementDecision)}
|
||||
* instead of calling this method and feeding the string back in as an explicit profile. Doing
|
||||
* that re-enters {@link #spawn(SpawnRequest)}'s explicit-profile branch, which disagrees with
|
||||
* the routing branch on purpose about what an excluded profile means: the routing branch (and
|
||||
* {@link #place}) falls through a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0
|
||||
* profile to the next candidate, while the explicit branch refuses outright — correct for an
|
||||
* operator who named that profile on purpose, wrong for a name that only ever came from placement
|
||||
* itself. That accidental refusal is exactly the regression fleetd #425 rework round 2 fixes:
|
||||
* the default implementation below delegates to {@link #place}, so the two can never drift apart,
|
||||
* but a caller that resolves through this method alone and spawns separately can still recreate
|
||||
* the round-1 defect for itself. (Before fleetd #435, this accident was also reachable through
|
||||
* {@code maxLoad} specifically, because {@code FixedPlacementPolicy} — the default policy — did
|
||||
* not evaluate it at all for automatic selection; #435 closed that gap, so a placement decision
|
||||
* can no longer be at cap in the first place. The refusal-vs-fall-through disagreement above is
|
||||
* the part that was never about {@code maxLoad} and is still real.)
|
||||
*
|
||||
* @throws RuntimeException (implementation-specific, typically a placement exception) if no
|
||||
* candidate in {@code role}'s pool is currently placeable
|
||||
*/
|
||||
default String routedProfileFor(MemberRole role) {
|
||||
return place(role).profile();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve, <em>without spawning</em>, the {@link PlacementDecision} an unqualified spawn of
|
||||
* {@code role} would make right now — the same candidate list, the same {@code
|
||||
* quarantined}/{@code coolingOff}/{@code modelOff} filtering, and the same {@code
|
||||
* PlacementPolicy} {@link #spawn(SpawnRequest)}'s blank-profile branch itself consults (fleetd
|
||||
* #425 rework).
|
||||
*
|
||||
* <p>Pair this with {@link #spawn(SpawnRequest, PlacementDecision)}, never with {@link
|
||||
* #spawn(SpawnRequest)} fed the decision's profile as an explicit name — see {@link
|
||||
* PlacementDecision}'s own javadoc for why that second form regressed.
|
||||
*
|
||||
* <p>Default implementation wraps {@link #defaultProfile()}, ignoring {@code role} and every
|
||||
* placement condition — the right answer for a launcher with no pool or placement-policy
|
||||
* concept of its own, matching {@link #defaultProfileFor}'s own default.
|
||||
*
|
||||
* <p>fleetd #453: same reasoning as {@link #defaultProfileFor}'s own #453 note — this default
|
||||
* is deliberately not abstract (no live defect today, correct for the sole single-adapter
|
||||
* implementer), but <strong>a launcher that ever routes more than one profile per role MUST
|
||||
* override this method too</strong>, or placement silently ignores {@code role} for it. See
|
||||
* {@code HerdrPeerLauncher.spawn(SpawnRequest, PlacementDecision)}'s javadoc, which names this
|
||||
* method as "unoverridden here" and why that is correct only for a single-profile adapter.
|
||||
*
|
||||
* @throws RuntimeException (implementation-specific, typically a placement exception) if no
|
||||
* candidate in {@code role}'s pool is currently placeable
|
||||
*/
|
||||
default PlacementDecision place(MemberRole role) {
|
||||
return new PlacementDecision(defaultProfile());
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn against an already-resolved {@link PlacementDecision} from {@link #place}, honoring it
|
||||
* completely: none of the conditions {@link #place} already applied — quarantine, cooling off,
|
||||
* {@code maxLoad} (evaluated by every placement policy including the default {@code fixed},
|
||||
* since fleetd #435), model-off — are re-evaluated here; {@code decision} already reflects them.
|
||||
* This is not skipping a check {@code place} left undone; it is not repeating one {@code place}
|
||||
* already did, and not re-opening the window between that decision and this spawn in which the
|
||||
* underlying state could otherwise move. This is what lets a resolve-then-spawn caller
|
||||
* ({@code SessionManager.acquireWithWorktree}, which must know the profile before it can
|
||||
* provision a worktree for it) and a plain blank-profile {@link #spawn(SpawnRequest)} caller
|
||||
* land on the exact same outcome for the exact same placement state (fleetd #425 rework,
|
||||
* round 2).
|
||||
*
|
||||
* <p>{@code req}'s own {@link SpawnRequest#profileName()} is ignored in favor of {@code
|
||||
* decision.profile()} — the caller is expected to have built {@code req} with a blank or
|
||||
* matching profile; passing a request that names a <em>different</em>, explicit profile than
|
||||
* the decision it is paired with is a caller bug this method does not attempt to detect.
|
||||
*
|
||||
* <p>No default implementation (fleetd #450): the two correct bodies disagree on purpose, so an
|
||||
* implementer must choose one rather than silently inherit whichever this interface happened to
|
||||
* provide. An implementer with no placement concept of its own — spawns a single profile, e.g.
|
||||
* {@code HerdrPeerLauncher} — should delegate to {@link #spawn(SpawnRequest)} with the decision's
|
||||
* profile named explicitly, since there is no separate routing path to honor there: the
|
||||
* explicit-profile branch it re-enters and the routing branch {@link #place} would have used are
|
||||
* the same thing. <strong>A launcher that routes across more than one profile — the way {@code
|
||||
* CompositePeerLauncher} routes across every configured adapter — MUST NOT re-enter {@link
|
||||
* #spawn(SpawnRequest)}.</strong> Doing so re-applies that single-argument method's
|
||||
* explicit-profile checks ({@code enforceNotQuarantined}, {@code enforceNotCoolingOff}, {@code
|
||||
* enforceMaxLoad}, {@code enforceModelEnabled} in {@code CompositePeerLauncher}), which can
|
||||
* refuse the very profile {@link #place} just chose, if the underlying placement state moved in
|
||||
* the window between the {@link #place} call and this one — the exact window this method and
|
||||
* {@link PlacementDecision} exist to close (fleetd #444). Before #450 this was a {@code default}
|
||||
* method that only {@code CompositePeerLauncher} overrode; a future placement-doing launcher
|
||||
* could have inherited the re-entering body silently and never known. Making it abstract turns
|
||||
* that silent inheritance into a compile error.
|
||||
*
|
||||
* @throws IllegalArgumentException if the decision names an unknown profile
|
||||
*/
|
||||
PeerHandle spawn(SpawnRequest req, PlacementDecision decision);
|
||||
|
||||
/**
|
||||
* Resolve the effective working directory for a spawn {@code req} without actually spawning.
|
||||
* Resolution order: requestedCwd → profile cwd → callerCwd → daemon cwd.
|
||||
@@ -192,4 +345,66 @@ public interface PeerLauncher {
|
||||
* @return {@code true} when a reset was sent and its status transition must settle before reuse
|
||||
*/
|
||||
boolean clearContext(String id);
|
||||
|
||||
/**
|
||||
* Model ids the operator has currently turned off in the central {@code models.allow:} list
|
||||
* (fleetd #422) — empty for a launcher with nothing to gate against. {@code fleet_profiles}/
|
||||
* {@code GET /profiles} (via {@code FleetMcp.profilesView}) call this to report which models
|
||||
* are off, and MUST read this exact accessor rather than deriving their own answer: the fleetd
|
||||
* #404 lesson is that a status field reading a different source than the behaviour it describes
|
||||
* can drift from what the gate ({@code CompositePeerLauncher.enforceModelEnabled} and its
|
||||
* candidate filter) actually enforces. A default of {@code Set.of()} keeps every other {@link
|
||||
* PeerLauncher} implementer (the herdr adapters, and the two test-fake implementers) unchanged.
|
||||
*
|
||||
* <p>fleetd #422 follow-up: this alone cannot tell "no {@code models:} block at all" from "a
|
||||
* {@code models:} block where nothing is currently off" — both report an empty set here. Delegates
|
||||
* to {@link #modelGateState()} so the two facts always come from the one read {@link
|
||||
* #modelGateState()}'s implementer makes; do not override this method separately from that one.
|
||||
*/
|
||||
default Set<String> disabledModels() {
|
||||
return modelGateState().off();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the central {@code models.allow:} gate (fleetd #422) is armed at all, together with
|
||||
* which model ids are currently off — fleetd #422 follow-up. {@link #disabledModels()} alone
|
||||
* cannot distinguish two states that both report an empty set: a host with no {@code models:}
|
||||
* block (nothing is gated, and nothing can be) and a host WITH a {@code models:} block where
|
||||
* nothing is currently turned off (the gate is armed and reporting zero). This method exists so
|
||||
* a caller — the startup log, {@code fleet_profiles}/{@code GET /profiles} — can tell the two
|
||||
* apart, the same reason {@code CompletionResolver.UnsetMeaning} exists: an accessor that can
|
||||
* legitimately report "empty" must never let a caller guess why.
|
||||
*
|
||||
* <p>Default {@link ModelGateState#notConfigured()} — every launcher without a {@code models:}
|
||||
* block to read from (the herdr adapters, and the two test-fake implementers), matching {@link
|
||||
* #disabledModels()}'s own default of an empty set.
|
||||
*/
|
||||
default ModelGateState modelGateState() {
|
||||
return ModelGateState.notConfigured();
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #422 follow-up: the result of {@link #modelGateState()} — see that method's javadoc
|
||||
* for why "armed" and "off" must be reported together from one read rather than as two
|
||||
* separately-derived facts that a reload landing between them could make disagree.
|
||||
*
|
||||
* @param configured {@code true} when a {@code models:} block exists at all (armed), regardless
|
||||
* of whether anything in it is currently turned off; {@code false} when there
|
||||
* is no block to gate against
|
||||
* @param off the model ids currently turned off; always empty when {@code configured} is
|
||||
* {@code false}
|
||||
*/
|
||||
record ModelGateState(boolean configured, Set<String> off) {
|
||||
public ModelGateState {
|
||||
off = Set.copyOf(off);
|
||||
}
|
||||
|
||||
public static ModelGateState notConfigured() {
|
||||
return new ModelGateState(false, Set.of());
|
||||
}
|
||||
|
||||
public static ModelGateState armed(Set<String> off) {
|
||||
return new ModelGateState(true, off);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -29,4 +29,9 @@ public record SpawnRequest(String profileName, String requestedCwd, String calle
|
||||
String sessionName, String resumeSessionId) {
|
||||
this(profileName, requestedCwd, callerCwd, sessionName, resumeSessionId, null);
|
||||
}
|
||||
|
||||
/** Return a copy of this request with {@code profileName} replaced by {@code profile}. */
|
||||
public SpawnRequest withProfile(String profile) {
|
||||
return new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, role);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,200 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashSet;
|
||||
import java.util.List;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
import java.util.function.LongSupplier;
|
||||
|
||||
/**
|
||||
* Fleetd #201 / #227: correlates classified backend errors (credential outage or provider 5xx,
|
||||
* produced elsewhere by the classifier that turns raw pane text into a typed event — this class
|
||||
* knows nothing about pane text, profiles, sessions, launchers, or leads) and decides, purely from
|
||||
* counts and timing, when a credential's backend is out.
|
||||
*
|
||||
* <p>Two classified errors on the <b>same credential</b> — never the profile name, never the error
|
||||
* text, see {@code FleetConfig.Profile#effectiveCredentialId()} like {@link BackendQuarantine} —
|
||||
* <b>from two distinct targets</b> inside a 60-second window is treated as an outage: {@link
|
||||
* #record} then returns an {@link Incident} and starts a 60-second cool-off for that credential. The
|
||||
* threshold counts distinct targets, not raw events, on purpose: the classifier is a heuristic and a
|
||||
* valid member report can quote an {@code API Error:} line, so one target repeating that line twice
|
||||
* must never remove fleet capacity by itself. Two different targets independently producing a
|
||||
* classified error is far less likely to be a coincidence, and a real backend outage hits every
|
||||
* target on that credential anyway, so this loses nothing against the case being protected against.
|
||||
* This is deliberately a different store from {@link BackendQuarantine}: that one holds a
|
||||
* 1800-second exhaustion cooldown for a spent credential, and reusing it here would both use the
|
||||
* wrong duration and report "backend exhausted" for what is a short transient fault.
|
||||
*
|
||||
* <p>Correlation and cool-off live in <b>one class</b> so that crossing the threshold and setting
|
||||
* the cool-off deadline happen as a single atomic update: every {@link #record} call goes through
|
||||
* {@link ConcurrentHashMap#compute}, which serializes on the credential's map bucket, so a
|
||||
* concurrent second and third event on the same credential can never both observe "about to cross"
|
||||
* and both mint an incident.
|
||||
*
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@link BackendQuarantine}), never read inline, so the window and cool-off are testable without a
|
||||
* real sleep.
|
||||
*/
|
||||
public final class BackendOutagePolicy {
|
||||
|
||||
/** Distinct targets a credential needs a classified error from to declare an outage. */
|
||||
public static final int THRESHOLD = 2;
|
||||
|
||||
/** How long a credential's evidence stays fresh, inclusive of both endpoints. */
|
||||
public static final long WINDOW_NANOS = 60_000_000_000L;
|
||||
|
||||
/** How long a credential sits out once the threshold is crossed. */
|
||||
public static final long COOLOFF_NANOS = 60_000_000_000L;
|
||||
|
||||
private final ConcurrentHashMap<String, CredentialState> states = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final AtomicLong incidentSequence = new AtomicLong();
|
||||
|
||||
public BackendOutagePolicy(LongSupplier nowNanos) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
}
|
||||
|
||||
/**
|
||||
* Records one classified backend error for {@code credentialId} against {@code target} (e.g. a
|
||||
* session or worker id — this class never interprets it, only collects it for the incident's
|
||||
* affected-targets list, and counts distinct targets toward the threshold) with {@code reason}
|
||||
* (the classifier's free-text reason, kept per event for whoever renders the eventual notice —
|
||||
* every event's reason is kept even when the same target repeats, so {@link Incident#reasons()}
|
||||
* can be longer than {@link Incident#evidenceCount()}).
|
||||
*
|
||||
* <p>Returns a populated {@link Incident} only at the exact moment a <b>second distinct target</b>
|
||||
* is seen for this credential inside the window — never before, and never again while the
|
||||
* resulting cool-off is active. A repeat error from a target already counted does not advance the
|
||||
* threshold. While a credential is cooling off, a fresh error is ignored outright: it neither
|
||||
* extends the deadline nor produces another incident. Once the cool-off has elapsed, the next
|
||||
* error clears the old evidence and starts a brand-new window — two fresh, distinct targets are
|
||||
* required to rearm.
|
||||
*/
|
||||
public Optional<Incident> record(String credentialId, String target, String reason) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
Objects.requireNonNull(target, "target");
|
||||
Objects.requireNonNull(reason, "reason");
|
||||
long now = nowNanos.getAsLong();
|
||||
|
||||
AtomicReference<Incident> minted = new AtomicReference<>();
|
||||
states.compute(credentialId, (_, existing) -> {
|
||||
CredentialState state = existing;
|
||||
|
||||
if (state != null && state.inCoolOff()) {
|
||||
if (now < state.coolOffUntilNanos) {
|
||||
return state; // still cooling off: no extension, no incident
|
||||
}
|
||||
state = null; // cool-off elapsed: evidence is cleared, rearm from scratch
|
||||
}
|
||||
|
||||
if (state == null || now - state.firstEventNanos > WINDOW_NANOS) {
|
||||
return CredentialState.first(now, target, reason);
|
||||
}
|
||||
|
||||
CredentialState advanced = state.withAdditionalEvidence(target, reason);
|
||||
if (advanced.evidenceCount() < THRESHOLD) {
|
||||
return advanced;
|
||||
}
|
||||
|
||||
long coolOffUntilNanos = now + COOLOFF_NANOS;
|
||||
minted.set(new Incident(
|
||||
"outage-" + credentialId + "-" + incidentSequence.incrementAndGet(),
|
||||
credentialId,
|
||||
Set.copyOf(advanced.targets),
|
||||
advanced.evidenceCount(),
|
||||
toSecondsRoundedUp(WINDOW_NANOS),
|
||||
toSecondsRoundedUp(COOLOFF_NANOS),
|
||||
List.copyOf(advanced.reasons)));
|
||||
return advanced.enteringCoolOff(coolOffUntilNanos);
|
||||
});
|
||||
|
||||
return Optional.ofNullable(minted.get());
|
||||
}
|
||||
|
||||
/** Seconds left on {@code credentialId}'s cool-off, or empty when it is not cooling off. */
|
||||
public OptionalLong remainingCoolOffSeconds(String credentialId) {
|
||||
Objects.requireNonNull(credentialId, "credentialId");
|
||||
CredentialState state = states.get(credentialId);
|
||||
if (state == null || !state.inCoolOff()) {
|
||||
return OptionalLong.empty();
|
||||
}
|
||||
long remaining = state.coolOffUntilNanos - nowNanos.getAsLong();
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
return (nanos + 999_999_999L) / 1_000_000_000L;
|
||||
}
|
||||
|
||||
/**
|
||||
* One credential crossing the outage threshold. {@code remainingCoolOffSeconds} is the cool-off
|
||||
* length as observed at the moment of minting — this incident is only ever produced right as the
|
||||
* cool-off starts, so it is always the full {@link #COOLOFF_NANOS} rounded up.
|
||||
*
|
||||
* <p>{@code evidenceCount} is {@code targets.size()} — the threshold is on distinct targets, not
|
||||
* raw events — while {@code reasons} keeps every event's reason, including repeats from a target
|
||||
* already counted. The two are deliberately different lengths: a single target hammering the same
|
||||
* classified error never grows {@code evidenceCount} past 1, but each occurrence still lands in
|
||||
* {@code reasons} for whoever renders the notice.
|
||||
*/
|
||||
public record Incident(
|
||||
String id,
|
||||
String credentialId,
|
||||
Set<String> targets,
|
||||
int evidenceCount,
|
||||
long windowSeconds,
|
||||
long remainingCoolOffSeconds,
|
||||
List<String> reasons) {
|
||||
}
|
||||
|
||||
/** Evidence accumulated for one credential since its window opened, or its active cool-off. */
|
||||
private static final class CredentialState {
|
||||
final long firstEventNanos;
|
||||
final Set<String> targets;
|
||||
final List<String> reasons;
|
||||
final long coolOffUntilNanos; // 0 means "not cooling off"
|
||||
|
||||
private CredentialState(long firstEventNanos, Set<String> targets, List<String> reasons,
|
||||
long coolOffUntilNanos) {
|
||||
this.firstEventNanos = firstEventNanos;
|
||||
this.targets = targets;
|
||||
this.reasons = reasons;
|
||||
this.coolOffUntilNanos = coolOffUntilNanos;
|
||||
}
|
||||
|
||||
static CredentialState first(long nowNanos, String target, String reason) {
|
||||
Set<String> targets = new LinkedHashSet<>();
|
||||
targets.add(target);
|
||||
List<String> reasons = new ArrayList<>();
|
||||
reasons.add(reason);
|
||||
return new CredentialState(nowNanos, targets, reasons, 0L);
|
||||
}
|
||||
|
||||
CredentialState withAdditionalEvidence(String target, String reason) {
|
||||
Set<String> newTargets = new LinkedHashSet<>(targets);
|
||||
newTargets.add(target);
|
||||
List<String> newReasons = new ArrayList<>(reasons);
|
||||
newReasons.add(reason);
|
||||
return new CredentialState(firstEventNanos, newTargets, newReasons, 0L);
|
||||
}
|
||||
|
||||
CredentialState enteringCoolOff(long coolOffUntilNanos) {
|
||||
return new CredentialState(firstEventNanos, targets, reasons, coolOffUntilNanos);
|
||||
}
|
||||
|
||||
/** Distinct targets seen so far — the threshold counts this, never {@code reasons.size()}. */
|
||||
int evidenceCount() {
|
||||
return targets.size();
|
||||
}
|
||||
|
||||
boolean inCoolOff() {
|
||||
return coolOffUntilNanos > 0;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -3,6 +3,7 @@ package dev.ltms.fleet.placement;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.Map;
|
||||
import java.util.Objects;
|
||||
import java.util.Optional;
|
||||
import java.util.OptionalLong;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -21,46 +22,158 @@ import java.util.function.LongSupplier;
|
||||
* <p>The clock is injected ({@link LongSupplier}, conventionally {@code System::nanoTime} like
|
||||
* {@code FleetHealthMonitor}), never read inline, so a quarantine's expiry is testable without a
|
||||
* real sleep.
|
||||
*
|
||||
* <h2>Escalation (fleetd #466)</h2>
|
||||
* A flat cooldown does not fit every exhaustion. A backend that reports "out of capacity for the
|
||||
* rest of the hour" recovers in one cooldown; a weekly subscription limit does not — it keeps
|
||||
* reporting exhausted on every attempt made before the window resets, so a flat 30-minute cooldown
|
||||
* (the default {@code cooldownNanos}) means roughly 336 pointless spawn attempts across a week, one
|
||||
* every cooldown.
|
||||
*
|
||||
* <p><strong>This class only ever sees the exhaustion signal.</strong> Its only production caller is
|
||||
* {@code Fleetd.exhaustionSink}, wired to fire on a {@code BACKEND_EXHAUSTED} classification alone.
|
||||
* The daemon's other outage state — a credential "cooling off" after repeated non-exhaustion
|
||||
* backend errors (an HTTP 5xx storm, say) — is a separate mechanism, {@code BackendOutagePolicy},
|
||||
* with its own short fixed 60s cooldown and no repeat tracking. The two are never merged: escalating
|
||||
* on a cooling-off signal would turn a transient 5xx storm into a multi-hour backoff, which is
|
||||
* exactly the failure this ticket is not asking for. Confirmed by reading every call site of
|
||||
* {@link #quarantine} — {@code BackendOutagePolicy} has its own {@code coolOff} method and never
|
||||
* calls this one.
|
||||
*
|
||||
* <p><strong>Mechanism</strong> — the {@link #withEscalation} constructors track, per credential, how
|
||||
* many times in a row {@link #quarantine} has been called without an intervening "quiet" gap.
|
||||
* Each call computes {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at
|
||||
* {@code maxCooldownNanos}. A call counts as a continuation of the same streak — {@code repeatCount}
|
||||
* increments — when it arrives no more than one base {@code cooldownNanos} after the previous
|
||||
* quarantine's deadline (this covers both "still quarantined" and "quarantine just expired and it
|
||||
* was exhausted again immediately"); otherwise the streak resets and this call is treated as a fresh
|
||||
* first occurrence at the base cooldown.
|
||||
*
|
||||
* <p><strong>Reset, honestly stated.</strong> The ideal reset signal is "the cooldown expired and the
|
||||
* next attempt succeeded" — but nothing in this codebase reports a spawn success back to this class
|
||||
* (checked: {@code SessionManager} and {@code CompositePeerLauncher} never call any method here
|
||||
* except {@link #quarantine}/{@link #isQuarantined}/{@link #remainingSeconds}, none of which is a
|
||||
* success hook). Lacking that signal, the reset used here is a time-based proxy: a base-cooldown's
|
||||
* worth of quiet — no exhaustion report for that credential — since the last quarantine ended. It is
|
||||
* not proof the credential started working again, only the best available evidence without adding an
|
||||
* active probe, which is out of scope by the operator's own design constraint (no automatic probing
|
||||
* of a limited backend).
|
||||
*
|
||||
* <p><strong>Ceiling.</strong> {@code maxCooldownNanos} bounds the growth — an unbounded backoff is a
|
||||
* permanent, unrecoverable-without-a-restart outage, which would be worse than the flat-rate bug this
|
||||
* escalation fixes. {@link #withEscalation(LongSupplier, long)} defaults the ceiling to
|
||||
* {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown (12x the 1800s default ≈ 6 hours), so a
|
||||
* chronically exhausted credential still gets re-tried roughly every 6 hours instead of every 30
|
||||
* minutes — about a dozen attempts a week instead of ~336.
|
||||
*
|
||||
* <p><strong>Backward compatibility.</strong> The original two-argument {@link #BackendQuarantine(
|
||||
* LongSupplier, long)} constructor is unchanged in behaviour: it is exactly {@code
|
||||
* withEscalation}'s mechanism with {@code backoffMultiplier = 1.0} and {@code maxCooldownNanos =
|
||||
* cooldownNanos}, which collapses the formula back to the original flat {@code now + cooldownNanos}
|
||||
* on every call regardless of history. Every existing call site (roughly 20 across the test suite,
|
||||
* plus {@link #none()}) keeps its current shape and behaviour unchanged.
|
||||
*/
|
||||
public final class BackendQuarantine {
|
||||
|
||||
private final ConcurrentHashMap<String, Long> quarantinedUntilNanos = new ConcurrentHashMap<>();
|
||||
/** Default growth per consecutive exhaustion streak — see the class doc's Mechanism section. */
|
||||
static final double DEFAULT_BACKOFF_MULTIPLIER = 2.0;
|
||||
/** Default ceiling, expressed as a multiple of the base cooldown — see the class doc's Ceiling section. */
|
||||
static final long DEFAULT_MAX_COOLDOWN_MULTIPLE = 12;
|
||||
|
||||
private final ConcurrentHashMap<String, QuarantineState> quarantines = new ConcurrentHashMap<>();
|
||||
private final LongSupplier nowNanos;
|
||||
private final long cooldownNanos;
|
||||
private final double backoffMultiplier;
|
||||
private final long maxCooldownNanos;
|
||||
/** True only for {@link #none()}. See {@link #quarantine} for why this exists. */
|
||||
private final boolean inert;
|
||||
|
||||
/** How many consecutive exhaustion reports a credential is on, and when the resulting cooldown ends. */
|
||||
private record QuarantineState(int repeatCount, long deadlineNanos) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Flat cooldown, unchanged from before fleetd #466 — every {@link #quarantine} call blocks the
|
||||
* credential for exactly {@code cooldownNanos}, regardless of how many times it was called
|
||||
* before. Equivalent to {@link #withEscalation} with no growth ({@code backoffMultiplier = 1.0})
|
||||
* and a ceiling equal to the base cooldown, so it degrades to the identical {@code now +
|
||||
* cooldownNanos} formula every call. Kept for the existing call sites that want a fixed cooldown
|
||||
* (and for tests exercising the fixed-cooldown shape in isolation); production wiring uses
|
||||
* {@link #withEscalation} instead.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos how long a fresh {@link #quarantine} call blocks the credential for;
|
||||
* must be positive
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, false);
|
||||
this(nowNanos, cooldownNanos, 1.0, cooldownNanos, false);
|
||||
}
|
||||
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, boolean inert) {
|
||||
/**
|
||||
* Escalating cooldown (fleetd #466) — see the class doc's Mechanism/Reset/Ceiling sections.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be
|
||||
* positive
|
||||
* @param backoffMultiplier growth per consecutive exhaustion; must be {@code >= 1.0} ({@code 1.0}
|
||||
* disables growth and is exactly the flat two-argument constructor)
|
||||
* @param maxCooldownNanos ceiling on the escalated cooldown; must be {@code >= cooldownNanos}
|
||||
*/
|
||||
public BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
|
||||
long maxCooldownNanos) {
|
||||
this(nowNanos, cooldownNanos, backoffMultiplier, maxCooldownNanos, false);
|
||||
}
|
||||
|
||||
private BackendQuarantine(LongSupplier nowNanos, long cooldownNanos, double backoffMultiplier,
|
||||
long maxCooldownNanos, boolean inert) {
|
||||
this.nowNanos = Objects.requireNonNull(nowNanos, "nowNanos");
|
||||
if (cooldownNanos <= 0) {
|
||||
throw new IllegalArgumentException("cooldownNanos must be positive: " + cooldownNanos);
|
||||
}
|
||||
if (backoffMultiplier < 1.0) {
|
||||
throw new IllegalArgumentException("backoffMultiplier must be >= 1.0: " + backoffMultiplier);
|
||||
}
|
||||
if (maxCooldownNanos < cooldownNanos) {
|
||||
throw new IllegalArgumentException(
|
||||
"maxCooldownNanos must be >= cooldownNanos: " + maxCooldownNanos + " < " + cooldownNanos);
|
||||
}
|
||||
this.cooldownNanos = cooldownNanos;
|
||||
this.backoffMultiplier = backoffMultiplier;
|
||||
this.maxCooldownNanos = maxCooldownNanos;
|
||||
this.inert = inert;
|
||||
}
|
||||
|
||||
/**
|
||||
* Escalating cooldown with the fleetd #466 default shape: cooldown doubles
|
||||
* ({@value #DEFAULT_BACKOFF_MULTIPLIER}x) per consecutive exhaustion streak, capped at
|
||||
* {@value #DEFAULT_MAX_COOLDOWN_MULTIPLE}x the base cooldown. This is what production wiring
|
||||
* ({@code Fleetd.main}) uses.
|
||||
*
|
||||
* @param nowNanos monotonic clock, injected for testability
|
||||
* @param cooldownNanos base cooldown, applied to a fresh (non-streak) exhaustion; must be positive
|
||||
*/
|
||||
public static BackendQuarantine withEscalation(LongSupplier nowNanos, long cooldownNanos) {
|
||||
return new BackendQuarantine(nowNanos, cooldownNanos, DEFAULT_BACKOFF_MULTIPLIER,
|
||||
cooldownNanos * DEFAULT_MAX_COOLDOWN_MULTIPLE, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Inert quarantine — {@link #quarantine} does nothing on this instance, so nothing is ever
|
||||
* quarantined. The explicit stand-in a caller (or a test not exercising this feature) passes
|
||||
* instead of a defaulting overload, exactly like {@code ExhaustedPatternLookup.none()}.
|
||||
*/
|
||||
public static BackendQuarantine none() {
|
||||
return new BackendQuarantine(() -> 0L, 1, true);
|
||||
return new BackendQuarantine(() -> 0L, 1, 1.0, 1, true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Quarantine {@code credentialId} for the configured cooldown, starting now. A repeat call while
|
||||
* already quarantined restarts the cooldown at full length — a fresh refusal is fresh evidence the
|
||||
* account is still exhausted, not a reason to let an earlier, shorter wait stand.
|
||||
* Quarantine {@code credentialId} starting now. On a flat instance (the two-argument
|
||||
* constructor) this always blocks for exactly {@code cooldownNanos}, restarting the cooldown at
|
||||
* full length on every call — a fresh refusal is fresh evidence the account is still exhausted,
|
||||
* not a reason to let an earlier, shorter wait stand. On an escalating instance ({@link
|
||||
* #withEscalation}) the cooldown grows with each call that arrives within one base cooldown of
|
||||
* the previous deadline, and resets to the base cooldown once a call arrives after a longer gap
|
||||
* — see the class doc.
|
||||
*
|
||||
* <p>On {@link #none()} this is a no-op. It has to be: that instance holds a clock frozen at 0,
|
||||
* so recording a deadline would produce a quarantine that never expires — a credential locked out
|
||||
@@ -73,7 +186,13 @@ public final class BackendQuarantine {
|
||||
if (inert) {
|
||||
return;
|
||||
}
|
||||
quarantinedUntilNanos.put(credentialId, nowNanos.getAsLong() + cooldownNanos);
|
||||
long now = nowNanos.getAsLong();
|
||||
quarantines.compute(credentialId, (id, prev) -> {
|
||||
int repeatCount = (prev == null || now - prev.deadlineNanos() > cooldownNanos)
|
||||
? 1
|
||||
: prev.repeatCount() + 1;
|
||||
return new QuarantineState(repeatCount, now + escalatedCooldownNanos(repeatCount));
|
||||
});
|
||||
}
|
||||
|
||||
/** Whether {@code credentialId} is quarantined right now. */
|
||||
@@ -87,6 +206,49 @@ public final class BackendQuarantine {
|
||||
return remaining > 0 ? OptionalLong.of(toSecondsRoundedUp(remaining)) : OptionalLong.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* Remaining seconds together with which consecutive exhaustion this is (fleetd #466 scope item
|
||||
* 2) — {@code repeatCount} 1 for a first occurrence, 2 for the second in a row, and so on; see
|
||||
* {@link #quarantine}'s class-doc Mechanism section for exactly when a call continues a streak
|
||||
* versus starts a fresh one.
|
||||
*
|
||||
* <p><strong>Read together, off the one {@link QuarantineState} entry {@link #quarantine} itself
|
||||
* wrote</strong> — a single {@code quarantines.get(credentialId)}, never a separate lookup or a
|
||||
* value re-derived from {@code remainingSeconds} (e.g. inverting {@link
|
||||
* #escalatedCooldownNanos}). That inversion is not just extra work to avoid: once a streak has
|
||||
* hit {@code maxCooldownNanos}, every further consecutive exhaustion reports the identical
|
||||
* cooldown, so a derivation that starts from the cooldown value cannot tell the 4th repeat from
|
||||
* the 9th — only the stored {@code repeatCount} can. This is the same rule {@code
|
||||
* CompositePeerLauncher.modelGateState()} documents for its own gate/report pair: the report
|
||||
* reads the exact accessor the behaviour reads, so it can never disagree with what actually
|
||||
* happened (the fleetd #404/#422 lesson). {@code fleet_profiles}/{@code fleet_list}/{@code GET
|
||||
* /profiles} all call this — never {@link #remainingSeconds} plus a second, independent count —
|
||||
* for exactly that reason.
|
||||
*
|
||||
* @return empty when {@code credentialId} is not currently quarantined (including on {@link
|
||||
* #none()}, which quarantines nothing)
|
||||
*/
|
||||
public Optional<Status> status(String credentialId) {
|
||||
QuarantineState state = quarantines.get(credentialId);
|
||||
if (state == null) {
|
||||
return Optional.empty();
|
||||
}
|
||||
long remaining = state.deadlineNanos() - nowNanos.getAsLong();
|
||||
return remaining > 0
|
||||
? Optional.of(new Status(toSecondsRoundedUp(remaining), state.repeatCount()))
|
||||
: Optional.empty();
|
||||
}
|
||||
|
||||
/**
|
||||
* @param remainingSeconds seconds left on the quarantine, identical to {@link
|
||||
* #remainingSeconds(String)}'s answer for the same credential at the
|
||||
* same instant
|
||||
* @param repeatCount 1 for a first occurrence, 2 for the second consecutive one, etc. —
|
||||
* see {@link #status(String)}
|
||||
*/
|
||||
public record Status(long remainingSeconds, int repeatCount) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Every currently-quarantined credential id and its remaining seconds (CB-578 stage B fleet
|
||||
* reporting) — expired entries are never included. Not pruned from the backing map here: it stays
|
||||
@@ -95,8 +257,8 @@ public final class BackendQuarantine {
|
||||
*/
|
||||
public Map<String, Long> activeRemainingSeconds() {
|
||||
Map<String, Long> out = new LinkedHashMap<>();
|
||||
quarantinedUntilNanos.forEach((credentialId, deadline) -> {
|
||||
long remaining = deadline - nowNanos.getAsLong();
|
||||
quarantines.forEach((credentialId, state) -> {
|
||||
long remaining = state.deadlineNanos() - nowNanos.getAsLong();
|
||||
if (remaining > 0) {
|
||||
out.put(credentialId, toSecondsRoundedUp(remaining));
|
||||
}
|
||||
@@ -105,8 +267,14 @@ public final class BackendQuarantine {
|
||||
}
|
||||
|
||||
private long remainingNanos(String credentialId) {
|
||||
Long deadline = quarantinedUntilNanos.get(credentialId);
|
||||
return deadline == null ? 0L : deadline - nowNanos.getAsLong();
|
||||
QuarantineState state = quarantines.get(credentialId);
|
||||
return state == null ? 0L : state.deadlineNanos() - nowNanos.getAsLong();
|
||||
}
|
||||
|
||||
/** {@code cooldownNanos * backoffMultiplier ^ (repeatCount - 1)}, capped at {@code maxCooldownNanos}. */
|
||||
private long escalatedCooldownNanos(int repeatCount) {
|
||||
double raw = cooldownNanos * Math.pow(backoffMultiplier, repeatCount - 1);
|
||||
return raw >= (double) maxCooldownNanos ? maxCooldownNanos : (long) raw;
|
||||
}
|
||||
|
||||
private static long toSecondsRoundedUp(long nanos) {
|
||||
|
||||
@@ -1,66 +1,151 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.List;
|
||||
|
||||
/**
|
||||
* Backward-compatible placement: an unqualified spawn always resolves to the configured default
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. This ignores caps and
|
||||
* reachability so that a pre-existing config behaves identically after upgrade.
|
||||
* profile, exactly as {@code CompositePeerLauncher} did before CB-518. Reachability is a narrower
|
||||
* exception (fleetd #315, below): a profile is never checked for reachability up front, only
|
||||
* skipped once it has already failed in <em>this same</em> spawn call's retry loop — see the
|
||||
* unreachable case below.
|
||||
*
|
||||
* <p>Two exceptions walk past the default instead of returning it unconditionally:
|
||||
* <p>Six exceptions walk past the default instead of returning it unconditionally:
|
||||
* <ul>
|
||||
* <li>Quarantine (CB-578 stage B): a quarantined default is a credential that just refused on
|
||||
* a usage limit, not a transient capacity or reachability concern.
|
||||
* <li>Cooling off (fleetd #201 Unit 5): a credential cooling off after repeated backend errors
|
||||
* ({@code BackendOutagePolicy}) — a separate, shorter-lived source from quarantine. When a
|
||||
* profile is both quarantined and cooling off, only the quarantine reason is reported
|
||||
* (exhaustion takes priority), matching {@code CompositePeerLauncher}'s explicit-spawn order.
|
||||
* <li>At cap (fleetd #435): a profile whose live count has reached its {@code maxLoad}
|
||||
* ({@link PlacementPolicyUtil#atCap}) — a documented, unconditional capacity limit (see
|
||||
* {@code FleetConfig.Profile#maxLoad}), so {@code fixed} must gate on it exactly as {@code
|
||||
* weighted}/{@code round-robin} already do via {@link PlacementPolicyUtil#available}. Before
|
||||
* this fix {@code fixed} built its own {@link PlacementCandidate} for the default with {@code
|
||||
* maxLoad} forced to {@code null}, so a capped default was chosen anyway on every unqualified
|
||||
* spawn — the cap was advisory, not enforced, for the one placement policy every config uses
|
||||
* by default. Reported only when quarantine and cooling off are both absent, matching {@code
|
||||
* CompositePeerLauncher}'s explicit-spawn check order (quarantine, then cooling off, then max
|
||||
* load, then model-off).
|
||||
* <li>Model off (fleetd #422): a profile whose {@code model:} the operator has turned off in
|
||||
* {@code models.allow:} — an operator decision, never a backend-reported outage, so it is a
|
||||
* fifth, independent source from quarantine, cooling off, and at-cap (never merged with any
|
||||
* of them), exactly as {@code CompositePeerLauncher.enforceModelEnabled} and {@link
|
||||
* PlacementPolicyUtil#available} treat it. When a profile is model-off <em>and</em> quarantined,
|
||||
* cooling off, or at cap, only the higher-priority reason is reported, matching {@code
|
||||
* CompositePeerLauncher}'s explicit-spawn check order (quarantine, then cooling off, then max
|
||||
* load, then model-off).
|
||||
* <li>Unreachable (fleetd #315): {@code CompositePeerLauncher.spawn} retries a failed candidate
|
||||
* on the next one and rebuilds the {@link PlacementContext} so {@code ctx.unreachable()}
|
||||
* names every profile that already failed with {@code PeerUnreachableException} in this same
|
||||
* call. Without this check {@code select} kept handing back the same dead default forever —
|
||||
* the retry loop's own comment says "so the policy excludes this profile", and this is what
|
||||
* makes that true for {@code fixed} too, matching {@code weighted}/{@code round-robin}
|
||||
* (both filter on {@code ctx.unreachable()} via {@link PlacementPolicyUtil#available}).
|
||||
* <li>Weight 0 (CB-554): {@code fixed} is still automatic selection, so a profile the operator
|
||||
* marked "never auto-select me" ({@code weight <= 0}) must be skipped here exactly as
|
||||
* {@code weighted}/{@code round-robin} skip it — an explicit {@code fleet_spawn} naming
|
||||
* the profile is unaffected, only this automatic fallback walk.
|
||||
* </ul>
|
||||
* A fleet where nothing is ever quarantined or weight-0 never exercises either path, so today's
|
||||
* behaviour is unchanged.
|
||||
* A fleet where nothing is ever quarantined, cooling off, at cap, model-off, unreachable, or
|
||||
* weight-0 never exercises any of these paths, so today's behaviour is unchanged — in particular,
|
||||
* the very first selection of a spawn call always sees an empty {@code unreachable} set, so the
|
||||
* first choice is untouched.
|
||||
*/
|
||||
final class FixedPlacementPolicy implements PlacementPolicy {
|
||||
|
||||
@Override
|
||||
public PlacementCandidate select(PlacementContext ctx) {
|
||||
String d = ctx.defaultProfile();
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !weightExcluded(ctx, d)) {
|
||||
if (d != null && !d.isBlank() && !ctx.quarantined().contains(d) && !ctx.coolingOff().contains(d)
|
||||
&& !ctx.modelOff().contains(d) && !ctx.unreachable().contains(d) && !weightExcluded(ctx, d)
|
||||
&& !capExcluded(ctx, d)) {
|
||||
return new PlacementCandidate(d, null, 1.0f, null);
|
||||
}
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !c.excluded()) {
|
||||
if (!ctx.quarantined().contains(c.profile()) && !ctx.coolingOff().contains(c.profile())
|
||||
&& !ctx.modelOff().contains(c.profile()) && !ctx.unreachable().contains(c.profile())
|
||||
&& !c.excluded() && !PlacementPolicyUtil.atCap(ctx, c)) {
|
||||
return new PlacementCandidate(c.profile(), null, c.weight(), c.maxLoad());
|
||||
}
|
||||
}
|
||||
if (d != null && !d.isBlank()) {
|
||||
boolean dQuarantined = ctx.quarantined().contains(d);
|
||||
// Exhaustion quarantine takes priority: reported only when quarantine is absent, so the
|
||||
// message never claims "cooling off" for a profile that is really backend-exhausted.
|
||||
boolean dCoolingOff = !dQuarantined && ctx.coolingOff().contains(d);
|
||||
// fleetd #435: at-cap sits between cooling off and model-off, matching
|
||||
// CompositePeerLauncher's explicit-spawn check order (quarantine, cooling off, max load,
|
||||
// then model-off) — reported only when quarantine/cooling-off are both absent.
|
||||
boolean dAtCap = !dQuarantined && !dCoolingOff && capExcluded(ctx, d);
|
||||
// fleetd #422: model-off is a fifth, independent source (an operator decision) — but
|
||||
// quarantine/cooling-off/at-cap still take priority when more than one applies, matching
|
||||
// CompositePeerLauncher's explicit-spawn check order (quarantine, cooling off, max load,
|
||||
// then model-off).
|
||||
boolean dModelOff = !dQuarantined && !dCoolingOff && !dAtCap && ctx.modelOff().contains(d);
|
||||
boolean dUnreachable = ctx.unreachable().contains(d);
|
||||
boolean dWeightExcluded = weightExcluded(ctx, d);
|
||||
if (dQuarantined && dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and has weight 0 (excluded from automatic selection), and no "
|
||||
+ "available candidate remains");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
throw new PlacementException("worker profile '" + d + "' has weight 0 (excluded "
|
||||
+ "from automatic selection) and no available candidate remains");
|
||||
}
|
||||
if (dQuarantined) {
|
||||
throw new PlacementException("worker profile '" + d + "' is quarantined (backend "
|
||||
+ "exhausted) and no un-quarantined candidate is available");
|
||||
if (dQuarantined || dCoolingOff || dAtCap || dModelOff || dUnreachable || dWeightExcluded) {
|
||||
List<String> reasons = new ArrayList<>();
|
||||
if (dQuarantined) {
|
||||
reasons.add("is quarantined (backend exhausted)");
|
||||
}
|
||||
if (dCoolingOff) {
|
||||
reasons.add("is cooling off after repeated backend errors");
|
||||
}
|
||||
if (dAtCap) {
|
||||
PlacementCandidate c = candidateFor(ctx, d);
|
||||
int live = ctx.liveCount().apply(d);
|
||||
reasons.add("is at maxLoad (" + live + " live >= " + c.maxLoad() + " cap)");
|
||||
}
|
||||
if (dModelOff) {
|
||||
reasons.add("names a model the operator has turned off in models.allow");
|
||||
}
|
||||
if (dUnreachable) {
|
||||
reasons.add("is unreachable");
|
||||
}
|
||||
if (dWeightExcluded) {
|
||||
reasons.add("has weight 0 (excluded from automatic selection)");
|
||||
}
|
||||
throw new PlacementException("worker profile '" + d + "' "
|
||||
+ String.join(" and ", reasons) + ", and no available candidate remains");
|
||||
}
|
||||
}
|
||||
if (!ctx.candidates().isEmpty()) {
|
||||
throw new PlacementException(
|
||||
"all worker profiles are excluded from automatic selection (quarantined or weight-0)");
|
||||
throw new PlacementException("all worker profiles are excluded from automatic "
|
||||
+ "selection (quarantined, cooling off, at cap, model-off, unreachable, or weight-0)");
|
||||
}
|
||||
throw new PlacementException("no worker profiles configured");
|
||||
}
|
||||
|
||||
/** Whether {@code profile} carries {@code weight <= 0} (CB-554) among {@code ctx}'s candidates. */
|
||||
private static boolean weightExcluded(PlacementContext ctx, String profile) {
|
||||
PlacementCandidate c = candidateFor(ctx, profile);
|
||||
return c != null && c.excluded();
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether {@code profile} has reached its {@code maxLoad} cap (fleetd #435), using the shared
|
||||
* {@link PlacementPolicyUtil#atCap} definition — the same one {@code weighted}/{@code
|
||||
* round-robin} already consult via {@link PlacementPolicyUtil#available}. Looked up by name,
|
||||
* the same way {@link #weightExcluded} is: the default fast path above builds its own {@link
|
||||
* PlacementCandidate} with {@code maxLoad} forced to {@code null} (it carries no cap of its
|
||||
* own), so the candidate actually configured for {@code profile} has to be found in {@code
|
||||
* ctx.candidates()} first.
|
||||
*/
|
||||
private static boolean capExcluded(PlacementContext ctx, String profile) {
|
||||
PlacementCandidate c = candidateFor(ctx, profile);
|
||||
return c != null && PlacementPolicyUtil.atCap(ctx, c);
|
||||
}
|
||||
|
||||
/** The configured candidate named {@code profile} in {@code ctx}, or {@code null} if none. */
|
||||
private static PlacementCandidate candidateFor(PlacementContext ctx, String profile) {
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.profile().equals(profile)) {
|
||||
return c.excluded();
|
||||
return c;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -14,10 +14,37 @@ import java.util.function.Function;
|
||||
* @param quarantined profiles whose credential is currently quarantined (CB-578 stage B) — a
|
||||
* {@code BACKEND_EXHAUSTED} classification put it, or a profile it shares a
|
||||
* credential with, on cooldown. Filtered the same way as {@code unreachable}.
|
||||
* @param coolingOff profiles whose credential is currently cooling off after repeated backend
|
||||
* errors (fleetd #201 Unit 5 — {@code BackendOutagePolicy}), a SEPARATE,
|
||||
* shorter-lived source from {@code quarantined}: a credential outage cools off
|
||||
* even when no member was ever exhausted. Deliberately its own set rather than
|
||||
* merged into {@code quarantined} — {@link PlacementPolicyUtil} needs to tell
|
||||
* the two apart so its refusal message says "cooling off", not "exhausted",
|
||||
* when only this one is active. A profile can be in both sets at once; when it
|
||||
* is, exhaustion quarantine is reported (it takes priority).
|
||||
* @param modelOff profiles whose {@code model:} is currently turned off in {@code
|
||||
* models.allow:} (fleetd #422) — an operator decision, not a backend-reported
|
||||
* outage, so a SEPARATE, independent source from both {@code quarantined} and
|
||||
* {@code coolingOff}. A profile can be in this set together with either (or
|
||||
* both) of the others; {@link PlacementPolicyUtil} counts it into its own
|
||||
* bucket rather than merging it into theirs, the same reason
|
||||
* {@code coolingOff} is kept apart from {@code quarantined}.
|
||||
*/
|
||||
public record PlacementContext(String defaultProfile,
|
||||
List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount,
|
||||
Set<String> unreachable,
|
||||
Set<String> quarantined) {
|
||||
Set<String> quarantined,
|
||||
Set<String> coolingOff,
|
||||
Set<String> modelOff) {
|
||||
|
||||
/**
|
||||
* Back-compat form before the fleetd #422 model on/off gate was added — no candidate's model
|
||||
* is off. Keeps pre-#422 call sites (tests included) compiling and behaving identically.
|
||||
*/
|
||||
public PlacementContext(String defaultProfile, List<PlacementCandidate> candidates,
|
||||
Function<String, Integer> liveCount, Set<String> unreachable,
|
||||
Set<String> quarantined, Set<String> coolingOff) {
|
||||
this(defaultProfile, candidates, liveCount, unreachable, quarantined, coolingOff, Set.of());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,49 @@
|
||||
package dev.ltms.fleet.placement;
|
||||
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
|
||||
/**
|
||||
* An already-completed placement choice — the outcome of one {@link PeerLauncher#place} call,
|
||||
* carried forward so a later {@link PeerLauncher#spawn(SpawnRequest, PlacementDecision)} can honor
|
||||
* it directly instead of re-resolving the profile a second time (fleetd #425 rework, round 2).
|
||||
*
|
||||
* <p>The problem this exists to close: a caller that must know the profile <em>before</em> it can
|
||||
* spawn — {@code SessionManager.acquireWithWorktree} provisions a worktree's {@code repoRoot} and
|
||||
* parity overlay for a specific profile before the peer process exists — used to resolve that name
|
||||
* with {@code PeerLauncher.routedProfileFor(role)} and then hand the SAME string back to {@link
|
||||
* PeerLauncher#spawn(SpawnRequest)} as an EXPLICIT profile. That re-resolution is not free: naming
|
||||
* a profile explicitly makes {@code CompositePeerLauncher.spawn} take its THROWING branch
|
||||
* ({@code enforceNotQuarantined}/{@code enforceNotCoolingOff}/{@code enforceMaxLoad}/{@code
|
||||
* enforceModelEnabled}), while an unqualified spawn's ROUTING branch never runs those checks at
|
||||
* all — it instead FALLS THROUGH to the next candidate on exactly the same conditions the throwing
|
||||
* branch refuses on. That disagreement is deliberate: an operator who names a profile should get a
|
||||
* refusal, not a silent substitution. The bug is turning the fall-through into a refusal by
|
||||
* accident — resolving a name through the routing side and then re-entering the refusing side with
|
||||
* it, for a decision the routing side had already approved by walking past everything else.
|
||||
* Before fleetd #435, this accident was also reachable through {@code maxLoad} specifically: the
|
||||
* default {@code fixed} placement policy did not evaluate {@code maxLoad} at all for automatic
|
||||
* selection, so a profile placement itself just approved could still die at {@code enforceMaxLoad}
|
||||
* one call later, purely because the caller's route to the spawn passed through an explicit
|
||||
* profile name instead of the routing branch — a failure a worktree-less unqualified spawn would
|
||||
* never hit. fleetd #435 closed that specific gap ({@code fixed} now evaluates {@code maxLoad}
|
||||
* exactly like every other placement policy), so a {@link PlacementDecision} can no longer be
|
||||
* at-cap in the first place — but the refusal-vs-fall-through disagreement above was never about
|
||||
* {@code maxLoad}, and resolving a name and re-entering the refusing branch with it is still wrong
|
||||
* for every OTHER condition placement filters on.
|
||||
*
|
||||
* <p>{@link PeerLauncher#spawn(SpawnRequest, PlacementDecision)} closes that by spawning through
|
||||
* the identical code path the routing branch itself uses, keyed off the SAME decision {@link
|
||||
* PeerLauncher#place} returned — no re-checking of any condition placement already evaluated. A
|
||||
* resolve-then-spawn caller and a blank-profile {@link PeerLauncher#spawn(SpawnRequest)} caller can
|
||||
* then never disagree about which conditions apply to the same placement state, and neither one
|
||||
* re-opens the window between the placement decision and the spawn in which the underlying state
|
||||
* could otherwise move.
|
||||
*
|
||||
* @param profile the profile this decision resolved to (may be {@code null} only when no profile is
|
||||
* configured at all — the same corner case {@link PeerLauncher#defaultProfile()}
|
||||
* already tolerates)
|
||||
*/
|
||||
public record PlacementDecision(String profile) {
|
||||
}
|
||||
@@ -11,26 +11,40 @@ final class PlacementPolicyUtil {
|
||||
private PlacementPolicyUtil() {
|
||||
}
|
||||
|
||||
/**
|
||||
* True when {@code c} has reached its {@code maxLoad} cap: {@code liveCount(c.profile()) >=
|
||||
* c.maxLoad()}. A {@code null} maxLoad means unlimited, so it is never at cap.
|
||||
*
|
||||
* <p>Extracted as the single shared definition of "at cap" (fleetd #435): before this fix it
|
||||
* was computed inline in both {@link #available} and {@link #emptyException}, and {@code
|
||||
* FixedPlacementPolicy} — not a caller of either — quietly kept its own {@code select} free of
|
||||
* any cap check at all, so a capped default profile was chosen anyway under the default
|
||||
* placement policy. Every automatic policy must call this, not re-derive it.
|
||||
*/
|
||||
static boolean atCap(PlacementContext ctx, PlacementCandidate c) {
|
||||
Integer cap = c.maxLoad();
|
||||
return cap != null && ctx.liveCount().apply(c.profile()) >= cap;
|
||||
}
|
||||
|
||||
/**
|
||||
* Candidates that are not weight-excluded (CB-554: explicit {@code weight <= 0}, checked
|
||||
* first because it is a static config choice rather than transient state), not
|
||||
* known-unreachable, not quarantined (CB-578 stage B), and have not reached their maxLoad.
|
||||
* A {@code null} maxLoad means unlimited.
|
||||
* known-unreachable, not quarantined (CB-578 stage B), not cooling off after repeated backend
|
||||
* errors (fleetd #201 Unit 5 — a separate, shorter-lived source from quarantine), not naming a
|
||||
* model the operator has turned off (fleetd #422 — a third, independent source: an operator
|
||||
* decision, never a backend-reported outage), and have not reached their maxLoad (see {@link
|
||||
* #atCap}). A {@code null} maxLoad means unlimited.
|
||||
*/
|
||||
static List<PlacementCandidate> available(PlacementContext ctx) {
|
||||
List<PlacementCandidate> out = new ArrayList<>();
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
if (c.excluded() || ctx.unreachable().contains(c.profile())
|
||||
|| ctx.quarantined().contains(c.profile())) {
|
||||
|| ctx.quarantined().contains(c.profile())
|
||||
|| ctx.coolingOff().contains(c.profile())
|
||||
|| ctx.modelOff().contains(c.profile())
|
||||
|| atCap(ctx, c)) {
|
||||
continue;
|
||||
}
|
||||
Integer cap = c.maxLoad();
|
||||
if (cap != null) {
|
||||
int live = ctx.liveCount().apply(c.profile());
|
||||
if (live >= cap) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
out.add(c);
|
||||
}
|
||||
return out;
|
||||
@@ -38,24 +52,33 @@ final class PlacementPolicyUtil {
|
||||
|
||||
/**
|
||||
* Build a clear exception describing why every candidate was dropped: all weight-0, all
|
||||
* quarantined, all at capacity, all unreachable, or a mix. Each candidate is counted into
|
||||
* exactly one bucket (weight-excluded takes priority) so a candidate excluded for more than
|
||||
* one reason is never double-counted.
|
||||
* quarantined, all cooling off, all model-off, all at capacity, all unreachable, or a mix. Each
|
||||
* candidate is counted into exactly one bucket (weight-excluded first, then quarantined, then
|
||||
* cooling off, then model-off) so a candidate excluded for more than one reason is never
|
||||
* double-counted — a candidate that is both quarantined (CB-578 stage B, backend exhausted) and
|
||||
* cooling off (fleetd #201 Unit 5, repeated backend errors) counts only as quarantined, matching
|
||||
* {@code CompositePeerLauncher}'s explicit-spawn ordering: exhaustion quarantine takes priority
|
||||
* when more than one applies.
|
||||
*/
|
||||
static PlacementException emptyException(PlacementContext ctx) {
|
||||
int weightExcluded = 0;
|
||||
int atCap = 0;
|
||||
int unreachable = 0;
|
||||
int quarantined = 0;
|
||||
int coolingOff = 0;
|
||||
int modelOff = 0;
|
||||
for (PlacementCandidate c : ctx.candidates()) {
|
||||
Integer cap = c.maxLoad();
|
||||
if (c.excluded()) {
|
||||
weightExcluded++;
|
||||
} else if (ctx.quarantined().contains(c.profile())) {
|
||||
quarantined++;
|
||||
} else if (ctx.coolingOff().contains(c.profile())) {
|
||||
coolingOff++;
|
||||
} else if (ctx.modelOff().contains(c.profile())) {
|
||||
modelOff++;
|
||||
} else if (ctx.unreachable().contains(c.profile())) {
|
||||
unreachable++;
|
||||
} else if (cap != null && ctx.liveCount().apply(c.profile()) >= cap) {
|
||||
} else if (atCap(ctx, c)) {
|
||||
atCap++;
|
||||
}
|
||||
}
|
||||
@@ -71,6 +94,14 @@ final class PlacementPolicyUtil {
|
||||
if (quarantined == total) {
|
||||
return new PlacementException("all worker profiles are quarantined (backend exhausted)");
|
||||
}
|
||||
if (coolingOff == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles are cooling off after repeated backend errors");
|
||||
}
|
||||
if (modelOff == total) {
|
||||
return new PlacementException(
|
||||
"all worker profiles name a model the operator has turned off");
|
||||
}
|
||||
if (atCap == total) {
|
||||
return new PlacementException("all worker profiles are at maxLoad");
|
||||
}
|
||||
@@ -79,7 +110,10 @@ final class PlacementPolicyUtil {
|
||||
}
|
||||
return new PlacementException("no worker profile available: " + atCap + " at maxLoad, "
|
||||
+ unreachable + " unreachable, " + quarantined + " quarantined, "
|
||||
+ coolingOff + " cooling off, "
|
||||
+ modelOff + " model-off, "
|
||||
+ weightExcluded + " weight-0, "
|
||||
+ (total - atCap - unreachable - quarantined - weightExcluded) + " remaining");
|
||||
+ (total - atCap - unreachable - quarantined - coolingOff - modelOff - weightExcluded)
|
||||
+ " remaining");
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,93 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.io.IOException;
|
||||
import java.util.Locale;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
|
||||
/**
|
||||
* Holds macOS idle sleep off by keeping a {@code caffeinate -i} child process alive for the life
|
||||
* of the returned {@link SleepAssertion}.
|
||||
*
|
||||
* <p>{@code -i} asserts only against <em>idle</em> sleep — it does not stop the lid closing or an
|
||||
* operator-requested sleep from taking effect. That is deliberate: this class exists to stop an
|
||||
* unattended host from sleeping out from under a member's long turn, never to override the
|
||||
* operator. {@code -s}/{@code -d} (which also block system/display sleep on demand) are
|
||||
* intentionally not used here.
|
||||
*
|
||||
* <p>{@link #acquire()} never throws. It returns {@code null} — a no-op — off macOS, and again if
|
||||
* starting the {@code caffeinate} child fails for any reason (binary missing, process table full,
|
||||
* …); either case is logged once at INFO, not on every occurrence, so a daemon that runs for
|
||||
* weeks with the tool unavailable does not fill its log.
|
||||
*/
|
||||
public final class CaffeinateSleepAssertionMechanism implements SleepAssertionMechanism {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(CaffeinateSleepAssertionMechanism.class);
|
||||
|
||||
private final AtomicBoolean loggedOnce = new AtomicBoolean(false);
|
||||
|
||||
/** {@code true} when running on macOS, the only platform {@code caffeinate} ships on. */
|
||||
public static boolean isSupportedPlatform() {
|
||||
return isSupportedPlatform(System.getProperty("os.name"));
|
||||
}
|
||||
|
||||
/** Package-visible so a test can drive the platform check without touching a real property. */
|
||||
static boolean isSupportedPlatform(String osName) {
|
||||
return osName != null && osName.toLowerCase(Locale.ROOT).contains("mac");
|
||||
}
|
||||
|
||||
@Override
|
||||
public SleepAssertion acquire() {
|
||||
if (!isSupportedPlatform()) {
|
||||
logOnce("not running on macOS (os.name={}); the idle-sleep guard is a no-op on this platform",
|
||||
System.getProperty("os.name"));
|
||||
return null;
|
||||
}
|
||||
try {
|
||||
Process process = new ProcessBuilder("caffeinate", "-i")
|
||||
.redirectOutput(ProcessBuilder.Redirect.DISCARD)
|
||||
.redirectError(ProcessBuilder.Redirect.DISCARD)
|
||||
.start();
|
||||
return new CaffeinateAssertion(process);
|
||||
} catch (IOException | RuntimeException e) {
|
||||
logOnce("could not start 'caffeinate -i' ({}); the host may idle-sleep while members are live",
|
||||
e.toString());
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
private void logOnce(String format, Object arg) {
|
||||
if (loggedOnce.compareAndSet(false, true)) {
|
||||
log.info("idle-sleep guard: " + format, arg);
|
||||
}
|
||||
}
|
||||
|
||||
/** Wraps the live {@code caffeinate} child; {@link #close} force-destroys it, idempotently. */
|
||||
private static final class CaffeinateAssertion implements SleepAssertion {
|
||||
|
||||
private final Process process;
|
||||
|
||||
CaffeinateAssertion(Process process) {
|
||||
this.process = process;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
if (!process.isAlive()) {
|
||||
return;
|
||||
}
|
||||
process.destroy();
|
||||
try {
|
||||
if (!process.waitFor(2, TimeUnit.SECONDS)) {
|
||||
process.destroyForcibly();
|
||||
}
|
||||
} catch (InterruptedException e) {
|
||||
Thread.currentThread().interrupt();
|
||||
process.destroyForcibly();
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,105 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.util.function.IntSupplier;
|
||||
|
||||
/**
|
||||
* Holds an OS-level assertion against idle sleep for exactly as long as at least one fleet
|
||||
* member is live.
|
||||
*
|
||||
* <p><strong>Why this exists:</strong> a fleetd host was measured idle-sleeping after as little
|
||||
* as one minute of inactivity (its {@code pmset -g custom} reports {@code sleep 1} on battery).
|
||||
* Overnight the daemon's AMQP link to the broker dropped 13 times, and cross-checking every drop
|
||||
* minute against {@code pmset -g log} found a sleep or wake event in the same minute or the one
|
||||
* before, every time. The AMQP churn is only the visible symptom — the real problem is that a
|
||||
* member mid-turn freezes with the host, and a long turn with nobody typing is exactly the case
|
||||
* that goes idle.
|
||||
*
|
||||
* <p><strong>How it tracks "live":</strong> this is driven by {@code SessionManager}'s existing
|
||||
* {@code onAcquire}/{@code onRelease} lifecycle hooks (added for CB-520/CB-516, previously wired
|
||||
* to nothing but the reply inbox) rather than a second member count kept in parallel. Wire it as:
|
||||
* <pre>{@code
|
||||
* IdleSleepGuard guard = new IdleSleepGuard(mechanism, sessions::size);
|
||||
* sessions.onAcquire(_ -> guard.recheck());
|
||||
* sessions.onRelease(_ -> guard.recheck());
|
||||
* }</pre>
|
||||
* Every acquire/release event re-reads {@code SessionManager#size()} — the same registry {@code
|
||||
* fleet_list}'s live/capacity numbers are themselves computed from — and only an actual 0→1 or
|
||||
* 1→0 crossing touches the OS. A listener exception is already caught and logged by {@code
|
||||
* SessionManager} itself (it must never let a listener failure block the acquire/release it is
|
||||
* reacting to), so {@link #recheck()} does not need its own top-level try/catch to honor that.
|
||||
*
|
||||
* <p><strong>Failure posture:</strong> every method here is safe to call whether or not {@link
|
||||
* SleepAssertionMechanism#acquire()} actually works. A mechanism that returns {@code null} (wrong
|
||||
* platform, missing tool, spawn failure) simply means this guard never holds anything — it never
|
||||
* throws and never blocks a spawn, a release, or shutdown.
|
||||
*/
|
||||
public final class IdleSleepGuard implements AutoCloseable {
|
||||
|
||||
private static final Logger log = LoggerFactory.getLogger(IdleSleepGuard.class);
|
||||
|
||||
private final SleepAssertionMechanism mechanism;
|
||||
private final IntSupplier liveCount;
|
||||
private final Object lock = new Object();
|
||||
private SleepAssertion held;
|
||||
|
||||
public IdleSleepGuard(SleepAssertionMechanism mechanism, IntSupplier liveCount) {
|
||||
this.mechanism = mechanism;
|
||||
this.liveCount = liveCount;
|
||||
}
|
||||
|
||||
/**
|
||||
* Re-read the live count and acquire or release the held assertion to match: nothing held and
|
||||
* at least one member live ⇒ acquire; something held and no member live ⇒ release. A steady
|
||||
* count (still zero, still positive) is a no-op either way, so a single spawn or release only
|
||||
* ever touches the OS on the crossing, not on every call.
|
||||
*/
|
||||
public void recheck() {
|
||||
synchronized (lock) {
|
||||
int live = liveCount.getAsInt();
|
||||
if (live > 0 && held == null) {
|
||||
held = mechanism.acquire();
|
||||
if (held != null) {
|
||||
log.debug("idle-sleep guard armed: {} live member(s)", live);
|
||||
}
|
||||
} else if (live == 0 && held != null) {
|
||||
releaseHeldLocked();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** {@code true} while an assertion is actually held. Exposed for tests. */
|
||||
boolean isHeld() {
|
||||
synchronized (lock) {
|
||||
return held != null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Release whatever is held, if anything. Idempotent and safe to call at any time, including
|
||||
* repeatedly — a daemon shutdown hook calls this unconditionally so no assertion (and no
|
||||
* {@code caffeinate} child) survives the process, even if the drain that would otherwise have
|
||||
* driven the live count to zero was itself interrupted or threw.
|
||||
*/
|
||||
@Override
|
||||
public void close() {
|
||||
synchronized (lock) {
|
||||
if (held != null) {
|
||||
releaseHeldLocked();
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/** Caller must hold {@link #lock}. */
|
||||
private void releaseHeldLocked() {
|
||||
try {
|
||||
held.close();
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("idle-sleep guard: failed to release its assertion cleanly: {}", e.toString());
|
||||
} finally {
|
||||
held = null;
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
/**
|
||||
* A held OS-level assertion against idle sleep. {@link #close} must be idempotent — safe to call
|
||||
* more than once — and must never throw, matching {@link IdleSleepGuard}'s "never break the
|
||||
* fleet" contract.
|
||||
*/
|
||||
public interface SleepAssertion extends AutoCloseable {
|
||||
@Override
|
||||
void close();
|
||||
}
|
||||
@@ -0,0 +1,21 @@
|
||||
package dev.ltms.fleet.power;
|
||||
|
||||
/**
|
||||
* The OS mechanism {@link IdleSleepGuard} uses to hold and release an idle-sleep assertion. This
|
||||
* is the seam a test exercises instead of the real effect (a live {@code caffeinate} child) — see
|
||||
* {@code IdleSleepGuardTest}.
|
||||
*
|
||||
* <p>Implementations must never throw. Every failure — wrong platform, missing tool, a spawn
|
||||
* error — must show up as {@link #acquire()} returning {@code null}, so a caller can treat "no
|
||||
* assertion held" and "the mechanism could not be used" identically and the fleet keeps running
|
||||
* either way.
|
||||
*/
|
||||
public interface SleepAssertionMechanism {
|
||||
|
||||
/**
|
||||
* Acquire a fresh assertion against idle sleep, or {@code null} when this mechanism is not
|
||||
* usable right now (wrong platform, the tool is missing, the child process could not start).
|
||||
* Never throws.
|
||||
*/
|
||||
SleepAssertion acquire();
|
||||
}
|
||||
@@ -9,14 +9,17 @@ import dev.ltms.fleet.auth.Principal;
|
||||
import dev.ltms.fleet.guard.GuardException;
|
||||
import dev.ltms.fleet.metrics.FleetMetrics;
|
||||
import dev.ltms.fleet.metrics.Metrics;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.herdr.Agent;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.HerdrException;
|
||||
import dev.ltms.fleet.inject.MemberPresence;
|
||||
import dev.ltms.fleet.member.MemberCredentialPolicyView;
|
||||
import dev.ltms.fleet.peer.PeerUnreachableException;
|
||||
import dev.ltms.fleet.placement.PlacementException;
|
||||
import dev.ltms.fleet.msg.MessageService;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import dev.ltms.fleet.session.ShuttingDownException;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.WorktreeRequest;
|
||||
@@ -31,6 +34,8 @@ import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.function.Function;
|
||||
import java.util.function.Predicate;
|
||||
import java.util.function.Supplier;
|
||||
import java.util.stream.Collectors;
|
||||
|
||||
/**
|
||||
@@ -44,6 +49,22 @@ import java.util.stream.Collectors;
|
||||
*/
|
||||
public final class FleetApp {
|
||||
|
||||
/** The authorization action the matching route handler hands to {@link #allow}. */
|
||||
static Authz.Action routeAction(String route) {
|
||||
return switch (route) {
|
||||
case "GET /metrics" -> Authz.Action.METRICS;
|
||||
case "POST /members" -> Authz.Action.SPAWN;
|
||||
case "DELETE /members/{paneId}" -> Authz.Action.STOP;
|
||||
case "POST /sessions/{id}/message" -> Authz.Action.SEND;
|
||||
case "POST /sessions/{id}/reply" -> Authz.Action.REPLY;
|
||||
case "GET /sessions/{id}/replies" -> Authz.Action.DRAIN;
|
||||
case "POST /sessions/{id}/ask" -> Authz.Action.ASK;
|
||||
case "GET /sessions", "GET /agents", "GET /members", "GET /profiles",
|
||||
"GET /member-credentials", "GET /sessions/{id}/status", "GET /tasks/{ticket}" -> Authz.Action.READ;
|
||||
default -> throw new IllegalArgumentException("route has no authorization gate: " + route);
|
||||
};
|
||||
}
|
||||
|
||||
/** Default blocking window for a message; kept under typical HTTP idle timeouts. */
|
||||
private static final long DEFAULT_MESSAGE_TIMEOUT_MS = 25_000;
|
||||
private static final long MAX_MESSAGE_TIMEOUT_MS = 120_000;
|
||||
@@ -54,14 +75,25 @@ public final class FleetApp {
|
||||
/** Context attribute under which the resolved caller is stashed by the auth filter. */
|
||||
private static final String CALLER = "fleetd.caller";
|
||||
|
||||
private final HerdrClient herdr;
|
||||
private final HerdrClient herdr; // lead daemon
|
||||
private final HerdrClient memberHerdr; // CB-185: member daemon (same object when unconfigured)
|
||||
private final PeerLauncher workers;
|
||||
private final SessionManager sessions; // CB-301: authoritative session registry
|
||||
private final MessageService messages;
|
||||
private final MemberPresence presence; // CB-113: which workers are MCP-connected (available)
|
||||
private final Predicate<String> deliverable;
|
||||
private final HttpServlet mcpServlet; // MCP Streamable-HTTP endpoint, mounted at /mcp (nullable)
|
||||
private final CallerResolver auth; // CB-501: null → authz not enforced (legacy behaviour)
|
||||
private final Metrics metrics; // CB-502: null → /metrics not exposed
|
||||
// fleetd #111: re-read per request, same hot-reload shape as every other live config read —
|
||||
// absent() (the honest "no policy configured" view) for every constructor that does not wire
|
||||
// a real one, so existing legacy call sites keep building without knowing this field exists.
|
||||
private final Supplier<MemberCredentialPolicyView> memberCredentials;
|
||||
// fleetd #297: the SAME shared sources FleetMcp.profiles/fleet_profiles reads (BackendQuarantine
|
||||
// and BackendOutagePolicy are each one instance for the whole daemon — see Fleetd wiring) so
|
||||
// GET /profiles cannot drift from fleet_profiles about which profile is quarantined/cooling off.
|
||||
// .none() (the honest "feature not wired" view) for every constructor that does not pass one.
|
||||
private final FleetMcp.QuarantineSource quarantine;
|
||||
private final FleetMcp.OutageSource outage;
|
||||
private final ObjectMapper mapper = new ObjectMapper();
|
||||
|
||||
/**
|
||||
@@ -70,9 +102,9 @@ public final class FleetApp {
|
||||
* behaviour without each needing an auth fixture.
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet) {
|
||||
this(herdr, workers, sessions, messages, presence, mcpServlet, null, null);
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet) {
|
||||
this(herdr, workers, sessions, messages, presence, mcpServlet, null, null, presence::isPresent);
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -82,16 +114,75 @@ public final class FleetApp {
|
||||
* the endpoint
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics) {
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics) {
|
||||
this(herdr, workers, sessions, messages, presence, mcpServlet, auth, metrics, presence::isPresent);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param deliverable the injector's readiness gate, shared so status reports its real result
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable) {
|
||||
this(herdr, herdr, workers, sessions, messages, presence, mcpServlet, auth, metrics, deliverable);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param herdr the lead daemon's client
|
||||
* @param memberHerdr the member daemon's client (CB-185); pass the same instance as
|
||||
* {@code herdr} for a single-daemon deployment — {@code healthz}/{@code
|
||||
* sessions} then make exactly one herdr call each, unchanged from before
|
||||
* the two-daemon router existed
|
||||
* @param deliverable the injector's readiness gate, shared so status reports its real result
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, MemberCredentialPolicyView::absent);
|
||||
}
|
||||
|
||||
/**
|
||||
* @param memberCredentials live {@code memberCredentials:} policy view (fleetd #111), re-read
|
||||
* per request for {@code GET /member-credentials}; production wiring
|
||||
* passes the same hot-reload shape as every other live config read
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials) {
|
||||
this(herdr, memberHerdr, workers, sessions, messages, presence, mcpServlet, auth, metrics,
|
||||
deliverable, memberCredentials, FleetMcp.QuarantineSource.none(), FleetMcp.OutageSource.none());
|
||||
}
|
||||
|
||||
/**
|
||||
* @param quarantine the SAME {@link FleetMcp.QuarantineSource} instance passed to {@code
|
||||
* FleetMcp} (fleetd #297), so {@code GET /profiles} reports the identical
|
||||
* exhaustion-quarantine facts as {@code fleet_profiles} rather than a second,
|
||||
* independently-computed copy
|
||||
* @param outage the SAME {@link FleetMcp.OutageSource} instance passed to {@code FleetMcp} —
|
||||
* see {@code quarantine}; a SEPARATE check from it, never merged in
|
||||
*/
|
||||
public FleetApp(HerdrClient herdr, HerdrClient memberHerdr, PeerLauncher workers, SessionManager sessions,
|
||||
MessageService messages, MemberPresence presence,
|
||||
HttpServlet mcpServlet, CallerResolver auth, Metrics metrics,
|
||||
Predicate<String> deliverable, Supplier<MemberCredentialPolicyView> memberCredentials,
|
||||
FleetMcp.QuarantineSource quarantine, FleetMcp.OutageSource outage) {
|
||||
this.herdr = herdr;
|
||||
this.memberHerdr = memberHerdr != null ? memberHerdr : herdr;
|
||||
this.workers = workers;
|
||||
this.sessions = sessions;
|
||||
this.messages = messages;
|
||||
this.presence = presence;
|
||||
this.deliverable = deliverable;
|
||||
this.mcpServlet = mcpServlet;
|
||||
this.auth = auth;
|
||||
this.metrics = metrics;
|
||||
this.memberCredentials = memberCredentials != null ? memberCredentials : MemberCredentialPolicyView::absent;
|
||||
this.quarantine = quarantine != null ? quarantine : FleetMcp.QuarantineSource.none();
|
||||
this.outage = outage != null ? outage : FleetMcp.OutageSource.none();
|
||||
}
|
||||
|
||||
/** Wire routes onto a fresh, unstarted Javalin instance. Caller starts it. */
|
||||
@@ -120,6 +211,7 @@ public final class FleetApp {
|
||||
app.get("/agents", this::agents);
|
||||
app.get("/members", this::listMembers); // CB-304: registry roster + live herdr status
|
||||
app.get("/profiles", this::profiles); // configured backend profiles
|
||||
app.get("/member-credentials", this::memberCredentials); // fleetd #111: policy names + counts, never a value
|
||||
app.post("/members", this::spawnMember); // optional ?role=&profile= or {"role":…,"profile":…}
|
||||
app.delete("/members/{paneId}", this::stopMember);
|
||||
app.post("/sessions/{id}/message", this::sendMessage); // fleet_send (primary; blocking, wait:false, or answer via turnId)
|
||||
@@ -173,36 +265,91 @@ public final class FleetApp {
|
||||
|
||||
/** Prometheus scrape endpoint (CB-502). */
|
||||
private void metrics(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.METRICS, null)) {
|
||||
if (!allow(ctx, routeAction("GET /metrics"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).contentType("text/plain; version=0.0.4; charset=utf-8").result(metrics.render());
|
||||
}
|
||||
|
||||
/** Liveness + herdr reachability. 200 when herdr answers ping, 503 otherwise. */
|
||||
/**
|
||||
* Liveness + herdr reachability. 200 only when BOTH daemons answer ping — 503 otherwise
|
||||
* (CB-185). With no {@code memberHerdrSocket} configured {@code memberHerdr == herdr}, so this
|
||||
* makes exactly the one {@code ping} call it always did and reports the same body; with a
|
||||
* second daemon configured, a member daemon that is down must not be masked by a healthy lead
|
||||
* daemon — every spawn goes through the member daemon and would otherwise fail silently behind
|
||||
* a green {@code /healthz}.
|
||||
*
|
||||
* <p>CB-185 blocker 2: the {@code herdr} key always carries the <em>lead</em> daemon's
|
||||
* version/protocol, unchanged, because two consumers — {@code scripts/redeploy-fleetd.sh} and
|
||||
* {@code scripts/rename-checkout.sh} — read this endpoint already (both only check the HTTP
|
||||
* status code and print the body verbatim; neither parses a specific field, so adding a key
|
||||
* alongside {@code herdr} is safe). But it is the <em>member</em> daemon's protocol that decides
|
||||
* whether a spawn works, so when a second daemon is configured its version/protocol is reported
|
||||
* too, under a separate {@code member} key — never folded into {@code herdr}, which would make a
|
||||
* mismatch invisible to whichever consumer only reads that key. If the two protocol numbers
|
||||
* differ, {@code protocolMismatch: true} calls it out explicitly rather than leaving it to be
|
||||
* spotted by comparing two numbers by eye.
|
||||
*/
|
||||
private void healthz(Context ctx) {
|
||||
JsonNode pong;
|
||||
try {
|
||||
JsonNode pong = herdr.call("ping");
|
||||
ctx.status(200).json(Map.of(
|
||||
"status", "ok",
|
||||
"herdr", Map.of(
|
||||
"version", pong.path("version").asText(""),
|
||||
"protocol", pong.path("protocol").asInt())));
|
||||
pong = herdr.call("ping");
|
||||
} catch (HerdrException e) {
|
||||
ctx.status(503).json(Map.of(
|
||||
"status", "degraded",
|
||||
"herdr", "unreachable",
|
||||
"detail", e.getMessage()));
|
||||
}
|
||||
}
|
||||
|
||||
/** Sessions view derived from herdr {@code workspace.list} (one workspace → one row). */
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
return;
|
||||
}
|
||||
JsonNode result = herdr.call("workspace.list");
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("status", "ok");
|
||||
body.put("herdr", Map.of(
|
||||
"version", pong.path("version").asText(""),
|
||||
"protocol", pong.path("protocol").asInt()));
|
||||
if (memberHerdr != herdr) {
|
||||
JsonNode memberPong;
|
||||
try {
|
||||
memberPong = memberHerdr.call("ping");
|
||||
} catch (HerdrException e) {
|
||||
ctx.status(503).json(Map.of(
|
||||
"status", "degraded",
|
||||
"herdr", "member unreachable",
|
||||
"detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
int leadProtocol = pong.path("protocol").asInt();
|
||||
int memberProtocol = memberPong.path("protocol").asInt();
|
||||
body.put("member", Map.of(
|
||||
"version", memberPong.path("version").asText(""),
|
||||
"protocol", memberProtocol));
|
||||
if (leadProtocol != memberProtocol) {
|
||||
body.put("protocolMismatch", true);
|
||||
}
|
||||
}
|
||||
ctx.status(200).json(body);
|
||||
}
|
||||
|
||||
/**
|
||||
* Sessions view derived from herdr {@code workspace.list} (one workspace → one row), merged
|
||||
* across both daemons (CB-185). With no {@code memberHerdrSocket} configured {@code
|
||||
* memberHerdr == herdr}, so this calls {@code workspace.list} exactly once, same as before the
|
||||
* router existed; with a second daemon configured, calling it twice would silently drop every
|
||||
* member workspace (they live on the member daemon only).
|
||||
*/
|
||||
private void sessions(Context ctx) {
|
||||
if (!allow(ctx, routeAction("GET /sessions"), null)) {
|
||||
return;
|
||||
}
|
||||
List<Map<String, Object>> out = new ArrayList<>();
|
||||
collectSessions(herdr, out);
|
||||
if (memberHerdr != herdr) {
|
||||
collectSessions(memberHerdr, out);
|
||||
}
|
||||
ctx.status(200).json(Map.of("sessions", out));
|
||||
}
|
||||
|
||||
private static void collectSessions(HerdrClient client, List<Map<String, Object>> out) {
|
||||
JsonNode result = client.call("workspace.list");
|
||||
for (JsonNode w : result.path("workspaces")) {
|
||||
out.add(Map.of(
|
||||
"id", w.path("workspace_id").asText(""),
|
||||
@@ -211,49 +358,103 @@ public final class FleetApp {
|
||||
"paneCount", w.path("pane_count").asInt(),
|
||||
"agentStatus", w.path("agent_status").asText("unknown")));
|
||||
}
|
||||
ctx.status(200).json(Map.of("sessions", out));
|
||||
}
|
||||
|
||||
/** Discovery: every agent herdr tracks, keyed by its Claude session UUID. */
|
||||
private void agents(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /agents"), null)) {
|
||||
return;
|
||||
}
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
try {
|
||||
ctx.status(200).json(Map.of("agents",
|
||||
workers.list().stream().map(Agent.class::cast).map(FleetApp::view).toList()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: workers.list() reaches herdr — a transport failure must land in the same
|
||||
// {error, detail} envelope every other failure path here uses, not escape as a bare
|
||||
// exception and leave Javalin's default handling to respond outside the JSON contract.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** CB-304: bridge-owned roster merged with live herdr status by paneId. */
|
||||
private void listMembers(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /members"), null)) {
|
||||
return;
|
||||
}
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
List<Map<String, Object>> out = sessions.roster().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
try {
|
||||
// CB-519: the registry key is a host-unique id, not the pane coordinate — join on terminal.
|
||||
Map<String, Agent> live = workers.list().stream()
|
||||
.map(Agent.class::cast)
|
||||
.filter(a -> a.terminalId() != null)
|
||||
.collect(Collectors.toMap(Agent::terminalId, Function.identity(), (_, b) -> b));
|
||||
// fleetd #209: this REST roster reports agentSessionId via SessionManager.rosterView, so it
|
||||
// uses the resolving roster read (caller-driven, not a timer) rather than the plain one.
|
||||
List<Map<String, Object>> out = sessions.rosterResolved().stream()
|
||||
.map(s -> SessionManager.rosterView(s, live.get(s.terminalId())))
|
||||
.toList();
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
// fleetd #199: the endpoint became /members in the CB-634 rename but the body key stayed
|
||||
// "workers", so a caller that read "members" saw an empty fleet and reported no members at
|
||||
// all. "members" is the canonical key; "workers" stays as a deprecated alias so an existing
|
||||
// REST consumer keeps working — the out-of-band path a lead falls back to when its MCP mount
|
||||
// drops reads this endpoint. Drop the alias once nothing reads it.
|
||||
body.put("members", out);
|
||||
body.put("workers", out);
|
||||
// CB-586: operator visibility for the refs/wip snapshot store without shelling into the
|
||||
// repo — how many snapshot refs exist and roughly what they cost. Present only once a
|
||||
// worktree session has established the repo, so a never-snapshotted fleet reports nothing.
|
||||
sessions.wipRefs().ifPresent(st -> body.put("wipRefs",
|
||||
Map.of("count", st.count(), "costBytes", st.costBytes())));
|
||||
ctx.status(200).json(body);
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #297: same reasoning as agents() above — this is the out-of-band roster a lead
|
||||
// falls back to when its MCP mount drops, so it must stay inside the JSON error contract
|
||||
// exactly when herdr is briefly unreachable, not escape as a bare exception.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
/** The configured worker profiles and which one a no-argument spawn uses. */
|
||||
/**
|
||||
* The configured worker profiles, which one a no-argument spawn uses, and (fleetd #297) the two
|
||||
* outage states {@code fleet_profiles} already reports: {@code quarantined} (CB-578 stage B —
|
||||
* the backend reported it out of capacity) and {@code coolingOff} (fleetd #201 Unit 5 — the
|
||||
* credential threw repeated non-exhaustion backend errors). Both are read from the SAME shared
|
||||
* {@link FleetMcp.QuarantineSource}/{@link FleetMcp.OutageSource} instances {@code FleetMcp}
|
||||
* reads, never recomputed, so the two doors cannot disagree about which profile is down and why.
|
||||
* Independent checks, so a profile can appear in both maps at once; each map is present only
|
||||
* when at least one profile is in that state.
|
||||
*/
|
||||
private void profiles(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /profiles"), null)) {
|
||||
return;
|
||||
}
|
||||
// fleetd #297: ONE body builder, shared with fleet_profiles. Handing both doors the same
|
||||
// QuarantineSource/OutageSource instances stops them reading different facts; rendering
|
||||
// through the same method stops them reporting those facts differently. Both are needed.
|
||||
ctx.status(200).json(FleetMcp.profilesView(workers, quarantine, outage));
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #111 (CB-608): the live {@code memberCredentials:} policy as names and counts —
|
||||
* NEVER a value. The daemon does not hold a credential's value in the first place (only the
|
||||
* name it is configured under), so there is nothing to redact here beyond what {@link
|
||||
* MemberCredentialPolicyView} already omits by construction. This is the source
|
||||
* {@code scripts/probe-member-credentials.sh} reads instead of carrying its own hardcoded
|
||||
* name list, which is exactly what let the list drift silently behind the real policy.
|
||||
*/
|
||||
private void memberCredentials(Context ctx) {
|
||||
if (!allow(ctx, routeAction("GET /member-credentials"), null)) {
|
||||
return;
|
||||
}
|
||||
MemberCredentialPolicyView view = memberCredentials.get();
|
||||
ctx.status(200).json(Map.of(
|
||||
"profiles", workers.profiles(),
|
||||
"default", workers.defaultProfile() == null ? "" : workers.defaultProfile()));
|
||||
"present", view.present(),
|
||||
"policy", view.policy() == null ? "" : view.policy(),
|
||||
"known", view.known(),
|
||||
"allowed", view.allowed(),
|
||||
"knownCount", view.knownCount(),
|
||||
"allowedCount", view.allowedCount(),
|
||||
"blockedCount", view.blockedCount()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -262,7 +463,7 @@ public final class FleetApp {
|
||||
* the subscription boundary, 400 for an unknown profile.
|
||||
*/
|
||||
private void spawnMember(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.SPAWN, null)) {
|
||||
if (!allow(ctx, routeAction("POST /members"), null)) {
|
||||
return;
|
||||
}
|
||||
String role = ctx.queryParam("role");
|
||||
@@ -301,6 +502,11 @@ public final class FleetApp {
|
||||
ctx.status(201).json(view(member));
|
||||
} catch (GuardException e) {
|
||||
ctx.status(403).json(Map.of("error", "subscription_boundary", "detail", e.getMessage()));
|
||||
} catch (ShuttingDownException e) {
|
||||
// fleetd #308: the daemon's shutdown drain has already started — 503, not a bare 500,
|
||||
// so this reads the same as PlacementException below: valid request, refused because
|
||||
// of a transient daemon state rather than a bad argument.
|
||||
ctx.status(503).json(Map.of("error", "shutting_down", "detail", e.getMessage()));
|
||||
} catch (PlacementException e) {
|
||||
// CB-599: no candidate had capacity (maxLoad, quarantine, or all-exhausted) — a benign,
|
||||
// likely-transient refusal, distinct from "profile does not exist" below. 503: the
|
||||
@@ -310,6 +516,12 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "unknown_profile", "detail", e.getMessage()));
|
||||
} catch (PeerUnreachableException e) {
|
||||
ctx.status(502).json(Map.of("error", "spawn_timeout", "detail", e.getMessage()));
|
||||
} catch (HerdrException e) {
|
||||
// fleetd #304: not every herdr failure on the spawn path is a readiness timeout, so
|
||||
// PeerUnreachableException above does not cover this. Without this catch the exception
|
||||
// escapes to Javalin's default 500, while fleet_spawn reports the same failure as a
|
||||
// clean named error (FleetMcp.spawn) — the #297 one-door-guarded shape.
|
||||
herdrError(ctx, e);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -330,13 +542,29 @@ public final class FleetApp {
|
||||
return (s == null || s.isBlank()) ? null : s;
|
||||
}
|
||||
|
||||
/** Tear a worker down by pane id. */
|
||||
/**
|
||||
* Tear a worker down by pane id.
|
||||
*
|
||||
* <p>fleetd #304: the {@code HerdrException} catch is not cosmetic. {@code release} deregisters
|
||||
* the session, notifies the release listener and preserves a dirty worktree <em>before</em> it
|
||||
* calls {@code launcher.stop}, so a throw from that stop arrives after the teardown the caller
|
||||
* asked for has already happened. Letting it escape gave Javalin's default 500, which tells the
|
||||
* caller to retry — and the retry finds nothing in the registry, reaches the same stop, and
|
||||
* throws again, so it can never succeed. {@code herdrError} instead answers 404 ("the pane is
|
||||
* gone, stop retrying") or 502 ("herdr is upstream and broken, a retry may help"), matching what
|
||||
* {@code fleet_stop} reports for the same failure.
|
||||
*/
|
||||
private void stopMember(Context ctx) {
|
||||
String paneId = ctx.pathParam("paneId");
|
||||
if (!allow(ctx, Authz.Action.STOP, paneId)) {
|
||||
if (!allow(ctx, routeAction("DELETE /members/{paneId}"), paneId)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
sessions.release(paneId);
|
||||
} catch (HerdrException e) {
|
||||
herdrError(ctx, e);
|
||||
return;
|
||||
}
|
||||
sessions.release(paneId);
|
||||
ctx.status(204);
|
||||
}
|
||||
|
||||
@@ -348,7 +576,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void sendMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.SEND, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/message"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -435,7 +663,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void askMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.ASK, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/ask"), id)) {
|
||||
return;
|
||||
}
|
||||
String question;
|
||||
@@ -468,13 +696,17 @@ public final class FleetApp {
|
||||
/**
|
||||
* The worker's structured reply ({@code fleet_reply}) — resolves the blocking send awaiting
|
||||
* on this session, or queues the reply in the inbox when no send is open (CB-307).
|
||||
*
|
||||
* <p>fleetd #365: the response body's {@code delivered} field used to be unconditionally
|
||||
* {@code true} for either case; it now reports whether a send/ticket was actually resolved,
|
||||
* with {@code outcome} naming which (see {@link MessageService.ReplyOutcome}).
|
||||
*/
|
||||
private void replyMessage(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
// The rule that matters: a worker may reply only as itself. Over MCP this was already true
|
||||
// structurally (identity comes from the connection, never an argument); over REST the path
|
||||
// id was simply trusted, so this is where the invariant actually gets enforced.
|
||||
if (!allow(ctx, Authz.Action.REPLY, id)) {
|
||||
if (!allow(ctx, routeAction("POST /sessions/{id}/reply"), id)) {
|
||||
return;
|
||||
}
|
||||
String content;
|
||||
@@ -484,8 +716,26 @@ public final class FleetApp {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", "body must be JSON"));
|
||||
return;
|
||||
}
|
||||
messages.reply(id, content);
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", true));
|
||||
// fleetd #302: content is required. `.path("content").asText("")` above turns a missing key
|
||||
// into "" rather than throwing, so without this check an empty/blank reply used to reach
|
||||
// messages.reply(...) and silently resolve the lead's waiter — the same class of bug as the
|
||||
// sibling "content is required" guards on sendMessage/askMessage below, except this one wrote
|
||||
// a WRONG value instead of failing loudly. The check lives in MessageService.reply so both
|
||||
// this door and FleetMcp.reply inherit the same rule; this catch only translates it into the
|
||||
// {error, detail} envelope this file uses everywhere else.
|
||||
MessageService.ReplyOutcome outcome;
|
||||
try {
|
||||
outcome = messages.reply(id, content);
|
||||
} catch (IllegalArgumentException e) {
|
||||
ctx.status(400).json(Map.of("error", "bad_request", "detail", e.getMessage()));
|
||||
return;
|
||||
}
|
||||
// fleetd #365: "delivered": true used to be unconditional here, whether the reply resolved
|
||||
// a waiting send or was merely queued in the inbox for a later drain — the same gap
|
||||
// FleetMcp.reply had over MCP. `delivered` now reflects which actually happened, and
|
||||
// `outcome` names the specific case (see MessageService.ReplyOutcome).
|
||||
ctx.status(200).json(Map.of("sessionId", id, "delivered", outcome.delivered(),
|
||||
"outcome", outcome.wireName()));
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -495,7 +745,7 @@ public final class FleetApp {
|
||||
*/
|
||||
private void drainReplies(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.DRAIN, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/replies"), id)) {
|
||||
return;
|
||||
}
|
||||
var replies = messages.drainReplies(id);
|
||||
@@ -507,20 +757,20 @@ public final class FleetApp {
|
||||
|
||||
/**
|
||||
* Live lifecycle status of a worker (MCP `fleet_status` wraps this in CB-105), plus its
|
||||
* <em>readiness</em> (CB-113): {@code ready} is true once the worker's Claude has connected the
|
||||
* bridge MCP — the reliable "available to receive a task" signal, unlike bare {@code idle}, which
|
||||
* is also true during boot.
|
||||
* <em>readiness</em>: {@code ready} is true when the injector can deliver to the target. A
|
||||
* spawned member must connect the bridge MCP first, while a known lead is ready without member
|
||||
* presence. This differs from bare {@code idle}, which is also true during member boot.
|
||||
*/
|
||||
private void sessionStatus(Context ctx) {
|
||||
String id = ctx.pathParam("id");
|
||||
if (!allow(ctx, Authz.Action.READ, id)) {
|
||||
if (!allow(ctx, routeAction("GET /sessions/{id}/status"), id)) {
|
||||
return;
|
||||
}
|
||||
try {
|
||||
Map<String, Object> body = new LinkedHashMap<>();
|
||||
body.put("sessionId", id);
|
||||
body.put("status", messages.status(id).name().toLowerCase());
|
||||
body.put("ready", presence.isPresent(id));
|
||||
body.put("ready", deliverable.test(id));
|
||||
// CB-582: a worker paused mid-turn in an async fleet_ask is otherwise invisible to a
|
||||
// status poll — surface the open question and how to answer it, same as fleet_poll's
|
||||
// Phase.ASKING view.
|
||||
@@ -538,7 +788,7 @@ public final class FleetApp {
|
||||
|
||||
/** Poll an async (wait:false) delegation by ticket. 404 for an unknown/expired ticket. */
|
||||
private void taskStatus(Context ctx) {
|
||||
if (!allow(ctx, Authz.Action.READ, null)) {
|
||||
if (!allow(ctx, routeAction("GET /tasks/{ticket}"), null)) {
|
||||
return;
|
||||
}
|
||||
MessageService.TaskView v = messages.poll(ctx.pathParam("ticket"));
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -46,7 +46,8 @@ public record MemberSession(
|
||||
String worktree,
|
||||
String branch,
|
||||
CharterReceipt charterReceipt,
|
||||
String agentSessionId) {
|
||||
String agentSessionId,
|
||||
String failureReason) {
|
||||
|
||||
/** One-shot worker lifecycle states. */
|
||||
public enum State {
|
||||
@@ -54,6 +55,7 @@ public record MemberSession(
|
||||
READY,
|
||||
BUSY,
|
||||
DONE,
|
||||
BACKEND_ERROR,
|
||||
FAILED,
|
||||
RELEASED
|
||||
}
|
||||
@@ -69,24 +71,51 @@ public record MemberSession(
|
||||
long lastActivityAtNanos, int turnCount, State state,
|
||||
String worktree, String branch) {
|
||||
this(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, null, null);
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, null, null, null);
|
||||
}
|
||||
|
||||
/** Backward-compatible shape without a backend failure reason. */
|
||||
public MemberSession(String paneId, String terminalId, String profile, MemberRole role,
|
||||
String cwd, String ownerTerminal, long spawnedAtNanos,
|
||||
long lastActivityAtNanos, int turnCount, State state,
|
||||
String worktree, String branch, CharterReceipt charterReceipt,
|
||||
String agentSessionId) {
|
||||
this(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, null);
|
||||
}
|
||||
|
||||
/** Return a copy of this session in {@code state}. */
|
||||
public MemberSession withState(State state) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with {@code lastActivityAtNanos} updated to {@code nowNanos}. */
|
||||
public MemberSession withActivity(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
nowNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with the turn count incremented and activity timestamped at {@code nowNanos}. */
|
||||
public MemberSession bumpTurn(long nowNanos) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
nowNanos, turnCount + 1, state, worktree, branch, charterReceipt, agentSessionId);
|
||||
nowNanos, turnCount + 1, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/**
|
||||
* Return a copy with {@code agentSessionId} resolved to a non-null value (fleetd #209). Some
|
||||
* adapters (opencode) cannot answer {@link dev.ltms.fleet.peer.PeerHandle#agentSessionId()} at
|
||||
* spawn time — the peer has not persisted its session record yet — so the id is discovered on
|
||||
* a later poll and swapped into the otherwise-immutable session via this wither.
|
||||
*/
|
||||
public MemberSession withAgentSessionId(String agentSessionId) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
|
||||
/** Return a copy with the durable backend failure detail. */
|
||||
public MemberSession withFailureReason(String failureReason) {
|
||||
return new MemberSession(paneId, terminalId, profile, role, cwd, ownerTerminal, spawnedAtNanos,
|
||||
lastActivityAtNanos, turnCount, state, worktree, branch, charterReceipt, agentSessionId, failureReason);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -10,6 +10,7 @@ import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import org.slf4j.Logger;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
@@ -21,6 +22,7 @@ import java.util.Optional;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.ConcurrentHashMap;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicBoolean;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.function.Consumer;
|
||||
import java.util.function.LongSupplier;
|
||||
@@ -46,12 +48,23 @@ public final class SessionManager implements TurnListener {
|
||||
private final PeerLauncher launcher;
|
||||
private final Worktrees worktrees;
|
||||
private final ConcurrentHashMap<String /*paneId*/, MemberSession> registry = new ConcurrentHashMap<>();
|
||||
/**
|
||||
* fleetd #209: the live {@link PeerHandle} for every registered pane, retained solely so
|
||||
* {@link #resolveAgentSessionId} can re-poll {@link PeerHandle#agentSessionId()} after spawn.
|
||||
* The handle used to go out of scope at the end of the spawn method, so a launcher that answers
|
||||
* the id lazily (opencode — the on-disk session row is written after the pane is created) could
|
||||
* never be re-asked, and {@code fleet_list}/{@code fleet_spawn resumeSessionId} never saw it.
|
||||
* Populated on every spawn path, removed on {@link #release}.
|
||||
*/
|
||||
private final ConcurrentHashMap<String /*paneId*/, PeerHandle> handles = new ConcurrentHashMap<>();
|
||||
private final MemberPresence presence;
|
||||
private final SecureRandom nonceRandom = new SecureRandom();
|
||||
private final AtomicLong nonceSeq = new AtomicLong();
|
||||
private final LongSupplier nowNanos;
|
||||
private final int contextCap;
|
||||
private final boolean clearAfterTurn;
|
||||
/** Null in production; test seam for the interval before an idle session's conditional release. */
|
||||
private final Consumer<MemberSession> beforeIdleRelease;
|
||||
private volatile MemberLifecycle memberLifecycle = MemberLifecycle.NONE;
|
||||
/**
|
||||
* CB-586: the repo root the fleet actually works in, remembered the first time a worktree
|
||||
@@ -62,6 +75,16 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private volatile String fleetRepoRoot;
|
||||
|
||||
/**
|
||||
* fleetd #308: flips true the instant {@link #drainAll} starts, before its registry snapshot
|
||||
* is even taken — so a spawn already in flight sees the refusal as early as a plain flag can
|
||||
* make it. This alone cannot close the race completely: a caller that read {@code false} just
|
||||
* before the flip can still land in the registry after the snapshot. {@link #drainAll}'s
|
||||
* post-loop sweep is what catches that straggler; the two mechanisms are deliberately paired,
|
||||
* see {@link #drainAll}'s javadoc.
|
||||
*/
|
||||
private final AtomicBoolean draining = new AtomicBoolean(false);
|
||||
|
||||
/** CB-520: notified with a terminalId on every acquire; no-op until wired. */
|
||||
private final List<Consumer<String>> acquireListeners = new java.util.concurrent.CopyOnWriteArrayList<>();
|
||||
/** CB-516: notified with a {@link ReleaseDetail} on every release; no-op until wired. */
|
||||
@@ -93,13 +116,23 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
public SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
int contextCap, boolean clearAfterTurn) {
|
||||
this(launcher, worktrees, nowNanos, contextCap, clearAfterTurn, null);
|
||||
}
|
||||
|
||||
/**
|
||||
* Package-private constructor for a deterministic reap/delivery race test. Production callers
|
||||
* use the constructor above, whose null hook adds no callback or lock to an ordinary reap.
|
||||
*/
|
||||
SessionManager(PeerLauncher launcher, Worktrees worktrees, LongSupplier nowNanos,
|
||||
int contextCap, boolean clearAfterTurn, Consumer<MemberSession> beforeIdleRelease) {
|
||||
this.launcher = launcher;
|
||||
this.worktrees = worktrees;
|
||||
this.presence = new PresenceFleet(this);
|
||||
this.nowNanos = nowNanos;
|
||||
this.contextCap = contextCap;
|
||||
this.clearAfterTurn = clearAfterTurn;
|
||||
this.beforeIdleRelease = beforeIdleRelease;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -172,47 +205,61 @@ public final class SessionManager implements TurnListener {
|
||||
public MemberSession acquire(String profile, MemberRole role, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
// fleetd #308: refuse before anything else runs — no slot reservation, no launcher spawn —
|
||||
// so a caller learns the daemon is going down instead of getting a session drainAll will
|
||||
// never see again. Checked here because every other acquire(...) overload delegates to
|
||||
// this one, so this is the single point every spawn path passes through.
|
||||
if (draining.get()) {
|
||||
throw new ShuttingDownException("fleetd is shutting down; refusing to spawn a session "
|
||||
+ "the shutdown drain would never see");
|
||||
}
|
||||
MemberRole memberRole = (role == null) ? MemberRole.DEV : role;
|
||||
requireResumeCapability(profile, resumeSessionId);
|
||||
// CB-619 / fleetd #123: an explicit profile bypasses placement (CompositePeerLauncher only
|
||||
// constrains an UNQUALIFIED spawn to the role's pool), so it is the one path that can ask
|
||||
// for a role with no slot to bind it to. Refuse before anything spawns. A blank profile is
|
||||
// left to placement, which already restricts an unqualified spawn to the role's pool.
|
||||
if (profile != null && !profile.isBlank()) {
|
||||
memberLifecycle.requireSlotFor(memberRole, profile);
|
||||
}
|
||||
MemberLifecycle.SlotReservation reservation = memberLifecycle.reserve(memberRole, profile);
|
||||
String launchProfile = reservation == null ? profile : reservation.profile();
|
||||
if (wt == null) {
|
||||
// CB-557: the role must ride on the SpawnRequest, not stay a local. The launcher needs it
|
||||
// to pick the profile out of that role's pool and to label the tab; a role kept only on
|
||||
// the MemberSession is recorded after the spawn it was supposed to steer.
|
||||
SpawnRequest req = new SpawnRequest(profile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
SpawnRequest req = new SpawnRequest(launchProfile, requestedCwd, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
PeerHandle handle;
|
||||
boolean bound = false;
|
||||
try {
|
||||
handle = launcher.spawn(req);
|
||||
String resolvedProfile = resolveProfile(handle, launchProfile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
bound = reservation == null || actualRole == MemberRole.ARCHITECT;
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(), handle.terminalId(), resolvedProfile, actualRole, cwd, ownerTerminal, now, now, 0,
|
||||
MemberSession.State.SPAWNING, null, null, handle.charterReceipt(), handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={}: {}", profile, memberRole, e.getMessage());
|
||||
if (!bound) memberLifecycle.release(reservation);
|
||||
log.warn("spawn failed for profile={} role={}: {}", launchProfile, memberRole, e.getMessage());
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, requestedCwd, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
now,
|
||||
0,
|
||||
MemberSession.State.SPAWNING,
|
||||
null,
|
||||
null,
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
log.debug("acquired session id={} terminal={} profile={} owner={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.ownerTerminal());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
return acquireWithWorktree(profile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId);
|
||||
try {
|
||||
return acquireWithWorktree(launchProfile, memberRole, requestedCwd, callerCwd, ownerTerminal, wt,
|
||||
sessionName, resumeSessionId, reservation);
|
||||
} catch (RuntimeException e) {
|
||||
memberLifecycle.release(reservation);
|
||||
throw e;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -257,6 +304,32 @@ public final class SessionManager implements TurnListener {
|
||||
*/
|
||||
private void release(String paneId, ReleaseCause cause) {
|
||||
MemberSession removed = registry.remove(paneId);
|
||||
releaseRemoved(paneId, removed, handles.remove(paneId), cause);
|
||||
}
|
||||
|
||||
/**
|
||||
* Tear a session down only while {@code expected} is still its registry value. A lifecycle
|
||||
* transition replaces the immutable record, so this prevents a reap based on an old READY or
|
||||
* DONE record from stopping a worker that delivery has made BUSY.
|
||||
*/
|
||||
private boolean releaseIfCurrent(MemberSession expected, ReleaseCause cause) {
|
||||
if (!registry.remove(expected.paneId(), expected)) {
|
||||
// A lifecycle transition replaced the record between the caller's check and this remove.
|
||||
// Log it: this race is by definition unobservable otherwise, and a reaper that silently
|
||||
// declines to reap is the hardest kind of behaviour to diagnose after the fact.
|
||||
log.debug("skipping reap of pane={}: its registry record changed after the idle check "
|
||||
+ "(most likely a delivery made it BUSY)", expected.paneId());
|
||||
return false;
|
||||
}
|
||||
releaseRemoved(expected.paneId(), expected, handles.remove(expected.paneId()), cause);
|
||||
return true;
|
||||
}
|
||||
|
||||
private void releaseRemoved(String paneId, MemberSession removed, PeerHandle removedHandle,
|
||||
ReleaseCause cause) {
|
||||
// fleetd #209: remove right alongside the registry entry so a released session's handle is
|
||||
// never leaked — but keep the local reference below, so the id can still be resolved for
|
||||
// the ReleaseDetail this teardown notifies with.
|
||||
boolean preserveWorktree = cause == ReleaseCause.SHUTDOWN;
|
||||
String snapshotRef = null;
|
||||
if (removed != null) {
|
||||
@@ -303,8 +376,11 @@ public final class SessionManager implements TurnListener {
|
||||
// too, so a failed ticket's detail can point a lead at the same tree to re-dispatch.
|
||||
// CB-584 (issue #65 criterion 5): carry agentSessionId alongside them, so a lead can
|
||||
// also resume the member's conversation, not just re-dispatch onto its files.
|
||||
notifyReleased(new ReleaseDetail(removed.terminalId(), removed.worktree(),
|
||||
removed.branch(), snapshotRef, removed.agentSessionId()));
|
||||
// fleetd #209: a late-resolving adapter (opencode) may only now have an id — resolve
|
||||
// one last time so a released member's detail carries the id it now has.
|
||||
MemberSession resolved = resolveAgentSessionId(removed, removedHandle);
|
||||
notifyReleased(new ReleaseDetail(resolved.terminalId(), resolved.worktree(),
|
||||
resolved.branch(), snapshotRef, resolved.agentSessionId()));
|
||||
}
|
||||
}
|
||||
// CB-581: the pane must always stop, even if the dirty check above threw. A session removed
|
||||
@@ -312,7 +388,63 @@ public final class SessionManager implements TurnListener {
|
||||
// slot that no longer appears in the roster and can never be reclaimed.
|
||||
launcher.stop(paneId);
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
// fleetd #316: the `dirty` read above ran while the worker could still write to this
|
||||
// worktree, so a stale `false` must not be trusted to authorise the --force removal
|
||||
// below. Re-read the worktree's state one more time, right here — immediately before
|
||||
// the one step that would destroy it, and only on the path that is actually about to
|
||||
// do that (invariant 4: no second unconditional `git status` on a release that already
|
||||
// decided to preserve). By now `launcher.stop` has returned, so this read reflects
|
||||
// whatever the worker managed to write up to and including its teardown, not whatever
|
||||
// it had written at release-start time.
|
||||
if (dirtyImmediatelyBeforeRemoval(removed)) {
|
||||
preserveWorktree = true;
|
||||
// The pre-stop snapshot above never ran for this session (the pre-stop read said
|
||||
// clean), so this is the only chance to get the newly-discovered work into
|
||||
// refs/wip/* rather than leaving the on-disk preserve as the sole copy. Best-effort,
|
||||
// like every other snapshot attempt — trySnapshot logs and swallows its own failure.
|
||||
String lateSnapshotRef = trySnapshot(removed, cause);
|
||||
log.warn("release {} preserves worktree {} for pane={} terminal={}: it reported "
|
||||
+ "clean before the pane stopped but dirty immediately before removal — the "
|
||||
+ "worker wrote to it during teardown, and --force removing it now would "
|
||||
+ "have destroyed that work{}",
|
||||
cause, removed.worktree(), paneId, removed.terminalId(),
|
||||
lateSnapshotRef == null ? "" : " (snapshotted to refs/wip/" + removed.branch()
|
||||
+ " commit=" + lateSnapshotRef + ")");
|
||||
}
|
||||
}
|
||||
if (removed != null && !preserveWorktree && removed.worktree() != null) {
|
||||
// fleetd #283: this is the one cleanup step in this method that used to be bare. By the
|
||||
// time it runs, the registry entry, the retained handle, and the pane are all already
|
||||
// gone — so a throw here (a stale index lock, a slow filesystem, `remove`'s own 30s exec
|
||||
// timeout) must not escape release(): there is no retry path (a second stop on this
|
||||
// paneId is a no-op), and the caller would otherwise see a "failed stop" for a session
|
||||
// that is in fact fully torn down. Log and swallow, matching every sibling step above.
|
||||
try {
|
||||
worktrees.remove(worktrees.repoRoot(removed.cwd()), removed.worktree());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("failed to remove worktree {} for pane={} terminal={} after release: the "
|
||||
+ "pane is already stopped and the session already deregistered, so this is "
|
||||
+ "not retryable — the directory must be reclaimed manually: {}",
|
||||
removed.worktree(), paneId, removed.terminalId(), e.toString());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* fleetd #316: the read that actually authorises {@code worktrees.remove}, taken with the
|
||||
* worker's pane already stopped. Fails toward preserving (returns {@code true}) on any
|
||||
* exception — the same rule the pre-stop check applies (CB-581): once we can no longer tell
|
||||
* whether the worktree is dirty, preserving costs disk while deleting on a guess can destroy
|
||||
* work that has no other copy.
|
||||
*/
|
||||
private boolean dirtyImmediatelyBeforeRemoval(MemberSession removed) {
|
||||
try {
|
||||
return worktrees.hasUncommitted(removed.worktree());
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("release could not re-check worktree {} for pane={} terminal={} immediately "
|
||||
+ "before removal; preserving it rather than risk destroying unsaved work: {}",
|
||||
removed.worktree(), removed.paneId(), removed.terminalId(), e.toString());
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -451,9 +583,67 @@ public final class SessionManager implements TurnListener {
|
||||
|
||||
private MemberSession acquireWithWorktree(String profile, MemberRole memberRole, String requestedCwd, String callerCwd,
|
||||
String ownerTerminal, WorktreeRequest wt,
|
||||
String sessionName, String resumeSessionId) {
|
||||
String preResolvedProfile = (profile == null || profile.isBlank())
|
||||
? launcher.defaultProfile() : profile;
|
||||
String sessionName, String resumeSessionId,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
// fleetd #425 rework (round 2): resolved through launcher.place(memberRole) — the same
|
||||
// candidate list, quarantine/cool-off/model-off filtering, and PlacementPolicy an unqualified
|
||||
// spawn of this role is actually judged against right now — never launcher.defaultProfile()
|
||||
// (MemberRole.DEV only, wrong for any other role) and never launcher.defaultProfileFor()
|
||||
// (the role's pool FIRST entry, blind to quarantine/cool-off/model-off: a first-round fix
|
||||
// used exactly this and regressed fleetd #429's "the fleet keeps working when a model is
|
||||
// turned off" guarantee — a quarantined or model-off pool-first profile made this throw
|
||||
// instead of routing around it, which an unqualified spawn is supposed to do). This same
|
||||
// resolved decision is reused below for repoRoot, parityOverlay, AND the spawn itself so the
|
||||
// worktree is always provisioned for the profile the member actually runs on — the two could
|
||||
// disagree before fleetd #425: this name picked repoRoot/overlay, but the spawn below passed
|
||||
// the ORIGINAL (blank) profile through to placement, which re-resolves live and can pick a
|
||||
// different profile if the pool changed between the two reads, or a genuinely different one
|
||||
// under weighted/round-robin placement.
|
||||
//
|
||||
// Round 1 of this rework fed the resolved name back into launcher.spawn(SpawnRequest) as an
|
||||
// EXPLICIT profile. That was a mistake this round corrects, and the mistake is not that the
|
||||
// two branches apply different checks — they are SUPPOSED to disagree: the routing branch a
|
||||
// blank spawn takes treats a quarantined/cooling-off/at-cap/model-off/unreachable/weight-0
|
||||
// profile as a reason to fall through to the next candidate, while CompositePeerLauncher's
|
||||
// THROWING branch (enforceNotQuarantined/enforceNotCoolingOff/enforceMaxLoad/
|
||||
// enforceModelEnabled) treats naming that same profile explicitly as a reason to refuse
|
||||
// outright. That is correct: an operator who names a profile should get a refusal, not a
|
||||
// silent substitution onto a different backend. The mistake was turning a fall-through into
|
||||
// a refusal by accident — resolving a name via the routing side and then re-entering the
|
||||
// refusing side with it, for a placement the routing side had already approved by walking
|
||||
// past everything else.
|
||||
//
|
||||
// Before fleetd #435, this accident was reachable through maxLoad specifically: the default
|
||||
// `fixed` placement policy did not evaluate maxLoad at all for automatic selection, so an
|
||||
// at-cap pool-first profile that placement itself would have picked for a plain unqualified
|
||||
// spawn could die at enforceMaxLoad one call later, purely because this method's route to
|
||||
// the spawn passed through an explicit profile name — a failure a worktree-less unqualified
|
||||
// spawn never hit. fleetd #435 closed that gap (`fixed` now evaluates maxLoad exactly like
|
||||
// every other placement policy), so that specific failure can no longer happen — a
|
||||
// PlacementDecision this method resolves can no longer be at-cap in the first place. What
|
||||
// this round's fix still buys, now that maxLoad can no longer cause the accident: it keeps
|
||||
// the PlacementDecision from place() and hands it to launcher.spawn(SpawnRequest,
|
||||
// PlacementDecision) for an unqualified request, which spawns through the SAME routing
|
||||
// branch a blank spawn uses — no enforce* check is newly applied, and the window between the
|
||||
// placement decision and the spawn (in which the pool, a config reload, or another spawn
|
||||
// landing on the same profile could otherwise move the state) never reopens. An
|
||||
// explicitly-named profile still goes through launcher.spawn(SpawnRequest) and its throwing
|
||||
// branch, unchanged — that caller asked for one profile by name and still gets everything
|
||||
// enforceNotQuarantined/enforceNotCoolingOff/enforceMaxLoad/enforceModelEnabled decide about
|
||||
// it, refusal included.
|
||||
//
|
||||
// The one cost that remains, unchanged from round 1: an unqualified worktree-provisioned
|
||||
// spawn does not get CompositePeerLauncher's cross-candidate retry on a live
|
||||
// PeerUnreachableException raised by the backend itself at spawn time (a transport-level
|
||||
// failure placement cannot see in advance) — spawn(req, decision) commits to the one profile
|
||||
// place() already chose, the same way an explicit-profile spawn commits to its one name. That
|
||||
// trade is deliberate: a worktree provisioned for the wrong backend (the #425 hazard) is worse
|
||||
// than a spawn that fails cleanly and can be retried by the caller. Nothing else is lost:
|
||||
// maxLoad, quarantine, cool-off and model-off all behave identically whether or not a
|
||||
// worktree was requested — that agreement is the invariant this rework exists to hold.
|
||||
boolean unqualifiedProfile = profile == null || profile.isBlank();
|
||||
PlacementDecision decision = unqualifiedProfile ? launcher.place(memberRole) : new PlacementDecision(profile);
|
||||
String preResolvedProfile = decision.profile();
|
||||
// CB-507: resolve through the launcher's CB-112 chain (requested → profile cwd → caller →
|
||||
// daemon cwd → "."), never the raw args. A plain REST spawn supplies neither a requested
|
||||
// nor a caller cwd, so taking the first non-blank of those two yielded null and put
|
||||
@@ -472,27 +662,57 @@ public final class SessionManager implements TurnListener {
|
||||
try {
|
||||
path = worktrees.add(repoRoot, branch, wt.baseRef());
|
||||
worktrees.overlayParity(repoRoot, path, launcher.parityOverlay(preResolvedProfile));
|
||||
handle = launcher.spawn(new SpawnRequest(profile, path, callerCwd, sessionName, resumeSessionId, memberRole));
|
||||
// fleetd #185 stage 3: MUST run after overlayParity, not folded into add() — overlayParity
|
||||
// copies more files into the worktree after add() returns, so sharing the group any earlier
|
||||
// leaves those overlay files operator-owned and read-only for a different-uid member.
|
||||
worktrees.shareWithGroup(repoRoot, path);
|
||||
// fleetd #425: preResolvedProfile, not the original (possibly blank) profile — see the
|
||||
// comment above where it is resolved. The overlay/repoRoot above and the spawn here must
|
||||
// name the same profile. An unqualified request stays unqualified here and is honored via
|
||||
// the PlacementDecision already captured above (spawn(req, decision) — the routing branch,
|
||||
// no enforce* re-check); an explicitly-named profile still goes through the single-arg
|
||||
// spawn(req) and its throwing branch, exactly as before this rework.
|
||||
SpawnRequest spawnReq = new SpawnRequest(unqualifiedProfile ? null : preResolvedProfile,
|
||||
path, callerCwd, sessionName, resumeSessionId, memberRole);
|
||||
handle = unqualifiedProfile ? launcher.spawn(spawnReq, decision) : launcher.spawn(spawnReq);
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("spawn failed for profile={} role={} branch={} path={}: {}",
|
||||
preResolvedProfile, memberRole, branch, path, e.getMessage());
|
||||
if (path != null) {
|
||||
// fleetd #283: this catch covers every failure AFTER worktrees.add() returned —
|
||||
// overlayParity, shareWithGroup, launcher.spawn itself — so by this point `branch`
|
||||
// was actually created in git. #274 fixed the sibling failure INSIDE add() by having
|
||||
// GitWorktrees.cleanupAfterAddFailure delete both the worktree and the branch it
|
||||
// provisioned; this path removed only the worktree and left the branch orphaned. A
|
||||
// spawn failure here is routine (a quarantined credential, a backend refusal), so
|
||||
// every occurrence leaked a `worker/<slug>-<nonce>` branch nothing ever pointed at
|
||||
// again. Reuse the same Worktrees.deleteBranch GitWorktrees already has, rather than
|
||||
// a second copy of the git command. Best-effort and log-only, like the worktree
|
||||
// removal right above it — neither cleanup step may mask the original exception.
|
||||
try {
|
||||
worktrees.remove(repoRoot, path);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up worktree {} after spawn error: {}", path, cleanup.getMessage());
|
||||
}
|
||||
try {
|
||||
worktrees.deleteBranch(repoRoot, branch);
|
||||
} catch (RuntimeException cleanup) {
|
||||
log.warn("failed to clean up branch {} after spawn error: {}", branch, cleanup.getMessage());
|
||||
}
|
||||
}
|
||||
throw e;
|
||||
}
|
||||
String resolvedProfile = resolveProfile(handle, profile);
|
||||
String resolvedProfile = resolveProfile(handle, preResolvedProfile);
|
||||
String cwd = launcher.effectiveCwd(new SpawnRequest(resolvedProfile, path, callerCwd));
|
||||
long now = nowNanos.getAsLong();
|
||||
// CB-619: see the no-worktree path above — bind before recording, and store the returned
|
||||
// actual role, so this session's role is never a lie about what it actually holds.
|
||||
MemberRole actualRole = acquired(memberRole, resolvedProfile, handle.terminalId(), reservation);
|
||||
MemberSession session = new MemberSession(
|
||||
handle.id(),
|
||||
handle.terminalId(),
|
||||
resolvedProfile,
|
||||
memberRole,
|
||||
actualRole,
|
||||
cwd,
|
||||
ownerTerminal,
|
||||
now,
|
||||
@@ -504,13 +724,25 @@ public final class SessionManager implements TurnListener {
|
||||
handle.charterReceipt(),
|
||||
handle.agentSessionId());
|
||||
registry.put(handle.id(), session);
|
||||
memberLifecycle.acquired(session.role(), session.profile(), session.terminalId());
|
||||
handles.put(handle.id(), handle);
|
||||
log.debug("acquired worktree session id={} terminal={} profile={} branch={} path={}",
|
||||
handle.id(), handle.terminalId(), session.profile(), session.branch(), session.worktree());
|
||||
notifyAcquired(session.terminalId());
|
||||
return session;
|
||||
}
|
||||
|
||||
private MemberRole acquired(MemberRole role, String profile, String terminal,
|
||||
MemberLifecycle.SlotReservation reservation) {
|
||||
if (reservation == null) {
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
if (memberLifecycle.bind(reservation, terminal)) {
|
||||
return role;
|
||||
}
|
||||
memberLifecycle.release(reservation);
|
||||
return memberLifecycle.acquired(role, profile, terminal);
|
||||
}
|
||||
|
||||
private String slug(String raw) {
|
||||
return raw == null ? "ticket" : raw.toLowerCase().replaceAll("[^a-z0-9]+", "-").replaceAll("^-+|-+$", "");
|
||||
}
|
||||
@@ -535,16 +767,73 @@ public final class SessionManager implements TurnListener {
|
||||
return launcher.defaultProfile();
|
||||
}
|
||||
|
||||
/** The session for {@code paneId}, if it is still registered and not released. */
|
||||
/**
|
||||
* The session for {@code paneId}, if it is still registered and not released. fleetd #209:
|
||||
* resolves a still-unknown {@code agentSessionId} against the retained handle before returning,
|
||||
* so {@code fleet_status} sees an id a lazy-resolving adapter has since written.
|
||||
*/
|
||||
public Optional<MemberSession> get(String paneId) {
|
||||
return Optional.ofNullable(registry.get(paneId));
|
||||
return Optional.ofNullable(registry.get(paneId)).map(this::resolveAgentSessionId);
|
||||
}
|
||||
|
||||
/** Fleet-owned roster: all registered sessions (acquired minus released). */
|
||||
/**
|
||||
* Fleet-owned roster: all registered sessions (acquired minus released). Deliberately does
|
||||
* <strong>not</strong> resolve {@code agentSessionId} (fleetd #209 follow-up) — this is the
|
||||
* roster supplier on the heartbeat and health-tick timers ({@code LeadHeartbeatLoop},
|
||||
* {@code FleetHealthMonitor} in {@code Fleetd}), on the placement/exhaustion paths, and on the
|
||||
* metrics scrape ({@code FleetMetrics}), all called far more often than any caller actually
|
||||
* reads {@code agentSessionId}. Resolving here would mean every tick opens a lazy-resolving
|
||||
* adapter's (opencode's) on-disk session store once per member whose id is still unknown — and
|
||||
* for a member whose id never appears, that cost never stops, for the life of the process. Use
|
||||
* {@link #rosterResolved()} instead wherever the id must be current.
|
||||
*/
|
||||
public List<MemberSession> roster() {
|
||||
return List.copyOf(registry.values());
|
||||
}
|
||||
|
||||
/**
|
||||
* {@link #roster()}, with each session's still-unknown {@code agentSessionId} re-resolved
|
||||
* against its retained handle (fleetd #209) — so a caller that actually reports the id (
|
||||
* {@code fleet_list}, the REST roster) sees one a lazy-resolving adapter (opencode) has since
|
||||
* written, rather than the null frozen in at spawn time. Reserved for caller-driven reads, not
|
||||
* timers: see {@link #roster()}'s javadoc for why the plain roster must stay non-resolving.
|
||||
*/
|
||||
public List<MemberSession> rosterResolved() {
|
||||
return registry.values().stream().map(this::resolveAgentSessionId).toList();
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve {@code session}'s {@code agentSessionId} if still unknown, re-polling the retained
|
||||
* {@link PeerHandle} for this pane (fleetd #209). A no-op — returning {@code session} unchanged
|
||||
* — once the id is already known, once no handle is retained for this pane (never spawned, or
|
||||
* already released), or if the handle throws while answering. A resolved id is best-effort
|
||||
* CAS-swapped into the registry via {@link #replace}; a lost race just means another caller
|
||||
* already applied the same update, so the resolved value is returned either way.
|
||||
*/
|
||||
private MemberSession resolveAgentSessionId(MemberSession session) {
|
||||
return resolveAgentSessionId(session, handles.get(session.paneId()));
|
||||
}
|
||||
|
||||
private MemberSession resolveAgentSessionId(MemberSession session, PeerHandle handle) {
|
||||
if (session.agentSessionId() != null || handle == null) {
|
||||
return session;
|
||||
}
|
||||
String resolved;
|
||||
try {
|
||||
resolved = handle.agentSessionId();
|
||||
} catch (RuntimeException e) {
|
||||
log.debug("agentSessionId lookup failed for pane={} terminal={}: {}",
|
||||
session.paneId(), session.terminalId(), e.toString());
|
||||
return session;
|
||||
}
|
||||
if (resolved == null) {
|
||||
return session;
|
||||
}
|
||||
MemberSession updated = session.withAgentSessionId(resolved);
|
||||
replace(session, updated); // best-effort; a lost CAS just means the resolved value stands anyway
|
||||
return updated;
|
||||
}
|
||||
|
||||
/**
|
||||
* CB-304 merged roster+live view. The registry is authoritative for worktree, branch,
|
||||
* profile, owner, and state; the optional live agent supplies the herdr-reported status.
|
||||
@@ -559,6 +848,9 @@ public final class SessionManager implements TurnListener {
|
||||
m.put("profile", session.profile());
|
||||
m.put("role", session.role() == null ? "dev" : session.role().wireName());
|
||||
m.put("state", session.state().name().toLowerCase());
|
||||
if (session.failureReason() != null) {
|
||||
m.put("failureReason", session.failureReason());
|
||||
}
|
||||
if (session.worktree() != null) {
|
||||
m.put("worktree", session.worktree());
|
||||
}
|
||||
@@ -619,6 +911,36 @@ public final class SessionManager implements TurnListener {
|
||||
completeTurn(target, false);
|
||||
}
|
||||
|
||||
/**
|
||||
* Record that a member backend failed after a delegated ticket expired. This CAS loop accepts
|
||||
* either side of the completion race: {@code BUSY -> BACKEND_ERROR} or
|
||||
* {@code DONE -> BACKEND_ERROR}. A released or unknown session is never recreated.
|
||||
*
|
||||
* @return {@code true} when the error is recorded on a live session
|
||||
*/
|
||||
public boolean onBackendError(String target, String reason) {
|
||||
while (true) {
|
||||
MemberSession current = findByTerminal(target);
|
||||
if (current == null) {
|
||||
log.warn("backend error for unknown member terminal={}", target);
|
||||
return false;
|
||||
}
|
||||
if (current.state() == MemberSession.State.RELEASED) {
|
||||
return false;
|
||||
}
|
||||
if (current.state() == MemberSession.State.BACKEND_ERROR) {
|
||||
return true;
|
||||
}
|
||||
MemberSession updated = current.withState(MemberSession.State.BACKEND_ERROR)
|
||||
.withFailureReason(reason).withActivity(nowNanos.getAsLong());
|
||||
if (replace(current, updated)) {
|
||||
log.warn("member terminal={} pane={} transitioned {} -> BACKEND_ERROR: {}",
|
||||
target, current.paneId(), current.state(), reason);
|
||||
return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean hasPostTurnAction(String target) {
|
||||
if (!clearAfterTurn) return false;
|
||||
@@ -637,10 +959,9 @@ public final class SessionManager implements TurnListener {
|
||||
if (current == null || current.state() != MemberSession.State.BUSY) return false;
|
||||
long now = nowNanos.getAsLong();
|
||||
MemberSession updated = current.withState(MemberSession.State.DONE).withActivity(now);
|
||||
if (replace(current, updated)) {
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
}
|
||||
if (!replace(current, updated)) return false;
|
||||
log.debug("session transitioned terminal={} pane={} BUSY -> DONE turn={}",
|
||||
target, current.paneId(), updated.turnCount());
|
||||
if (contextCap > 0 && updated.turnCount() >= contextCap) {
|
||||
release(current.paneId());
|
||||
return false;
|
||||
@@ -687,14 +1008,18 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
long idleNanos = now - s.lastActivityAtNanos();
|
||||
if (idleNanos > idleTtlNanos) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
// CB-581: one session that fails to release must not abort the whole reaping pass —
|
||||
// match drainAll's per-session try/catch so the rest of the roster still gets reaped.
|
||||
try {
|
||||
release(s.paneId());
|
||||
reaped++;
|
||||
if (beforeIdleRelease != null) {
|
||||
beforeIdleRelease.accept(s);
|
||||
}
|
||||
if (releaseIfCurrent(s, ReleaseCause.COMPLETED)) {
|
||||
log.debug("reaping idle session terminal={} pane={}: idle {}s exceeds the {}s ttl",
|
||||
s.terminalId(), s.paneId(), TimeUnit.NANOSECONDS.toSeconds(idleNanos),
|
||||
TimeUnit.NANOSECONDS.toSeconds(idleTtlNanos));
|
||||
reaped++;
|
||||
}
|
||||
} catch (RuntimeException e) {
|
||||
log.warn("reap failed for pane={} terminal={} worktree={}; continuing with "
|
||||
+ "remaining sessions", s.paneId(), s.terminalId(), s.worktree(), e);
|
||||
@@ -705,20 +1030,60 @@ public final class SessionManager implements TurnListener {
|
||||
}
|
||||
|
||||
/**
|
||||
* Gracefully drain all registered sessions on daemon shutdown. For each session that is
|
||||
* {@code BUSY}, poll up to {@code timeoutNanos} for it to leave {@code BUSY}, then release it
|
||||
* regardless. Non-busy sessions are released immediately. A failure releasing one session is
|
||||
* logged and does not abort the rest.
|
||||
* Gracefully drain all registered sessions on daemon shutdown. Non-busy sessions are released
|
||||
* immediately; a {@code BUSY} one is polled until it leaves {@code BUSY}, then released
|
||||
* regardless. A failure releasing one session is logged and does not abort the rest.
|
||||
*
|
||||
* <p>{@code timeoutNanos} is a budget for the WHOLE drain, not a grace period per session: the
|
||||
* deadline is taken once, before the loop. So the first BUSY session can spend all of it, and a
|
||||
* later BUSY one is then released with no wait at all. That is deliberate. This drain is only
|
||||
* one phase of shutdown — {@code Fleetd} closes the message service, the push loop, the
|
||||
* heartbeat, MCP and the router after it — and the whole sequence has to finish inside
|
||||
* launchd's exit window. A per-session grace would let N busy members drain for N * the
|
||||
* timeout, overrun that window, and get the daemon SIGKILLed part-way through; the members not
|
||||
* yet reached would then get no clean release, no preserved-worktree log, and no snapshot.
|
||||
* Cutting one turn short is the cheaper failure, and it is not silent: an abandoned BUSY
|
||||
* session is preserved, snapshotted, and logged at WARN by {@code logPreservedForShutdown}.
|
||||
*
|
||||
* <p>CB-544: this is a {@link ReleaseCause#SHUTDOWN} release — the worker's pane is stopped
|
||||
* (the process must end) but its worktree is preserved and its path logged. Shutdown is never
|
||||
* a reason to delete a worker's only copy of its uncommitted work. A session still {@code BUSY}
|
||||
* when the timeout expired is abandoned mid-turn and logged loudly so an operator can find its
|
||||
* kept worktree.
|
||||
*
|
||||
* <p>fleetd #308: {@code roster()} is a one-shot snapshot (see its javadoc), and nothing used
|
||||
* to stop a new session from registering after it was taken — {@link #acquire} stayed open for
|
||||
* as long as this drain waited on a {@code BUSY} session, up to the whole {@code timeoutNanos}
|
||||
* budget. Two things close that window, deliberately paired because neither alone is complete:
|
||||
* {@link #draining} is flipped true before the snapshot is even taken, so {@link #acquire}
|
||||
* refuses (invariant 3: loudly, via {@link ShuttingDownException}) as much of the window as a
|
||||
* plain flag can close; and the sweep below re-reads the registry once the initial snapshot has
|
||||
* fully drained and drains whatever a straggler — a caller that read the flag as {@code false}
|
||||
* a moment before it flipped — still managed to register. The sweep shares the same
|
||||
* {@code deadline} rather than getting its own: {@code timeoutNanos} is a budget for the WHOLE
|
||||
* drain (see above), and a straggler must not buy the drain more time than the flag it lost the
|
||||
* race against would have. In the ordinary case the sweep finds nothing and costs one empty
|
||||
* {@link #roster()} call.
|
||||
*/
|
||||
void drainAll(long timeoutNanos) {
|
||||
long deadline = System.nanoTime() + timeoutNanos;
|
||||
for (MemberSession s : roster()) {
|
||||
draining.set(true);
|
||||
drainSnapshot(roster(), deadline);
|
||||
List<MemberSession> stragglers = roster();
|
||||
if (!stragglers.isEmpty()) {
|
||||
log.warn("drain sweep found {} session(s) registered after the drain snapshot was "
|
||||
+ "taken (raced past the shutdown guard); draining them too", stragglers.size());
|
||||
drainSnapshot(stragglers, deadline);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Drain exactly the sessions in {@code snapshot}, waiting out a {@code BUSY} one against the
|
||||
* shared whole-drain {@code deadline} before releasing it. Shared by {@link #drainAll}'s main
|
||||
* pass and its post-loop straggler sweep (fleetd #308) so both honor the same one budget.
|
||||
*/
|
||||
private void drainSnapshot(List<MemberSession> snapshot, long deadline) {
|
||||
for (MemberSession s : snapshot) {
|
||||
try {
|
||||
if (s.state() == MemberSession.State.BUSY) {
|
||||
while (System.nanoTime() < deadline) {
|
||||
|
||||
@@ -0,0 +1,18 @@
|
||||
package dev.ltms.fleet.session;
|
||||
|
||||
/**
|
||||
* Thrown by {@link SessionManager#acquire} when a spawn is requested after the daemon's shutdown
|
||||
* drain has already begun (fleetd #308).
|
||||
*
|
||||
* <p>{@link SessionManager#drainAll} snapshots the registry once and tears down exactly what is
|
||||
* in that snapshot. A session registered after the snapshot is invisible to the drain loop: its
|
||||
* pane is left running and its worktree is never preserved, and nothing else ever reclaims
|
||||
* either — the daemon's in-memory registry dies with the process. Refusing the spawn here,
|
||||
* loudly, is what stops that session from ever being created in the first place, rather than
|
||||
* silently handing the caller a session the daemon can no longer manage.
|
||||
*/
|
||||
public final class ShuttingDownException extends RuntimeException {
|
||||
public ShuttingDownException(String message) {
|
||||
super(message);
|
||||
}
|
||||
}
|
||||
@@ -11,6 +11,15 @@ public interface Worktrees {
|
||||
/** git -C <repoRoot> worktree remove --force <path>. Idempotent (already-gone tolerated). */
|
||||
void remove(String repoRoot, String worktreePath);
|
||||
|
||||
/**
|
||||
* git -C {@code repoRoot} branch -D {@code branch}. Force-deletes a branch that has no other
|
||||
* owner — used only on the failed-provisioning path (fleetd #274, #283), never on a normal
|
||||
* release: {@link SessionManager#release} deliberately leaves a released session's branch
|
||||
* behind so a lead can still recover the work, and this method must never be called from
|
||||
* that path.
|
||||
*/
|
||||
void deleteBranch(String repoRoot, String branch);
|
||||
|
||||
/**
|
||||
* True when the worktree holds uncommitted changes the bridge cannot see: tracked
|
||||
* modifications, staged files, or untracked files. {@code git status --porcelain} is the
|
||||
@@ -98,4 +107,21 @@ public interface Worktrees {
|
||||
/** CB-586: the operator-visible census of {@code refs/wip/*} in one repository. */
|
||||
record WipRefStats(int count, long costBytes) {
|
||||
}
|
||||
|
||||
/**
|
||||
* Make {@code repoRoot}'s git store and {@code worktreePath} writable by the configured group
|
||||
* (fleetd #185 stage 3), so a member spawned as a different OS user (see
|
||||
* {@code memberHerdrSocket}) can write its own worktree, its per-worktree git metadata, and
|
||||
* its own commit objects. No-op when no group is configured.
|
||||
*
|
||||
* <p><strong>This isolates credentials, not the repository.</strong> A member in the group can
|
||||
* still write the operator's git objects and refs in the shared repo — this only fixes file
|
||||
* ownership/permissions so a different-uid member can work at all, it grants no narrower access
|
||||
* than that.
|
||||
*
|
||||
* @param repoRoot the repository whose git store ({@code .git/objects}, {@code refs},
|
||||
* {@code logs}, {@code worktrees}, {@code packed-refs}) needs sharing
|
||||
* @param worktreePath the linked worktree's own directory
|
||||
*/
|
||||
void shareWithGroup(String repoRoot, String worktreePath);
|
||||
}
|
||||
|
||||
@@ -0,0 +1,150 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.List;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #395: {@code exhaustedPattern} (see {@link FleetConfig.Profile#exhaustedPattern}) is
|
||||
* deliberately opt-in — an unset one leaves usage-limit detection silently OFF for that profile,
|
||||
* and nothing quarantines its credential. {@link Fleetd#reportExhaustedPatternGap} must say so at
|
||||
* startup, naming every unarmed profile, and must never fire when every profile is armed. Mirrors
|
||||
* {@link MemberCredentialsGapReportTest}'s pattern, capturing the real log via a
|
||||
* {@link ListAppender}.
|
||||
*
|
||||
* <p>The 8-profile shape in {@link #theLiveEightProfileShapeWarnsExactlyTheSixUnarmedProfiles} is
|
||||
* the live {@code fleetd.yaml} shape measured 2026-09-10 (fleetd #395's own ticket): 6 of 8
|
||||
* profiles unarmed, 2 of those 6 ({@code opus}, {@code sonnet}) running on the operator's Claude
|
||||
* subscription. {@code fleetd.yaml} itself is gitignored and unavailable to this test, so the
|
||||
* shape is reproduced as a throwaway config in a {@code @TempDir} rather than read off disk.
|
||||
*/
|
||||
class ExhaustedPatternGapReportTest {
|
||||
|
||||
private static FleetConfig load(Path dir, String yaml) throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, yaml);
|
||||
return FleetConfig.load(f);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
/** Every profile name mentioned by a WARN-level log line, across every WARN this call produced. */
|
||||
private static List<String> warnMessages(ListAppender<ILoggingEvent> appender) {
|
||||
return appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.WARN)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.toList();
|
||||
}
|
||||
|
||||
@Test
|
||||
void theLiveEightProfileShapeWarnsExactlyTheSixUnarmedProfiles(@TempDir Path dir) throws Exception {
|
||||
// Reproduces the live shape measured 2026-09-10: 8 profiles, 2 armed (sol, terra), 6
|
||||
// unarmed (local, local-direct, gx, opus, sonnet, xf) — 2 of the unarmed 6 (opus, sonnet)
|
||||
// are subscription: true.
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
local:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
local-direct:
|
||||
baseUrl: http://gx01.gw:8000
|
||||
gx:
|
||||
kind: opencode
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
opus:
|
||||
subscription: true
|
||||
model: claude-opus-5
|
||||
sonnet:
|
||||
subscription: true
|
||||
model: claude-sonnet-5
|
||||
sol:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
terra:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
xf:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportExhaustedPatternGap(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
List<String> warns = warnMessages(appender);
|
||||
assertFalse(warns.isEmpty(), "6 of 8 profiles are unarmed — at least one WARN must fire");
|
||||
|
||||
// Exactly one WARN aggregates every unarmed profile, naming all 6 and none of the 2 armed.
|
||||
String aggregate = warns.stream()
|
||||
.filter(m -> m.contains("no exhaustedPattern configured"))
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new AssertionError("expected an aggregate unarmed-profiles WARN: " + warns));
|
||||
for (String unarmed : List.of("local", "local-direct", "gx", "opus", "sonnet", "xf")) {
|
||||
assertTrue(aggregate.contains(unarmed), "aggregate WARN must name '" + unarmed + "': " + aggregate);
|
||||
}
|
||||
for (String armed : List.of("sol", "terra")) {
|
||||
assertFalse(aggregate.contains(armed), "aggregate WARN must NOT name armed profile '" + armed + "': " + aggregate);
|
||||
}
|
||||
|
||||
// A second, louder WARN calls out the subscription profiles specifically.
|
||||
String subscriptionWarn = warns.stream()
|
||||
.filter(m -> m.contains("metered Claude plan"))
|
||||
.findFirst()
|
||||
.orElseThrow(() -> new AssertionError("expected a subscription-specific WARN: " + warns));
|
||||
assertTrue(subscriptionWarn.contains("opus"), subscriptionWarn);
|
||||
assertTrue(subscriptionWarn.contains("sonnet"), subscriptionWarn);
|
||||
assertFalse(subscriptionWarn.contains("local-direct"),
|
||||
"the subscription WARN must not name a non-subscription profile: " + subscriptionWarn);
|
||||
}
|
||||
|
||||
@Test
|
||||
void everyProfileArmedProducesNoWarningAtAll(@TempDir Path dir) throws Exception {
|
||||
FleetConfig cfg = load(dir, """
|
||||
profiles:
|
||||
sol:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
terra:
|
||||
baseUrl: https://llm.ltms.dev/v1
|
||||
exhaustedPattern: "The usage limit has been reached"
|
||||
opus:
|
||||
subscription: true
|
||||
model: claude-opus-5
|
||||
exhaustedPattern: "5-hour limit reached"
|
||||
""");
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
Fleetd.reportExhaustedPatternGap(cfg);
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
assertTrue(warnMessages(appender).isEmpty(),
|
||||
"every profile is armed — a checker that warns anyway always fires: " + warnMessages(appender));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,59 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.inject.BackendErrorPatternLookup;
|
||||
import dev.ltms.fleet.peer.MemberRole;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.regex.Pattern;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertNull;
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: {@link Fleetd#backendErrorPatternLookup} is the factory that
|
||||
* replaced the local lambda {@code Fleetd.main} used to build {@code backendErrorPatterns} — one
|
||||
* of the two arguments {@code CompletionResolver} lost cleanly (0 compile errors, every test still
|
||||
* green) when this ticket's measurement dropped it alongside {@code backendErrorSink}. This class
|
||||
* proves the factory's own behaviour; {@code FleetdCompletionResolverWiringTest} proves {@code
|
||||
* main} still passes its result into {@code CompletionResolver}.
|
||||
*/
|
||||
class FleetdBackendErrorPatternLookupTest {
|
||||
|
||||
private static MemberSession session(String terminal, String profile) {
|
||||
return new MemberSession("pane-" + terminal, terminal, profile, MemberRole.DEV,
|
||||
"/cwd", null, 0L, 0L, 0, MemberSession.State.READY, null, null);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target on a profile with a configured pattern resolves to that pattern")
|
||||
void configuredProfileResolves() {
|
||||
Map<String, Pattern> byProfile = Map.of("terra", Pattern.compile("(?i)503"));
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(() -> List.of(session("term1", "terra")), byProfile);
|
||||
|
||||
assertEquals("(?i)503", lookup.patternFor("term1").pattern());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target on a profile with no configured pattern resolves to null")
|
||||
void unconfiguredProfileResolvesToNull() {
|
||||
Map<String, Pattern> byProfile = Map.of("terra", Pattern.compile("x"));
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(() -> List.of(session("term1", "sol")), byProfile);
|
||||
|
||||
assertNull(lookup.patternFor("term1"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("an unknown target resolves to null")
|
||||
void unknownTargetResolvesToNull() {
|
||||
BackendErrorPatternLookup lookup =
|
||||
Fleetd.backendErrorPatternLookup(List::of, Map.of("terra", Pattern.compile("x")));
|
||||
|
||||
assertNull(lookup.patternFor("term_stranger"));
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,275 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import com.fasterxml.jackson.databind.JsonNode;
|
||||
import com.fasterxml.jackson.databind.ObjectMapper;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.HerdrClient;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.BackendErrorSink;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.member.CompositePeerLauncher;
|
||||
import dev.ltms.fleet.mcp.PrimaryRegistry;
|
||||
import dev.ltms.fleet.msg.InMemoryReplyInbox;
|
||||
import dev.ltms.fleet.msg.ReplyPushLoop;
|
||||
import dev.ltms.fleet.peer.Capability;
|
||||
import dev.ltms.fleet.peer.PeerHandle;
|
||||
import dev.ltms.fleet.peer.PeerLauncher;
|
||||
import dev.ltms.fleet.peer.SpawnRequest;
|
||||
import dev.ltms.fleet.placement.BackendOutagePolicy;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.placement.PlacementDecision;
|
||||
import dev.ltms.fleet.placement.PlacementPolicies;
|
||||
import dev.ltms.fleet.session.MemberSession;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.AfterEach;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import java.util.ArrayList;
|
||||
import java.util.LinkedHashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.CopyOnWriteArrayList;
|
||||
import java.util.concurrent.CountDownLatch;
|
||||
import java.util.concurrent.Executors;
|
||||
import java.util.concurrent.ScheduledExecutorService;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
import java.util.concurrent.atomic.AtomicLong;
|
||||
import java.util.concurrent.atomic.AtomicReference;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #248 / fleetd#201 Unit 5: {@link Fleetd#backendErrorSink} is the factory that replaced
|
||||
* the local lambda {@code Fleetd.main} used to build {@code backendErrorSink} — the other half of
|
||||
* the pair this ticket's measurement dropped cleanly (0 compile errors, every test still green).
|
||||
*
|
||||
* <p>Before this ticket, the closest thing to coverage was {@code
|
||||
* dev.ltms.fleet.inject.BackendOutageFlowTest}, whose own class doc said it "mirrors {@code
|
||||
* Fleetd.main}'s {@code backendErrorSink} lambda line-for-line" — a hand-copy that proves itself,
|
||||
* never that {@code main} still wires the real thing. This class exercises the actual production
|
||||
* factory instead. {@code FleetdCompletionResolverWiringTest} proves {@code main} still passes its
|
||||
* result into {@code CompletionResolver}.
|
||||
*/
|
||||
class FleetdBackendErrorSinkTest {
|
||||
|
||||
private final List<ScheduledExecutorService> schedulers = new ArrayList<>();
|
||||
|
||||
@AfterEach
|
||||
void tearDown() {
|
||||
schedulers.forEach(ScheduledExecutorService::shutdownNow);
|
||||
}
|
||||
|
||||
private static FleetConfig.Profile stubWorker(String profile, String credentialId) {
|
||||
return new FleetConfig.Profile(profile, "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, null, null,
|
||||
null, null, credentialId, null);
|
||||
}
|
||||
|
||||
private static Map<String, FleetConfig.Profile> orderedProfiles() {
|
||||
Map<String, FleetConfig.Profile> m = new LinkedHashMap<>();
|
||||
m.put("terra", stubWorker("terra", "shared-openai"));
|
||||
m.put("sol", stubWorker("sol", "shared-openai"));
|
||||
return m;
|
||||
}
|
||||
|
||||
/** Minimal recording {@code HerdrClient} for the LEAD pane — mirrors ReplyPushLoopTest's own. */
|
||||
private static final class RecordingLeadClient implements HerdrClient {
|
||||
private static final ObjectMapper MAPPER = new ObjectMapper();
|
||||
private final List<Object> prompts = new CopyOnWriteArrayList<>();
|
||||
volatile CountDownLatch sendLatch = new CountDownLatch(1);
|
||||
|
||||
@Override
|
||||
public JsonNode call(String method, Object params) {
|
||||
if ("agent.get".equals(method)) {
|
||||
return MAPPER.createObjectNode().set("agent", MAPPER.createObjectNode()
|
||||
.put("terminal_id", "term_primary").put("agent_status", "idle"));
|
||||
}
|
||||
if ("agent.prompt".equals(method)) {
|
||||
prompts.add(params);
|
||||
sendLatch.countDown();
|
||||
}
|
||||
return MAPPER.createObjectNode();
|
||||
}
|
||||
|
||||
@Override
|
||||
public void close() {
|
||||
}
|
||||
|
||||
int sendCount() {
|
||||
return prompts.size();
|
||||
}
|
||||
}
|
||||
|
||||
/** A {@link PeerLauncher} that never actually spawns — enough to construct a bare {@link SessionManager}. */
|
||||
private static final class NeverSpawnsLauncher implements PeerLauncher {
|
||||
@Override
|
||||
public Set<Capability> capabilities() {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<Capability> capabilitiesFor(String profileName) {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public PeerHandle spawn(SpawnRequest req, PlacementDecision decision) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public Set<String> profiles() {
|
||||
return Set.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public String defaultProfile() {
|
||||
return null;
|
||||
}
|
||||
|
||||
@Override
|
||||
public String effectiveCwd(SpawnRequest req) {
|
||||
throw new UnsupportedOperationException("not reachable — this test never acquires a session");
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<String> parityOverlay(String profileName) {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public List<?> list() {
|
||||
return List.of();
|
||||
}
|
||||
|
||||
@Override
|
||||
public int reapOrphanWorkers() {
|
||||
return 0;
|
||||
}
|
||||
|
||||
@Override
|
||||
public void stop(String id) {
|
||||
}
|
||||
|
||||
@Override
|
||||
public boolean clearContext(String id) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a target with no resolvable profile logs and returns without recording an incident (never throws)")
|
||||
void unresolvableProfileDoesNotRecordOrThrow() {
|
||||
SessionManager sessions = new SessionManager(new NeverSpawnsLauncher());
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(() -> 0L);
|
||||
RecordingLeadClient leadClient = new RecordingLeadClient();
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
PrimaryRegistry registry = new PrimaryRegistry(null);
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
ReplyPushLoop pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
|
||||
BackendErrorSink sink = Fleetd.backendErrorSink(sessions, Map::of, outagePolicy, () -> pushLoop);
|
||||
sink.onBackendError("term_unmapped", "matched line", "503 Service Unavailable");
|
||||
|
||||
assertTrue(outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty(),
|
||||
"no credential is ever resolvable here, so nothing must be recorded");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("two distinct targets classified through the real factory start an incident and cool the credential")
|
||||
void twoDistinctTargetsStartAnIncident() throws Exception {
|
||||
FakeHerdr herdr = new FakeHerdr()
|
||||
.readText("⏺ 503 Service Unavailable: upstream credential rejected\n❯ ");
|
||||
Map<String, FleetConfig.Profile> profiles = orderedProfiles();
|
||||
AtomicLong clockNanos = new AtomicLong(0L);
|
||||
BackendOutagePolicy outagePolicy = new BackendOutagePolicy(clockNanos::get);
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr),
|
||||
new WorkspaceControl(herdr), new SubscriptionGuard(Set.of("gx00.gw")),
|
||||
profiles, "terra", _ -> "tok");
|
||||
CompositePeerLauncher workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.weighted(), _ -> 0, null, BackendQuarantine.none(), outagePolicy);
|
||||
SessionManager sessions = new SessionManager(workers);
|
||||
MemberSession s1 = sessions.acquire("terra", null, null, null);
|
||||
MemberSession s2 = sessions.acquire("terra", null, null, null);
|
||||
|
||||
PrimaryRegistry registry = new PrimaryRegistry(null);
|
||||
registry.recordDelegation(s1.terminalId(), "term_primary");
|
||||
registry.recordDelegation(s2.terminalId(), "term_primary");
|
||||
InMemoryReplyInbox inbox = new InMemoryReplyInbox();
|
||||
inbox.own(s1.terminalId());
|
||||
inbox.own(s2.terminalId());
|
||||
RecordingLeadClient leadClient = new RecordingLeadClient();
|
||||
ScheduledExecutorService scheduler = Executors.newSingleThreadScheduledExecutor();
|
||||
schedulers.add(scheduler);
|
||||
ReplyPushLoop pushLoop = new ReplyPushLoop(registry, new AgentControl(leadClient), inbox, scheduler, 3, 50);
|
||||
AtomicReference<ReplyPushLoop> pushLoopRef = new AtomicReference<>(pushLoop);
|
||||
|
||||
// The exact object under test: Fleetd's real production factory, not a hand copy.
|
||||
BackendErrorSink sink = Fleetd.backendErrorSink(sessions, () -> profiles, outagePolicy, pushLoopRef::get);
|
||||
|
||||
sink.onBackendError(s1.terminalId(), "matched line", "503 Service Unavailable");
|
||||
assertTrue(outagePolicy.remainingCoolOffSeconds("shared-openai").isEmpty(),
|
||||
"one distinct target must not start a cool-off");
|
||||
|
||||
sink.onBackendError(s2.terminalId(), "matched line", "503 Service Unavailable");
|
||||
|
||||
assertTrue(leadClient.sendLatch.await(3, TimeUnit.SECONDS),
|
||||
"the second distinct target must cross the threshold and nudge the lead");
|
||||
var remaining = outagePolicy.remainingCoolOffSeconds("shared-openai");
|
||||
assertTrue(remaining.isPresent(), "two distinct targets must start a cool-off");
|
||||
assertEquals(1, leadClient.sendCount());
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a backend-error session no longer blocks the real maxLoad spawn gate")
|
||||
void backendErrorSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
assertTrue(sessions.onBackendError(failed.terminalId(), "backend exited"));
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a backend error");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a failed session no longer blocks the real maxLoad spawn gate")
|
||||
void failedSessionDoesNotBlockFreshSpawnAtMaxLoad() {
|
||||
SessionManager sessions = capacityLimitedSessions();
|
||||
MemberSession failed = sessions.acquire("terra", null, null, null);
|
||||
|
||||
sessions.onTurnFailed(failed.terminalId());
|
||||
|
||||
MemberSession fresh = sessions.acquire("terra", null, null, null);
|
||||
assertEquals("terra", fresh.profile(), "the real maxLoad gate grants a fresh spawn after a failed turn");
|
||||
}
|
||||
|
||||
private static SessionManager capacityLimitedSessions() {
|
||||
FakeHerdr herdr = new FakeHerdr();
|
||||
FleetConfig.Profile profile = new FleetConfig.Profile("terra", "http://gx00.gw:8000", "coder",
|
||||
null, "FLEETD_WORKER_TOKEN", List.of("claude"), "tab", "fleetd-workers",
|
||||
"w #{n}", null, null, null, null, null, null, null, 1.0f, 1);
|
||||
Map<String, FleetConfig.Profile> profiles = Map.of("terra", profile);
|
||||
ClaudeCodeLauncher adapter = new ClaudeCodeLauncher(new AgentControl(herdr), new WorkspaceControl(herdr),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), profiles, "terra", _ -> "tok");
|
||||
AtomicReference<SessionManager> sessionsRef = new AtomicReference<>();
|
||||
CompositePeerLauncher workers = new CompositePeerLauncher(List.of(adapter), "terra", profiles,
|
||||
PlacementPolicies.fixed(), name -> Fleetd.liveSessionCount(sessionsRef.get().roster(), name));
|
||||
SessionManager sessions = new SessionManager(workers);
|
||||
sessionsRef.set(sessions);
|
||||
return sessions;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,65 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #466 follow-up: {@code Fleetd.main} builds the daemon's one {@code BackendQuarantine}
|
||||
* from {@link dev.ltms.fleet.placement.BackendQuarantine#withEscalation(java.util.function.LongSupplier,
|
||||
* long)} — the escalating factory — rather than the plain two-argument constructor, which is still a
|
||||
* flat cooldown (kept for backward compatibility, see that class's doc). {@code
|
||||
* BackendQuarantineTest} proves {@code withEscalation} itself escalates, is ceilinged, and resets;
|
||||
* it says nothing about which one {@code main} actually calls.
|
||||
*
|
||||
* <p>Measured directly: reverting {@code main} to {@code new BackendQuarantine(System::nanoTime,
|
||||
* TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()))} — the pre-#466 flat call — compiles
|
||||
* with 0 errors and leaves the entire 1608-test suite (including every {@code BackendQuarantineTest}
|
||||
* case) green, because no other test constructs its {@code BackendQuarantine} through {@code main};
|
||||
* every one of them builds its own instance directly. That silent regression is exactly the shape
|
||||
* {@link FleetdLeadSeatWiringTest} and {@link FleetdCompletionResolverWiringTest} already guard
|
||||
* against for their own constructor arguments — this is the same class of gap for fleetd #466's
|
||||
* factory choice, following their approach.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* BackendQuarantine} and never runs {@code main} — a green result here proves only that the exact
|
||||
* text {@code main} calls {@code BackendQuarantine.withEscalation(...)} rather than the flat
|
||||
* constructor. It does not prove that call actually executes at startup (no test here starts the
|
||||
* daemon), and it does not prove the escalation reaches a real backend or credential — only
|
||||
* {@code BackendQuarantineTest} proves the factory's own behaviour, and only a live daemon proves
|
||||
* the wiring runs.
|
||||
*/
|
||||
class FleetdBackendQuarantineWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main's BackendQuarantine local is still built from BackendQuarantine.withEscalation(...)")
|
||||
void mainStillWiresTheEscalatingQuarantineFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendQuarantine quarantine = BackendQuarantine.withEscalation(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"Fleetd.main's BackendQuarantine local must still be built from "
|
||||
+ "BackendQuarantine.withEscalation(System::nanoTime, "
|
||||
+ "TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds())). Reverting to the flat "
|
||||
+ "two-argument constructor (fleetd #466's measured regression) compiles with 0 errors "
|
||||
+ "and leaves the whole suite green, including every BackendQuarantineTest case that "
|
||||
+ "proves the escalation itself works — this source check is what must go red instead. "
|
||||
+ "A reverted daemon would go back to retrying a weekly subscription limit on every "
|
||||
+ "flat ~30-minute cooldown, about 336 times across the week.");
|
||||
|
||||
// Negative form of the same check: the pre-#466 flat call, if it ever reappears at this
|
||||
// declaration, must not be mistaken for the escalating one by a looser positive-only check.
|
||||
assertFalse(source.contains(
|
||||
"BackendQuarantine quarantine = new BackendQuarantine(System::nanoTime,\n"
|
||||
+ " TimeUnit.SECONDS.toNanos(cfg.quarantineCooldownSeconds()));"),
|
||||
"main's BackendQuarantine local must never regress to the flat two-argument constructor");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,155 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
|
||||
/**
|
||||
* fleetd #416: {@code fleet_list}'s {@code CapacitySource.configuredProfiles} must enumerate the
|
||||
* <em>startup</em> profile set, not the live, hot-reloaded one.
|
||||
*
|
||||
* <p>{@code profiles} as a whole is a {@code DEFERRED} key ({@link ConfigRef#DEFERRED_KEYS}):
|
||||
* {@code HerdrPeerLauncher} takes {@code Map.copyOf(profiles)} once at construction, so a profile
|
||||
* only added to the hot-reloaded map can never actually be spawned. Before this fix, {@code
|
||||
* Fleetd.main} built {@code CapacitySource} with {@code () -> config.get().profiles().keySet()} —
|
||||
* the live map — so {@code fleet_list} would report a freshly hot-reloaded profile as available
|
||||
* ({@code free > 0}) while {@code fleet_spawn} on that same profile failed with
|
||||
* {@code unknown worker profile}. Measured on another host: adding a throwaway profile and letting
|
||||
* it hot-reload gave {@code fleet_list} -> {@code free: 3} and {@code fleet_spawn} ->
|
||||
* {@code error: unknown worker profile}.
|
||||
*
|
||||
* <p>This test needs a reload, the same reason {@link FleetdExhaustionDetectionArmedWiringTest}
|
||||
* does: at startup the two snapshots agree, so a test of only a newly started daemon would not
|
||||
* detect a live {@code config.get()} lookup for the set.
|
||||
*
|
||||
* <p><b>maxLoad must stay hot.</b> It is read off {@code config.get()} exactly like
|
||||
* {@code credentialId} ({@link ConfigRef} documents both as hot, "read live off the config
|
||||
* supplier ... exactly like weight/maxLoad"), so a reload that only changes an existing profile's
|
||||
* {@code maxLoad} — no add/remove — must still change what {@code fleet_list} reports without a
|
||||
* restart. A fix that freezes the whole {@code CapacitySource} against {@code cfg} (rather than
|
||||
* only its {@code configuredProfiles} set) would trade this bug for its mirror image and is pinned
|
||||
* wrong by {@link #reloadedMaxLoadStillChangesWhatFleetListReports}.
|
||||
*/
|
||||
class FleetdCapacitySourceWiringTest {
|
||||
|
||||
private static final String STARTUP = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 3
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_NEW_PROFILE = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 3
|
||||
ghost404:
|
||||
baseUrl: http://gx00.gw:8001
|
||||
model: ghost404
|
||||
maxLoad: 3
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_CHANGED_MAX_LOAD = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
maxLoad: 9
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile present only in the live (hot-reloaded) config is NOT listed")
|
||||
void liveOnlyProfileIsNotListed(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
Files.writeString(file, WITH_NEW_PROFILE);
|
||||
assertTrue(config.reload().applied());
|
||||
// The live snapshot now has the new profile — proves the reload really happened and this
|
||||
// test is not accidentally passing because nothing changed.
|
||||
assertTrue(config.get().profiles().containsKey("ghost404"));
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
|
||||
assertFalse(source.configuredProfiles().get().contains("ghost404"),
|
||||
"a profile added only to the hot-reloaded config must not be listed by fleet_list — "
|
||||
+ "HerdrPeerLauncher never learns about it until a restart, so fleet_spawn on it "
|
||||
+ "would fail with 'unknown worker profile' while fleet_list claimed it free");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile present in the startup set IS listed")
|
||||
void startupProfileIsListed(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
|
||||
// fleetd #416, both-directions requirement: a test that only ever passes an empty/absent
|
||||
// startup set (the case above) cannot tell a correct lookup from one that is permanently
|
||||
// empty (e.g. a mutation replacing the supplier with Set::of). This is the direction that
|
||||
// fails if the fix regresses to reporting nothing at all.
|
||||
assertTrue(source.configuredProfiles().get().contains("terra"),
|
||||
"a profile present in the startup snapshot must still be listed by fleet_list");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a hot maxLoad edit still changes what fleet_list reports")
|
||||
void reloadedMaxLoadStillChangesWhatFleetListReports(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, STARTUP);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = new ConfigRef(file, cfg);
|
||||
|
||||
FleetMcp.CapacitySource source = Fleetd.capacitySource(config, cfg, _ -> 0);
|
||||
assertEquals(3, source.maxLoad().apply("terra"),
|
||||
"sanity: maxLoad reads 3 from the startup config before any reload");
|
||||
|
||||
Files.writeString(file, WITH_CHANGED_MAX_LOAD);
|
||||
assertTrue(config.reload().applied());
|
||||
|
||||
assertEquals(9, source.maxLoad().apply("terra"),
|
||||
"maxLoad must stay hot — the SAME CapacitySource instance must reflect a reloaded "
|
||||
+ "maxLoad without a restart, exactly like credentialId. Freezing the whole "
|
||||
+ "CapacitySource against the startup snapshot (rather than only its "
|
||||
+ "configuredProfiles set) would trade fleetd #416 for its mirror image.");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,89 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #248: this is the test that was actually missing. {@code Fleetd.main} builds its {@code
|
||||
* CompletionResolver} from an 8-argument constructor, and the ticket's own measurement proved two
|
||||
* ways to silently unwire it — both compiled with 0 errors and left every existing test green:
|
||||
*
|
||||
* <ul>
|
||||
* <li>replacing the worktree/branch argument (the 8th) with {@code _ -> null} — drops
|
||||
* fleetd#241's fallback-report location entirely;</li>
|
||||
* <li>replacing {@code backendErrorPatterns, backendErrorSink} (5th/6th) with {@code
|
||||
* BackendErrorPatternLookup.legacy(), BackendErrorSink.none()} — drops fleetd#201 Unit 5's
|
||||
* backend-error classification and cool-off entirely.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>Neither mutation could be caught by any test that constructs its own {@code
|
||||
* CompletionResolver} (every test before this one did exactly that) or by a test of {@link
|
||||
* Fleetd#worktreeBranchLookup}, {@link Fleetd#backendErrorPatternLookup}, or {@link
|
||||
* Fleetd#backendErrorSink} in isolation (see {@code FleetdWorktreeBranchLookupTest}, {@code
|
||||
* FleetdBackendErrorPatternLookupTest}, {@code FleetdBackendErrorSinkTest}) — those prove the
|
||||
* factories work, never that {@code main} still calls them. This class is a plain source-text
|
||||
* assertion on {@code Fleetd.java} — crude, but honest about what it checks, and it turns red the
|
||||
* instant the wiring is dropped, mirroring the same fallback shape {@link
|
||||
* FleetdFleetAppConstructionTest} already uses for a different constructor argument.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* CompletionResolver} and never runs {@code main}.
|
||||
*/
|
||||
class FleetdCompletionResolverWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still names backendErrorPatterns and backendErrorSink")
|
||||
void backendErrorArgumentsAreStillNamedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"exhaustionSink, backendErrorPatterns, backendErrorSink, System::nanoTime,"),
|
||||
"CompletionResolver's construction call must still pass backendErrorPatterns and "
|
||||
+ "backendErrorSink as its 5th/6th arguments. Replacing them with "
|
||||
+ "BackendErrorPatternLookup.legacy()/BackendErrorSink.none() (fleetd #248's measured "
|
||||
+ "mutation) compiles with 0 errors and leaves every behavioural test green — this "
|
||||
+ "source check is what must go red instead.");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] CompletionResolver's construction call still passes worktreeBranchLookup(sessions::roster)")
|
||||
void worktreeBranchLookupIsStillPassedAtTheCallSite() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains("worktreeBranchLookup(sessions::roster)"),
|
||||
"CompletionResolver's construction call must still pass worktreeBranchLookup(sessions::roster) "
|
||||
+ "as its 8th (last) argument. Replacing it with the inert `_ -> null` (fleetd #248's "
|
||||
+ "other measured mutation) compiles with 0 errors and leaves every behavioural test "
|
||||
+ "green — this source check is what must go red instead.");
|
||||
assertFalse(source.contains("System::nanoTime,\n _ -> null"),
|
||||
"the worktree/branch argument must never regress to the inert `_ -> null` literal");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorPatterns is assigned from the extracted backendErrorPatternLookup(...) factory")
|
||||
void backendErrorPatternsComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorPatternLookup backendErrorPatterns = backendErrorPatternLookup(sessions::roster,"),
|
||||
"backendErrorPatterns must be assigned from Fleetd.backendErrorPatternLookup(...), not an "
|
||||
+ "inline lambda that a source check on the CompletionResolver call alone cannot see "
|
||||
+ "through");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] backendErrorSink is assigned from the extracted backendErrorSink(...) factory")
|
||||
void backendErrorSinkComesFromTheFactory() throws Exception {
|
||||
String source = fleetdSource();
|
||||
assertTrue(source.contains(
|
||||
"BackendErrorSink backendErrorSink = backendErrorSink(sessions, () -> config.get().profiles(),"),
|
||||
"backendErrorSink must be assigned from Fleetd.backendErrorSink(...), not an inline lambda "
|
||||
+ "that a source check on the CompletionResolver call alone cannot see through");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,106 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertEquals;
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertNotNull;
|
||||
import static org.junit.jupiter.api.Assertions.assertSame;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #474: proves the exact wiring {@code Fleetd.main} uses to construct its live {@code
|
||||
* ConfigRef} — {@code new ConfigRef(configPath, cfg, Fleetd::assertChartersNameOnlyRegisteredTools)}
|
||||
* — actually makes {@link ConfigRef#reload()} refuse a charter that names an MCP tool the server
|
||||
* does not register, the same way {@code Fleetd.main} itself refuses one at startup (see {@code
|
||||
* FleetdStartupValidationTest#mainRefusesACharterNamingAnUnregisteredTool}).
|
||||
*
|
||||
* <p>{@code dev.ltms.fleet.config.ConfigRefTest} pins the same behaviour through a locally-built
|
||||
* {@code Consumer<FleetConfig>} adapter that calls the same production {@code CharterToolSurface}
|
||||
* method, because that test lives in {@code dev.ltms.fleet.config} and cannot see {@code
|
||||
* Fleetd#assertChartersNameOnlyRegisteredTools} (package-private to {@code dev.ltms.fleet}). This
|
||||
* class is the companion proof that lives where the real method reference is visible, so the literal
|
||||
* expression {@code Fleetd::assertChartersNameOnlyRegisteredTools} — not just an equivalent — is
|
||||
* what gets exercised. {@code Fleetd.main} itself cannot be driven this far in a unit test: every
|
||||
* fixture in {@code FleetdStartupValidationTest} is deliberately invalid so {@code main} throws
|
||||
* before opening a socket, binding Javalin, or doing anything else with a real side effect, so a
|
||||
* test cannot get {@code main} far enough to hold a running daemon it could then reload — this test
|
||||
* builds the {@code ConfigRef} the same way {@code main} does and drives {@link ConfigRef#reload()}
|
||||
* directly instead, the same shape {@code FleetdExhaustionDetectionArmedWiringTest} and its
|
||||
* siblings already use for the rest of {@code Fleetd.main}'s wiring.
|
||||
*/
|
||||
class FleetdConfigRefCharterToolSurfaceWiringTest {
|
||||
|
||||
private static final String BASE = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
sonnet:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: sonnet
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
@Test
|
||||
void reloadRefusesACharterNamingAnUnregisteredToolThroughFleetdsOwnWiring(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: |
|
||||
Send the final handoff through fleet_reply.
|
||||
""");
|
||||
ConfigRef config = new ConfigRef(f, FleetConfig.load(f),
|
||||
Fleetd::assertChartersNameOnlyRegisteredTools);
|
||||
FleetConfig before = config.get();
|
||||
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: |
|
||||
Send the final handoff through bridge_send.
|
||||
""");
|
||||
ConfigRef.Outcome out = config.reload();
|
||||
|
||||
assertFalse(out.applied());
|
||||
assertNotNull(out.error());
|
||||
assertTrue(out.error().contains("dev"), out.error());
|
||||
assertTrue(out.error().contains("bridge_send"), out.error());
|
||||
assertSame(before, config.get());
|
||||
}
|
||||
|
||||
@Test
|
||||
void reloadAcceptsACharterNamingOnlyRegisteredToolsThroughFleetdsOwnWiring(@TempDir Path dir)
|
||||
throws Exception {
|
||||
Path f = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: old charter
|
||||
""");
|
||||
ConfigRef config = new ConfigRef(f, FleetConfig.load(f),
|
||||
Fleetd::assertChartersNameOnlyRegisteredTools);
|
||||
|
||||
Files.writeString(f, BASE + """
|
||||
fleet:
|
||||
charters:
|
||||
dev: |
|
||||
Send the final handoff through fleet_reply.
|
||||
""");
|
||||
ConfigRef.Outcome out = config.reload();
|
||||
|
||||
assertTrue(out.applied());
|
||||
assertEquals("config reloaded", out.summary());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,80 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #474 follow-up: {@code Fleetd.main} builds its live {@code ConfigRef} from the
|
||||
* three-argument constructor, {@code new ConfigRef(configPath, cfg,
|
||||
* Fleetd::assertChartersNameOnlyRegisteredTools)}, so a reload runs the same charter tool-surface
|
||||
* gate startup does (see {@link dev.ltms.fleet.config.ConfigRef}'s class doc, "fleetd #474" bullet).
|
||||
* {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} prove the
|
||||
* three-argument constructor and the {@code Fleetd.assertChartersNameOnlyRegisteredTools} adapter
|
||||
* work correctly together — both build their OWN {@code ConfigRef} with that constructor, so neither
|
||||
* proves {@code main} still CHOOSES the three-argument form over the plain two-argument {@code new
|
||||
* ConfigRef(configPath, cfg)}.
|
||||
*
|
||||
* <p>Measured directly: reverting {@code Fleetd.java}'s {@code config} local to the two-argument
|
||||
* constructor compiles with 0 errors and leaves the entire 1633-test suite green — including every
|
||||
* {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} case — because
|
||||
* neither of those tests constructs its {@code ConfigRef} through {@code main}; both build their own
|
||||
* instance directly, wired with the check by hand. That silent regression is exactly the shape
|
||||
* {@link FleetdBackendQuarantineWiringTest}, {@link FleetdLeadSeatWiringTest} and {@link
|
||||
* FleetdCompletionResolverWiringTest} already guard against for their own constructor arguments —
|
||||
* this class is the same class of gap for fleetd #474's {@code extraValidation} argument, following
|
||||
* their approach.
|
||||
*
|
||||
* <p><b>This test checks source text, not runtime behaviour.</b> It never constructs a {@code
|
||||
* ConfigRef} and never runs {@code main} — a green result here proves only that the exact text
|
||||
* {@code main} contains is the three-argument construction with {@code
|
||||
* Fleetd::assertChartersNameOnlyRegisteredTools}. It does not prove that call actually executes at
|
||||
* startup (no test here starts the daemon), and it does not prove the reload gate itself works —
|
||||
* only {@code ConfigRefTest} and {@code FleetdConfigRefCharterToolSurfaceWiringTest} prove the
|
||||
* behaviour; only a live daemon proves the wiring runs.
|
||||
*/
|
||||
class FleetdConfigRefWiringTest {
|
||||
|
||||
private static String fleetdSource() throws Exception {
|
||||
return Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("[SOURCE TEXT] main still builds config from the three-argument ConfigRef constructor")
|
||||
void mainStillWiresTheThreeArgumentConfigRefConstructor() throws Exception {
|
||||
String source = fleetdSource();
|
||||
|
||||
// A broken read (wrong working directory, wrong path, a file that came back empty) would
|
||||
// make the assertFalse below pass vacuously — a "clean" negative check that actually
|
||||
// checked nothing. Guard against that first, with an anchor that has nothing to do with
|
||||
// this mutation, so a bad read fails loudly here instead of silently proving nothing below.
|
||||
assertTrue(source.contains("public final class Fleetd"),
|
||||
"fleetdSource() did not read anything usable — src/main/java/dev/ltms/fleet/Fleetd.java "
|
||||
+ "did not come back containing its own class declaration. The assertFalse below "
|
||||
+ "would pass vacuously on a broken read; fix the read before trusting this test.");
|
||||
|
||||
assertTrue(source.contains(
|
||||
"ConfigRef config = new ConfigRef(configPath, cfg, "
|
||||
+ "Fleetd::assertChartersNameOnlyRegisteredTools);"),
|
||||
"Fleetd.main's config local must still be built from the three-argument ConfigRef "
|
||||
+ "constructor, with Fleetd::assertChartersNameOnlyRegisteredTools as "
|
||||
+ "extraValidation. Reverting to the plain two-argument constructor (fleetd #474's "
|
||||
+ "measured M2 regression) compiles with 0 errors and leaves the whole suite green — "
|
||||
+ "including ConfigRefTest and FleetdConfigRefCharterToolSurfaceWiringTest, because "
|
||||
+ "neither builds its ConfigRef through main — this source check is what must go "
|
||||
+ "red instead. A reverted daemon would accept, through a reload with no restart, "
|
||||
+ "exactly the charter that #469/#474 already refuse at startup.");
|
||||
|
||||
// Negative form of the same check: the pre-#474 two-argument call, if it ever reappears at
|
||||
// this declaration, must not be mistaken for the three-argument one by a looser
|
||||
// positive-only check — this is the M2 mutation this test exists to kill.
|
||||
assertFalse(source.contains("ConfigRef config = new ConfigRef(configPath, cfg);"),
|
||||
"main's config local must never regress to the plain two-argument ConfigRef "
|
||||
+ "constructor — that drops the reload-path charter check (fleetd #474's measured "
|
||||
+ "M2 mutation) with no other test catching it");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,31 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-185: {@code ConnectionIdentity} must resolve a caller's pane on EITHER herdr daemon (a
|
||||
* lead's MCP connection resolves against the lead daemon; a member's against the member daemon).
|
||||
* Pinning {@code PaneLocator} to {@code memberHerdr} alone — the bug this guards against — leaves
|
||||
* every lead's own connection unresolvable ({@code callerTerminal == null}) the moment
|
||||
* {@code memberHerdrSocket} names a second daemon, which breaks {@code fleet_reply}/{@code
|
||||
* fleet_ask} and {@code fleet_whoami} for a lead. A unit test on {@link
|
||||
* dev.ltms.fleet.herdr.PaneLocator} alone (see {@code PaneLocatorTest}) proves the class CAN
|
||||
* search two clients, but not that {@code Fleetd.main} actually wires it that way — hence this
|
||||
* source-level assertion, the same technique {@code FleetdHerdrControlConstructionTest} uses.
|
||||
*/
|
||||
class FleetdConnectionIdentityConstructionTest {
|
||||
@Test
|
||||
void connectionIdentitySearchesBothDaemonsNotJustTheMemberOne() throws Exception {
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new PaneLocator(memberHerdr)"),
|
||||
"PaneLocator must not be pinned to the member daemon alone — a lead's own "
|
||||
+ "connection resolves against the LEAD daemon and would never be found");
|
||||
assertTrue(source.contains("new PaneLocator(herdr, memberHerdr)"),
|
||||
"PaneLocator must search the lead daemon first, then the member daemon");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,137 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.inject.LiveExhaustedPatterns;
|
||||
import dev.ltms.fleet.mcp.FleetMcp;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.Map;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #446: {@code exhaustionDetectionArmed} must describe the LIVE config, not a startup
|
||||
* snapshot — the opposite of what fleetd #404 asked for. #404 pinned {@code exhaustedPattern} as
|
||||
* compiled once at startup (matching {@code CompletionResolver}'s then-frozen behaviour); #446's
|
||||
* whole point is that both sides — the classification {@code CompletionResolver} enforces and the
|
||||
* {@code exhaustionDetectionArmed} field this reports — now read the SAME live {@link
|
||||
* LiveExhaustedPatterns} instance, so a reload arms or disarms detection with no restart, and the
|
||||
* report can never disagree with the behaviour (the fleetd #404 rule, now upheld for real instead
|
||||
* of by freezing both sides).
|
||||
*
|
||||
* <p>This test needs a reload. At startup the two snapshots agree, so a test of only a newly
|
||||
* started daemon would not detect a live {@code config.get()} lookup in the report field.
|
||||
*
|
||||
* <p><b>The hotness proof criterion 1 demands:</b> run this against the fixed code (green) and
|
||||
* against the pre-#446 compile site — {@code Fleetd.main}'s {@code exhaustedPatternsByProfile}
|
||||
* compiled once into a {@code Map<String, Pattern>} at startup, handed to {@code quarantineSource}
|
||||
* — and show it fails there. See the class doc history above: that shape is exactly what
|
||||
* {@code reloadedPatternArmsDetectionWithNoRestart} below is written to catch, and it is the test
|
||||
* that used to assert the opposite (see git history for this file's pre-#446 version, which
|
||||
* asserted {@code assertFalse(...)} on the identical scenario this now asserts {@code assertTrue}
|
||||
* on).
|
||||
*/
|
||||
class FleetdExhaustionDetectionArmedWiringTest {
|
||||
|
||||
private static final String NO_PATTERN = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static final String WITH_PATTERN = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: terra
|
||||
exhaustedPattern: "usage limit"
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
@Test
|
||||
@DisplayName("reloading a profile's exhaustedPattern IN arms exhaustionDetectionArmed, no restart")
|
||||
void reloadedPatternArmsDetectionWithNoRestart(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, NO_PATTERN);
|
||||
ConfigRef config = new ConfigRef(file, FleetConfig.load(file));
|
||||
|
||||
FleetMcp.QuarantineSource before = Fleetd.quarantineSource(config, BackendQuarantine.none(),
|
||||
new LiveExhaustedPatterns(() -> config.get().profiles()), Map.of());
|
||||
assertFalse(before.exhaustedPatternArmed().apply("terra"),
|
||||
"no exhaustedPattern configured yet — must report unarmed");
|
||||
|
||||
Files.writeString(file, WITH_PATTERN);
|
||||
assertTrue(config.reload().applied());
|
||||
assertTrue(config.get().profiles().get("terra").hasExhaustedPattern());
|
||||
|
||||
// Re-read exhaustedPatternArmed off the SAME QuarantineSource built BEFORE the reload — no
|
||||
// new object, no new wiring — to prove the field itself is live, not merely that a freshly
|
||||
// built source would be. This is the exact axis the pre-#446 code fails on: the same
|
||||
// Function<String,Boolean> lambda, called again after a reload, sees the new answer only if
|
||||
// it re-reads config.get() on every call rather than a value captured earlier.
|
||||
assertTrue(before.exhaustedPatternArmed().apply("terra"),
|
||||
"exhaustionDetectionArmed must read the LIVE config on every call — fleetd #446 made "
|
||||
+ "this hot; it must reflect a reload with no restart and no new QuarantineSource");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("reloading exhaustedPattern OUT disarms exhaustionDetectionArmed, no restart")
|
||||
void reloadedPatternRemovalDisarmsDetectionWithNoRestart(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, WITH_PATTERN);
|
||||
ConfigRef config = new ConfigRef(file, FleetConfig.load(file));
|
||||
|
||||
FleetMcp.QuarantineSource source = Fleetd.quarantineSource(config, BackendQuarantine.none(),
|
||||
new LiveExhaustedPatterns(() -> config.get().profiles()), Map.of());
|
||||
assertTrue(source.exhaustedPatternArmed().apply("terra"),
|
||||
"exhaustedPattern is configured from the start — must report armed");
|
||||
|
||||
Files.writeString(file, NO_PATTERN);
|
||||
assertTrue(config.reload().applied());
|
||||
|
||||
assertFalse(source.exhaustedPatternArmed().apply("terra"),
|
||||
"removing exhaustedPattern and reloading must disarm detection with no restart — an "
|
||||
+ "operator turning detection off (e.g. while debugging a false positive) "
|
||||
+ "must see that reflected immediately, exactly like arming it is");
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile with no exhaustedPattern configured is reported unarmed")
|
||||
void aProfileWithNoPatternIsUnarmed(@TempDir Path dir) throws Exception {
|
||||
// fleetd #404's second direction, carried forward: this test alone fails if
|
||||
// exhaustedPatternArmed degenerates to a constant `profile -> true` — it needs at least one
|
||||
// profile that is genuinely unarmed to catch that.
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, WITH_PATTERN);
|
||||
ConfigRef config = new ConfigRef(file, FleetConfig.load(file));
|
||||
|
||||
FleetMcp.QuarantineSource source = Fleetd.quarantineSource(config, BackendQuarantine.none(),
|
||||
new LiveExhaustedPatterns(() -> config.get().profiles()), Map.of());
|
||||
|
||||
assertTrue(source.exhaustedPatternArmed().apply("terra"),
|
||||
"a profile whose exhaustedPattern is configured must report armed");
|
||||
assertFalse(source.exhaustedPatternArmed().apply("sonnet"),
|
||||
"a profile absent from config entirely must not report armed");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,175 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import ch.qos.logback.classic.Level;
|
||||
import ch.qos.logback.classic.Logger;
|
||||
import ch.qos.logback.classic.spi.ILoggingEvent;
|
||||
import ch.qos.logback.core.read.ListAppender;
|
||||
import dev.ltms.fleet.config.ConfigRef;
|
||||
import dev.ltms.fleet.config.FleetConfig;
|
||||
import dev.ltms.fleet.guard.SubscriptionGuard;
|
||||
import dev.ltms.fleet.herdr.AgentControl;
|
||||
import dev.ltms.fleet.herdr.FakeHerdr;
|
||||
import dev.ltms.fleet.herdr.WorkspaceControl;
|
||||
import dev.ltms.fleet.inject.ExhaustionSink;
|
||||
import dev.ltms.fleet.member.ClaudeCodeLauncher;
|
||||
import dev.ltms.fleet.placement.BackendQuarantine;
|
||||
import dev.ltms.fleet.session.SessionManager;
|
||||
import org.junit.jupiter.api.DisplayName;
|
||||
import org.junit.jupiter.api.Test;
|
||||
import org.junit.jupiter.api.io.TempDir;
|
||||
import org.slf4j.LoggerFactory;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import java.util.HashMap;
|
||||
import java.util.List;
|
||||
import java.util.Map;
|
||||
import java.util.Set;
|
||||
import java.util.concurrent.TimeUnit;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* fleetd #446 follow-up (round 3): round 2's {@link FleetdUsageLimitFixWarningTest} calls {@link
|
||||
* Fleetd#usageLimitFixWarning}/{@link Fleetd#usageLimitFixWarningNoModel} directly, which pins
|
||||
* the two methods' TEXT but cannot see whether {@link Fleetd#exhaustionSink}'s {@code log.warn}
|
||||
* call actually invokes either one, or invokes the right one. A mutation battery run against
|
||||
* round 2's merge (1592 green tests) proved both gaps real:
|
||||
*
|
||||
* <ul>
|
||||
* <li><b>Cell A</b> — replacing the whole ternary result with a literal string
|
||||
* ({@code "usage-limit fix: MUTANT"}) left every test green;</li>
|
||||
* <li><b>Cell B</b> — swapping the ternary's two branches, so a profile WITH a {@code model:}
|
||||
* gets the no-model fallback message and vice versa, was measured unmeasured by the lead
|
||||
* but the existing test file shows no call that would notice it either.</li>
|
||||
* </ul>
|
||||
*
|
||||
* <p>This class drives {@link Fleetd#exhaustionSink} — the actual factory {@code main} wires,
|
||||
* since round 3's extraction pulled it out of the inline lambda for exactly this reason — and
|
||||
* asserts on the REAL text a {@link ListAppender} attached to {@code Fleetd}'s own logger
|
||||
* captures, mirroring {@code ExhaustedPatternGapReportTest}'s idiom. Chosen over a source-text
|
||||
* assertion (the {@code FleetMcpAuthzTest} {@code everyRegisteredToolHasItsHandlerActionPinned}
|
||||
* idiom) because that route can prove Cell A (the ternary is still there, still built from the
|
||||
* two named methods) but not Cell B (which branch a given profile actually reaches) — this route
|
||||
* answers both from one mechanism, since it reads what the sink actually logged for each shape of
|
||||
* profile.
|
||||
*
|
||||
* <p>Each test filters for the WARN line starting {@code "usage-limit fix:"} specifically — {@link
|
||||
* Fleetd#exhaustionSink} also logs a separate {@code "credential '...' quarantined for..."} WARN
|
||||
* on every call, and asserting against the wrong one would pass or fail for the wrong reason. That
|
||||
* prefix survives even under the M5 mutation above (the mutant keeps the tag, only the body
|
||||
* becomes {@code "MUTANT"}), so the filter itself is not what either mutation defeats.
|
||||
*/
|
||||
class FleetdExhaustionSinkWarningTest {
|
||||
|
||||
private static final String YAML = """
|
||||
bind:
|
||||
host: 127.0.0.1
|
||||
port: 8765
|
||||
herdrSocket: ~/.config/herdr/herdr.sock
|
||||
profiles:
|
||||
terra:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
model: claude-opus-5
|
||||
gx:
|
||||
baseUrl: http://gx00.gw:8000
|
||||
guard:
|
||||
offSubscriptionHosts:
|
||||
- gx00.gw
|
||||
""";
|
||||
|
||||
private static SessionManager emptyRosterSessions() {
|
||||
FakeHerdr h = new FakeHerdr();
|
||||
FleetConfig.Profile dummy = new FleetConfig.Profile(
|
||||
"dummy", "http://gx00.gw:8000", "coder", null, "FLEETD_WORKER_TOKEN", null,
|
||||
"tab", "fleetd-workers", "worker: {profile} #{n}", null, null, null);
|
||||
ClaudeCodeLauncher launcher = new ClaudeCodeLauncher(new AgentControl(h), new WorkspaceControl(h),
|
||||
new SubscriptionGuard(Set.of("gx00.gw")), Map.of(dummy.profile(), dummy), dummy.profile(), _ -> "tok");
|
||||
// Never acquires a session — exhaustionSink's roster lookup is expected to miss and fall
|
||||
// back to the profileHint argument, exactly like OpenCodeLauncher's real call site does
|
||||
// (fleetd #234) — so the sink under test never needs a populated roster.
|
||||
return new SessionManager(launcher);
|
||||
}
|
||||
|
||||
private static ListAppender<ILoggingEvent> attach() {
|
||||
Logger logger = (Logger) LoggerFactory.getLogger(Fleetd.class);
|
||||
ListAppender<ILoggingEvent> appender = new ListAppender<>();
|
||||
appender.start();
|
||||
logger.addAppender(appender);
|
||||
return appender;
|
||||
}
|
||||
|
||||
private static void detach(ListAppender<ILoggingEvent> appender) {
|
||||
((Logger) LoggerFactory.getLogger(Fleetd.class)).detachAppender(appender);
|
||||
}
|
||||
|
||||
/** The one WARN line {@code exhaustionSink} builds from the ternary under test, or {@code null}. */
|
||||
private static String fixWarning(ListAppender<ILoggingEvent> appender) {
|
||||
List<String> warns = appender.list.stream()
|
||||
.filter(e -> e.getLevel() == Level.WARN)
|
||||
.map(ILoggingEvent::getFormattedMessage)
|
||||
.filter(m -> m.startsWith("usage-limit fix:"))
|
||||
.toList();
|
||||
assertTrue(warns.size() == 1,
|
||||
"expected exactly one 'usage-limit fix:' WARN per onExhausted call, got " + warns.size()
|
||||
+ ": " + warns);
|
||||
return warns.getFirst();
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile WITH a configured model gets the actionable models.allow fix, not the fallback")
|
||||
void profileWithModelGetsTheActionableFix(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, YAML);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = ConfigRef.fixed(cfg);
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
Map<String, String> reasonByCredential = new HashMap<>();
|
||||
ExhaustionSink sink = Fleetd.exhaustionSink(emptyRosterSessions(), config, quarantine,
|
||||
reasonByCredential, cfg);
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
sink.onExhausted("term_x", "The usage limit has been reached", "terra");
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
String warning = fixWarning(appender);
|
||||
assertTrue(warning.contains("terra"), "must name the profile: " + warning);
|
||||
assertTrue(warning.contains("claude-opus-5"), "must name the model: " + warning);
|
||||
assertTrue(warning.contains("enabled: false"), "must name the action: " + warning);
|
||||
assertTrue(warning.contains("models.allow"), "must name where the action goes: " + warning);
|
||||
assertFalse(warning.contains("has no model: configured"),
|
||||
"a profile WITH a model must not get the no-model fallback text: " + warning);
|
||||
}
|
||||
|
||||
@Test
|
||||
@DisplayName("a profile with NO configured model gets the fallback, never a fix that does not exist")
|
||||
void profileWithNoModelGetsTheFallbackNotAFix(@TempDir Path dir) throws Exception {
|
||||
Path file = dir.resolve("fleetd.yaml");
|
||||
Files.writeString(file, YAML);
|
||||
FleetConfig cfg = FleetConfig.load(file);
|
||||
ConfigRef config = ConfigRef.fixed(cfg);
|
||||
BackendQuarantine quarantine = new BackendQuarantine(() -> 0L, TimeUnit.MINUTES.toNanos(30));
|
||||
Map<String, String> reasonByCredential = new HashMap<>();
|
||||
ExhaustionSink sink = Fleetd.exhaustionSink(emptyRosterSessions(), config, quarantine,
|
||||
reasonByCredential, cfg);
|
||||
|
||||
ListAppender<ILoggingEvent> appender = attach();
|
||||
try {
|
||||
sink.onExhausted("term_y", "The usage limit has been reached", "gx");
|
||||
} finally {
|
||||
detach(appender);
|
||||
}
|
||||
|
||||
String warning = fixWarning(appender);
|
||||
assertTrue(warning.contains("gx"), "must name the profile: " + warning);
|
||||
assertTrue(warning.contains("has no model: configured"),
|
||||
"must say there is no model to gate by name: " + warning);
|
||||
assertFalse(warning.contains("enabled: false"),
|
||||
"a profile with no model: configured has no models.allow entry to flip — must "
|
||||
+ "not claim one exists: " + warning);
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,29 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
import static org.junit.jupiter.api.Assertions.assertTrue;
|
||||
|
||||
/**
|
||||
* CB-185: {@code FleetApp} must be constructed with BOTH herdr clients (the lead's and the
|
||||
* member's), never the raw lead-only {@code herdr}. Passing only {@code herdr} — the bug this
|
||||
* guards against — makes {@code GET /healthz} green while the member daemon is down (so every
|
||||
* spawn fails invisibly) and silently drops every member workspace from {@code GET /sessions}.
|
||||
* A behavioural test on {@code FleetApp} alone (see {@code FleetAppTwoDaemonTest}) proves the
|
||||
* class merges/gates correctly when given two clients, but not that {@code Fleetd.main} actually
|
||||
* passes it two — hence this source-level assertion, mirroring
|
||||
* {@code FleetdHerdrControlConstructionTest}.
|
||||
*/
|
||||
class FleetdFleetAppConstructionTest {
|
||||
@Test
|
||||
void fleetAppIsConstructedWithBothHerdrDaemons() throws Exception {
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new FleetApp(herdr, workers,"),
|
||||
"FleetApp must not be constructed with the lead-only herdr client");
|
||||
assertTrue(source.contains("new FleetApp(herdr, memberHerdr, workers,"),
|
||||
"FleetApp must be constructed with both the lead and the member herdr client");
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,17 @@
|
||||
package dev.ltms.fleet;
|
||||
|
||||
import java.nio.file.Files;
|
||||
import java.nio.file.Path;
|
||||
import org.junit.jupiter.api.Test;
|
||||
|
||||
import static org.junit.jupiter.api.Assertions.assertFalse;
|
||||
|
||||
class FleetdHerdrControlConstructionTest {
|
||||
@Test
|
||||
void fleetdDelegatesStatefulControlsToTheRouter() throws Exception {
|
||||
// AgentControl caches paneByTerminal, so the router must be its only production factory.
|
||||
String source = Files.readString(Path.of("src/main/java/dev/ltms/fleet/Fleetd.java"));
|
||||
assertFalse(source.contains("new AgentControl("));
|
||||
assertFalse(source.contains("new WorkspaceControl("));
|
||||
}
|
||||
}
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user